diff --git a/.gitattributes b/.gitattributes
index a6344aac8c09253b3b630fb776ae94478aa0275b..6e925fa7f24368fe75448c8fc482c470796d9475 100644
--- a/.gitattributes
+++ b/.gitattributes
@@ -33,3 +33,224 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
+debug_dino_init.jpg filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/UI/Inter_18pt-Bold.ttf filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_100.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_101.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_102.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_103.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_104.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_105.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_106.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_107.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_108.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_109.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_110.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_111.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_112.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_113.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_114.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_115.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_116.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_117.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_118.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_119.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_120.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_121.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_122.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_123.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_124.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_125.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_126.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_127.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_128.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_129.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_130.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_131.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_132.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_133.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_134.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_135.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_136.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_137.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_138.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_139.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_140.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_141.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_142.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_143.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_144.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_145.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_146.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_147.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_148.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_149.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_150.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_151.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_152.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_153.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_154.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_155.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_156.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_157.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_158.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_159.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_160.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_161.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_162.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_163.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_164.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_165.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_166.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_167.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_168.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_169.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_170.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_171.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_172.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_173.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_174.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_175.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_176.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_177.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_178.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_179.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_180.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_181.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_182.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_183.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_184.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_185.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_186.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_187.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_188.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_189.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_190.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_191.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_192.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_193.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_194.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_195.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_196.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_197.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_198.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_199.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_200.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_201.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_202.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_203.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_204.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_205.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_206.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_207.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_208.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_209.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_210.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_211.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_212.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_213.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_214.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_215.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_216.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_217.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_218.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_219.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_220.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_221.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_222.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_223.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_224.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_225.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_226.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_227.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_228.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_229.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_230.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_231.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_232.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_233.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_234.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_235.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_236.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_237.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_238.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_239.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_240.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_241.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_242.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_243.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_244.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_245.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_246.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_247.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_248.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_249.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_250.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_251.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_252.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_253.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_254.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_255.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_256.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_257.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_258.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_259.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_260.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_261.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_262.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_263.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_264.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_265.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_266.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_267.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_268.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_269.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_54.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_55.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_56.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_57.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_58.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_59.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_60.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_61.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_62.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_63.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_64.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_65.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_66.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_67.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_68.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_69.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_70.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_71.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_72.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_73.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_74.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_75.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_76.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_77.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_78.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_79.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_80.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_81.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_82.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_83.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_84.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_85.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_86.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_87.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_88.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_89.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_90.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_91.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_92.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_93.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_94.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_95.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_96.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_97.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_98.png filter=lfs diff=lfs merge=lfs -text
+third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_99.png filter=lfs diff=lfs merge=lfs -text
+third_party/hamer/example_data/test1.jpg filter=lfs diff=lfs merge=lfs -text
+third_party/hamer/example_data/test4.jpg filter=lfs diff=lfs merge=lfs -text
+third_party/hamer/example_data/test5.jpg filter=lfs diff=lfs merge=lfs -text
diff --git a/.gitignore b/.gitignore
new file mode 100644
index 0000000000000000000000000000000000000000..b2d60c811864c81715d86f14564df73417068878
--- /dev/null
+++ b/.gitignore
@@ -0,0 +1,16 @@
+dataset-generator/
+out/
+__pycache__/
+*.pyc
+outputs/
+out/
+__pycache__/
+*.pyc
+processed_dataset/
+mmpose/
+gvhmr.egg-info/
+Grounded-SAM-2/
+.cache/
+third-party/
+assets/
+*.mp4
\ No newline at end of file
diff --git a/LICENSE b/LICENSE
new file mode 100644
index 0000000000000000000000000000000000000000..0e17249dc9ad40f52cf22cc6eebddb34e1e4da0e
--- /dev/null
+++ b/LICENSE
@@ -0,0 +1,36 @@
+NVIDIA License
+
+1. Definitions
+
+“Licensor” means any person or entity that distributes its Work.
+“Work” means (a) the original work of authorship made available under this license, which may include software, documentation, or other files, and (b) any additions to or derivative works thereof that are made available under this license.
+The terms “reproduce,” “reproduction,” “derivative works,” and “distribution” have the meaning as provided under U.S. copyright law; provided, however, that for the purposes of this license, derivative works shall not include works that remain separable from, or merely link (or bind by name) to the interfaces of, the Work.
+Works are “made available” under this license by including in or with the Work either (a) a copyright notice referencing the applicability of this license to the Work, or (b) a copy of this license.
+
+2. License Grant
+
+2.1 Copyright Grant. Subject to the terms and conditions of this license, each Licensor grants to you a perpetual, worldwide, non-exclusive, royalty-free, copyright license to use, reproduce, prepare derivative works of, publicly display, publicly perform, sublicense and distribute its Work and any resulting derivative works in any form.
+
+3. Limitations
+
+3.1 Redistribution. You may reproduce or distribute the Work only if (a) you do so under this license, (b) you include a complete copy of this license with your distribution, and (c) you retain without modification any copyright, patent, trademark, or attribution notices that are present in the Work.
+
+3.2 Derivative Works. You may specify that additional or different terms apply to the use, reproduction, and distribution of your derivative works of the Work (“Your Terms”) only if (a) Your Terms provide that the use limitation in Section 3.3 applies to your derivative works, and (b) you identify the specific derivative works that are subject to Your Terms. Notwithstanding Your Terms, this license (including the redistribution requirements in Section 3.1) will continue to apply to the Work itself.
+
+3.3 Use Limitation. The Work and any derivative works thereof only may be used or intended for use non-commercially. Notwithstanding the foregoing, NVIDIA Corporation and its affiliates may use the Work and any derivative works commercially. As used herein, “non-commercially” means for non-commercial academic purposes only.
+
+3.4 Patent Claims. If you bring or threaten to bring a patent claim against any Licensor (including any claim, cross-claim or counterclaim in a lawsuit) to enforce any patents that you allege are infringed by any Work, then your rights under this license from such Licensor (including the grant in Section 2.1) will terminate immediately.
+
+3.5 Trademarks. This license does not grant any rights to use any Licensor’s or its affiliates’ names, logos, or trademarks, except as necessary to reproduce the notices described in this license.
+
+3.6 Termination. If you violate any term of this license, then your rights under this license (including the grant in Section 2.1) will terminate immediately.
+
+4. Disclaimer of Warranty.
+
+THE WORK IS PROVIDED “AS IS” WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, EITHER EXPRESS OR IMPLIED, INCLUDING WARRANTIES OR CONDITIONS OF
+MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, TITLE OR NON-INFRINGEMENT. YOU BEAR THE RISK OF UNDERTAKING ANY ACTIVITIES UNDER THIS LICENSE.
+
+5. Limitation of Liability.
+
+EXCEPT AS PROHIBITED BY APPLICABLE LAW, IN NO EVENT AND UNDER NO LEGAL THEORY, WHETHER IN TORT (INCLUDING NEGLIGENCE), CONTRACT, OR OTHERWISE SHALL ANY LICENSOR BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES ARISING OUT OF OR RELATED TO THIS LICENSE, THE USE OR INABILITY TO USE THE WORK (INCLUDING BUT NOT LIMITED TO LOSS OF GOODWILL, BUSINESS INTERRUPTION, LOST PROFITS OR DATA, COMPUTER FAILURE OR MALFUNCTION, OR ANY OTHER DAMAGES OR LOSSES), EVEN IF THE LICENSOR HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH DAMAGES.
+
diff --git a/README.md b/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..d89c2fed4218ac2626923fcb8061aea5e243bc88
--- /dev/null
+++ b/README.md
@@ -0,0 +1,75 @@
+
+
GEM: A Generalist Model for Human Motion
+
+ Jiefeng Li
+ ·
+ Jinkun Cao
+ ·
+ Haotian Zhang
+ ·
+ Davis Rempe
+ ·
+ Jan Kautz
+ ·
+ Umar Iqbal
+ ·
+ Ye Yuan
+
+ ICCV 2025 (Highlight)
+
+

+
+
+
+
+
+
+
+
+**GEM** is a generalist model for human motion that handles multiple tasks with a single model, supporting diverse conditioning signals including video, keypoints, text, audio, and 3D keyframes.
+
+---
+
+## 📰 News
+- **[December 2025]** 📢 GENMO has been renamed to **GEM**.
+- **[October 2025]** 📢 The **GEM** codebase is **released!**
+ Stay tuned for the pretrained models and evaluation scripts.
+ Follow the [project page](https://research.nvidia.com/labs/dair/gem/) for updates and announcements.
+
+
+---
+
+
+## 🚀 Highlights
+
+GEM introduces a **unified generative framework** that connects motion estimation and generation through shared objectives.
+
+- **Unified framework:** Reframes motion estimation as *constrained generation*, allowing a single model to perform both tasks.
+- **Regression × Diffusion synergy:** Combines the accuracy of regression models with the diversity of diffusion-based generation.
+- **Estimation-guided training:** Trains effectively on in-the-wild datasets using only 2D or textual supervision.
+- **Multimodal conditioning:** Supports video, text, audio, 2D/3D keyframes, or even time-varying mixed inputs (e.g., video → text → video).
+- **Arbitrary-length motion:** Generates continuous, coherent sequences of any duration in one diffusion pass.
+- **State-of-the-art performance:** Achieves leading results on diverse motion estimation and generation benchmarks.
+
+For more details, visit the **[GEM project page →](https://research.nvidia.com/labs/dair/gem/)**
+
+---
+
+### Pretrained Models
+You can download pretrained models from [Google Drive](https://drive.google.com/file/d/1b1E84G7S0h2n5o0RmrcmKOhRKukOjgsJ/view?usp=sharing).
+
+## 📖 Paper & Citation
+
+**Paper:**
+[GENMO: A GENeralist Model for Human MOtion](https://arxiv.org/abs/2505.01425)
+*Jiefeng Li, Jinkun Cao, Haotian Zhang, Davis Rempe, Jan Kautz, Umar Iqbal, Ye Yuan*
+ICCV, 2025
+
+**BibTeX:**
+```bibtex
+@inproceedings{genmo2025,
+ title = {GENMO: A GENeralist Model for Human MOtion},
+ author = {Li, Jiefeng and Cao, Jinkun and Zhang, Haotian and Rempe, Davis and Kautz, Jan and Iqbal, Umar and Yuan, Ye},
+ booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
+ year = {2025}
+}
diff --git a/REPROCESS_AND_TEST.md b/REPROCESS_AND_TEST.md
new file mode 100644
index 0000000000000000000000000000000000000000..bc613898a6f889d8b6916801d36baceb6d5932c1
--- /dev/null
+++ b/REPROCESS_AND_TEST.md
@@ -0,0 +1,139 @@
+# Quick Reprocess and Test Instructions
+
+## Summary of Changes
+
+I've fixed the **rotation computation order** issue in `process_dataset.py`:
+
+### The Problem
+- Old: Compute relative rotation in Unity space, then convert to CV
+- Issue: This breaks GENMO's expected formula `R_c = R_w2c @ R_pel_w`
+
+### The Fix
+- New: Convert rotations to CV **first**, then compute relative rotation
+- This matches GENMO's training data convention
+
+## Step-by-Step Instructions
+
+### 1. Clean Old Processed Data
+```bash
+cd /root/miko/puni/train/PromptHMR/GENMO
+rm -rf processed_dataset/genmo_features/*.pt
+```
+
+### 2. Reprocess Your Dataset
+```bash
+# Replace paths as needed
+python third_party/GVHMR/tools/demo/process_dataset.py \
+ --input /path/to/your/unity_export \
+ --output processed_dataset \
+ --genmo --vitpose --smplx \
+ --consistency_check \
+ --consistency_check_frames 5
+```
+
+**Expected output:**
+- `[Kabsch] Frame 0 consistency: Rotation err = 0.00°, Translation err = <0.1m`
+- Processing should complete without errors
+
+### 3. Run Diagnosis
+```bash
+python diagnose_data.py
+```
+
+**Expected results (V2 fix):**
+```
+In-camera orientation errors (mean ± std):
+ Yaw: <5.00° ± <2.00° (was 9.44°)
+ Pitch: <5.00° ± <2.00° (was 10.95°)
+ Roll: <5.00° ± <2.00° (was 168.77° ← THE BUG!)
+
+World orientation errors (mean ± std):
+ Yaw: <5.00° ± <2.00° (was 55.10°)
+ Pitch: <5.00° ± <2.00° (was 4.12°)
+ Roll: <5.00° ± <2.00° (was 3.65°)
+
+Body pose error: <10.00° ± <20.00° (max: <100°)
+```
+
+### 4. If Errors Are Still High...
+
+The rotation fix addresses the **rotation computation order**, but if errors persist, check:
+
+#### A. Body Pose Export
+Your Unity export's `smplx_pose` might have issues. Check:
+```bash
+# Test if body_pose matches between Unity and processed data
+python test_single_frame.py
+```
+
+#### B. SMPL Model Mismatch
+GENMO uses `supermotion_v437coco17`. Verify your Unity uses the same:
+- Check betas (should be 10D, matching `shape.npz`)
+- Check body_pose structure (should be 63D = 21 joints × 3)
+
+#### C. Coordinate System Issues
+If roll is exactly 180° off, you might need the Z-180° fix after all (but only for specific camera setups).
+
+## What Changed in process_dataset.py
+
+### Line 267-281: Incam Rotation
+```python
+# OLD (BROKEN):
+R_rel_unity = R_cam_w.T @ R_pel_w
+R_cv = C @ R_rel_unity @ C
+
+# NEW (FIXED):
+R_cam_w_cv = C @ R_cam_w_unity @ C
+R_w2c_cv = R_cam_w_cv.T
+R_pel_w_cv = C @ R_pel_w_unity @ C
+R_pel_c_cv = R_w2c_cv @ R_pel_w_cv # GENMO's formula!
+```
+
+### Line 602-614: World Rotation
+```python
+# OLD: Recompute from Unity quaternions
+# NEW: Use pre-converted CV rotations
+R_c2w_cv = p['R_w2c_cv'].T
+R_pelvis_w_cv = R_c2w_cv @ R_pelvis_c_cv
+```
+
+### Line 665-671: Camera Matrix
+```python
+# OLD: Convert Unity T_wc with C4 @ T @ C4
+# NEW: Use pre-computed R_w2c_cv
+cam_T_w2c_cv[:3, :3] = p["R_w2c_cv"]
+```
+
+## Training
+
+After reprocessing with the fix, training should:
+- ✅ Start with loss ~1-5 (not 12)
+- ✅ Decrease steadily (not explode to 100+)
+- ✅ Converge to ~0.5-2.0 after sufficient epochs
+
+If loss still explodes:
+1. Check learning rate (might be too high for fine-tuning)
+2. Check data augmentation settings
+3. Verify batch size matches pretrained model's training setup
+
+## Files Modified
+
+1. [process_dataset.py](third_party/GVHMR/tools/demo/process_dataset.py)
+ - Lines 267-281: Fixed incam rotation derivation
+ - Lines 306-318: Added R_w2c_cv to return dict
+ - Lines 602-614: Use pre-converted rotations
+ - Lines 665-671: Use pre-computed camera matrix
+
+## Next Steps
+
+1. ✅ Reprocess dataset
+2. ✅ Verify diagnosis shows <5° errors
+3. ✅ Start training
+4. 📊 Monitor loss curve (should decrease, not explode)
+
+---
+
+**Quick Check**: If `diagnose_data.py` still shows 168° roll error after reprocessing, the changes didn't apply. Check:
+- Did you edit the correct `process_dataset.py` file?
+- Did you delete old `.pt` files before reprocessing?
+- Did the reprocessing script complete without errors?
diff --git a/ROTATION_FIX_SUMMARY.md b/ROTATION_FIX_SUMMARY.md
new file mode 100644
index 0000000000000000000000000000000000000000..d05dd08a1fdc1017edf3c6a33c0b5884ca09e569
--- /dev/null
+++ b/ROTATION_FIX_SUMMARY.md
@@ -0,0 +1,121 @@
+# Unity Dataset Rotation Fix - Summary
+
+## Problem Identified
+
+Your training loss was exploding (12 → 100+) because of a **180° rotation mismatch** between your processed data and the pretrained GENMO model's expected convention.
+
+### Diagnosis Results (Before Fix)
+- **In-camera roll error: 348.31°** ≈ -11.87° (180° flip issue)
+- **World yaw error: 32.81°**
+- **Body pose max error: 163.37°**
+- **Training loss: Exploding from 12 to 100+**
+
+### Root Cause
+The Z-180° rotation fix in [process_dataset.py:276](third_party/GVHMR/tools/demo/process_dataset.py#L276) was being applied during data processing, but the pretrained GENMO model was trained on data **without** this fix. This created a systematic rotation offset.
+
+## Changes Made
+
+### 1. Removed Z-180° Fix from Incam Rotation (Line 273-278)
+```python
+# OLD (BROKEN):
+R_final = R_cv @ R.from_euler("z", 180, degrees=True).as_matrix()
+global_orient_aa = R.from_matrix(R_final).as_rotvec()
+
+# NEW (FIXED):
+global_orient_aa = R.from_matrix(R_cv).as_rotvec()
+```
+
+### 2. Removed Z-180° Fix from World Rotation (Line 593-611)
+```python
+# OLD (BROKEN):
+fix_rot = R.from_euler("z", 180, degrees=True).as_matrix()
+R_cam_w_cv = fix_rot @ (C @ R_cam_w_unity @ C)
+pelvis_pos_w_cv = fix_rot @ pos_cv_raw
+
+# NEW (FIXED):
+R_cam_w_cv = C @ R_cam_w_unity @ C
+pelvis_pos_w_cv = pos_cv_raw
+```
+
+### 3. Removed Z-180° Fix from Camera Matrix (Line 662-667)
+```python
+# OLD (BROKEN):
+cam_T_wc_cv = fix_mat @ (C4 @ cam_T_wc @ C4)
+
+# NEW (FIXED):
+cam_T_wc_cv = C4 @ cam_T_wc @ C4
+```
+
+## Verification Steps
+
+### 1. Reprocess Your Dataset
+```bash
+# Delete old processed data
+rm -rf processed_dataset/genmo_features/*.pt
+
+# Reprocess with the fixed script
+python third_party/GVHMR/tools/demo/process_dataset.py \
+ --input path/to/unity_export \
+ --output processed_dataset \
+ --genmo --vitpose --smplx \
+ --consistency_check
+```
+
+### 2. Run Diagnosis Again
+```bash
+python diagnose_data.py
+```
+
+### Expected Results (After Fix)
+- **In-camera roll error: <10°** (instead of 348°)
+- **World orientation errors: <5°** for all axes
+- **Body pose error: <10° mean**
+- **Training loss: Should stabilize around 0.5-2.0**
+
+### 3. Resume Training
+```bash
+# Your training should now converge properly
+python train.py --config configs/genmo_lg.yaml
+```
+
+## Why Kabsch Consistency Check Still Passed
+
+The Kabsch alignment check (0.00° error) only verifies **internal geometric consistency** between incam and world SMPL parameters using your exported camera transforms. It does NOT check if your data matches the pretrained model's convention.
+
+Think of it like this:
+- ✅ Your Unity → SMPL conversion is geometrically correct
+- ❌ But the coordinate convention doesn't match GENMO's training data
+
+## Additional Notes
+
+### If Errors Persist After Reprocessing
+
+1. **Check body pose errors**: If still >20° mean, your Unity export's `smplx_pose` might have issues
+2. **Check world translation**: Should be normalized (first frame at origin ± offset)
+3. **Verify Unity quaternion order**: Should be XYZW (not WXYZ)
+
+### Understanding the Z-180° Fix
+
+The Z-180° rotation is sometimes needed when converting between:
+- Unity's left-handed Y-up coordinate system
+- CV convention's right-handed Y-down system
+
+However, the pretrained GENMO model was trained with data that did NOT apply this fix after the basic Unity→CV conversion (Y-flip via `C = diag([1, -1, 1])`). Your processing pipeline was applying an extra 180° rotation that the model wasn't expecting.
+
+## Modified Files
+
+1. [third_party/GVHMR/tools/demo/process_dataset.py](third_party/GVHMR/tools/demo/process_dataset.py)
+ - Line 273-278: Removed Z-180° from incam rotation
+ - Line 593-611: Removed Z-180° from world rotation derivation
+ - Line 662-667: Removed Z-180° from camera matrices
+
+## Next Steps
+
+1. ✅ Reprocess dataset with fixed script
+2. ✅ Verify with `diagnose_data.py` (should show <10° errors)
+3. ✅ Resume training (loss should stabilize)
+4. 🎯 If training still diverges, check learning rate and batch size
+
+---
+
+**tl;dr**: The pretrained model expects rotations without Z-180° fix. Removed the fix from 3 places in process_dataset.py. Reprocess your data and training should converge.
diff --git a/ROTATION_FIX_V2_SUMMARY.md b/ROTATION_FIX_V2_SUMMARY.md
new file mode 100644
index 0000000000000000000000000000000000000000..87947f99bf3c0f6018fef9c8d85e59919733a1ba
--- /dev/null
+++ b/ROTATION_FIX_V2_SUMMARY.md
@@ -0,0 +1,196 @@
+# Unity Dataset Rotation Fix V2 - The Real Issue
+
+## Problem Evolution
+
+### First Attempt Results (After removing Z-180° fix)
+- **In-camera roll error: 168.77°** (improved from 348°, but still wrong)
+- **World yaw error: 55.10°**
+- Still not matching GENMO's expected convention
+
+## Root Cause - Matrix Multiplication Order
+
+After deep investigation of GENMO's codebase, the issue was found in **how we compute the incam rotation**:
+
+### GENMO's Expected Formula
+From `/third_party/GVHMR/hmr4d/utils/geo/hmr_global.py:100`:
+```python
+R_c = matrix_to_axis_angle(R_w2c @ R_w) # Camera rotation = R_w2c @ World_rotation
+```
+
+### Our Old Method (WRONG)
+```python
+# Step 1: Compute relative rotation in Unity space
+R_rel_unity = R_cam_w_unity.T @ R_pel_w_unity
+
+# Step 2: Convert to CV
+R_cv = C @ R_rel_unity @ C
+
+# Problem: Coordinate conversion happens AFTER computing relative rotation
+# This breaks the math because rotation composition is not commutative with basis changes
+```
+
+### New Method (CORRECT)
+```python
+# Step 1: Convert BOTH rotations to CV convention FIRST
+R_cam_w_cv = C @ R_cam_w_unity @ C # Camera-to-world in CV
+R_pel_w_cv = C @ R_pel_w_unity @ C # Pelvis-to-world in CV
+
+# Step 2: Compute relative rotation IN CV space
+R_w2c_cv = R_cam_w_cv.T # World-to-camera in CV
+R_pel_c_cv = R_w2c_cv @ R_pel_w_cv # GENMO's formula in CV space
+```
+
+## Why Order Matters
+
+The key insight: **R @ (C @ M @ C) ≠ C @ (R @ M) @ C** when changing coordinate systems.
+
+When you:
+1. ❌ Compute rotation in Unity space, then convert to CV → Wrong
+2. ✅ Convert rotations to CV, then compute relative rotation → Correct
+
+This is because rotation composition depends on the coordinate basis. The formula `R_w2c @ R_pel_w` assumes BOTH matrices are in the SAME coordinate system (CV).
+
+## Changes Made
+
+### 1. Fixed Incam Rotation Computation (Lines 267-281)
+
+```python
+# Get raw Unity quaternions
+cam_rot_w_quat = np.array(row["cam_rot_world"], dtype=np.float64)
+R_cam_w_unity = R.from_quat(cam_rot_w_quat).as_matrix()
+pel_rot_w_quat = np.array(row["pelvis_rot_world"], dtype=np.float64)
+R_pel_w_unity = R.from_quat(pel_rot_w_quat).as_matrix()
+
+# Convert to CV convention FIRST, then compute relative rotation
+# Model expects: global_orient_c = R_w2c @ R_pel_w (in CV convention)
+R_cam_w_cv = C @ R_cam_w_unity @ C # Camera-to-world in CV
+R_w2c_cv = R_cam_w_cv.T # World-to-camera in CV
+R_pel_w_cv = C @ R_pel_w_unity @ C # Pelvis-to-world in CV
+
+# Compute incam rotation in CV convention
+R_pel_c_cv = R_w2c_cv @ R_pel_w_cv # This matches GENMO's formula!
+global_orient_aa = R.from_matrix(R_pel_c_cv).as_rotvec().astype(np.float32)
+```
+
+### 2. Updated Return Values to Include CV Rotations (Lines 306-318)
+
+```python
+return {
+ "global_orient": global_orient_aa,
+ "body_pose": body_pose,
+ "betas": betas10,
+ "R_w2c_cv": R_w2c_cv, # Add for reuse
+ "R_pel_w_cv": R_pel_w_cv, # Add for reuse
+ ...
+}
+```
+
+### 3. Fixed World Rotation Derivation (Lines 602-614)
+
+```python
+# Use pre-converted rotations from parse_smpl_inputs_from_row
+for p in parsed:
+ R_pelvis_c_cv = R.from_rotvec(p['global_orient']).as_matrix()
+ R_c2w_cv = p['R_w2c_cv'].T # Camera-to-world (inverse of w2c)
+ R_pelvis_w_cv = R_c2w_cv @ R_pelvis_c_cv
+ all_go_w.append(R.from_matrix(R_pelvis_w_cv).as_rotvec())
+```
+
+### 4. Fixed Camera Matrix Construction (Lines 665-671)
+
+```python
+# Use pre-computed CV-convention rotation (not recompute from Unity!)
+cam_T_w2c_cv = np.eye(4, dtype=np.float32)
+cam_T_w2c_cv[:3, :3] = p["R_w2c_cv"].astype(np.float32)
+cam_pos_cv = C @ p["cam_pos_world"]
+cam_T_w2c_cv[:3, 3] = (-p["R_w2c_cv"] @ cam_pos_cv).astype(np.float32)
+```
+
+## Verification Steps
+
+### 1. Reprocess Dataset
+```bash
+# Clean old data
+rm -rf processed_dataset/genmo_features/*.pt
+
+# Reprocess
+python third_party/GVHMR/tools/demo/process_dataset.py \
+ --input path/to/unity_export \
+ --output processed_dataset \
+ --genmo --vitpose --smplx \
+ --consistency_check
+```
+
+### 2. Run Diagnosis
+```bash
+python diagnose_data.py
+```
+
+### Expected Results (After V2 Fix)
+- **In-camera orientation errors: <5°** for all axes (yaw, pitch, roll)
+- **World orientation errors: <5°** for all axes
+- **Body pose mean error: <5°**
+- **World translation error: <0.1m** (excluding the intentional 1.34m Y-offset)
+- **Training loss: Should converge to 0.5-2.0** instead of exploding
+
+## Understanding the Math
+
+### Why R_w2c @ R_pel_w?
+
+Think of applying rotations to a vector:
+1. Start with pelvis-local vector: `v_pelvis`
+2. Rotate to world: `v_world = R_pel_w @ v_pelvis`
+3. Rotate to camera: `v_camera = R_w2c @ v_world`
+4. Combine: `v_camera = R_w2c @ (R_pel_w @ v_pelvis) = (R_w2c @ R_pel_w) @ v_pelvis`
+
+So: `R_pel_c = R_w2c @ R_pel_w` (rotation composition follows the transformation chain)
+
+### Why Convert to CV First?
+
+Because GENMO was trained on data where ALL rotations are in CV convention. If you compute relative rotations in Unity space then convert, the mathematical relationship changes due to the basis transformation.
+
+It's like computing `A + B` vs. `f(A) + f(B)` - only works if `f` is linear (which coordinate transforms are for INDIVIDUAL rotations, but NOT for rotation composition).
+
+## Modified Files
+
+1. [third_party/GVHMR/tools/demo/process_dataset.py](third_party/GVHMR/tools/demo/process_dataset.py)
+ - Lines 267-281: Fixed incam rotation (convert to CV FIRST)
+ - Lines 306-318: Added R_w2c_cv and R_pel_w_cv to return dict
+ - Lines 602-614: Use pre-converted rotations for world derivation
+ - Lines 665-671: Use pre-computed R_w2c_cv for camera matrix
+
+## Technical Details
+
+### Coordinate Systems Involved
+
+1. **Unity World**: Left-handed, Y-up
+ - Camera: `cam_rot_world` (quaternion XYZW)
+ - Pelvis: `pelvis_rot_world` (quaternion XYZW)
+
+2. **CV Convention**: Right-handed, Y-down
+ - Conversion: `R_cv = C @ R_unity @ C` where `C = diag([1, -1, 1])`
+ - Flips Y-axis to convert handedness
+
+3. **SMPL**: Uses axis-angle representation (3D vectors)
+ - Magnitude = rotation angle (radians)
+ - Direction = rotation axis (right-hand rule)
+
+### Rotation Representation Chain
+
+```
+Unity Quat → Matrix → CV Matrix → Composition → CV Matrix → Axis-Angle
+ (XYZW) (3x3) (3x3) (R_w2c@R_w) (3x3) (3D vec)
+```
+
+Each step must preserve the rotation semantics in the target coordinate system.
+
+## Next Steps
+
+1. ✅ Reprocess dataset with V2 fix
+2. ✅ Verify with `diagnose_data.py` (expect <5° errors)
+3. ✅ Resume training
+4. 🎯 Monitor first 1000 steps - loss should decrease steadily
+
+---
+
+**tl;dr**: The issue was computing relative rotations in Unity space then converting to CV. Fixed by converting BOTH rotations to CV FIRST, then computing `R_pel_c = R_w2c @ R_pel_w` as GENMO expects. This matches the mathematical formula in the trained model.
diff --git a/SyntheticRecorder.cs b/SyntheticRecorder.cs
new file mode 100644
index 0000000000000000000000000000000000000000..ea8fa0708e20460302dfe96fb76bab2f3c55d839
--- /dev/null
+++ b/SyntheticRecorder.cs
@@ -0,0 +1,681 @@
+using UnityEngine;
+using System;
+using System.IO;
+using System.Collections;
+using System.Collections.Generic;
+using Newtonsoft.Json;
+using UnityEngine.SceneManagement;
+using System.Diagnostics; // Required for FFmpeg Process
+
+public class SyntheticRecorder : MonoBehaviour
+{
+ // --- CONFIGURATION CLASSES ---
+ [System.Serializable]
+ public class AvatarConfig
+ {
+ public string avatarName = "Avatar";
+ public GameObject avatarObject;
+ [Header("Animation")]
+ public Animator animator;
+ [Header("Retargeting Link")]
+ public HybridPoseCopier retargeter;
+ public List extraMeshes = new List();
+ public float specificPadding = 40f;
+ [Header("Keypoint Markers (COCO-17 order)")]
+ public List customMarkers = new List();
+ }
+
+ // --- JSON STRUCTURES ---
+ public class SequenceData { public List frames; }
+ public class FrameData { public int i; public float[] p, t, b; public int s; }
+
+ public class OutputMeta
+ {
+ public int frame_index;
+ public string image_path;
+ public string avatar_name;
+ public int face_id;
+ public int left_hand_id;
+ public int right_hand_id;
+ public float[] bbox;
+ public float[] kpts_2d;
+ public int[] kpts_vis;
+ public float[] bbox_clip;
+ public float[] cam_intrinsics;
+ // RAW UNITY TRANSFORMS - Python derives incam/global from these
+ public float[] cam_pos_world;
+ public float[] cam_rot_world;
+ public float[] pelvis_pos_world;
+ public float[] pelvis_rot_world;
+ // INCAM TRANSLATION (pre-converted to CV Y-flip)
+ public float[] smpl_incam_transl;
+ public float[] smpl_root_incam_transl;
+ public float smpl_root_world_scale;
+ public float[] kpts_3d_world;
+ public float[] smplx_pose;
+ public float[] smplx_betas;
+ }
+
+ [Header("Settings")]
+ public string inputFolderPath = "Assets/StreamingAssets";
+ public string outputFolder = "C:/Temp/SyntheticDataset";
+ public bool startRecordingOnPlay = true;
+ public bool showDebugUI = true;
+ [Tooltip("If true, saves depth_xxxxx.png files to check what the occlusion camera sees.")]
+ public bool saveDebugDepthImages = true;
+
+ [Header("Video Settings")]
+ public string ffmpegPath = "ffmpeg";
+ public int frameRate = 30;
+
+ [Header("Compression (Twitch VOD Simulation)")]
+ [Tooltip("Target Bitrate in kbps. 6000 is High Quality 1080p. 2500 is messy 720p.")]
+ public int targetBitrateKbps = 2500;
+ [Tooltip("GOP (Group of Pictures) size in seconds. Twitch uses 2 seconds.")]
+ public float gopSizeSeconds = 2.0f;
+
+ [Header("Parallel Processing")]
+ public int workerId = 0;
+ public int totalWorkers = 1;
+
+ [Header("Sequence Naming")]
+ public string sequenceName = "";
+ private string _currentInputJsonPath = "";
+
+ [Header("Occlusion Settings")]
+ public float occlusionBias = 0.02f;
+
+ [Header("References")]
+ public GameObject characterRoot;
+ public Camera vtuberCamera;
+ public SyntheticCameraDriver cameraDriver;
+
+ [Header("Randomization")]
+ public List avatarList = new List();
+ public List worldSceneNames = new List();
+ public LoadSceneMode worldSceneLoadMode = LoadSceneMode.Additive;
+ public bool setLoadedWorldSceneActive = true;
+ public string worldMainCameraName = "Main Camera";
+ public string spawnPointToken = "SpawnPoint";
+
+ [Header("Animation Indices")]
+ public int faceMaxId = 5;
+ public int handsMaxId = 5;
+ public int minSwitchFrames = 30;
+ public int maxSwitchFrames = 120;
+ public string paramFaceIndex = "FaceIndex";
+ public string paramLeftHandIndex = "LeftHandIndex";
+ public string paramRightHandIndex = "RightHandIndex";
+
+ [Header("BBOX Accuracy")]
+ public bool useBakedSkinnedMeshForBbox = true;
+ public int bakedVertexStride = 8;
+
+ [Header("Calibration")]
+ public float movementScale = 1.0f;
+ public Vector3 translationOffset = new Vector3(0, 0.05f, 0);
+ public Vector3 globalCoordinateCorrection = new Vector3(-90, 180, 0);
+
+ // --- PRIVATE STATE ---
+ private float _activePadding = 40f;
+ private SequenceData _data;
+ private Transform[] _bones;
+ private Transform _pelvisBone;
+ private List _activeMarkers = new List();
+ private HybridPoseCopier _activeRetargeter;
+ private Transform _activeAvatarRoot = null;
+ private Animator _activeAnimator = null;
+ private string _activeAvatarName = "";
+ private readonly List _activeBboxRenderers = new List();
+ private Mesh _bakeMesh;
+ private readonly List _bakedVerts = new List(8192);
+ private Texture2D _greenTex, _redTex, _occTex;
+ private Rect _cachedBbox = new Rect(0, 0, 0, 0);
+ private bool _cachedHasBbox = false;
+ private int[] _cachedMarkerVis = null;
+ private string _currentlyLoadedWorldScene = "";
+
+ private Shader _autoDepthShader;
+ private const int JOINT_COUNT = 22;
+ private static readonly string[] BONE_NAMES = {
+ "pelvis", "left_hip", "right_hip", "spine1", "left_knee", "right_knee", "spine2",
+ "left_ankle", "right_ankle", "spine3", "left_foot", "right_foot", "neck", "left_collar",
+ "right_collar", "head", "left_shoulder", "right_shoulder", "left_elbow", "right_elbow",
+ "left_wrist", "right_wrist"
+ };
+
+ void Start()
+ {
+ Screen.SetResolution(1280, 720, FullScreenMode.Windowed);
+
+ EnsureDepthShaderExists();
+ _autoDepthShader = Shader.Find("Custom/AutoLinearDepth");
+ if (!_autoDepthShader) UnityEngine.Debug.LogError("Could not load the auto-generated depth shader!");
+
+ _greenTex = new Texture2D(1, 1); _greenTex.SetPixel(0, 0, Color.green); _greenTex.Apply();
+ _redTex = new Texture2D(1, 1); _redTex.SetPixel(0, 0, Color.red); _redTex.Apply();
+ _occTex = new Texture2D(1, 1); _occTex.SetPixel(0, 0, new Color(1, 0, 0, 0.5f)); _occTex.Apply();
+
+ _bakeMesh = new Mesh();
+ _bakeMesh.MarkDynamic();
+
+ if (startRecordingOnPlay)
+ StartCoroutine(ProcessBatch());
+ }
+
+ private void EnsureDepthShaderExists()
+ {
+ string path = "Assets/SyntheticDepth.shader";
+ if (File.Exists(path)) return;
+
+ string shaderCode = @"
+Shader ""Custom/AutoLinearDepth""
+{
+ SubShader
+ {
+ Tags { ""RenderType""="""" ""Queue""=""Geometry"" ""ForceNoShadowCasting""=""True"" }
+ Cull Off
+ ZWrite On
+ ZTest LEqual
+ Pass
+ {
+ CGPROGRAM
+ #pragma vertex vert
+ #pragma fragment frag
+ #include ""UnityCG.cginc""
+ struct appdata { float4 vertex : POSITION; };
+ struct v2f { float4 pos : SV_POSITION; float depth : TEXCOORD0; };
+ v2f vert (appdata v) { v2f o; o.pos = UnityObjectToClipPos(v.vertex); o.depth = -UnityObjectToViewPos(v.vertex).z; return o; }
+ float4 frag (v2f i) : SV_Target { return float4(i.depth, 0, 0, 1); }
+ ENDCG
+ }
+ }
+}";
+ File.WriteAllText(path, shaderCode);
+ #if UNITY_EDITOR
+ UnityEditor.AssetDatabase.Refresh();
+ #endif
+ UnityEngine.Debug.Log("Created Aggressive AutoLinearDepth shader at " + path);
+ }
+
+ private IEnumerator ProcessBatch()
+ {
+ string fullInputPath = Path.IsPathRooted(inputFolderPath) ? inputFolderPath : Path.Combine(Application.dataPath, "..", inputFolderPath);
+ if (!Directory.Exists(fullInputPath)) { UnityEngine.Debug.LogError("Input folder missing"); yield break; }
+
+ string[] allFiles = Directory.GetFiles(fullInputPath, "*.json");
+ Array.Sort(allFiles);
+
+ List myFiles = new List();
+ int safeTotalWorkers = Mathf.Max(1, totalWorkers);
+
+ for (int i = 0; i < allFiles.Length; i++)
+ if (i % safeTotalWorkers == workerId) myFiles.Add(allFiles[i]);
+
+ foreach (string file in myFiles)
+ {
+ _currentInputJsonPath = file;
+ sequenceName = Path.GetFileNameWithoutExtension(file);
+
+ Resources.UnloadUnusedAssets();
+ System.GC.Collect();
+
+ yield return StartCoroutine(LoadRandomWorldRoutine());
+
+ RandomizeAvatarAndGatherRenderers();
+ FindAndCacheBones();
+ ApplyRandomSpawnPoint(SceneManager.GetActiveScene());
+
+ yield return StartCoroutine(RecordSingleSequence());
+ }
+
+ #if UNITY_EDITOR
+ UnityEditor.EditorApplication.isPlaying = false;
+ #else
+ Application.Quit();
+ #endif
+ }
+
+ private IEnumerator LoadRandomWorldRoutine()
+ {
+ if (worldSceneNames == null || worldSceneNames.Count == 0) yield break;
+
+ if (!string.IsNullOrEmpty(_currentlyLoadedWorldScene) && worldSceneLoadMode == LoadSceneMode.Additive)
+ {
+ AsyncOperation unloadOp = SceneManager.UnloadSceneAsync(_currentlyLoadedWorldScene);
+ while (unloadOp != null && !unloadOp.isDone) yield return null;
+ }
+
+ string chosen = worldSceneNames[UnityEngine.Random.Range(0, worldSceneNames.Count)].Trim();
+ _currentlyLoadedWorldScene = chosen;
+
+ AsyncOperation loadOp = SceneManager.LoadSceneAsync(chosen, worldSceneLoadMode);
+ while (!loadOp.isDone) yield return null;
+ yield return null;
+
+ Scene loaded = SceneManager.GetSceneByName(chosen);
+ if (!loaded.IsValid()) loaded = SceneManager.GetSceneByPath(chosen);
+
+ if (loaded.IsValid() && loaded.isLoaded)
+ {
+ if (setLoadedWorldSceneActive) SceneManager.SetActiveScene(loaded);
+ BindToWorldMainCameraOrLog(loaded);
+ }
+ }
+
+ IEnumerator RecordSingleSequence()
+ {
+ _data = JsonConvert.DeserializeObject(File.ReadAllText(_currentInputJsonPath));
+
+ if (!Directory.Exists(outputFolder)) Directory.CreateDirectory(outputFolder);
+ string seqImageDir = Path.Combine(outputFolder, "images", sequenceName);
+ if (!Directory.Exists(seqImageDir)) Directory.CreateDirectory(seqImageDir);
+
+ // --- FFmpeg Setup for VOD SIMULATION ---
+ string videoPath = Path.Combine(outputFolder, $"video_{sequenceName}.mp4").Replace("\\", "/");
+
+ // VOD SIMULATION LOGIC:
+ // 1. -b:v {bitrate}k -> Forces the encoder to target a specific bandwidth
+ // 2. -maxrate {bitrate}k -> Prevents it from spiking quality during high motion (causes artifacts)
+ // 3. -bufsize {bitrate*2}k -> Standard buffer size for streaming
+ // 4. -g {gop} -> Sets Keyframe Interval. Twitch uses 2 seconds fixed.
+ // 5. -preset ultrafast -> Keeps Unity realtime, but relies on bitrate starvation to cause the artifacts
+
+ int gopFrames = Mathf.RoundToInt(frameRate * gopSizeSeconds);
+
+ string ffmpegArgs = $"-y -f rawvideo -vcodec rawvideo -pix_fmt rgb24 " +
+ $"-s {Screen.width}x{Screen.height} -r {frameRate} -i - " +
+ $"-vf vflip " +
+ $"-c:v libx264 " +
+ $"-pix_fmt yuv420p " +
+ $"-preset ultrafast " +
+ $"-b:v {targetBitrateKbps}k -maxrate {targetBitrateKbps}k -bufsize {targetBitrateKbps * 2}k " +
+ $"-g {gopFrames} " +
+ $"\"{videoPath}\"";
+
+ ProcessStartInfo psi = new ProcessStartInfo
+ {
+ FileName = ffmpegPath,
+ Arguments = ffmpegArgs,
+ UseShellExecute = false,
+ RedirectStandardInput = true,
+ CreateNoWindow = true
+ };
+
+ Process ffmpegProcess = null;
+ try
+ {
+ ffmpegProcess = Process.Start(psi);
+ }
+ catch(Exception e)
+ {
+ UnityEngine.Debug.LogError($"Failed to start FFmpeg. Is it in PATH? Error: {e.Message}");
+ yield break;
+ }
+
+ Texture2D screenTex = new Texture2D(Screen.width, Screen.height, TextureFormat.RGB24, false);
+ RenderTexture depthRT = new RenderTexture(Screen.width, Screen.height, 24, RenderTextureFormat.RFloat);
+ Texture2D depthReadTex = new Texture2D(Screen.width, Screen.height, TextureFormat.RFloat, false);
+
+ string jsonlPath = Path.Combine(outputFolder, $"sequence_{sequenceName}.jsonl");
+
+ int framesUntilSwitch = 0;
+ int currentFaceId = 0, currentLeftHandId = 0, currentRightHandId = 0;
+
+ using (var sw = new StreamWriter(jsonlPath, false))
+ {
+ for (int i = 0; i < _data.frames.Count; i++)
+ {
+ if (framesUntilSwitch <= 0)
+ {
+ framesUntilSwitch = UnityEngine.Random.Range(minSwitchFrames, maxSwitchFrames + 1);
+ currentFaceId = UnityEngine.Random.Range(0, faceMaxId + 1);
+ currentLeftHandId = UnityEngine.Random.Range(0, handsMaxId + 1);
+ currentRightHandId = UnityEngine.Random.Range(0, handsMaxId + 1);
+ if (_activeAnimator != null)
+ {
+ _activeAnimator.SetInteger(paramFaceIndex, currentFaceId);
+ _activeAnimator.SetInteger(paramLeftHandIndex, currentLeftHandId);
+ _activeAnimator.SetInteger(paramRightHandIndex, currentRightHandId);
+ }
+ }
+ framesUntilSwitch--;
+
+ ApplyFrame(_data.frames[i]);
+ if (cameraDriver != null) cameraDriver.OnFrame(i);
+ if (_activeRetargeter != null) _activeRetargeter.ManualUpdatePose();
+
+ Physics.SyncTransforms();
+
+ vtuberCamera.clearFlags = CameraClearFlags.SolidColor;
+ vtuberCamera.backgroundColor = Color.black;
+ vtuberCamera.cullingMask = ~0;
+
+ yield return new WaitForEndOfFrame();
+
+ if (screenTex.width != Screen.width || screenTex.height != Screen.height)
+ screenTex.Reinitialize(Screen.width, Screen.height);
+
+ screenTex.ReadPixels(new Rect(0, 0, Screen.width, Screen.height), 0, 0);
+ screenTex.Apply();
+
+ byte[] rawFrame = screenTex.GetRawTextureData();
+ if (ffmpegProcess != null && !ffmpegProcess.HasExited)
+ {
+ try {
+ ffmpegProcess.StandardInput.BaseStream.Write(rawFrame, 0, rawFrame.Length);
+ ffmpegProcess.StandardInput.BaseStream.Flush();
+ } catch (Exception ex) {
+ UnityEngine.Debug.LogError("FFmpeg write error: " + ex.Message);
+ }
+ }
+
+ RenderTexture origRT = vtuberCamera.targetTexture;
+ CameraClearFlags origFlags = vtuberCamera.clearFlags;
+ Color origBG = vtuberCamera.backgroundColor;
+
+ vtuberCamera.targetTexture = depthRT;
+ vtuberCamera.clearFlags = CameraClearFlags.SolidColor;
+ vtuberCamera.backgroundColor = new Color(1000f, 0, 0, 0);
+
+ if (_autoDepthShader != null) vtuberCamera.RenderWithShader(_autoDepthShader, "");
+ else vtuberCamera.Render();
+
+ RenderTexture.active = depthRT;
+ if (depthReadTex.width != Screen.width || depthReadTex.height != Screen.height)
+ depthReadTex.Reinitialize(Screen.width, Screen.height);
+ depthReadTex.ReadPixels(new Rect(0, 0, Screen.width, Screen.height), 0, 0);
+ depthReadTex.Apply();
+
+ if (saveDebugDepthImages)
+ {
+ Texture2D visualDepth = new Texture2D(Screen.width, Screen.height, TextureFormat.RGB24, false);
+ Color[] rawPixels = depthReadTex.GetPixels();
+ Color[] visPixels = new Color[rawPixels.Length];
+ float displayRange = 3.0f;
+ for (int k = 0; k < rawPixels.Length; k++)
+ {
+ float d = rawPixels[k].r;
+ if (d > 999f) visPixels[k] = Color.white;
+ else
+ {
+ float norm = Mathf.Clamp01(d / displayRange);
+ visPixels[k] = new Color(norm, norm, norm);
+ }
+ }
+ visualDepth.SetPixels(visPixels);
+ visualDepth.Apply();
+ string depthFile = $"depth_{i:D5}.png";
+ File.WriteAllBytes(Path.Combine(seqImageDir, depthFile), visualDepth.EncodeToPNG());
+ Destroy(visualDepth);
+ }
+
+ vtuberCamera.targetTexture = origRT;
+ vtuberCamera.clearFlags = origFlags;
+ vtuberCamera.backgroundColor = origBG;
+ RenderTexture.active = null;
+
+ ComputeBoundingBoxCached();
+
+ float H = Screen.height; float W = Screen.width;
+ // Compute intrinsics in CV convention (Y-down, origin top-left)
+ // Use Y-up point to get Unity-convention focal length, then negate for CV
+ Vector3 camP0_W = vtuberCamera.transform.TransformPoint(new Vector3(0f, 0f, 1f));
+ Vector3 camPx_W = vtuberCamera.transform.TransformPoint(new Vector3(1f, 0f, 1f));
+ Vector3 camPy_W = vtuberCamera.transform.TransformPoint(new Vector3(0f, 1f, 1f)); // Y-up in Unity
+ Vector3 s0 = vtuberCamera.WorldToScreenPoint(camP0_W);
+ Vector3 sx = vtuberCamera.WorldToScreenPoint(camPx_W);
+ Vector3 sy = vtuberCamera.WorldToScreenPoint(camPy_W);
+ float cx = s0.x;
+ float cy = H - s0.y; // Convert to CV (origin at top-left)
+ float fx = sx.x - s0.x;
+ float fy = (H - s0.y) - (H - sy.y); // In CV, fy should be positive when Y-up maps to screen-down
+
+ Rect rFull = _cachedHasBbox ? _cachedBbox : new Rect(0, 0, 0, 0);
+ float bbox_x = rFull.x; float bbox_y = H - (rFull.y + rFull.height);
+ float bbox_w = rFull.width; float bbox_h = rFull.height;
+ float clip_x0 = Mathf.Clamp(bbox_x, 0, W);
+ float clip_y0 = Mathf.Clamp(bbox_y, 0, H);
+ float clip_w = Mathf.Max(0, Mathf.Clamp(bbox_x + bbox_w, 0, W) - clip_x0);
+ float clip_h = Mathf.Max(0, Mathf.Clamp(bbox_y + bbox_h, 0, H) - clip_y0);
+
+ var kpts2D = new List();
+ var kptsVis = new List();
+ var kpts3D = new List();
+
+ if (_activeMarkers != null)
+ {
+ for (int mi = 0; mi < _activeMarkers.Count; mi++)
+ {
+ Transform t = _activeMarkers[mi];
+ if (t == null) { continue; }
+
+ Vector3 wPos = t.position;
+ kpts3D.Add(wPos.x); kpts3D.Add(wPos.y); kpts3D.Add(wPos.z);
+
+ Vector3 sPos = vtuberCamera.WorldToScreenPoint(wPos);
+ float x_px = sPos.x;
+ float y_px = H - sPos.y;
+ kpts2D.Add(x_px); kpts2D.Add(y_px);
+
+ int vis = 0;
+ if (sPos.z > 0 && x_px >= 0 && x_px < W && sPos.y >= 0 && sPos.y < H)
+ {
+ int checkRadius = 2;
+ float requiredVisibilityRatio = 0.5f;
+ int totalSamples = 0;
+ int visibleSamples = 0;
+ float markerDistance = sPos.z;
+
+ for (int ox = -checkRadius; ox <= checkRadius; ox++)
+ {
+ for (int oy = -checkRadius; oy <= checkRadius; oy++)
+ {
+ int px = (int)sPos.x + ox;
+ int py = (int)sPos.y + oy;
+ if (px >= 0 && px < W && py >= 0 && py < H)
+ {
+ totalSamples++;
+ float pixelDepth = depthReadTex.GetPixel(px, py).r;
+ if (pixelDepth >= (markerDistance - occlusionBias))
+ visibleSamples++;
+ }
+ }
+ }
+ if (totalSamples > 0)
+ {
+ float visibilityPct = (float)visibleSamples / totalSamples;
+ vis = (visibilityPct >= requiredVisibilityRatio) ? 2 : 1;
+ }
+ else vis = 1;
+ }
+ kptsVis.Add(vis);
+ }
+ }
+
+ _cachedMarkerVis = kptsVis.ToArray();
+ Transform pelvis = (_bones != null && _bones.Length > 0) ? _bones[0] : null;
+ Vector3 pelvisPos = (pelvis != null) ? pelvis.position : Vector3.zero;
+ Quaternion pelvisWorld = (pelvis != null) ? pelvis.rotation : Quaternion.identity;
+
+ // --- SIMPLIFIED EXPORT: Raw Unity transforms only ---
+ // Python will derive incam and global consistently from these raw values
+ // using the approach in process_dataset_fromincam.py
+
+ // INCAM TRANSLATION: Camera-relative position (Y-flipped for CV)
+ Vector3 incamPos_unity = vtuberCamera.transform.InverseTransformPoint(pelvisPos);
+ Vector3 incamPos = new Vector3(incamPos_unity.x, -incamPos_unity.y, incamPos_unity.z);
+ Vector3 rootIncamPos_unity = (characterRoot != null) ? vtuberCamera.transform.InverseTransformPoint(characterRoot.transform.position) : Vector3.zero;
+ Vector3 rootIncamPos = new Vector3(rootIncamPos_unity.x, -rootIncamPos_unity.y, rootIncamPos_unity.z);
+
+ var meta = new OutputMeta
+ {
+ frame_index = i,
+ image_path = videoPath,
+ avatar_name = _activeAvatarName,
+ face_id = currentFaceId,
+ left_hand_id = currentLeftHandId,
+ right_hand_id = currentRightHandId,
+ bbox = new float[] { bbox_x, bbox_y, bbox_w, bbox_h },
+ bbox_clip = new float[] { clip_x0, clip_y0, clip_w, clip_h },
+ kpts_2d = kpts2D.ToArray(), kpts_vis = kptsVis.ToArray(),
+ cam_intrinsics = new float[] { fx, fy, cx, cy },
+ // RAW UNITY TRANSFORMS - Python derives everything from these
+ cam_pos_world = new float[] { vtuberCamera.transform.position.x, vtuberCamera.transform.position.y, vtuberCamera.transform.position.z },
+ cam_rot_world = new float[] { vtuberCamera.transform.rotation.x, vtuberCamera.transform.rotation.y, vtuberCamera.transform.rotation.z, vtuberCamera.transform.rotation.w },
+ pelvis_pos_world = new float[] { pelvisPos.x, pelvisPos.y, pelvisPos.z },
+ pelvis_rot_world = new float[] { pelvisWorld.x, pelvisWorld.y, pelvisWorld.z, pelvisWorld.w },
+ // INCAM TRANSLATION (only translation needs pre-conversion for CV Y-flip)
+ smpl_incam_transl = new float[] { incamPos.x, incamPos.y, incamPos.z },
+ smpl_root_incam_transl = new float[] { rootIncamPos.x, rootIncamPos.y, rootIncamPos.z },
+ smpl_root_world_scale = (characterRoot != null) ? characterRoot.transform.lossyScale.x : 1f,
+ kpts_3d_world = kpts3D.ToArray(),
+ smplx_pose = _data.frames[i].p, smplx_betas = _data.frames[i].b
+ // REMOVED: smpl_incam_quat, smpl_global_orient_unity, smpl_global_transl_unity
+ // Python derives incam/global rotation from pelvis_rot_world + cam_rot_world
+ };
+ sw.WriteLine(JsonConvert.SerializeObject(meta));
+ }
+ }
+
+ if (ffmpegProcess != null && !ffmpegProcess.HasExited)
+ {
+ ffmpegProcess.StandardInput.Close();
+ ffmpegProcess.WaitForExit();
+ ffmpegProcess.Close();
+ }
+
+ if(screenTex) Destroy(screenTex);
+ if(depthRT) Destroy(depthRT);
+ if(depthReadTex) Destroy(depthReadTex);
+ }
+
+ void ApplyFrame(FrameData f)
+ {
+ if (f.p == null || characterRoot == null) return;
+ Quaternion correction = Quaternion.Euler(globalCoordinateCorrection);
+ characterRoot.transform.localPosition = (correction * (new Vector3(-f.t[0], f.t[1], f.t[2]) * movementScale)) + translationOffset;
+ int floatIdx = 0;
+ for (int i = 0; i < JOINT_COUNT; i++)
+ {
+ if (floatIdx + 2 >= f.p.Length) break;
+ float x = f.p[floatIdx++], y = f.p[floatIdx++], z = f.p[floatIdx++];
+ float angle = Mathf.Sqrt(x * x + y * y + z * z);
+ Quaternion q = Quaternion.identity;
+ if (angle > 1e-6f) { float c = Mathf.Cos(angle * 0.5f), s = Mathf.Sin(angle * 0.5f); q = new Quaternion(-(x / angle) * s, (y / angle) * s, (z / angle) * s, -c); }
+ if (_bones != null && i < _bones.Length && _bones[i] != null) _bones[i].localRotation = (i == 0) ? (correction * q) : q;
+ }
+ if (cameraDriver != null) cameraDriver.SetStyleFromFrameData(f.s);
+ }
+
+ private void FindAndCacheBones()
+ {
+ _bones = new Transform[BONE_NAMES.Length];
+ for (int i = 0; i < BONE_NAMES.Length; i++) _bones[i] = FindDeep(characterRoot.transform, BONE_NAMES[i]);
+ _pelvisBone = (_bones != null && _bones.Length > 0) ? _bones[0] : null;
+ }
+
+ private static Transform FindDeep(Transform root, string name)
+ {
+ if (root.name == name) return root;
+ foreach (Transform child in root) { var res = FindDeep(child, name); if (res) return res; }
+ return null;
+ }
+
+ private void RandomizeAvatarAndGatherRenderers()
+ {
+ if (avatarList == null || avatarList.Count == 0) return;
+ _activeBboxRenderers.Clear(); _activeMarkers.Clear();
+ int randomIndex = UnityEngine.Random.Range(0, avatarList.Count);
+ AvatarConfig selected = avatarList[randomIndex];
+ _activeAvatarName = selected.avatarName;
+ _activeAnimator = selected.animator;
+ if (_activeAnimator == null && selected.avatarObject != null) _activeAnimator = selected.avatarObject.GetComponent();
+ for (int i = 0; i < avatarList.Count; i++) if (avatarList[i].avatarObject != null) avatarList[i].avatarObject.SetActive(i == randomIndex);
+ _activePadding = selected.specificPadding; _activeRetargeter = selected.retargeter;
+ if (selected.avatarObject != null) _activeAvatarRoot = selected.avatarObject.transform;
+ if (selected.customMarkers != null) _activeMarkers.AddRange(selected.customMarkers);
+ if (characterRoot != null) { foreach (var r in characterRoot.GetComponentsInChildren(true)) _activeBboxRenderers.Add(r); }
+ foreach (GameObject extra in selected.extraMeshes) if (extra) { foreach (var cr in extra.GetComponentsInChildren(true)) if (!_activeBboxRenderers.Contains(cr)) _activeBboxRenderers.Add(cr); }
+ }
+
+ private void ApplyRandomSpawnPoint(Scene worldScene)
+ {
+ if (!worldScene.IsValid() || !worldScene.isLoaded) return;
+ List spawns = new List();
+ foreach (GameObject root in worldScene.GetRootGameObjects()) { foreach (Transform child in root.GetComponentsInChildren(true)) if (child.name.Contains(spawnPointToken)) spawns.Add(child); }
+ if (spawns.Count > 0)
+ {
+ Transform chosen = spawns[UnityEngine.Random.Range(0, spawns.Count)];
+ this.transform.position = chosen.position; this.transform.rotation = chosen.rotation;
+ if (characterRoot != null) { characterRoot.transform.localPosition = Vector3.zero; characterRoot.transform.localRotation = Quaternion.identity; }
+ }
+ }
+
+ private void BindToWorldMainCameraOrLog(Scene worldScene)
+ {
+ Camera found = null;
+ foreach (GameObject root in worldScene.GetRootGameObjects())
+ {
+ foreach (Transform t in root.GetComponentsInChildren(true))
+ if (t.name == worldMainCameraName && t.GetComponent()) { found = t.GetComponent(); break; }
+ if (found) break;
+ }
+ if (!found) return;
+ Vector3 ls = found.transform.lossyScale;
+ if (Mathf.Abs(ls.x - 1f) > 1e-4f || Mathf.Abs(ls.y - 1f) > 1e-4f || Mathf.Abs(ls.z - 1f) > 1e-4f)
+ {
+ found.transform.SetParent(null, true);
+ found.transform.localScale = Vector3.one;
+ }
+ if (vtuberCamera && vtuberCamera != found) vtuberCamera.enabled = false;
+ vtuberCamera = found;
+ cameraDriver = found.GetComponent() ?? found.gameObject.AddComponent();
+ cameraDriver.BindAndInit(found, characterRoot.transform);
+ }
+
+ private bool ComputeBoundingBoxCached()
+ {
+ _cachedHasBbox = false; _cachedBbox = new Rect(0, 0, 0, 0);
+ if (vtuberCamera == null || _activeBboxRenderers.Count == 0) return false;
+ float minVX = float.MaxValue, maxVX = float.MinValue, minVY = float.MaxValue, maxVY = float.MinValue;
+ bool foundAny = false;
+ int stride = Mathf.Max(1, bakedVertexStride);
+ foreach (var rend in _activeBboxRenderers)
+ {
+ if (!rend) continue;
+ if (useBakedSkinnedMeshForBbox && rend is SkinnedMeshRenderer smr)
+ {
+ _bakeMesh.Clear(); smr.BakeMesh(_bakeMesh); _bakedVerts.Clear(); _bakeMesh.GetVertices(_bakedVerts);
+ Matrix4x4 localToWorldNoScale = Matrix4x4.TRS(smr.transform.position, smr.transform.rotation, Vector3.one);
+ for (int vi = 0; vi < _bakedVerts.Count; vi += stride)
+ {
+ Vector3 vp = vtuberCamera.WorldToViewportPoint(localToWorldNoScale.MultiplyPoint3x4(_bakedVerts[vi]));
+ if (vp.z <= 0f) continue;
+ foundAny = true; minVX = Math.Min(minVX, vp.x); maxVX = Math.Max(maxVX, vp.x); minVY = Math.Min(minVY, vp.y); maxVY = Math.Max(maxVY, vp.y);
+ }
+ }
+ else
+ {
+ Bounds b = rend.bounds; Vector3 c = b.center, e = b.extents;
+ Vector3[] corners = { c+new Vector3(-e.x,-e.y,-e.z), c+new Vector3(-e.x,-e.y,e.z), c+new Vector3(-e.x,e.y,-e.z), c+new Vector3(-e.x,e.y,e.z), c+new Vector3(e.x,-e.y,-e.z), c+new Vector3(e.x,-e.y,e.z), c+new Vector3(e.x,e.y,-e.z), c+new Vector3(e.x,e.y,e.z) };
+ foreach (var corner in corners)
+ {
+ Vector3 vp = vtuberCamera.WorldToViewportPoint(corner);
+ if (vp.z <= 0f) continue;
+ foundAny = true; minVX = Math.Min(minVX, vp.x); maxVX = Math.Max(maxVX, vp.x); minVY = Math.Min(minVY, vp.y); maxVY = Math.Max(maxVY, vp.y);
+ }
+ }
+ }
+ if (!foundAny) return false;
+ _cachedBbox = new Rect(minVX * Screen.width - _activePadding, minVY * Screen.height - _activePadding, (maxVX - minVX) * Screen.width + _activePadding * 2, (maxVY - minVY) * Screen.height + _activePadding * 2);
+ _cachedHasBbox = true; return true;
+ }
+
+ void OnGUI()
+ {
+ if (!showDebugUI || vtuberCamera == null) return;
+ if (_cachedHasBbox) { Rect r = _cachedBbox; float invY = Screen.height - (r.y + r.height); GUI.DrawTexture(new Rect(r.x, invY, r.width, 3), _greenTex); GUI.DrawTexture(new Rect(r.x, invY + r.height, r.width, 3), _greenTex); GUI.DrawTexture(new Rect(r.x, invY, 3, r.height), _greenTex); GUI.DrawTexture(new Rect(r.x + r.width, invY, 3, r.height), _greenTex); }
+ if (_activeMarkers != null) { for (int mi = 0; mi < _activeMarkers.Count; mi++) { if (!_activeMarkers[mi]) continue; Vector3 sc = vtuberCamera.WorldToScreenPoint(_activeMarkers[mi].position); if (sc.z > 0) { int vis = (_cachedMarkerVis != null && mi < _cachedMarkerVis.Length) ? _cachedMarkerVis[mi] : 2; GUI.DrawTexture(new Rect(sc.x - 2, Screen.height - sc.y - 2, 4, 4), vis == 1 ? _occTex : _greenTex); } } }
+ }
+}
\ No newline at end of file
diff --git a/TODO.md b/TODO.md
new file mode 100644
index 0000000000000000000000000000000000000000..7dd6812fec3eb435b365d1e4d9c704c23d369880
--- /dev/null
+++ b/TODO.md
@@ -0,0 +1,21 @@
+Hands (plug-and-play 3D, no fitting): HaMeR
+
+Use HaMeR (Hand Mesh Recovery).
+
+Input: hand bounding box + hand side (L/R) + image
+
+https://geopavlakos.github.io/hamer/?utm_source=chatgpt.com
+
+Hand4Whole exists, but it’s a full pipeline; HaMeR is the cleanest “hands-only module.”
+
+Face to Emotion state for labeling facial expressions.
+
+ViT Facial Expression Recognition
+HuggingFace (same model, easier)
+
+👉 https://huggingface.co/nateraw/vit-base-facial-expression-recognition
+
+DeepFace (simple, classic, works)
+👉 https://github.com/serengil/deepface
+
+CLIP zero-shot emotion classification
diff --git a/_DATA/hamer_demo_data.tar.gz b/_DATA/hamer_demo_data.tar.gz
new file mode 100644
index 0000000000000000000000000000000000000000..5f560aae485edc3d790ef1cc6beb51e13a5a0297
--- /dev/null
+++ b/_DATA/hamer_demo_data.tar.gz
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:bb4573f1ed131923eb6fa2e9537766b367095b71afc536e46767b95e0debc37f
+size 217497600
diff --git a/bedlam_GT.npz b/bedlam_GT.npz
new file mode 100644
index 0000000000000000000000000000000000000000..840a877bdca3eb44ba3e35a6efc435ad109ffaa8
--- /dev/null
+++ b/bedlam_GT.npz
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:719aa90ff80e2339aa4e7a96ccf243ae0e78860c4735baec32c8ab452bd98d19
+size 212875004
diff --git a/configs/__init__.py b/configs/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..58e91e0e40b6205baf87e1440cda68b977471f15
--- /dev/null
+++ b/configs/__init__.py
@@ -0,0 +1,30 @@
+import argparse
+import os
+
+from hydra import compose, initialize_config_module
+from hydra.core.config_store import ConfigStore
+
+os.environ["HYDRA_FULL_ERROR"] = "1"
+
+MainStore = ConfigStore.instance()
+
+
+def parse_args_to_cfg():
+ """
+ Use minimal Hydra API to parse args and return cfg.
+ This function don't do _run_hydra which create log file hierarchy.
+ """
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--config-name", "-cn", default="train")
+ parser.add_argument(
+ "overrides",
+ nargs="*",
+ help="Any key=value arguments to override config values (use dots for.nested=overrides)",
+ )
+ args = parser.parse_args()
+
+ # Cfg
+ with initialize_config_module(version_base="1.3", config_module="configs"):
+ cfg = compose(config_name=args.config_name, overrides=args.overrides)
+
+ return cfg
diff --git a/configs/callbacks/ckpt_saver/every10000s_top100.yaml b/configs/callbacks/ckpt_saver/every10000s_top100.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..f379fee9f967423ae13730351ed5b9488377c4f5
--- /dev/null
+++ b/configs/callbacks/ckpt_saver/every10000s_top100.yaml
@@ -0,0 +1,5 @@
+every10000s_top100:
+ _target_: genmo.callbacks.simple_ckpt_saver.SimpleCkptSaver
+ output_dir: ${output_dir}/checkpoints/
+ every_n_steps: 10000
+ save_top_k: 100
diff --git a/configs/callbacks/lr_monitor/pl.yaml b/configs/callbacks/lr_monitor/pl.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..5d5416b7be73bd6207492e53f44e227aedd65094
--- /dev/null
+++ b/configs/callbacks/lr_monitor/pl.yaml
@@ -0,0 +1,2 @@
+pl:
+ _target_: pytorch_lightning.callbacks.lr_monitor.LearningRateMonitor
diff --git a/configs/callbacks/metric/metric_3dpw.yaml b/configs/callbacks/metric/metric_3dpw.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..6f8f71666b797a02584874597fc48834084c32cb
--- /dev/null
+++ b/configs/callbacks/metric/metric_3dpw.yaml
@@ -0,0 +1,2 @@
+metric_3dpw:
+ _target_: genmo.callbacks.metric.metric_3dpw.MetricMocap
diff --git a/configs/callbacks/metric/metric_3dpw_occ.yaml b/configs/callbacks/metric/metric_3dpw_occ.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..5aebfa3eb816153ecd8a6fb05e9e1018c491c310
--- /dev/null
+++ b/configs/callbacks/metric/metric_3dpw_occ.yaml
@@ -0,0 +1,2 @@
+metric_3dpw_occ:
+ _target_: genmo.callbacks.metric.metric_3dpw_occ.MetricMocap
diff --git a/configs/callbacks/metric/metric_aistpp.yaml b/configs/callbacks/metric/metric_aistpp.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..f6ec8bb9d3eb6076e5d01627a72e7d9bf16c74b6
--- /dev/null
+++ b/configs/callbacks/metric/metric_aistpp.yaml
@@ -0,0 +1,2 @@
+metric_aistpp:
+ _target_: genmo.callbacks.metric.metric_aistpp.MetricMusic
diff --git a/configs/callbacks/metric/metric_emdb1.yaml b/configs/callbacks/metric/metric_emdb1.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..1772c811cff8f1d4b1f7138405b0129401ceb299
--- /dev/null
+++ b/configs/callbacks/metric/metric_emdb1.yaml
@@ -0,0 +1,4 @@
+metric_emdb1:
+ _target_: genmo.callbacks.metric.metric_emdb.MetricMocap
+ emdb_split: 1
+ occ: false
diff --git a/configs/callbacks/metric/metric_emdb2.yaml b/configs/callbacks/metric/metric_emdb2.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..5bdc898d59e4ff9bc41d78ecdb312bc88ad95fc2
--- /dev/null
+++ b/configs/callbacks/metric/metric_emdb2.yaml
@@ -0,0 +1,4 @@
+metric_emdb2:
+ _target_: genmo.callbacks.metric.metric_emdb.MetricMocap
+ emdb_split: 2
+ occ: false
diff --git a/configs/callbacks/metric/metric_rich.yaml b/configs/callbacks/metric/metric_rich.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..9c238e6ab0b4b6fe6815ae9eef100aa9f41585f9
--- /dev/null
+++ b/configs/callbacks/metric/metric_rich.yaml
@@ -0,0 +1,3 @@
+metric_rich:
+ _target_: genmo.callbacks.metric.metric_rich.MetricMocap
+ occ: false
diff --git a/configs/callbacks/metric/metric_unity.yaml b/configs/callbacks/metric/metric_unity.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..6878a8b0cce675e1b949267e30e5cb9a7a6d4b50
--- /dev/null
+++ b/configs/callbacks/metric/metric_unity.yaml
@@ -0,0 +1,4 @@
+metric_unity:
+ _target_: genmo.callbacks.metric.metric_unity.MetricUnity
+ # Disable the old scenepic HTML viz by default (use `vis/vis_unity_val` instead).
+ vis_every_n_val: 1000000000
diff --git a/configs/callbacks/prog_bar/prog_reporter_ed1.yaml b/configs/callbacks/prog_bar/prog_reporter_ed1.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..72507833d3746cdcd41d337566afdb4b71a46597
--- /dev/null
+++ b/configs/callbacks/prog_bar/prog_reporter_ed1.yaml
@@ -0,0 +1,5 @@
+prog_reporter_ed1:
+ _target_: genmo.callbacks.prog_bar.ProgressReporter
+ log_every_percent: 0.1
+ exp_name: ${exp_name}
+ data_name: ${data_name}
diff --git a/configs/callbacks/train_speed_timer/base.yaml b/configs/callbacks/train_speed_timer/base.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..8fd28a7ee6e8a0873a37c749987541eed5943d47
--- /dev/null
+++ b/configs/callbacks/train_speed_timer/base.yaml
@@ -0,0 +1,3 @@
+base:
+ _target_: genmo.callbacks.train_speed_timer.TrainSpeedTimer
+ N_avg: 5
diff --git a/configs/callbacks/vis/vis_music.yaml b/configs/callbacks/vis/vis_music.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..a8d8a8bc8de8e7e016b13f08de5c3ed029580d56
--- /dev/null
+++ b/configs/callbacks/vis/vis_music.yaml
@@ -0,0 +1,2 @@
+vis_music:
+ _target_: genmo.callbacks.vis.vis_music.VisMusic
diff --git a/configs/callbacks/vis/vis_speech.yaml b/configs/callbacks/vis/vis_speech.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..4c19661356679f6d194d92d634a417fe6c67e31e
--- /dev/null
+++ b/configs/callbacks/vis/vis_speech.yaml
@@ -0,0 +1,2 @@
+vis_speech:
+ _target_: genmo.callbacks.vis.vis_speech.VisSpeech
diff --git a/configs/callbacks/vis/vis_text.yaml b/configs/callbacks/vis/vis_text.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..6efd239e959a5edfc4ecdde0f7631dc13c947b0a
--- /dev/null
+++ b/configs/callbacks/vis/vis_text.yaml
@@ -0,0 +1,2 @@
+vis_text:
+ _target_: genmo.callbacks.vis.vis_text.VisText
diff --git a/configs/callbacks/vis/vis_unity_val.yaml b/configs/callbacks/vis/vis_unity_val.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..7db1f7008463ba22d15f87c772be99882c6df346
--- /dev/null
+++ b/configs/callbacks/vis/vis_unity_val.yaml
@@ -0,0 +1,17 @@
+vis_unity_val:
+ _target_: genmo.callbacks.vis.vis_unity_val.VisUnityVal
+ enabled: false
+ every_n_epochs: 1
+ num_batches: 1
+ # Which val batches to render: "first" or "random".
+ batch_select: "first"
+ batch_select_seed: 123
+ num_frames: 30
+ render_incam: true
+ render_global: true
+ use_gt_betas_for_pred: true
+ global_root_relative: false
+ crf: 23
+ save_dir: ${output_dir}/vis
+ pred_color: [176, 100, 244]
+ gt_color: [0, 255, 0]
diff --git a/configs/data/collate_cfg/default.yaml b/configs/data/collate_cfg/default.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..ce9a93aea964a8b01cff0c6ec0c7fd857beab5bc
--- /dev/null
+++ b/configs/data/collate_cfg/default.yaml
@@ -0,0 +1,23 @@
+max_motion_frames: ${data.dataset_opts.max_motion_frames}
+default_frame_feature_dim:
+ music_array: [1024]
+ music_embed: [35]
+ music_beats: []
+ audio_array: []
+ use_det_kp: []
+
+default_seq_feature_dim:
+ text_embed: [50, 1024]
+
+default_seq_feature_length_multiplier:
+ audio_array: 600
+
+default_feature_val:
+ caption: ""
+ music_fps: 30
+ audio_fps: 30
+ has_text: False
+ # has_audio: False
+ # has_music: False
+
+default_feature_type: {}
diff --git a/configs/data/mocap/trainX_testY.yaml b/configs/data/mocap/trainX_testY.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..d80295cf28d7cf1349f175d45eb1b4b456e14cd8
--- /dev/null
+++ b/configs/data/mocap/trainX_testY.yaml
@@ -0,0 +1,21 @@
+defaults:
+ - collate_cfg: default
+
+# definition of lightning datamodule (dataset + dataloader)
+_target_: genmo.datamodule.mocap_trainX_testY.DataModule
+
+dataset_opts:
+ train: ${train_datasets}
+ val: ${test_datasets}
+ max_motion_frames: 120
+
+loader_opts:
+ train:
+ batch_size: 128
+ num_workers: 8
+ val:
+ batch_size: 1
+ num_workers: 1
+ encoded_music_dim: ${pipeline.args.encoded_music_dim}
+
+limit_each_trainset: null
diff --git a/configs/demo.yaml b/configs/demo.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..792b8aad7ec6d2fcfe9c0f562661bd3180d944ba
--- /dev/null
+++ b/configs/demo.yaml
@@ -0,0 +1,85 @@
+defaults:
+ # pytorch-lightning
+ - data: ???
+ - model: ???
+ - /text_encoder@model.model_cfg.text_encoder: t5_3b
+ - callbacks: null
+
+ # system
+ - hydra: default
+
+ # utility groups that changes a lot
+ - pipeline: null
+ - network: null
+ - optimizer: null
+ - scheduler: null
+ - train_datasets: null
+ - test_datasets: null
+ - endecoder: null # normalize/unnormalize data
+ - refiner: null
+
+ # global-override
+ - exp: mixed # set "data, model and callbacks" in yaml
+ - global/task: null # dump/test
+ - global/hsearch: null # hyper-param search
+ - global/debug: null # debug mode
+ - _self_
+
+# ================================ #
+# global setting #
+# ================================ #
+
+
+# expirement information
+task: fit # [fit, predict]
+exp_name_base: ???
+exp_name_var: ""
+exp_name: ${exp_name_base}_${exp_name_var}
+data_name: ???
+
+# utilities in the entry file
+# output_dir: "outputs/${data_name}/${exp_name}"
+resume_mode: null
+seed: 42
+
+version: null
+ckpt_dir: outputs/${data_name}/${exp_name}/
+remote_results_path: /lustre/fsw/portfolios/nvr/projects/nvr_torontoai_humanmotionfm/workspaces/motiondiff/motiondiff_results/jiefengl/gvhmr
+ckpt_path: null
+
+###
+# W&B logging removed from this repo; TensorBoard is used by `scripts/train.py`.
+rsync_ckpt: true
+
+
+# ================================ #
+# global setting #
+# ================================ #
+
+video_name: ???
+output_root: outputs/demo
+output_dir: "${output_root}/${text1_video_name}"
+preprocess_dir: ${output_dir}/preprocess
+video_path: "${output_dir}/0_input_video.mp4"
+
+# Options
+text1: null
+text1_file: null
+text1_video_path: null
+text1_video_name: null
+text_length: 300
+static_cam: False
+verbose: False
+
+paths:
+ bbx: ${preprocess_dir}/bbx.pt
+ bbx_xyxy_video_overlay: ${preprocess_dir}/bbx_xyxy_video_overlay.mp4
+ vit_features: ${preprocess_dir}/vit_features.pt
+ vimo_pred: ${preprocess_dir}/vimo_pred.pt
+ vitpose: ${preprocess_dir}/vitpose.pt
+ vitpose_video_overlay: ${preprocess_dir}/vitpose_video_overlay.mp4
+ hmr4d_results: ${output_dir}/hmr4d_results.pt
+ incam_video: ${output_dir}/1_incam.mp4
+ global_video: ${output_dir}/2_global.mp4
+ incam_global_horiz_video: ${output_dir}/3_incam_global_horiz.mp4
+ slam: ${preprocess_dir}/camera.npy
diff --git a/configs/diffusion/ddim.yaml b/configs/diffusion/ddim.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..f62baeebb6544a3f18ce85dc8503ac962049dd20
--- /dev/null
+++ b/configs/diffusion/ddim.yaml
@@ -0,0 +1,8 @@
+sampler: ddim
+train_timestep_respacing: ""
+test_timestep_respacing: "50"
+schedule_sampler_type: uniform
+noise_schedule: cosine
+sigma_small: true
+guidance_param: 1.0
+ddim_eta: 0.0
diff --git a/configs/endecoder/unity.yaml b/configs/endecoder/unity.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..2f64cd8d01e577e69cb388d9ac56b3561310e128
--- /dev/null
+++ b/configs/endecoder/unity.yaml
@@ -0,0 +1,2 @@
+_target_: genmo.network.endecoder.EnDecoder
+stats_name: MM_UNITY
diff --git a/configs/endecoder/v1_amass_local_bedlam_cam.yaml b/configs/endecoder/v1_amass_local_bedlam_cam.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..f3254bb0614bfd2ac4fe20e5810da02b0a14f6b3
--- /dev/null
+++ b/configs/endecoder/v1_amass_local_bedlam_cam.yaml
@@ -0,0 +1,2 @@
+_target_: genmo.network.endecoder.EnDecoder
+stats_name: MM_V1_AMASS_LOCAL_BEDLAM_CAM
diff --git a/configs/exp/genmo_lg.yaml b/configs/exp/genmo_lg.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..f8e6ec8a565e73f4fbf9ff3aa36a437802db78e9
--- /dev/null
+++ b/configs/exp/genmo_lg.yaml
@@ -0,0 +1,64 @@
+# @package _global_
+defaults:
+ - /diffusion@model_cfg.diffusion: ddim
+ - override /data: mocap/trainX_testY
+ - override /model: genmo
+ - override /network: diffusion
+ - override /pipeline: dual_mode
+ - override /endecoder: v1_amass_local_bedlam_cam
+ - override /optimizer: adamw_2e-4
+ - override /scheduler: epoch_half_200_350
+ - override /train_datasets:
+ - amass_train_v11
+ - humanml3d_static_train
+ - bedlam_v2
+ - h36m_v1
+ - 3dpw_v1
+ - 3dpw_occ_v1
+ - aistpp_train
+ - beat2_static_train
+ - override /test_datasets:
+ # - aistpp_test
+ - humanml3d_eval
+ - emdb1_fliptest
+ - emdb2_fliptest
+ - rich_test
+ - 3dpw_fliptest
+ - 3dpw_occ_fliptest
+ - override /callbacks:
+ - ckpt_saver/every10000s_top100
+ - prog_bar/prog_reporter_ed1
+ - train_speed_timer/base
+ - lr_monitor/pl
+ - vis/vis_text
+ - metric/metric_emdb1
+ - metric/metric_emdb2
+ - metric/metric_rich
+ - metric/metric_3dpw
+ - metric/metric_3dpw_occ
+ # - metric_aistpp
+ - _self_
+
+exp_name_base: ${hydra:runtime.choices.exp}
+exp_name_var: ""
+exp_name: ${exp_name_base}_${exp_name_var}
+data_name: genmo_mixed
+
+multicond_args: null
+
+pl_trainer:
+ precision: 16-mixed
+ log_every_n_steps: 10
+ gradient_clip_val: 0.5
+ max_epochs: null
+ check_val_every_n_epoch: null
+ val_check_interval: 3000
+ max_steps: 200000
+ devices: 1
+ strategy: ddp_find_unused_parameters_true
+
+logger:
+ _target_: pytorch_lightning.loggers.tensorboard.TensorBoardLogger
+ save_dir: ${output_dir}
+ name: ""
+ version: ""
diff --git a/configs/finetune_unity.yaml b/configs/finetune_unity.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..ae00a6962c1a3bb03004766964c4d3a171b7df5a
--- /dev/null
+++ b/configs/finetune_unity.yaml
@@ -0,0 +1,143 @@
+# genmo/configs/finetune_unity.yaml
+defaults:
+ - train
+ - override /exp: genmo_lg
+ - override /optimizer: adamw_5e-5
+ - override /scheduler: cosine_50
+ # Keep only generic callbacks; drop dataset-specific metrics/visualizers.
+ - override /callbacks:
+ - ckpt_saver/every10000s_top100
+ - prog_bar/prog_reporter_ed1
+ - train_speed_timer/base
+ - lr_monitor/pl
+ - metric/metric_unity
+ - vis/vis_unity_val
+ - _self_
+
+# Fix logging path mismatch by forcing filename to be local to the run dir
+hydra:
+ job_logging:
+ handlers:
+ file:
+ filename: train.log
+
+# Define mandatory variables and sync output_dir with Hydra run dir
+data_name: "unity"
+exp_name_base: "finetune"
+# Keep `output_dir` from `configs/train.yaml` to avoid a Hydra/OmegaConf interpolation cycle:
+# `hydra.run.dir` -> `${output_dir}` (configs/hydra/default.yaml) and `output_dir` -> `${hydra:run.dir}` would recurse.
+
+# Save a checkpoint every N epochs.
+callbacks:
+ ckpt_saver:
+ every10000s_top100:
+ every_n_steps: null
+ every_n_epochs: 100
+ save_top_k: 1
+ vis:
+ vis_unity_val:
+ enabled: true
+ batch_select: "random"
+ batch_select_seed: 123
+ pad_incam_canvas: false
+ incam_background: "video"
+
+train_datasets:
+ unity:
+ _target_: genmo.datasets.unity_dataset.UnityDataset
+ root: "./processed_dataset"
+ # ORIGINAL dataset folder (same as `third_party/GVHMR/process_data.sh --input ...`) for mp4 backgrounds.
+ raw_root: "/mnt/c/Temp/SyntheticDataset"
+ split: "train"
+ motion_frames: 120 # Must be >= 91 for augmentation (L - 90 > 0)
+ # Keep the exact coordinate convention exported by `third_party/GVHMR/tools/demo/process_dataset.py`.
+ # Any extra basis swap here will desync `T_w2c`/cam velocities from the stored features/crops.
+ convert_world_to_az: false
+ # Use detector/VitPose kp2d as conditioning (inference-style).
+ vitpose_like: true
+ kp2d_clamp_to_image: false
+ kp2d_zero_oof: true
+ # Explicitly disable datasets inherited from `exp=genmo_lg`.
+ amass_train_v11: null
+ humanml3d_static_train: null
+ bedlam_v2: null
+ h36m_v1: null
+ 3dpw_v1: null
+ 3dpw_occ_v1: null
+ aistpp_train: null
+ beat2_static_train: null
+
+test_datasets:
+ unity_val:
+ _target_: genmo.datasets.unity_dataset.UnityDataset
+ root: "./processed_dataset"
+ raw_root: "/mnt/c/Temp/SyntheticDataset"
+ split: "train"
+ motion_frames: 120 # Must match train setting
+ convert_world_to_az: false
+ vitpose_like: true
+ kp2d_clamp_to_image: false
+ kp2d_zero_oof: true
+ # Explicitly disable test datasets inherited from `exp=genmo_lg`.
+ humanml3d_eval: null
+ emdb1_fliptest: null
+ emdb2_fliptest: null
+ rich_test: null
+ 3dpw_fliptest: null
+ 3dpw_occ_fliptest: null
+
+# Fine-tuning Hyperparameters
+# Lightning Trainer settings
+pl_trainer:
+ max_epochs: 50 # More epochs for full adaptation
+ check_val_every_n_epoch: 1
+ log_every_n_steps: 1
+ precision: 16-mixed # Must match checkpoint (was trained with fp16)
+ gradient_clip_val: 0.5 # Tighter clipping (was 1.0)
+ val_check_interval: 1.0
+ limit_val_batches: 1.0
+ accumulate_grad_batches: 4 # Effective batch size = 4 * batch_size
+
+# Fine-tune stability overrides:
+# - Regression-only prevents diffusion loss from destabilizing global trajectory on small datasets.
+# - Disable heavy masking/occlusion augmentation used for large-scale pretraining.
+model:
+ model_cfg:
+ train_modes: ["regression"]
+ mask_transl_vel_y: false
+ condition_mask:
+ mask_img_prob: 0.0
+ mask_cam_prob: 0.0
+ mask_cfg:
+ drop_prob: 0.0
+ body_mask_cfg:
+ drop_prob: 0.0
+
+# Disable huge reprojection/vertex losses for Unity fine-tune; keep global rollout + static-conf.
+pipeline:
+ args:
+ transl_w_xz_only: false
+ pp_ground: false
+ weights:
+ cr_j3d: 0.0
+ # Keep incam stable: supervise pred_cam (via gt transl_c -> gt_pred_cam).
+ transl_c: 1.0
+ cr_verts: 0.0
+ j2d: 0.0
+ j2d_17: 0.0
+ verts2d: 0.0
+ transl_w: 1.0
+ static_conf_bce: 1.0
+ gogv_mult: 1.0
+ transl_vel_mult: 1.0
+
+# Override the default dktaloader settings from `exp=genmo_lg` (it uses batch_size=128
+# and the DataModule uses `drop_last=True`, which yields 0 batches for small Unity sets).
+data:
+ loader_opts:
+ train:
+ batch_size: 2
+ num_workers: 2
+ val:
+ batch_size: 2
+ num_workers: 2
diff --git a/configs/hydra/default.yaml b/configs/hydra/default.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..3cadfdefe6165080f754ceb3495f2786ed365f86
--- /dev/null
+++ b/configs/hydra/default.yaml
@@ -0,0 +1,19 @@
+# enable color logging
+defaults:
+ - override hydra_logging: colorlog
+ - override job_logging: colorlog
+
+job_logging:
+ formatters:
+ simple:
+ datefmt: "%m/%d %H:%M:%S"
+ format: "[%(asctime)s][%(levelname)s] %(message)s"
+ colorlog:
+ datefmt: "%m/%d %H:%M:%S"
+ format: "[%(cyan)s%(asctime)s%(reset)s][%(log_color)s%(levelname)s%(reset)s] %(message)s"
+ handlers:
+ file:
+ filename: ${output_dir}/${hydra.job.name}.log
+
+run:
+ dir: ${output_dir}
diff --git a/configs/infer_video.yaml b/configs/infer_video.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..b4e717b51704e69c2a96ba32f4a20291daca2264
--- /dev/null
+++ b/configs/infer_video.yaml
@@ -0,0 +1,85 @@
+defaults:
+ # pytorch-lightning / hydra wiring (kept for compatibility with `exp=...` configs)
+ - data: ???
+ - model: ???
+ - callbacks: null
+ - hydra: default
+ - pipeline: null
+ - network: null
+ - optimizer: null
+ - scheduler: null
+ - train_datasets: null
+ - test_datasets: null
+ - endecoder: null
+ - refiner: null
+
+ # pick an experiment preset (sets data/model/network/pipeline/etc)
+ - exp: genmo_lg
+ - _self_
+
+# Video -> SMPL-X inference (GENMO/GEM)
+video_path: null
+video_name: null
+
+output_root: outputs/infer_video
+output_dir: ${output_root}/${video_name}
+preprocess_dir: ${output_dir}/preprocess
+
+# Checkpoint
+ckpt_path: null
+
+# Inference options
+static_cam: true
+use_kp2d: true
+postproc: true
+use_sam_masking: true # Apply SAM masks to VitPose and ViT features (removes background/other people)
+run_hamer: false # Run HaMeR for hand mesh recovery (adds hand poses to SMPL-X)
+resample_to_30fps: true
+verbose: false
+
+# Rendering
+render_incam: true
+render_global: true
+render_side_by_side: true
+render_crf: 23
+
+# Optional: face visibility + emotion classification from face crop.
+# Uses COCO17 head keypoints to estimate a face bbox; classifier is loaded from HF cache by default (offline-friendly).
+emotion:
+ enabled: true
+ model_id: clip:openai/clip-vit-base-patch32
+ cache_dir: ./third_party/GVHMR/.cache/huggingface
+ local_files_only: true
+ min_kpt_conf: 0.3
+ min_visible_kpts: 3
+ face_bbox_scale: 2.0
+ # Optional alternative output location; `${preprocess_dir}/emotion.jsonl` is always preferred for overlays.
+ output_path: null
+
+# Debug dump: save inputs + model outputs for offline analysis.
+dump_io: false
+dump_io_path: ${output_dir}/debug_io.pt
+
+# Optional visualization: draw estimated camera axes in the global render.
+draw_camera_axes: false
+# Camera pose convention for `paths.slam` (affects camera-axis visualization only):
+# - auto: choose the one closest to the person root each frame
+# - w2c: interpret trajectory as world->camera
+# - c2w: interpret trajectory as camera->world
+camera_pose_convention: auto
+camera_axis_length: 0.5
+camera_axis_width: 3
+
+paths:
+ input_video: ${output_dir}/0_input_video.mp4
+ video_30fps: ${output_dir}/0_input_video_30fps.mp4
+ bbx: ${preprocess_dir}/bbx.pt
+ vitpose: ${preprocess_dir}/vitpose.pt
+ vit_features: ${preprocess_dir}/vit_features.pt
+ hmr4d_results: ${output_dir}/hmr4d_results.pt
+ incam_video: ${output_dir}/1_incam.mp4
+ global_video: ${output_dir}/2_global.mp4
+ incam_global_horiz_video: ${output_dir}/3_incam_global_horiz.mp4
+# Disable external logging by default for a local demo script.
+###
+# W&B logging removed from this repo; TensorBoard is used by `scripts/train.py` for training runs.
diff --git a/configs/model/genmo.yaml b/configs/model/genmo.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..20789b900794453f505f5292910a70c4356fd509
--- /dev/null
+++ b/configs/model/genmo.yaml
@@ -0,0 +1,45 @@
+_target_: genmo.genmo.GENMO
+
+pipeline: ${pipeline}
+optimizer: ${optimizer}
+scheduler: ${scheduler}
+
+model_cfg:
+ train_modes: ["regression", "diffusion"]
+ noisy_2d_obs: true
+ kp2d_noise_scale: 0.5
+ perframe_condition_exists: true
+ train2d_mask_invis_obs: true
+ mask_occluded_imgfeats: true
+ cond_merge_strategy: "add"
+ use_cond_exists_as_input: true
+ normalize_cam_angvel: true
+
+ diffusion:
+ test_timestep_respacing: "50"
+ guidance_param: 2.5
+
+ text_encoder:
+ load_llm: false
+ llm_version: "t5-3b"
+ max_text_len: 50
+
+ condition_mask:
+ mask_img_prob: 0.5
+ mask_cam_prob: 1.0
+ reuse_regression_mask: false
+ regression_no_img_mask: true
+
+ mask_cfg:
+ drop_prob: 0.75
+ max_num_drops: 3
+ min_drop_nframes: 1
+ max_drop_nframes: 30
+ body_mask_cfg:
+ drop_prob: 0.75
+ joint_drop_prob: 0.25
+ max_num_drops: 3
+ min_drop_nframes: 1
+ max_drop_nframes: 30
+ music_mask_prob: 0.1
+ audio_mask_prob: 0.1
diff --git a/configs/network/diffusion.yaml b/configs/network/diffusion.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..62febb1b9076a009b69bf7ba70ed6d5e8c1b9a8c
--- /dev/null
+++ b/configs/network/diffusion.yaml
@@ -0,0 +1,25 @@
+_target_: genmo.network.genmo_diffusion.GENMODiffusion
+args: ${pipeline.args}
+latent_dim: ${.model_cfg.denoiser.latent_dim}
+cond_merge_strategy: "add"
+music_mask_prob: ${.model_cfg.denoiser.music_mask_prob}
+speech_mask_prob: ${.model_cfg.denoiser.speech_mask_prob}
+encoded_music_dim: ${pipeline.args.encoded_music_dim}
+model_cfg:
+ diffusion: ${model_cfg.diffusion}
+ denoiser:
+ _target_: genmo.network.genmo_denoiser.NetworkEncoderRoPE
+ output_dim: 151
+ xt_dim: ${.output_dim}
+ njoints: ${.xt_dim}
+ text_mask_prob: 0.1
+ music_mask_prob: 0.1
+ speech_mask_prob: 0.1
+ use_text_pos_enc: true
+ text_encoder_cfg:
+ mode: all
+ cross_attn_type: mha
+ latent_dim: 1024
+ num_layers: 16
+ num_heads: 8
+ mlp_ratio: 4
diff --git a/configs/optimizer/adamw_2e-4.yaml b/configs/optimizer/adamw_2e-4.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..6554f2bbee50d913f010cd840b8f1905d9edf382
--- /dev/null
+++ b/configs/optimizer/adamw_2e-4.yaml
@@ -0,0 +1,2 @@
+_target_: torch.optim.AdamW
+lr: 2e-4
diff --git a/configs/optimizer/adamw_5e-5.yaml b/configs/optimizer/adamw_5e-5.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..4d07c8c1203564afad3fb818a06818357f8d1ef2
--- /dev/null
+++ b/configs/optimizer/adamw_5e-5.yaml
@@ -0,0 +1,2 @@
+_target_: torch.optim.AdamW
+lr: 5e-5
diff --git a/configs/pipeline/dual_mode.yaml b/configs/pipeline/dual_mode.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..c80bcb9851bb4b29e6dee3955498b5639b08578a
--- /dev/null
+++ b/configs/pipeline/dual_mode.yaml
@@ -0,0 +1,37 @@
+_target_: genmo.pipeline.genmo_pipeline.Pipeline
+args_denoiser3d: ${network}
+args:
+ endecoder_opt: ${endecoder}
+ use_regression_outputs_prob: 0.
+ use_cfg_sampler_for_gen: true
+ inpaint_x_start_gt: false
+ regression_only: true
+ encoded_music_dim: 35
+ multicond_args: ${multicond_args}
+ infer_version: 2
+ weights:
+ cr_j3d: 500.
+ transl_c: 1.
+ cr_verts: 500.
+ j2d: 1000.
+ j2d_17: 1000.
+ verts2d: 1000.
+
+ proj_gt_j2d_to_bi01: true
+
+ transl_w: 1.
+ static_conf_bce: 1.
+
+ static_conf:
+ vel_thr: 0.15
+
+ in_attr:
+ - obs
+ - f_cliffcam
+ - f_imgseq
+ - f_cam_angvel
+ - encoded_music
+ - encoded_audio
+ mask_out_attr: [] # ${.in_attr}
+ out_attr:
+ pred_cam: 3
diff --git a/configs/scheduler/cosine_50.yaml b/configs/scheduler/cosine_50.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..f7b33ab13d89f6b29ccc97e6dd32fa4ff41aec51
--- /dev/null
+++ b/configs/scheduler/cosine_50.yaml
@@ -0,0 +1,6 @@
+scheduler:
+ _target_: torch.optim.lr_scheduler.CosineAnnealingLR
+ T_max: 50
+ eta_min: 1e-6
+interval: epoch
+frequency: 1
diff --git a/configs/scheduler/epoch_half_200_350.yaml b/configs/scheduler/epoch_half_200_350.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..e1c1fdae29a66904c4243ec75e1323049d9b9e87
--- /dev/null
+++ b/configs/scheduler/epoch_half_200_350.yaml
@@ -0,0 +1,6 @@
+scheduler:
+ _target_: torch.optim.lr_scheduler.MultiStepLR
+ milestones: [200, 350]
+ gamma: 0.5
+interval: epoch
+frequency: 1
diff --git a/configs/test_datasets/3dpw_fliptest.yaml b/configs/test_datasets/3dpw_fliptest.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..1c19addea2c2aa78c8fb2821e0d121b6d24bc04b
--- /dev/null
+++ b/configs/test_datasets/3dpw_fliptest.yaml
@@ -0,0 +1,3 @@
+3dpw_fliptest:
+ _target_: genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset
+ flip_test: true
diff --git a/configs/test_datasets/3dpw_occ_fliptest.yaml b/configs/test_datasets/3dpw_occ_fliptest.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..b18e7294af9bd0133b51b650029c824f5ae56fcf
--- /dev/null
+++ b/configs/test_datasets/3dpw_occ_fliptest.yaml
@@ -0,0 +1,3 @@
+3dpw_occ_fliptest:
+ _target_: genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset
+ flip_test: true
diff --git a/configs/test_datasets/emdb1_fliptest.yaml b/configs/test_datasets/emdb1_fliptest.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..ea4ed5d8a10abef14ce936b6f505c258ca7fbffa
--- /dev/null
+++ b/configs/test_datasets/emdb1_fliptest.yaml
@@ -0,0 +1,4 @@
+emdb1_fliptest:
+ _target_: genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset
+ split: 1
+ flip_test: true
diff --git a/configs/test_datasets/emdb2_fliptest.yaml b/configs/test_datasets/emdb2_fliptest.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..5dfa63536e8accf8d34f719085334825fbe97c77
--- /dev/null
+++ b/configs/test_datasets/emdb2_fliptest.yaml
@@ -0,0 +1,4 @@
+emdb2_fliptest:
+ _target_: genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset
+ split: 2
+ flip_test: true
diff --git a/configs/test_datasets/humanml3d_eval.yaml b/configs/test_datasets/humanml3d_eval.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..93b172388ba789ecbc84ab6d6d971a3379b62c66
--- /dev/null
+++ b/configs/test_datasets/humanml3d_eval.yaml
@@ -0,0 +1,7 @@
+humanml3d_eval:
+ _target_: genmo.datasets.pure_motion.humanml3d.Humanml3dDataset
+ eval_gen_only: true
+ cam_augmentation: v11
+ use_random_subset: true
+ random_subset_size: 2
+ random_subset_seed: 7
diff --git a/configs/test_datasets/rich_test.yaml b/configs/test_datasets/rich_test.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..03fd77da693fcd0e03f0855a988da9f7c6401f9f
--- /dev/null
+++ b/configs/test_datasets/rich_test.yaml
@@ -0,0 +1,2 @@
+rich_test:
+ _target_: genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset
diff --git a/configs/text_encoder/t5_3b.yaml b/configs/text_encoder/t5_3b.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..07f1b822754be2398d71da391a288c7ded9774c2
--- /dev/null
+++ b/configs/text_encoder/t5_3b.yaml
@@ -0,0 +1,3 @@
+load_llm: true
+llm_version: "t5-3b"
+max_text_len: 50
\ No newline at end of file
diff --git a/configs/train.yaml b/configs/train.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..23de9b3286a13193bacb626fb10e034c0232a330
--- /dev/null
+++ b/configs/train.yaml
@@ -0,0 +1,55 @@
+defaults:
+ - _self_
+ # pytorch-lightning
+ - data: ???
+ - model: ???
+ - callbacks: null
+
+ # system
+ - hydra: default
+
+ # utility groups that changes a lot
+ - pipeline: null
+ - network: null
+ - optimizer: null
+ - scheduler: null
+ - train_datasets: null
+ - test_datasets: null
+ - endecoder: null # normalize/unnormalize data
+ - refiner: null
+
+ # global-override
+ - exp: mixed # set "data, model and callbacks" in yaml
+ - global/task: null # dump/test
+ - global/hsearch: null # hyper-param search
+ - global/debug: null # debug mode
+
+# ================================ #
+# global setting #
+# ================================ #
+# expirement information
+task: fit # [fit, predict]
+exp_name_base: ???
+exp_name_var: ""
+exp_name: ${exp_name_base}_${exp_name_var}
+data_name: ???
+num_test_data: 32
+
+# utilities in the entry file
+output_dir: "outputs/${data_name}/${exp_name}"
+ckpt_path: null
+resume_mode: null
+seed: 42
+
+# lightning default settings
+pl_trainer:
+ devices: 1
+ num_sanity_val_steps: 0 # disable sanity check
+ precision: 32
+ inference_mode: False
+
+logger:
+ _target_: pytorch_lightning.loggers.tensorboard.TensorBoardLogger
+ save_dir: ${output_dir}
+ name: ""
+ version: ""
diff --git a/configs/train_datasets/3dpw_occ_v1.yaml b/configs/train_datasets/3dpw_occ_v1.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..6c2b3a8654f6e967ddd2db5253df9a84590c1941
--- /dev/null
+++ b/configs/train_datasets/3dpw_occ_v1.yaml
@@ -0,0 +1,2 @@
+3dpw_occ_v1:
+ _target_: genmo.datasets.threedpw.threedpw_occ_motion_train.ThreedpwOccSmplDataset
diff --git a/configs/train_datasets/3dpw_v1.yaml b/configs/train_datasets/3dpw_v1.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..b6014b73d515e27d31032314f8f95c2f4cd9b6b9
--- /dev/null
+++ b/configs/train_datasets/3dpw_v1.yaml
@@ -0,0 +1,2 @@
+3dpw_v1:
+ _target_: genmo.datasets.threedpw.threedpw_motion_train.ThreedpwSmplDataset
diff --git a/configs/train_datasets/aistpp_train.yaml b/configs/train_datasets/aistpp_train.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..d9afd308a8f2c97864a43aa8d16ace58d8b45a4e
--- /dev/null
+++ b/configs/train_datasets/aistpp_train.yaml
@@ -0,0 +1,7 @@
+aistpp_train:
+ _target_: genmo.datasets.aistplusplus.aistplusplus.AISTPlusPlusSmplDataset
+ split: train
+ motion_frames: 120
+ lazy_load: false
+ eval_gen_only: true
+ feat_version: v2
diff --git a/configs/train_datasets/amass_train_v11.yaml b/configs/train_datasets/amass_train_v11.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..6a5625b234d1ab5994deaa93f386d14ad57888d4
--- /dev/null
+++ b/configs/train_datasets/amass_train_v11.yaml
@@ -0,0 +1,9 @@
+amass_train_v11:
+ _target_: genmo.datasets.pure_motion.amass.AmassDataset
+
+ motion_frames: 120
+ l_factor: 1.5
+ skip_moyo: True
+ cam_augmentation: v11
+ random1024: False
+ limit_size: null
diff --git a/configs/train_datasets/beat2_static_train.yaml b/configs/train_datasets/beat2_static_train.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..686533094926f21b4a0ba4971f7b3d5e600fef95
--- /dev/null
+++ b/configs/train_datasets/beat2_static_train.yaml
@@ -0,0 +1,6 @@
+beat2_static_train:
+ _target_: genmo.datasets.beat2.beat2.BEAT2SmplDataset
+ split: train
+ cam_augmentation: static
+ motion_frames: 120
+ lazy_load: false
diff --git a/configs/train_datasets/bedlam_v2.yaml b/configs/train_datasets/bedlam_v2.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..840d8754568892270e575b4c3425a71e815edf92
--- /dev/null
+++ b/configs/train_datasets/bedlam_v2.yaml
@@ -0,0 +1,2 @@
+bedlam_v2:
+ _target_: genmo.datasets.bedlam.bedlam.BedlamDatasetV2
diff --git a/configs/train_datasets/h36m_v1.yaml b/configs/train_datasets/h36m_v1.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..aa4b58998be00a501431b7b515da74ccb52b5f62
--- /dev/null
+++ b/configs/train_datasets/h36m_v1.yaml
@@ -0,0 +1,7 @@
+h36m_v1:
+ _target_: genmo.datasets.h36m.h36m.H36mSmplDataset
+
+ root: inputs/H36M/hmr4d_support
+ original_coord: az
+ motion_frames: 120
+ lazy_load: false
diff --git a/configs/train_datasets/humanml3d_static_train.yaml b/configs/train_datasets/humanml3d_static_train.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..7522d5b5ddc71c880eabdb1941dd3efdf9478c6a
--- /dev/null
+++ b/configs/train_datasets/humanml3d_static_train.yaml
@@ -0,0 +1,6 @@
+humanml3d_static_train:
+ _target_: genmo.datasets.pure_motion.humanml3d.Humanml3dDataset
+
+ motion_frames: 120
+ cam_augmentation: static
+ split: train
diff --git a/configs/train_datasets/unity_finetune.yaml b/configs/train_datasets/unity_finetune.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..b06c228907a16550a90f26f8aabfc4380974f6fb
--- /dev/null
+++ b/configs/train_datasets/unity_finetune.yaml
@@ -0,0 +1,6 @@
+# genmo/configs/dataset/unity_finetune.yaml
+_target_: genmo.datasets.unity_dataset.UnityDataset
+
+root: "./third_party/GVHMR/processed_dataset"
+split: "train"
+motion_frames: 120
diff --git a/debug_compare_unity_pt.py b/debug_compare_unity_pt.py
new file mode 100644
index 0000000000000000000000000000000000000000..5be32729be655e7b03eefa36ebd363647a957382
--- /dev/null
+++ b/debug_compare_unity_pt.py
@@ -0,0 +1,263 @@
+#!/usr/bin/env python3
+"""
+Compare Unity GENMO-exported .pt inputs between two sequences.
+
+Goal: find *input* differences (K/bbox/kp2d/cam vel) that could plausibly cause
+incam instability/oscillation for a specific clip.
+
+Run:
+ /root/miniconda3/envs/gvhmr/bin/python debug_compare_unity_pt.py \\
+ --a 101_biboo_birthday_speech_explosion_2 \\
+ --b 107_biboo_birthday_speech_explosion_8
+"""
+
+from __future__ import annotations
+
+import argparse
+from dataclasses import dataclass
+from pathlib import Path
+
+import numpy as np
+import torch
+
+from genmo.utils.pylogger import Log
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import (
+ compute_bbox_info_bedlam,
+ get_bbx_xys,
+ get_a_pred_cam,
+ safely_render_x3d_K,
+)
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+@dataclass
+class SeqStats:
+ vid: str
+ L: int
+ W: int
+ H: int
+ fx_mean: float
+ fx_std: float
+ bbx_saved_c_std: tuple[float, float]
+ bbx_saved_s_std: float
+ bbx_gt_c_std: tuple[float, float]
+ bbx_gt_s_std: float
+ bbx_delta_c_mean: tuple[float, float]
+ bbx_delta_c_std: tuple[float, float]
+ bbx_delta_s_mean: float
+ bbx_delta_s_std: float
+ fcliff_saved_std: tuple[float, float, float]
+ fcliff_gt_std: tuple[float, float, float]
+ fcliff_delta_std: tuple[float, float, float]
+ kp2d_conf_gt05_frac: float
+ kp2d_oof_conf_gt05_frac: float
+ gt_pred_cam_std: tuple[float, float, float]
+ gt_pred_cam_d_std: tuple[float, float, float]
+ gt_pred_cam_d_p95: tuple[float, float, float]
+ gt_pred_cam_norm_std: float
+ gt_pred_cam_d_norm_p95: float
+
+
+def _as_np(x: torch.Tensor) -> np.ndarray:
+ return x.detach().cpu().numpy()
+
+
+def _infer_wh_from_K(K_fullimg: np.ndarray) -> tuple[int, int]:
+ cx = float(np.median(K_fullimg[:, 0, 2]))
+ cy = float(np.median(K_fullimg[:, 1, 2]))
+ W = int(round(cx * 2.0))
+ H = int(round(cy * 2.0))
+ return max(W, 1), max(H, 1)
+
+
+def compute_seq_stats(pt_path: Path, smplx_model) -> SeqStats:
+ data = torch.load(pt_path, map_location="cpu", weights_only=False)
+ vid = pt_path.stem
+ bbx_saved = _as_np(data["bbx_xys"]).astype(np.float64) # (L,3)
+ K = _as_np(data["K_fullimg"]).astype(np.float64) # (L,3,3)
+ kp2d = _as_np(data.get("kp2d", torch.zeros((bbx_saved.shape[0], 17, 3)))).astype(
+ np.float64
+ )
+ transl_c = _as_np(data["smpl_params_c"]["transl"]).astype(np.float64) # (L,3)
+
+ L = int(bbx_saved.shape[0])
+ W, H = _infer_wh_from_K(K)
+ fx = K[:, 0, 0]
+
+ # Compute GT-projected bbox from verts (same logic used during training when bbx is missing).
+ smpl_params_c = data["smpl_params_c"]
+ with torch.no_grad():
+ out = smplx_model(
+ global_orient=smpl_params_c["global_orient"].float(),
+ body_pose=smpl_params_c["body_pose"].float(),
+ betas=smpl_params_c["betas"].float(),
+ transl=smpl_params_c["transl"].float(),
+ )
+ verts = out.vertices # (L, V, 3)
+
+ verts_b = verts[None] # (1,L,V,3)
+ K_b = torch.from_numpy(K).float()[None] # (1,L,3,3)
+ i_x2d = safely_render_x3d_K(verts_b, K_b, thr=0.3) # (1,L,V,2)
+ bbx_gt = get_bbx_xys(i_x2d, do_augment=False)[0].detach().cpu().numpy().astype(np.float64) # (L,3)
+
+ # bbox stats
+ bbx_saved_c = bbx_saved[:, :2]
+ bbx_saved_s = bbx_saved[:, 2]
+ bbx_gt_c = bbx_gt[:, :2]
+ bbx_gt_s = bbx_gt[:, 2]
+
+ bbx_delta_c = bbx_saved_c - bbx_gt_c
+ bbx_delta_s = bbx_saved_s - bbx_gt_s
+
+ # f_cliffcam stats (this is what the network sees)
+ bbx_saved_t = torch.from_numpy(bbx_saved).float()
+ bbx_gt_t = torch.from_numpy(bbx_gt).float()
+ K_t = torch.from_numpy(K).float()
+ fcliff_saved = compute_bbox_info_bedlam(bbx_saved_t, K_t).numpy().astype(np.float64)
+ fcliff_gt = compute_bbox_info_bedlam(bbx_gt_t, K_t).numpy().astype(np.float64)
+ fcliff_delta = fcliff_saved - fcliff_gt
+
+ # Conditioning target used by incam translation loss: gt_pred_cam (s,tx,ty)
+ # (see `third_party/.../hmr_cam.py:get_a_pred_cam`).
+ gt_pred_cam = get_a_pred_cam(
+ torch.from_numpy(transl_c).float(),
+ bbx_saved_t,
+ K_t,
+ ).numpy().astype(np.float64) # (L,3)
+ d_gt_pred_cam = np.diff(gt_pred_cam, axis=0)
+ gt_pred_cam_std = gt_pred_cam.std(axis=0)
+ gt_pred_cam_norm_std = float(np.linalg.norm(gt_pred_cam - gt_pred_cam.mean(axis=0), axis=1).std())
+ gt_pred_cam_d_std = d_gt_pred_cam.std(axis=0)
+ gt_pred_cam_d_p95 = np.percentile(np.abs(d_gt_pred_cam), 95, axis=0)
+ gt_pred_cam_d_norm_p95 = float(np.percentile(np.linalg.norm(d_gt_pred_cam, axis=1), 95))
+
+ # kp2d sanity: how much of provided kp2d is confidently in-frame?
+ conf = kp2d[..., 2]
+ x = kp2d[..., 0]
+ y = kp2d[..., 1]
+ conf_gt05 = conf > 0.5
+ conf_gt05_frac = float(conf_gt05.mean()) if conf.size else 0.0
+ oof = (x < 0.0) | (x > (W - 1.0)) | (y < 0.0) | (y > (H - 1.0))
+ oof_conf = oof & conf_gt05
+ oof_conf_frac = float(oof_conf.sum() / max(conf_gt05.sum(), 1.0))
+
+ return SeqStats(
+ vid=vid,
+ L=L,
+ W=W,
+ H=H,
+ fx_mean=float(fx.mean()),
+ fx_std=float(fx.std()),
+ bbx_saved_c_std=(float(bbx_saved_c[:, 0].std()), float(bbx_saved_c[:, 1].std())),
+ bbx_saved_s_std=float(bbx_saved_s.std()),
+ bbx_gt_c_std=(float(bbx_gt_c[:, 0].std()), float(bbx_gt_c[:, 1].std())),
+ bbx_gt_s_std=float(bbx_gt_s.std()),
+ bbx_delta_c_mean=(
+ float(bbx_delta_c[:, 0].mean()),
+ float(bbx_delta_c[:, 1].mean()),
+ ),
+ bbx_delta_c_std=(float(bbx_delta_c[:, 0].std()), float(bbx_delta_c[:, 1].std())),
+ bbx_delta_s_mean=float(bbx_delta_s.mean()),
+ bbx_delta_s_std=float(bbx_delta_s.std()),
+ fcliff_saved_std=(
+ float(fcliff_saved[:, 0].std()),
+ float(fcliff_saved[:, 1].std()),
+ float(fcliff_saved[:, 2].std()),
+ ),
+ fcliff_gt_std=(
+ float(fcliff_gt[:, 0].std()),
+ float(fcliff_gt[:, 1].std()),
+ float(fcliff_gt[:, 2].std()),
+ ),
+ fcliff_delta_std=(
+ float(fcliff_delta[:, 0].std()),
+ float(fcliff_delta[:, 1].std()),
+ float(fcliff_delta[:, 2].std()),
+ ),
+ kp2d_conf_gt05_frac=conf_gt05_frac,
+ kp2d_oof_conf_gt05_frac=oof_conf_frac,
+ gt_pred_cam_std=(float(gt_pred_cam_std[0]), float(gt_pred_cam_std[1]), float(gt_pred_cam_std[2])),
+ gt_pred_cam_d_std=(float(gt_pred_cam_d_std[0]), float(gt_pred_cam_d_std[1]), float(gt_pred_cam_d_std[2])),
+ gt_pred_cam_d_p95=(float(gt_pred_cam_d_p95[0]), float(gt_pred_cam_d_p95[1]), float(gt_pred_cam_d_p95[2])),
+ gt_pred_cam_norm_std=gt_pred_cam_norm_std,
+ gt_pred_cam_d_norm_p95=gt_pred_cam_d_norm_p95,
+ )
+
+
+def _print_stats(s: SeqStats) -> None:
+ Log.info(f"=== {s.vid} ===")
+ Log.info(f"L={s.L} W×H={s.W}×{s.H} fx_mean/std={s.fx_mean:.3f}/{s.fx_std:.3f}")
+ Log.info(
+ "bbx_saved std: center(x/y)=(%.2f,%.2f) size=%.2f"
+ % (*s.bbx_saved_c_std, s.bbx_saved_s_std)
+ )
+ Log.info(
+ "bbx_gt std: center(x/y)=(%.2f,%.2f) size=%.2f"
+ % (*s.bbx_gt_c_std, s.bbx_gt_s_std)
+ )
+ Log.info(
+ "bbx(saved-gt) mean: center(x/y)=(%.2f,%.2f) size=%.2f"
+ % (*s.bbx_delta_c_mean, s.bbx_delta_s_mean)
+ )
+ Log.info(
+ "bbx(saved-gt) std : center(x/y)=(%.2f,%.2f) size=%.2f"
+ % (*s.bbx_delta_c_std, s.bbx_delta_s_std)
+ )
+ Log.info(
+ "f_cliff std saved=(%.4f,%.4f,%.4f) gt=(%.4f,%.4f,%.4f) delta_std=(%.4f,%.4f,%.4f)"
+ % (
+ *s.fcliff_saved_std,
+ *s.fcliff_gt_std,
+ *s.fcliff_delta_std,
+ )
+ )
+ Log.info(
+ "kp2d conf>0.5 frac=%.3f oof|conf>0.5 frac=%.3f"
+ % (s.kp2d_conf_gt05_frac, s.kp2d_oof_conf_gt05_frac)
+ )
+ Log.info(
+ "gt_pred_cam std(s/tx/ty)=(%.4f,%.4f,%.4f) norm_std=%.4f"
+ % (*s.gt_pred_cam_std, s.gt_pred_cam_norm_std)
+ )
+ Log.info(
+ "d(gt_pred_cam) std(s/tx/ty)=(%.4f,%.4f,%.4f) p95_abs(s/tx/ty)=(%.4f,%.4f,%.4f) p95_norm=%.4f"
+ % (*s.gt_pred_cam_d_std, *s.gt_pred_cam_d_p95, s.gt_pred_cam_d_norm_p95)
+ )
+
+
+def main() -> None:
+ ap = argparse.ArgumentParser()
+ ap.add_argument("--root", default="processed_dataset/genmo_features")
+ ap.add_argument("--a", required=True)
+ ap.add_argument("--b", required=True)
+ args = ap.parse_args()
+
+ root = Path(args.root)
+ pt_a = root / f"{args.a}.pt"
+ pt_b = root / f"{args.b}.pt"
+ if not pt_a.exists():
+ raise FileNotFoundError(pt_a)
+ if not pt_b.exists():
+ raise FileNotFoundError(pt_b)
+
+ smplx_model = make_smplx("supermotion").eval()
+
+ s_a = compute_seq_stats(pt_a, smplx_model)
+ s_b = compute_seq_stats(pt_b, smplx_model)
+
+ _print_stats(s_a)
+ _print_stats(s_b)
+
+ Log.info("=== delta (A - B) ===")
+ Log.info(
+ "fcliff_delta_std A vs B: (%.4f,%.4f,%.4f) vs (%.4f,%.4f,%.4f)"
+ % (*s_a.fcliff_delta_std, *s_b.fcliff_delta_std)
+ )
+ Log.info(
+ "bbx(saved-gt) center std A vs B: (%.2f,%.2f) vs (%.2f,%.2f)"
+ % (*s_a.bbx_delta_c_std, *s_b.bbx_delta_c_std)
+ )
+
+
+if __name__ == "__main__":
+ main()
diff --git a/debug_dino_init.jpg b/debug_dino_init.jpg
new file mode 100644
index 0000000000000000000000000000000000000000..3bf5312e65379cf327a7e567dbfb7cf2885e9ec7
--- /dev/null
+++ b/debug_dino_init.jpg
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c05612fc81427f6617daefa1328f20491e9c66e9c4d02b99d1c02a03eae4b8dc
+size 269726
diff --git a/debug_overlay.py b/debug_overlay.py
new file mode 100644
index 0000000000000000000000000000000000000000..d11582d18050443ba215059285460aa643badd05
--- /dev/null
+++ b/debug_overlay.py
@@ -0,0 +1,286 @@
+import os
+import cv2
+import torch
+import numpy as np
+
+# ---- paths (edit these) ----
+OUT_DIR = "./outputs/infer_video/VRM_JG6Z7WA_3780.00_3804.00"
+VIDEO_PATH = os.path.join(OUT_DIR, "1_incam.mp4")
+
+BBX_PATH = os.path.join(OUT_DIR, "preprocess", "bbx.pt")
+VITPOSE_PATH = os.path.join(OUT_DIR, "preprocess", "vitpose.pt")
+
+OUT_BBOX_ONLY = os.path.join(OUT_DIR, "debug_bbox_only_on_incam.mp4")
+OUT_BBOX_KP = os.path.join(OUT_DIR, "debug_bbox_kp_on_incam.mp4")
+# ---------------------------
+
+COCO17_NAMES = [
+ "nose", "l_eye", "r_eye", "l_ear", "r_ear",
+ "l_sho", "r_sho", "l_elb", "r_elb", "l_wri", "r_wri",
+ "l_hip", "r_hip", "l_knee", "r_knee", "l_ank", "r_ank"
+]
+
+
+def to_numpy(x):
+ if isinstance(x, torch.Tensor):
+ return x.detach().cpu().numpy()
+ return np.array(x)
+
+
+def xyxy_to_xys(bbx_xyxy_t: torch.Tensor) -> torch.Tensor:
+ """(L,4) xyxy -> (L,3) (cx,cy,s) where s is square side = max(w,h)."""
+ x1, y1, x2, y2 = bbx_xyxy_t.unbind(-1)
+ cx = (x1 + x2) * 0.5
+ cy = (y1 + y2) * 0.5
+ w = (x2 - x1).clamp(min=1.0)
+ h = (y2 - y1).clamp(min=1.0)
+ s = torch.maximum(w, h)
+ return torch.stack([cx, cy, s], dim=-1)
+
+
+def xys_to_xyxy(bbx_xys_t: torch.Tensor) -> torch.Tensor:
+ """(L,3) (cx,cy,s) -> (L,4) xyxy of square."""
+ cx, cy, s = bbx_xys_t.unbind(-1)
+ hs = s * 0.5
+ x1 = cx - hs
+ y1 = cy - hs
+ x2 = cx + hs
+ y2 = cy + hs
+ return torch.stack([x1, y1, x2, y2], dim=-1)
+
+
+def draw_bbox_xyxy(frame, xyxy, color=(0, 255, 0), thickness=2):
+ x1, y1, x2, y2 = [int(round(float(v))) for v in xyxy]
+ cv2.rectangle(frame, (x1, y1), (x2, y2), color, thickness)
+ return frame
+
+
+def draw_kps_with_names(frame, kps_xy, conf=None, radius=3, show_conf=True, conf_thr=0.0):
+ """
+ kps_xy: (J,2) in image pixels
+ conf: (J,) optional
+ - Always renders the label (including confidence) so you can verify confidence behavior.
+ - Text is BLACK with a white background box for readability.
+ """
+ H, W = frame.shape[:2]
+
+ for j, (x, y) in enumerate(kps_xy):
+ x_i, y_i = int(round(float(x))), int(round(float(y)))
+ if x_i < 0 or y_i < 0 or x_i >= W or y_i >= H:
+ continue
+
+ # Confidence
+ if conf is None:
+ c = 1.0
+ else:
+ c = float(conf[j])
+
+ ok = (c >= conf_thr)
+
+ # Draw joint marker (keep your preferred colors; this is just for visibility)
+ if conf is None:
+ pt_color = (0, 0, 255) # red
+ else:
+ pt_color = (0, 0, 255) if ok else (150, 150, 150) # red if ok, gray if low
+
+ cv2.circle(frame, (x_i, y_i), radius, pt_color, -1)
+
+ # Label
+ name = COCO17_NAMES[j] if j < len(COCO17_NAMES) else f"j{j}"
+ label = f"{name} {c:.2f}" if (conf is not None and show_conf) else name
+
+ # Position label
+ org = (x_i + 4, y_i - 6)
+
+ # Compute text size for background box
+ font = cv2.FONT_HERSHEY_SIMPLEX
+ font_scale = 0.40
+ thickness = 1
+ (tw, th), baseline = cv2.getTextSize(label, font, font_scale, thickness)
+
+ # Background rectangle (white), then black text on top
+ x0, y0 = org[0], org[1] - th
+ x1, y1 = org[0] + tw, org[1] + baseline
+
+ # Clamp background box within frame
+ x0 = max(0, min(W - 1, x0))
+ y0 = max(0, min(H - 1, y0))
+ x1 = max(0, min(W - 1, x1))
+ y1 = max(0, min(H - 1, y1))
+
+ cv2.rectangle(frame, (x0, y0), (x1, y1), (255, 255, 255), -1) # filled white
+ cv2.putText(frame, label, org, font, font_scale, (0, 0, 0), thickness, cv2.LINE_AA) # black text
+
+ return frame
+
+
+
+def convert_kp_to_image_pixels(kp, bbx_xys, crop_size=256):
+ """
+ Convert kp to full-image pixel coords using HMR-style crop mapping:
+ x_img = cx + x_norm * (s/2)
+ y_img = cy + y_norm * (s/2)
+
+ Supports:
+ - kp in [-1,1] (normalized crop coords)
+ - kp in crop pixels [0..crop_size-1]
+ - kp already in image pixels (then returned unchanged)
+ """
+ kp = np.asarray(kp, dtype=np.float32) # (L,J,2)
+ bbx_xys = np.asarray(bbx_xys, dtype=np.float32) # (L,3)
+
+ kp_min = float(np.nanmin(kp))
+ kp_max = float(np.nanmax(kp))
+
+ # Decide mode
+ if kp_min >= -1.5 and kp_max <= 1.5:
+ mode = "norm_pm1" # [-1,1]
+ kp_norm = kp
+ elif kp_min >= -5.0 and kp_max <= (crop_size + 5.0):
+ mode = "crop_pixels"
+ # crop pixels -> [-1,1]
+ denom = (crop_size - 1.0)
+ kp_norm = (kp / denom) * 2.0 - 1.0
+ else:
+ mode = "image_pixels"
+ return kp, mode, (kp_min, kp_max)
+
+ cx = bbx_xys[:, 0:1] # (L,1)
+ cy = bbx_xys[:, 1:2] # (L,1)
+ s = bbx_xys[:, 2:3] # (L,1)
+ hs = s * 0.5
+
+ x_img = cx + kp_norm[..., 0] * hs
+ y_img = cy + kp_norm[..., 1] * hs
+ kp_img = np.stack([x_img, y_img], axis=-1)
+ return kp_img, mode, (kp_min, kp_max)
+
+
+def main():
+ # ---- Load bbox ----
+ bbx = torch.load(BBX_PATH, map_location="cpu")
+
+ bbx_xyxy_t = bbx.get("bbx_xyxy", None)
+ bbx_xys_t = bbx.get("bbx_xys", None)
+
+ if bbx_xyxy_t is None and bbx_xys_t is None:
+ raise ValueError("bbx.pt must contain 'bbx_xyxy' and/or 'bbx_xys'.")
+
+ if bbx_xys_t is None and bbx_xyxy_t is not None:
+ bbx_xys_t = xyxy_to_xys(bbx_xyxy_t)
+
+ if bbx_xyxy_t is None and bbx_xys_t is not None:
+ bbx_xyxy_t = xys_to_xyxy(bbx_xys_t)
+
+ bbx_xyxy = to_numpy(bbx_xyxy_t) # (L,4)
+ bbx_xys = to_numpy(bbx_xys_t) # (L,3)
+
+ print("bbx_xyxy shape:", bbx_xyxy.shape, "bbx_xys shape:", bbx_xys.shape)
+
+ # ---- Load vitpose ----
+ vitpose = torch.load(VITPOSE_PATH, map_location="cpu")
+
+ conf = None
+ if isinstance(vitpose, dict):
+ kp = None
+ for k in ["kp2d", "keypoints", "kps", "joints_2d", "vitpose"]:
+ if k in vitpose:
+ kp = vitpose[k]
+ break
+ if kp is None:
+ print("vitpose.pt keys:", list(vitpose.keys()))
+ raise ValueError("Couldn't find keypoints in vitpose dict.")
+ kp = to_numpy(kp)
+
+ for k in ["conf", "confidence", "scores", "kp2d_conf", "keypoint_scores"]:
+ if k in vitpose:
+ conf = to_numpy(vitpose[k])
+ break
+ else:
+ kp = to_numpy(vitpose)
+
+ if kp.ndim != 3:
+ raise ValueError(f"Unexpected kp shape: {kp.shape} (expected L x J x 2/3)")
+
+ if kp.shape[-1] == 3 and conf is None:
+ conf = kp[..., 2]
+ kp = kp[..., :2]
+ elif kp.shape[-1] != 2:
+ raise ValueError(f"Unexpected kp last dim: {kp.shape[-1]} (expected 2 or 3)")
+
+ # ---- Open video ----
+ cap = cv2.VideoCapture(VIDEO_PATH)
+ if not cap.isOpened():
+ raise RuntimeError(f"Failed to open video: {VIDEO_PATH}")
+
+ fps = cap.get(cv2.CAP_PROP_FPS) or 30.0
+ W = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
+ H = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
+ print("Video:", VIDEO_PATH, "W,H:", W, H, "fps:", fps)
+
+ # ---- Align lengths ----
+ L = min(len(bbx_xyxy), len(bbx_xys), kp.shape[0])
+ print("Using L =", L)
+
+ # ---- Convert keypoints to image pixels if needed ----
+ kp_img, mode, (kp_min, kp_max) = convert_kp_to_image_pixels(kp[:L], bbx_xys[:L], crop_size=256)
+ print(f"kp stats min/max: {kp_min:.3f} / {kp_max:.3f} -> interpreted as mode: {mode}")
+
+ # Basic bbox sanity
+ centers = np.stack(
+ [(bbx_xyxy[:L, 0] + bbx_xyxy[:L, 2]) * 0.5,
+ (bbx_xyxy[:L, 1] + bbx_xyxy[:L, 3]) * 0.5],
+ axis=-1
+ )
+ center_speed = np.linalg.norm(centers[1:] - centers[:-1], axis=-1)
+ if len(center_speed) > 0:
+ print("bbox center jump px (p50/p90/max):",
+ float(np.percentile(center_speed, 50)),
+ float(np.percentile(center_speed, 90)),
+ float(center_speed.max()))
+
+ if conf is not None:
+ conf_use = conf[:L]
+ print("kp conf (mean/p10):",
+ float(np.mean(conf_use)),
+ float(np.percentile(conf_use, 10)))
+
+ # ---- Writers ----
+ fourcc = cv2.VideoWriter_fourcc(*"mp4v")
+ w_bbox = cv2.VideoWriter(OUT_BBOX_ONLY, fourcc, fps, (W, H))
+ w_kp = cv2.VideoWriter(OUT_BBOX_KP, fourcc, fps, (W, H))
+
+ t = 0
+ while t < L:
+ ok, frame = cap.read()
+ if not ok:
+ break
+
+ f1 = frame.copy()
+ f2 = frame.copy()
+
+ draw_bbox_xyxy(f1, bbx_xyxy[t])
+ draw_bbox_xyxy(f2, bbx_xyxy[t])
+
+ c_t = conf[t] if conf is not None else None
+ draw_kps_with_names(f2, kp_img[t], conf=c_t, show_conf=True)
+
+ cv2.putText(f1, f"t={t}", (10, 25), cv2.FONT_HERSHEY_SIMPLEX, 0.8, (255, 255, 255),
+ 2, cv2.LINE_AA)
+ cv2.putText(f2, f"t={t} mode={mode}", (10, 25), cv2.FONT_HERSHEY_SIMPLEX, 0.8, (255, 255, 255),
+ 2, cv2.LINE_AA)
+
+ w_bbox.write(f1)
+ w_kp.write(f2)
+ t += 1
+
+ cap.release()
+ w_bbox.release()
+ w_kp.release()
+
+ print("Saved:", OUT_BBOX_ONLY)
+ print("Saved:", OUT_BBOX_KP)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/debug_seq.py b/debug_seq.py
new file mode 100644
index 0000000000000000000000000000000000000000..861fad7171ca8eb7cfcfa0ce8038691c6d6b17c3
--- /dev/null
+++ b/debug_seq.py
@@ -0,0 +1,117 @@
+import os
+import json
+import numpy as np
+import torch
+from scipy.spatial.transform import Rotation as R
+from hmr4d.utils.smplx_utils import make_smplx
+
+def test_kabsch():
+ input_dir = "/mnt/c/Temp/SyntheticDataset"
+ seqs = ["sequence_100_biboo_birthday_speech_explosion_1.jsonl", "sequence_101_biboo_birthday_speech_explosion_2.jsonl"]
+
+ device = "cuda"
+ model = make_smplx("supermotion").to(device).eval()
+
+ fix_rot = R.from_euler("z", 180, degrees=True).as_matrix()
+ C = np.diag([1.0, -1.0, 1.0])
+ C4 = np.diag([1.0, -1.0, 1.0, 1.0])
+ fix_mat = np.eye(4); fix_mat[:3, :3] = fix_rot
+
+ for seq in seqs:
+ p = os.path.join(input_dir, seq)
+ with open(p, "r") as f: lines = f.readlines()
+
+ row = json.loads(lines[1]) # frame 0
+
+ # parse unity
+ R_cam_w = R.from_quat(row["cam_rot_world"]).as_matrix()
+ R_pel_w = R.from_quat(row["pelvis_rot_world"]).as_matrix()
+ R_rel_unity = R_cam_w.T @ R_pel_w
+ R_cv = C @ R_rel_unity @ C
+ R_final = R_cv @ R.from_euler("z", 180, degrees=True).as_matrix()
+ go_c = R.from_matrix(R_final).as_rotvec().astype(np.float32)
+
+ pelvis_cam_cv = np.asarray(row["smpl_incam_transl"], dtype=np.float64).reshape(3)
+ root_cam_cv = np.asarray(row.get("smpl_root_incam_transl", [0.0, 0.0, 0.0]), dtype=np.float64).reshape(3)
+ pelvis_cam_cv = pelvis_cam_cv # + np.array([0.0, -0.02, 0.0], dtype=np.float64)
+ target_cam_cv = pelvis_cam_cv
+
+ R_cam_w_cv = fix_rot @ (C @ R_cam_w @ C)
+ R_pelvis_w_cv = R_cam_w_cv @ R_final
+ go_w = R.from_matrix(R_pelvis_w_cv).as_rotvec().astype(np.float32)
+
+ pos_cv_raw = (C @ row['pelvis_pos_world'])
+ pelvis_pos_w_cv = fix_rot @ pos_cv_raw
+
+ pose = np.asarray(row["smplx_pose"], dtype=np.float32)
+ body_pose = pose[3:66].astype(np.float32)
+ betas = np.zeros(10, dtype=np.float32)
+
+ # compute pel0
+ with torch.no_grad():
+ bt = torch.from_numpy(betas[None]).to(device)
+ bp = torch.from_numpy(body_pose[None]).to(device)
+
+ go_c_t = torch.from_numpy(go_c[None]).to(device)
+ pel0_c = model(betas=bt, global_orient=go_c_t, body_pose=bp).joints[0, 0].cpu().numpy()
+
+ go_w_t = torch.from_numpy(go_w[None]).to(device)
+ pel0_w = model(betas=bt, global_orient=go_w_t, body_pose=bp).joints[0, 0].cpu().numpy()
+
+ tr_c = target_cam_cv - pel0_c
+ tr_w = pelvis_pos_w_cv - pel0_w
+
+ # math check
+ R_c_v = R.from_rotvec(go_c).as_matrix()
+ R_w_v = R.from_rotvec(go_w).as_matrix()
+ R_w2c_v = R_c_v @ R_w_v.T
+
+ j0_c = pel0_c + tr_c
+ j0_w = pel0_w + tr_w
+ t_w2c_derived = j0_c - R_w2c_v @ j0_w
+
+ cam_T_wc = np.eye(4)
+ cam_T_wc[:3, :3] = R_cam_w
+ cam_T_wc[:3, 3] = row["cam_pos_world"]
+ cam_T_wc_cv = fix_mat @ (C4 @ cam_T_wc @ C4)
+ T_w2c_exported = np.linalg.inv(cam_T_wc_cv)
+
+ # Kabsch verify manually
+ T_c2w_exported = np.linalg.inv(T_w2c_exported)
+
+ with torch.no_grad():
+ tr_c_t = torch.from_numpy(tr_c[None]).to(device)
+ j_c = model(betas=bt, global_orient=go_c_t, body_pose=bp, transl=tr_c_t).joints[0, :22].cpu().numpy()
+
+ tr_w_t = torch.from_numpy(tr_w[None]).to(device)
+ j_w = model(betas=bt, global_orient=go_w_t, body_pose=bp, transl=tr_w_t).joints[0, :22].cpu().numpy()
+
+ src_mean = j_c.mean(axis=0)
+ dst_mean = j_w.mean(axis=0)
+ X = j_c - src_mean
+ Y = j_w - dst_mean
+ H = X.T @ Y
+ U, _, Vt = np.linalg.svd(H)
+ R_align = Vt.T @ U.T
+ if np.linalg.det(R_align) < 0:
+ Vt[-1, :] *= -1
+ R_align = Vt.T @ U.T
+ t_align = dst_mean - src_mean @ R_align
+
+ T_c2w_kabsch = np.eye(4)
+ T_c2w_kabsch[:3, :3] = R_align
+ T_c2w_kabsch[:3, 3] = t_align
+
+ t_err = np.linalg.norm(T_c2w_kabsch[:3, 3] - T_c2w_exported[:3, 3])
+
+ print(f"[{seq}]")
+ print(f" t_err: {t_err:.3f}m")
+ print(f" t_c2w_kabsch: {T_c2w_kabsch[:3, 3]}")
+ print(f" T_c2w_exported: {T_c2w_exported[:3, 3]}")
+ print(f" diff(kabsch - exported): {T_c2w_kabsch[:3, 3] - T_c2w_exported[:3, 3]}")
+ print(f" t_w2c_derived: {t_w2c_derived}")
+ print(f" T_w2c_exported: {T_w2c_exported[:3, 3]}")
+ print(f" delta_T_w2c (derived - exported): {t_w2c_derived - T_w2c_exported[:3, 3]}")
+
+if __name__ == "__main__":
+ test_kabsch()
diff --git a/debug_tensors.py b/debug_tensors.py
new file mode 100644
index 0000000000000000000000000000000000000000..4048e4c45fcd3b0d01bfce345e76d5288203c8c2
--- /dev/null
+++ b/debug_tensors.py
@@ -0,0 +1,51 @@
+import torch
+import numpy as np
+from pathlib import Path
+import sys
+
+# Add Genmo to Path
+repo_root = str(Path(__file__).resolve().parents[0])
+if repo_root not in sys.path:
+ sys.path.insert(0, repo_root)
+
+ROOT = "./processed_dataset"
+pt_files = sorted(list(Path(ROOT).glob("genmo_features/*.pt")))
+
+def debug_print():
+ if len(pt_files) == 0:
+ return
+
+ file_path = pt_files[0]
+ data = torch.load(file_path, map_location="cpu")
+
+ print("----- CONFIGURATION DUMP -----")
+ print(f"Data Keys: {data.keys()}")
+
+ T_w2c = data["T_w2c"]
+ tr_w = data["smpl_params_w"]["transl"]
+ tr_c = data["smpl_params_c"]["transl"]
+ go_w = data["smpl_params_w"]["global_orient"]
+ go_c = data["smpl_params_c"]["global_orient"]
+ world_offset = data.get("world_offset", None)
+
+ print(f"World Offset Applied: {world_offset}")
+
+ # Check Kabsch Error manually
+ from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+ from scipy.spatial.transform import Rotation as R
+ smplx = make_smplx("supermotion").eval().cpu()
+
+ print(f"\nT_w2c Shape: {T_w2c.shape}")
+ print(f"T_w2c Frame 0:\n{T_w2c[0]}")
+
+ # Calculate global SMPL from T_w2c ^ -1 @ incam:
+ inv_T = torch.inverse(T_w2c[0])
+
+ tr_c_homo = torch.cat([tr_c[0], torch.tensor([1.0])])
+ expected_tr_w = (inv_T @ tr_c_homo)[:3]
+
+ print(f"tr_w = {tr_w[0]}")
+ print(f"Expecttr_w = {expected_tr_w}")
+
+if __name__ == "__main__":
+ debug_print()
diff --git a/debug_tr_w.py b/debug_tr_w.py
new file mode 100644
index 0000000000000000000000000000000000000000..9281acd543df8c45bcbb000e8e91887e4bb71e6c
--- /dev/null
+++ b/debug_tr_w.py
@@ -0,0 +1,18 @@
+import torch
+import numpy as np
+from pathlib import Path
+import sys
+
+# Add Genmo to Path
+repo_root = str(Path(__file__).resolve().parents[0])
+if repo_root not in sys.path:
+ sys.path.insert(0, repo_root)
+
+# Verify Kabsch in metric evaluation
+from third_party.GVHMR.hmr4d.utils.geo_transform import apply_T_on_points
+
+def dump_train_script():
+ pass
+
+if __name__ == "__main__":
+ pass
diff --git a/debug_unity_data.py b/debug_unity_data.py
new file mode 100644
index 0000000000000000000000000000000000000000..10c4720899e637d0d8ebb41937232879453c0670
--- /dev/null
+++ b/debug_unity_data.py
@@ -0,0 +1,184 @@
+#!/usr/bin/env python3
+"""
+Diagnostic script to check Unity dataset coordinate consistency.
+
+Checks:
+- Rotation consistency between `smpl_params_c`, `smpl_params_w`, and `T_w2c`
+- `cam_angvel` matches the convention used in preprocessing
+- SMPL forward-kinematics consistency between camera/world parameters
+
+Run with the gvhmr env python:
+ /root/miniconda3/envs/gvhmr/bin/python debug_unity_data.py
+"""
+import torch
+import numpy as np
+from pathlib import Path
+from scipy.spatial.transform import Rotation as R
+
+def axis_angle_to_matrix(aa):
+ """Convert axis-angle to rotation matrix (numpy)."""
+ return R.from_rotvec(aa).as_matrix()
+
+def check_single_sequence(pt_path):
+ """Check a single .pt file for coordinate consistency."""
+ print(f"\n{'='*80}")
+ print(f"Checking: {pt_path.name}")
+ print(f"{'='*80}")
+
+ data = torch.load(pt_path, map_location="cpu")
+
+ # Extract key data
+ smpl_c = data["smpl_params_c"]
+ smpl_w = data["smpl_params_w"]
+ T_w2c = data["T_w2c"].numpy()
+
+ # Check first frame
+ idx = 0
+ print(f"\n[Frame {idx}]")
+
+ # Ground truth
+ go_c_gt = smpl_c["global_orient"][idx].numpy() # (3,) axis-angle
+ go_w_gt = smpl_w["global_orient"][idx].numpy() # (3,) axis-angle
+
+ # Convert to matrices
+ R_c_gt = axis_angle_to_matrix(go_c_gt) # Pelvis in camera frame
+ R_w_gt = axis_angle_to_matrix(go_w_gt) # Pelvis in world frame
+ R_w2c = T_w2c[idx, :3, :3] # World to camera
+ R_c2w = R_w2c.T # Camera to world
+
+ # Verify: R_w = R_c2w @ R_c
+ R_w_reconstructed = R_c2w @ R_c_gt
+
+ # Compare
+ R_diff = R_w_reconstructed @ R_w_gt.T
+ angle_err_deg = np.linalg.norm(R.from_matrix(R_diff).as_rotvec()) * 180.0 / np.pi
+
+ print(f"Ground truth global_orient_c (axis-angle): {go_c_gt}")
+ print(f"Ground truth global_orient_w (axis-angle): {go_w_gt}")
+ print(f"\nReconstruction test: R_w = R_c2w @ R_c")
+ print(f" Rotation error: {angle_err_deg:.4f}°")
+
+ if angle_err_deg > 1.0:
+ print(f" ❌ ERROR: Rotation mismatch > 1°!")
+ print(f" R_w (ground truth):\n{R_w_gt}")
+ print(f" R_w (reconstructed):\n{R_w_reconstructed}")
+ else:
+ print(f" ✅ OK: Rotations are consistent")
+
+ # Check cam_angvel computation (should match preprocess convention)
+ print(f"\n[Camera Angular Velocity Check]")
+ cam_ok = True
+ if "cam_angvel" in data:
+ cam_angvel = data["cam_angvel"] # (L, 6) - 6D rotation
+ print(f" cam_angvel shape: {cam_angvel.shape}")
+ print(f" cam_angvel[0]: {cam_angvel[0].numpy()}")
+
+ # Manually compute cam_angvel and compare.
+ # Convention (see `tools/demo/process_dataset.py:compute_velocity`):
+ # cam_angvel[0] = [1,0,0, 0,1,0] (identity, rotation6d)
+ # cam_angvel[i] = rot6d(R_i @ R_{i-1}^T)
+ from genmo.utils.rotation_conversions import matrix_to_rotation_6d
+ R_w2c_t = torch.from_numpy(T_w2c[:, :3, :3]).float()
+ L = int(R_w2c_t.shape[0])
+ cam_angvel_manual = torch.zeros((L, 6), dtype=torch.float32)
+ cam_angvel_manual[0] = cam_angvel_manual.new_tensor([1.0, 0.0, 0.0, 0.0, 1.0, 0.0])
+ if L > 1:
+ R_diff_manual = R_w2c_t[1:] @ R_w2c_t[:-1].transpose(-1, -2)
+ cam_angvel_manual[1:] = matrix_to_rotation_6d(R_diff_manual)
+
+ diff = (cam_angvel - cam_angvel_manual).abs().max()
+ print(f" Manual vs stored cam_angvel max diff: {diff:.6f}")
+ if diff > 1e-4:
+ print(f" ❌ WARNING: cam_angvel mismatch!")
+ cam_ok = False
+ else:
+ print(f" ✅ OK: cam_angvel matches manual computation")
+
+ # Check SMPL forward kinematics consistency
+ print(f"\n[SMPL FK Check]")
+ fk_ok = True
+ try:
+ from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+ smplx_model = make_smplx("supermotion").eval()
+
+ with torch.no_grad():
+ # Incam SMPL
+ out_c = smplx_model(
+ global_orient=smpl_c["global_orient"][idx:idx+1],
+ body_pose=smpl_c["body_pose"][idx:idx+1],
+ betas=smpl_c["betas"][idx:idx+1],
+ transl=smpl_c["transl"][idx:idx+1]
+ )
+ joints_c = out_c.joints[0, :22].numpy() # (22, 3)
+
+ # Global SMPL
+ out_w = smplx_model(
+ global_orient=smpl_w["global_orient"][idx:idx+1],
+ body_pose=smpl_w["body_pose"][idx:idx+1],
+ betas=smpl_w["betas"][idx:idx+1],
+ transl=smpl_w["transl"][idx:idx+1]
+ )
+ joints_w = out_w.joints[0, :22].numpy() # (22, 3)
+
+ # Transform camera->world using T_w2c (world->camera):
+ # x_c = R_w2c x_w + t_w2c
+ # => x_w = R_w2c^T (x_c - t_w2c)
+ t_w2c = T_w2c[idx, :3, 3]
+ joints_c2w = (R_c2w @ (joints_c - t_w2c).T).T
+
+ # Compare
+ joint_err = np.linalg.norm(joints_c2w - joints_w, axis=-1).mean()
+ print(f" Mean joint error (incam→world vs world GT): {joint_err:.4f}m")
+
+ if joint_err > 0.05:
+ print(f" ❌ ERROR: Joint mismatch > 5cm!")
+ fk_ok = False
+ else:
+ print(f" ✅ OK: SMPL joints are consistent")
+
+ except Exception as e:
+ print(f" ⚠️ Could not run SMPL FK check: {e}")
+ fk_ok = False
+
+ # Consider the clip consistent only if all checks are reasonable.
+ ok_rot = angle_err_deg < 1.0
+ return ok_rot and cam_ok and fk_ok
+
+def main():
+ dataset_root = Path("./processed_dataset")
+ feat_dir = dataset_root / "genmo_features"
+
+ if not feat_dir.exists():
+ print(f"Error: {feat_dir} not found!")
+ return
+
+ pt_files = sorted(list(feat_dir.glob("*.pt")))
+ print(f"Found {len(pt_files)} sequences")
+
+ if len(pt_files) == 0:
+ print("No .pt files found!")
+ return
+
+ # Check first 3 sequences
+ num_check = min(3, len(pt_files))
+ all_ok = True
+
+ for i in range(num_check):
+ ok = check_single_sequence(pt_files[i])
+ all_ok = all_ok and ok
+
+ print(f"\n{'='*80}")
+ if all_ok:
+ print("✅ All checks passed! Data appears consistent.")
+ print("\nIf training still has high loss, the issue is likely:")
+ print(" 1. Model architecture/hyperparameters")
+ print(" 2. Normalization statistics mismatch")
+ print(" 3. Sequence length handling during training")
+ else:
+ print("❌ Data consistency issues found!")
+ print("\nThis explains the high training loss.")
+ print("You need to fix the coordinate system in process_dataset.py")
+ print(f"{'='*80}\n")
+
+if __name__ == "__main__":
+ main()
diff --git a/diagnose_data.py b/diagnose_data.py
new file mode 100644
index 0000000000000000000000000000000000000000..184c5a8d11d7d7bdc51ccb6b0c4660aa41bc30e8
--- /dev/null
+++ b/diagnose_data.py
@@ -0,0 +1,288 @@
+#!/usr/bin/env python3
+"""
+Run pretrained GENMO model on Unity GT processed data and compare predictions vs GT.
+This diagnosis shows if the pretrained model can correctly predict poses from your data.
+"""
+import torch
+import numpy as np
+from pathlib import Path
+from scipy.spatial.transform import Rotation as R
+import sys
+import hydra
+from omegaconf import DictConfig, OmegaConf
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import get_tgtcoord_rootparam
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+# Ensure repo root is on sys.path
+REPO_ROOT = Path(__file__).resolve().parent
+if str(REPO_ROOT) not in sys.path:
+ sys.path.insert(0, str(REPO_ROOT))
+
+
+def main():
+ # Load processed Unity data
+ pt_files = sorted(Path('./processed_dataset/genmo_features').glob('*.pt'))
+ if not pt_files:
+ print("No .pt files found!")
+ return
+
+ pt_path = pt_files[0]
+ print(f"=== Loading: {pt_path.name} ===\n")
+ data = torch.load(pt_path, map_location='cpu', weights_only=False)
+
+ # Limit to manageable length
+ L = min(120, data['f_imgseq'].shape[0])
+
+ # Load configuration properly using hydra compose
+ print("Loading GENMO model configuration...")
+ from hydra import compose, initialize_config_dir
+
+ config_dir = str(Path(__file__).parent / 'configs')
+
+ # Initialize hydra with the config directory
+ with initialize_config_dir(version_base="1.3", config_dir=config_dir):
+ # Compose config with exp=genmo_lg to get all defaults
+ cfg = compose(config_name="infer_video", overrides=["exp=genmo_lg"])
+
+ # Set checkpoint path - use the one from inference.sh
+ ckpt_path = 's050000.ckpt'
+ ckpt_path = Path(ckpt_path)
+
+ if not ckpt_path.exists():
+ print(f"ERROR: Checkpoint not found at {ckpt_path}")
+ print("Available checkpoints found:")
+ for ckpt in Path('.').glob('*.ckpt'):
+ print(f" - {ckpt}")
+ return
+
+ print(f"Using checkpoint: {ckpt_path}\n")
+ ckpt_path = str(ckpt_path)
+
+ # Load the pretrained GENMO model
+ print("Loading pretrained GENMO model...")
+ model = hydra.utils.instantiate(cfg.model, _recursive_=False)
+ model.load_pretrained_model(ckpt_path)
+ model = model.eval().cuda()
+
+ # Prepare input data in the format expected by model.predict()
+ # Slice all data to length L and compute R_w2c from T_w2c
+ # Convert all to float32 to avoid dtype mismatch
+ T_w2c = data['T_w2c'][:L].float()
+ R_w2c = T_w2c[:, :3, :3] # Extract rotation from transformation matrix
+
+ input_data = {
+ 'meta': data.get('meta', [{'vid': pt_path.stem}]),
+ 'caption': data.get('caption', ''),
+ 'has_text': data.get('has_text', torch.tensor([False])),
+ 'length': torch.tensor(L),
+ 'bbx_xys': data['bbx_xys'][:L].float(),
+ 'K_fullimg': data['K_fullimg'][:L].float(),
+ 'f_imgseq': data['f_imgseq'][:L].float(),
+ 'kp2d': data['kp2d'][:L].float(),
+ 'cam_angvel': data['cam_angvel'][:L].float(),
+ 'cam_tvel': (data['cam_tvel'][:L] if 'cam_tvel' in data else torch.zeros(L, 3)).float(),
+ 'R_w2c': R_w2c,
+ 'T_w2c': T_w2c,
+ 'gt_T_w2c': data.get('gt_T_w2c', T_w2c).float(),
+ 'mask': data.get('mask', {
+ 'valid': torch.ones(L),
+ 'has_img_mask': torch.ones(L).bool(),
+ 'has_2d_mask': torch.ones(L).bool(),
+ 'has_cam_mask': torch.ones(L).bool(),
+ 'has_audio_mask': torch.zeros(L).bool(),
+ 'has_music_mask': torch.zeros(L).bool(),
+ }),
+ }
+
+ # Run model inference
+ print("Running model inference...")
+ with torch.inference_mode():
+ pred = model.predict(input_data, static_cam=False, postproc=True)
+
+ # Extract predictions
+ pred_params_c = pred['smpl_params_incam'] # predicted in-camera params
+ pred_params_g = pred['smpl_params_global'] # predicted global params
+
+ # Get GT values
+ gt_go_c = data['smpl_params_c']['global_orient'][:L].cpu().numpy() # (L, 3)
+ gt_go_w = data['smpl_params_w']['global_orient'][:L].cpu().numpy() # (L, 3)
+ gt_transl_c = data['smpl_params_c']['transl'][:L].cpu().numpy() # (L, 3)
+ gt_transl_w = data['smpl_params_w']['transl'][:L].cpu().numpy() # (L, 3)
+ gt_body_pose = data['smpl_params_c']['body_pose'][:L].cpu().numpy() # (L, 63)
+
+ # Get predicted values (convert to axis-angle if needed)
+ from genmo.utils.rotation_conversions import axis_angle_to_matrix, matrix_to_axis_angle
+
+ pred_go_c = pred_params_c['global_orient'].cpu()
+ if pred_go_c.ndim == 3 and pred_go_c.shape[-1] == 3: # already axis-angle
+ pred_go_c = pred_go_c.numpy()
+ elif pred_go_c.ndim == 4: # rotation matrix
+ pred_go_c = matrix_to_axis_angle(pred_go_c.squeeze(1)).numpy()
+
+ pred_go_g = pred_params_g['global_orient'].cpu()
+ if pred_go_g.ndim == 3 and pred_go_g.shape[-1] == 3:
+ pred_go_g = pred_go_g.numpy()
+ elif pred_go_g.ndim == 4:
+ pred_go_g = matrix_to_axis_angle(pred_go_g.squeeze(1)).numpy()
+
+ pred_body_pose = pred_params_c['body_pose'].cpu()
+ if pred_body_pose.ndim == 4:
+ pred_body_pose = matrix_to_axis_angle(pred_body_pose).numpy().reshape(L, -1)
+ else:
+ pred_body_pose = pred_body_pose.numpy()
+
+ pred_transl_c = pred_params_c['transl'].cpu().numpy()
+ pred_transl_g = pred_params_g['transl'].cpu().numpy()
+
+ # Convert predicted global params from AY -> ANY for fair comparison if needed.
+ pred_go_g_any = pred_go_g
+ pred_transl_g_any = pred_transl_g
+ try:
+ pred_go_g_any_t, pred_transl_g_any_t, _ = get_tgtcoord_rootparam(
+ torch.from_numpy(pred_go_g).float(),
+ torch.from_numpy(pred_transl_g).float(),
+ tsf="ay->any",
+ )
+ pred_go_g_any = pred_go_g_any_t.numpy()
+ pred_transl_g_any = pred_transl_g_any_t.numpy()
+ except Exception:
+ pass
+
+ world_offset = None
+ if "world_offset" in data:
+ world_offset = data["world_offset"].cpu().numpy()
+
+ # Print comparisons for frame 0
+ print("\n" + "="*60)
+ print("FRAME 0 COMPARISON: PREDICTION vs GROUND TRUTH")
+ print("="*60)
+
+ # In-camera orientation
+ gt_euler_c = R.from_rotvec(gt_go_c[0]).as_euler('YXZ', degrees=True)
+ pred_euler_c = R.from_rotvec(pred_go_c[0]).as_euler('YXZ', degrees=True)
+
+ print("\n--- In-Camera Global Orientation ---")
+ print(f"GT (YXZ deg): yaw={gt_euler_c[0]:7.2f}, pitch={gt_euler_c[1]:7.2f}, roll={gt_euler_c[2]:7.2f}")
+ print(f"Pred (YXZ deg): yaw={pred_euler_c[0]:7.2f}, pitch={pred_euler_c[1]:7.2f}, roll={pred_euler_c[2]:7.2f}")
+ go_c_diff = pred_euler_c - gt_euler_c
+ print(f"Diff (deg): yaw={go_c_diff[0]:7.2f}, pitch={go_c_diff[1]:7.2f}, roll={go_c_diff[2]:7.2f}")
+
+ # World orientation
+ gt_euler_w = R.from_rotvec(gt_go_w[0]).as_euler('YXZ', degrees=True)
+ pred_euler_g = R.from_rotvec(pred_go_g[0]).as_euler('YXZ', degrees=True)
+
+ print("\n--- World/Global Orientation ---")
+ print(f"GT (YXZ deg): yaw={gt_euler_w[0]:7.2f}, pitch={gt_euler_w[1]:7.2f}, roll={gt_euler_w[2]:7.2f}")
+ print(f"Pred (YXZ deg): yaw={pred_euler_g[0]:7.2f}, pitch={pred_euler_g[1]:7.2f}, roll={pred_euler_g[2]:7.2f}")
+ go_g_diff = pred_euler_g - gt_euler_w
+ # Normalize yaw to [-180, 180]
+ if go_g_diff[0] > 180: go_g_diff[0] -= 360
+ if go_g_diff[0] < -180: go_g_diff[0] += 360
+ print(f"Diff (deg): yaw={go_g_diff[0]:7.2f}, pitch={go_g_diff[1]:7.2f}, roll={go_g_diff[2]:7.2f}")
+ pred_euler_g_any = R.from_rotvec(pred_go_g_any[0]).as_euler('YXZ', degrees=True)
+ print(f"Pred AY->ANY (YXZ deg): yaw={pred_euler_g_any[0]:7.2f}, pitch={pred_euler_g_any[1]:7.2f}, roll={pred_euler_g_any[2]:7.2f}")
+
+ # Translation
+ print("\n--- In-Camera Translation ---")
+ print(f"GT: [{gt_transl_c[0, 0]:7.3f}, {gt_transl_c[0, 1]:7.3f}, {gt_transl_c[0, 2]:7.3f}]")
+ print(f"Pred: [{pred_transl_c[0, 0]:7.3f}, {pred_transl_c[0, 1]:7.3f}, {pred_transl_c[0, 2]:7.3f}]")
+ transl_c_diff = pred_transl_c[0] - gt_transl_c[0]
+ print(f"Diff: [{transl_c_diff[0]:7.3f}, {transl_c_diff[1]:7.3f}, {transl_c_diff[2]:7.3f}]")
+
+ print("\n--- World Translation ---")
+ print(f"GT: [{gt_transl_w[0, 0]:7.3f}, {gt_transl_w[0, 1]:7.3f}, {gt_transl_w[0, 2]:7.3f}]")
+ print(f"Pred: [{pred_transl_g[0, 0]:7.3f}, {pred_transl_g[0, 1]:7.3f}, {pred_transl_g[0, 2]:7.3f}]")
+ transl_g_diff = pred_transl_g[0] - gt_transl_w[0]
+ print(f"Diff: [{transl_g_diff[0]:7.3f}, {transl_g_diff[1]:7.3f}, {transl_g_diff[2]:7.3f}]")
+ print(f"Pred AY->ANY: [{pred_transl_g_any[0, 0]:7.3f}, {pred_transl_g_any[0, 1]:7.3f}, {pred_transl_g_any[0, 2]:7.3f}]")
+ if world_offset is not None:
+ gt_transl_w_world = gt_transl_w + world_offset[None]
+ print(f"GT + world_offset: [{gt_transl_w_world[0, 0]:7.3f}, {gt_transl_w_world[0, 1]:7.3f}, {gt_transl_w_world[0, 2]:7.3f}]")
+
+ # GT internal consistency check: smpl_params_w + T_w2c -> smpl_params_c
+ gt_go_c_t = data['smpl_params_c']['global_orient'][:L].float()
+ gt_go_w_t = data['smpl_params_w']['global_orient'][:L].float()
+ gt_tr_c_t = data['smpl_params_c']['transl'][:L].float()
+ gt_tr_w_t = data['smpl_params_w']['transl'][:L].float()
+ T_w2c_t = data['T_w2c'][:L].float()
+
+ R_w2c_t = T_w2c_t[:, :3, :3]
+ t_w2c_t = T_w2c_t[:, :3, 3]
+ R_w_t = axis_angle_to_matrix(gt_go_w_t)
+ R_c_from_w = torch.matmul(R_w2c_t, R_w_t)
+ R_c_gt = axis_angle_to_matrix(gt_go_c_t)
+ R_diff = torch.matmul(R_c_from_w.transpose(-1, -2), R_c_gt)
+ rot_err_deg = torch.norm(matrix_to_axis_angle(R_diff), dim=-1) * (180.0 / np.pi)
+
+ tr_c_from_w = (R_w2c_t @ gt_tr_w_t.unsqueeze(-1)).squeeze(-1) + t_w2c_t
+ tr_err = torch.norm(tr_c_from_w - gt_tr_c_t, dim=-1)
+
+ print("\n--- GT Consistency Check (smpl_params_w + T_w2c -> smpl_params_c) ---")
+ print(f"Rotation error: mean={rot_err_deg.mean():.3f}° | max={rot_err_deg.max():.3f}°")
+ print(f"Translation error (no offset): mean={tr_err.mean():.3f}m | max={tr_err.max():.3f}m")
+
+ # Consistency with SMPLX root offset
+ try:
+ smplx = make_smplx("supermotion").to(gt_go_w_t).eval()
+ betas0 = data['smpl_params_w']['betas'][0].float()[None]
+ offset = smplx.get_skeleton(betas0)[0, 0]
+ tr_c_from_w_off = (R_w2c_t @ (gt_tr_w_t + offset).unsqueeze(-1)).squeeze(-1) + t_w2c_t - offset
+ tr_err_off = torch.norm(tr_c_from_w_off - gt_tr_c_t, dim=-1)
+ print(f"Translation error (with offset): mean={tr_err_off.mean():.3f}m | max={tr_err_off.max():.3f}m")
+ except Exception:
+ print("Translation error (with offset): skipped (smplx model unavailable)")
+
+ # Body pose
+ bp_diff = np.degrees(np.abs(pred_body_pose[0] - gt_body_pose[0]))
+ print(f"\n--- Body Pose ---")
+ print(f"Max difference: {bp_diff.max():.2f}°")
+ print(f"Mean difference: {bp_diff.mean():.2f}°")
+
+ # Summary across all frames
+ print("\n" + "="*60)
+ print("SUMMARY ACROSS ALL FRAMES")
+ print("="*60)
+
+ # Compute errors across all frames
+ all_go_c_errors = []
+ all_go_g_errors = []
+ for i in range(min(L, len(gt_go_c))):
+ gt_e_c = R.from_rotvec(gt_go_c[i]).as_euler('YXZ', degrees=True)
+ pred_e_c = R.from_rotvec(pred_go_c[i]).as_euler('YXZ', degrees=True)
+ all_go_c_errors.append(np.abs(pred_e_c - gt_e_c))
+
+ gt_e_g = R.from_rotvec(gt_go_w[i]).as_euler('YXZ', degrees=True)
+ pred_e_g = R.from_rotvec(pred_go_g[i]).as_euler('YXZ', degrees=True)
+ diff_g = pred_e_g - gt_e_g
+ # Normalize yaw
+ if diff_g[0] > 180: diff_g[0] -= 360
+ if diff_g[0] < -180: diff_g[0] += 360
+ all_go_g_errors.append(np.abs(diff_g)) # Fixed: was all_go_c_errors
+
+ all_go_c_errors = np.array(all_go_c_errors)
+ all_go_g_errors = np.array(all_go_g_errors)
+
+ print(f"\nIn-camera orientation errors (mean ± std):")
+ print(f" Yaw: {all_go_c_errors[:, 0].mean():6.2f}° ± {all_go_c_errors[:, 0].std():6.2f}°")
+ print(f" Pitch: {all_go_c_errors[:, 1].mean():6.2f}° ± {all_go_c_errors[:, 1].std():6.2f}°")
+ print(f" Roll: {all_go_c_errors[:, 2].mean():6.2f}° ± {all_go_c_errors[:, 2].std():6.2f}°")
+
+ print(f"\nWorld orientation errors (mean ± std):")
+ print(f" Yaw: {all_go_g_errors[:, 0].mean():6.2f}° ± {all_go_g_errors[:, 0].std():6.2f}°")
+ print(f" Pitch: {all_go_g_errors[:, 1].mean():6.2f}° ± {all_go_g_errors[:, 1].std():6.2f}°")
+ print(f" Roll: {all_go_g_errors[:, 2].mean():6.2f}° ± {all_go_g_errors[:, 2].std():6.2f}°")
+
+ transl_c_errors = np.linalg.norm(pred_transl_c[:L] - gt_transl_c[:L], axis=1)
+ print(f"\nIn-camera translation error: {transl_c_errors.mean():.4f}m ± {transl_c_errors.std():.4f}m")
+
+ transl_g_errors = np.linalg.norm(pred_transl_g[:L] - gt_transl_w[:L], axis=1)
+ print(f"World translation error: {transl_g_errors.mean():.4f}m ± {transl_g_errors.std():.4f}m")
+
+ all_bp_diff = np.degrees(np.abs(pred_body_pose[:L] - gt_body_pose[:L]))
+ print(f"\nBody pose error: {all_bp_diff.mean():.2f}° ± {all_bp_diff.std():.2f}° (max: {all_bp_diff.max():.2f}°)")
+
+ print("\n" + "="*60)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/diagnose_output.log b/diagnose_output.log
new file mode 100644
index 0000000000000000000000000000000000000000..4de4f0e6938d7d1fb9d4732d395e512f4b3e21bd
--- /dev/null
+++ b/diagnose_output.log
@@ -0,0 +1,63 @@
+[[36m01/08 05:07:54[0m][[32mINFO[0m] [PL-Trainer] Loading ckpt: e004-s000005.ckpt[0m
+[[36m01/08 05:07:54[0m][[32mINFO[0m] [PL-Trainer] Loading ckpt: e004-s000005.ckpt[0m
+
+=== Loading: 0_biboo_birthday_speech.pt ===
+
+Loading GENMO model configuration...
+Using checkpoint: e004-s000005.ckpt
+
+Loading pretrained GENMO model...
+Gen only test timestep respacing: 50
+Running model inference...
+Preproc taken: 0.08589911460876465
+Demo taken: 3.6113734245300293
+
+============================================================
+FRAME 0 COMPARISON: PREDICTION vs GROUND TRUTH
+============================================================
+
+--- In-Camera Global Orientation ---
+GT (YXZ deg): yaw=-119.65, pitch= -0.06, roll=-178.54
+Pred (YXZ deg): yaw=-133.52, pitch= -12.25, roll= 169.91
+Diff (deg): yaw= -13.87, pitch= -12.19, roll= 348.45
+
+--- World/Global Orientation ---
+GT (YXZ deg): yaw= 169.81, pitch= 0.18, roll= 1.24
+Pred (YXZ deg): yaw= 134.80, pitch= 4.41, roll= -3.63
+Diff (deg): yaw= -35.01, pitch= 4.23, roll= -4.87
+
+--- In-Camera Translation ---
+GT: [ -0.045, 0.911, 1.153]
+Pred: [ -0.023, 0.821, 1.147]
+Diff: [ 0.022, -0.090, -0.005]
+
+--- World Translation ---
+GT: [ 0.000, 0.000, 0.000]
+Pred: [ -0.001, 1.237, 0.003]
+Diff: [ -0.001, 1.237, 0.003]
+
+--- Body Pose ---
+Max difference: 101.85°
+Mean difference: 8.71°
+
+============================================================
+SUMMARY ACROSS ALL FRAMES
+============================================================
+
+In-camera orientation errors (mean ± std):
+ Yaw: 11.03° ± 1.44°
+ Pitch: 12.49° ± 1.06°
+ Roll: 348.56° ± 0.37°
+
+World orientation errors (mean ± std):
+ Yaw: 37.47° ± 1.39°
+ Pitch: 4.45° ± 0.55°
+ Roll: 4.69° ± 0.32°
+
+In-camera translation error: 0.0987m ± 0.0070m
+World translation error: 1.2386m ± 0.0017m
+
+Body pose error: 12.39° ± 22.74° (max: 171.51°)
+
+============================================================
+
diff --git a/e004-s000005.ckpt b/e004-s000005.ckpt
new file mode 100644
index 0000000000000000000000000000000000000000..d3749da84614e2cc2f24f73a906fec25e6aa956f
--- /dev/null
+++ b/e004-s000005.ckpt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:da3173cd5990f638d6b8dd60708800b70c50fc6a6dadda1e31ae30800782a837
+size 2097279492
diff --git a/example_optim_mld.py b/example_optim_mld.py
new file mode 100644
index 0000000000000000000000000000000000000000..74d0a49d772d3e6db906198ff89087a83674b7d4
--- /dev/null
+++ b/example_optim_mld.py
@@ -0,0 +1,394 @@
+from __future__ import annotations
+
+import os
+import pdb
+import random
+import time
+from typing import Literal
+from dataclasses import dataclass, asdict, make_dataclass
+
+import numpy as np
+import torch
+import torch.nn as nn
+import torch.optim as optim
+import tyro
+import yaml
+from torch.distributions.normal import Normal
+from torch.utils.tensorboard import SummaryWriter
+from pathlib import Path
+from tqdm import tqdm
+import pickle
+import json
+import copy
+
+from model.mld_denoiser import DenoiserMLP, DenoiserTransformer
+from model.mld_vae import AutoMldVae
+from data_loaders.humanml.data.dataset import WeightedPrimitiveSequenceDataset, SinglePrimitiveDataset
+from utils.smpl_utils import *
+from utils.misc_util import encode_text, compose_texts_with_and
+from pytorch3d import transforms
+from diffusion import gaussian_diffusion as gd
+from diffusion.respace import SpacedDiffusion, space_timesteps
+from diffusion.resample import create_named_schedule_sampler
+
+from mld.train_mvae import Args as MVAEArgs
+from mld.train_mvae import DataArgs, TrainArgs
+from mld.train_mld import DenoiserArgs, MLDArgs, create_gaussian_diffusion, DenoiserMLPArgs, DenoiserTransformerArgs
+from mld.rollout_mld import load_mld, ClassifierFreeWrapper
+
+debug = 0
+
+@dataclass
+class OptimArgs:
+ seed: int = 0
+ torch_deterministic: bool = True
+ device: str = "cuda"
+ save_dir = None
+
+ denoiser_checkpoint: str = ''
+ optim_input: str = ''
+ text_prompt: str = None
+
+ respacing: str = 'ddim10'
+ guidance_param: float = 5.0
+ export_smpl: int = 0
+ zero_noise: int = 0
+ use_predicted_joints: int = 0
+ batch_size: int = 1
+ result_dir: str = 'inbetween'
+ seed_type: str= 'history'
+
+ optim_lr: float = 0.01
+ optim_steps: int = 300
+ optim_unit_grad: int = 1
+ optim_anneal_lr: int = 1
+ weight_jerk: float = 0.0
+ weight_floor: float = 0.0
+ init_noise_scale: float = 1.0
+
+
+def calc_jerk(joints):
+ vel = joints[:, 1:] - joints[:, :-1] # --> B x T-1 x 22 x 3
+ acc = vel[:, 1:] - vel[:, :-1] # --> B x T-2 x 22 x 3
+ jerk = acc[:, 1:] - acc[:, :-1] # --> B x T-3 x 22 x 3
+ jerk = torch.sqrt((jerk ** 2).sum(dim=-1)) # --> B x T-3 x 22, compute L1 norm of jerk
+ jerk = jerk.amax(dim=[1, 2]) # --> B, Get the max of the jerk across all joints and frames
+
+ return jerk.mean()
+
+def optimize(text_prompt, canonicalized_primitive_dict, goal_joints, joints_mask, denoiser_args, denoiser_model, vae_args, vae_model, diffusion, dataset, optim_args):
+ device = optim_args.device
+ batch_size = optim_args.batch_size
+ future_length = dataset.future_length
+ history_length = dataset.history_length
+ primitive_length = history_length + future_length
+ start_idx = history_length - 1 if optim_args.seed_type == 'repeat' else 0
+ end_idx = start_idx + seq_length - 1
+ assert 'ddim' in optim_args.respacing
+ sample_fn = diffusion.ddim_sample_loop_full_chain
+
+ texts = []
+ if ',' in text_prompt: # contain a time line of multipel actions
+ num_rollout = 0
+ for segment in text_prompt.split(','):
+ action, num_mp = segment.split('*')
+ action = compose_texts_with_and(action.split(' and '))
+ texts = texts + [action] * int(num_mp)
+ num_rollout += int(num_mp)
+ else:
+ action, num_rollout = text_prompt.split('*')
+ action = compose_texts_with_and(action.split(' and '))
+ num_rollout = int(num_rollout)
+ for _ in range(num_rollout):
+ texts.append(action)
+ all_text_embedding = encode_text(dataset.clip_model, texts, force_empty_zero=True).to(dtype=torch.float32,
+ device=device)
+ primitive_utility = dataset.primitive_utility
+
+ out_path = optim_args.save_dir
+ filename = f'guidance{optim_args.guidance_param}_seed{optim_args.seed}'
+ if text_prompt != '':
+ filename = text_prompt[:40].replace(' ', '_').replace('.', '') + '_' + filename
+ if optim_args.respacing != '':
+ filename = f'{optim_args.respacing}_{filename}'
+ # if optim_args.smooth:
+ # filename = f'smooth_{filename}'
+ if optim_args.zero_noise:
+ filename = f'zero_noise_{filename}'
+ if optim_args.use_predicted_joints:
+ filename = f'use_pred_joints_{filename}'
+ filename = f'scale{optim_args.init_noise_scale}_floor{optim_args.weight_floor}_jerk{optim_args.weight_jerk}_{filename}'
+ out_path = out_path / optim_args.result_dir / f'{optim_args.seed_type}seed' / filename
+ out_path.mkdir(parents=True, exist_ok=True)
+
+ batch = dataset.get_batch(batch_size=optim_args.batch_size)
+ input_motions, model_kwargs = batch[0]['motion_tensor_normalized'], {'y': batch[0]}
+ del model_kwargs['y']['motion_tensor_normalized']
+ gender = model_kwargs['y']['gender'][0]
+ betas = model_kwargs['y']['betas'][:, :primitive_length, :].to(device) # [B, H+F, 10]
+ pelvis_delta = primitive_utility.calc_calibrate_offset({
+ 'betas': betas[:, 0, :],
+ 'gender': gender,
+ })
+ # print(input_motions, model_kwargs)
+ input_motions = input_motions.to(device) # [B, D, 1, T]
+ motion_tensor = input_motions.squeeze(2).permute(0, 2, 1) # [B, T, D]
+ history_motion_gt = motion_tensor[:, :history_length, :] # [B, H, D]
+ if text_prompt == '':
+ optim_args.guidance_param = 0. # Force unconditioned generation
+
+ def rollout(noise):
+ motion_sequences = None
+ history_motion = history_motion_gt
+ transf_rotmat = torch.eye(3, device=device, dtype=torch.float32).unsqueeze(0).repeat(batch_size, 1, 1)
+ transf_transl = torch.zeros(3, device=device, dtype=torch.float32).reshape(1, 1, 3).repeat(batch_size, 1, 1)
+ for segment_id in range(num_rollout):
+ text_embedding = all_text_embedding[segment_id].expand(batch_size, -1) # [B, 512]
+ guidance_param = torch.ones(batch_size, *denoiser_args.model_args.noise_shape).to(device=device) * optim_args.guidance_param
+ y = {
+ 'text_embedding': text_embedding,
+ 'history_motion_normalized': history_motion,
+ 'scale': guidance_param,
+ }
+
+ x_start_pred = sample_fn(
+ denoiser_model,
+ (batch_size, *denoiser_args.model_args.noise_shape),
+ clip_denoised=False,
+ model_kwargs={'y': y},
+ skip_timesteps=0, # 0 is the default value - i.e. don't skip any step
+ init_image=None,
+ progress=False,
+ noise=noise[segment_id],
+ ) # [B, T=1, D]
+ # x_start_pred = x_start_pred.clamp(min=-3, max=3)
+ # print('x_start_pred:', x_start_pred.mean(), x_start_pred.std(), x_start_pred.min(), x_start_pred.max())
+ latent_pred = x_start_pred.permute(1, 0, 2) # [T=1, B, D]
+ future_motion_pred = vae_model.decode(latent_pred, history_motion, nfuture=future_length,
+ scale_latent=denoiser_args.rescale_latent) # [B, F, D], normalized
+
+ future_frames = dataset.denormalize(future_motion_pred)
+ new_history_frames = future_frames[:, -history_length:, :]
+
+ """transform primitive to world coordinate, prepare for serialization"""
+ if segment_id == 0: # add init history motion
+ future_frames = torch.cat([dataset.denormalize(history_motion), future_frames], dim=1)
+ future_feature_dict = primitive_utility.tensor_to_dict(future_frames)
+ future_feature_dict.update(
+ {
+ 'transf_rotmat': transf_rotmat,
+ 'transf_transl': transf_transl,
+ 'gender': gender,
+ 'betas': betas[:, :future_length, :] if segment_id > 0 else betas[:, :primitive_length, :],
+ 'pelvis_delta': pelvis_delta,
+ }
+ )
+ future_primitive_dict = primitive_utility.feature_dict_to_smpl_dict(future_feature_dict)
+ future_primitive_dict = primitive_utility.transform_primitive_to_world(future_primitive_dict)
+ if motion_sequences is None:
+ motion_sequences = future_primitive_dict
+ else:
+ for key in ['transl', 'global_orient', 'body_pose', 'betas', 'joints']:
+ motion_sequences[key] = torch.cat([motion_sequences[key], future_primitive_dict[key]], dim=1) # [B, T, ...]
+
+ """update history motion seed, update global transform"""
+ history_feature_dict = primitive_utility.tensor_to_dict(new_history_frames)
+ history_feature_dict.update(
+ {
+ 'transf_rotmat': transf_rotmat,
+ 'transf_transl': transf_transl,
+ 'gender': gender,
+ 'betas': betas[:, :history_length, :],
+ 'pelvis_delta': pelvis_delta,
+ }
+ )
+ canonicalized_history_primitive_dict, blended_feature_dict = primitive_utility.get_blended_feature(
+ history_feature_dict, use_predicted_joints=optim_args.use_predicted_joints)
+ transf_rotmat, transf_transl = canonicalized_history_primitive_dict['transf_rotmat'], \
+ canonicalized_history_primitive_dict['transf_transl']
+ history_motion = primitive_utility.dict_to_tensor(blended_feature_dict)
+ history_motion = dataset.normalize(history_motion) # [B, T, D]
+
+ motion_sequences['texts'] = texts
+ return motion_sequences
+
+ optim_steps = optim_args.optim_steps
+ lr = optim_args.optim_lr
+ noise = torch.randn(num_rollout, batch_size, *denoiser_args.model_args.noise_shape,
+ device=device, dtype=torch.float32)
+ # noise = noise.clip(min=-1, max=1)
+ noise = noise * optim_args.init_noise_scale
+ noise.requires_grad_(True)
+ reduction_dims = list(range(1, len(noise.shape)))
+ criterion = torch.nn.HuberLoss(reduction='mean', delta=1.0)
+
+ optimizer = torch.optim.Adam([noise], lr=lr)
+ for i in tqdm(range(optim_steps)):
+ optimizer.zero_grad()
+ if optim_args.optim_anneal_lr:
+ frac = 1.0 - i / optim_steps
+ lrnow = frac * lr
+ optimizer.param_groups[0]["lr"] = lrnow
+
+ motion_sequences = rollout(noise)
+ # joints_diff = (motion_sequences['joints'][:, seq_length - 1, joints_mask] - goal_joints[:, joints_mask]) ** 2
+ # joints_diff = torch.sqrt(joints_diff.sum(dim=-1)).mean(dim=1).mean(dim=0)
+ # loss_joints = joints_diff
+ # print('joints shape:', motion_sequences['joints'].shape, goal_joints.shape, joints_mask.shape)
+ loss_joints = criterion(motion_sequences['joints'][:, end_idx, joints_mask], goal_joints[:, joints_mask])
+ loss_jerk = calc_jerk(motion_sequences['joints'][:, start_idx:end_idx + 1])
+ floor_height = motion_sequences['joints'][:, 0, FOOT_JOINTS_IDX, 2].amin(dim=-1) # [B], assuming first frame on floor
+ foot_height = motion_sequences['joints'][:, start_idx:end_idx + 1, FOOT_JOINTS_IDX, 2].amin(dim=-1) # [B, T]
+ loss_floor = -(foot_height - floor_height.unsqueeze(1)).clamp(max=0).mean()
+ loss = loss_joints + optim_args.weight_jerk * loss_jerk + optim_args.weight_floor * loss_floor
+ loss.backward()
+ if optim_args.optim_unit_grad:
+ noise.grad.data /= noise.grad.norm(p=2, dim=reduction_dims, keepdim=True).clamp(min=1e-6)
+ optimizer.step()
+ # print(f'[{i}/{optim_steps}] loss: {loss.item()} joints_diff: {loss_joints.item()} jerk: {loss_jerk.item()} floor: {loss_floor.item()}')
+ print(f'[{i}/{optim_steps}] loss: {loss.item()} joints_diff: {loss_joints.item()} jerk: {loss_jerk.item()} floor: {loss_floor.item()}')
+
+ motion_sequences = rollout(noise)
+ # export input sequence
+ sequence = {
+ 'texts': texts,
+ 'gender': canonicalized_primitive_dict['gender'],
+ 'betas': canonicalized_primitive_dict['betas'][0],
+ 'transl': canonicalized_primitive_dict['transl'][0],
+ 'global_orient': canonicalized_primitive_dict['global_orient'][0],
+ 'body_pose': canonicalized_primitive_dict['body_pose'][0],
+ 'joints': canonicalized_primitive_dict['joints'][0],
+ 'history_length': history_length,
+ 'future_length': future_length,
+ 'mocap_framerate': dataset.target_fps,
+ }
+ if optim_args.seed_type == 'history':
+ for key in ['betas', 'transl', 'global_orient', 'body_pose', 'joints']:
+ sequence[key][history_length:-1] = sequence[key][history_length]
+ tensor_dict_to_device(sequence, 'cpu')
+ with open(os.path.join(out_path, f'input.pkl'), 'wb') as f:
+ pickle.dump(sequence, f)
+
+ for idx in range(optim_args.batch_size):
+ sequence = {
+ 'texts': texts,
+ 'gender': motion_sequences['gender'],
+ 'betas': motion_sequences['betas'][idx, start_idx:end_idx + 1],
+ 'transl': motion_sequences['transl'][idx, start_idx:end_idx + 1],
+ 'global_orient': motion_sequences['global_orient'][idx, start_idx:end_idx + 1],
+ 'body_pose': motion_sequences['body_pose'][idx, start_idx:end_idx + 1],
+ 'joints': motion_sequences['joints'][idx, start_idx:end_idx + 1],
+ 'history_length': history_length,
+ 'future_length': future_length,
+ 'mocap_framerate': dataset.target_fps,
+ }
+ tensor_dict_to_device(sequence, 'cpu')
+ with open(out_path / f'sample_{idx}.pkl', 'wb') as f:
+ pickle.dump(sequence, f)
+
+ # export smplx sequences for blender
+ if optim_args.export_smpl:
+ poses = transforms.matrix_to_axis_angle(
+ torch.cat([sequence['global_orient'].reshape(-1, 1, 3, 3), sequence['body_pose']], dim=1)
+ ).reshape(-1, 22 * 3)
+ poses = torch.cat([poses, torch.zeros(poses.shape[0], 99).to(dtype=poses.dtype, device=poses.device)],
+ dim=1)
+ data_dict = {
+ 'mocap_framerate': dataset.target_fps, # 30
+ 'gender': sequence['gender'],
+ 'betas': sequence['betas'][0, :10].detach().cpu().numpy(),
+ 'poses': poses.detach().cpu().numpy(),
+ 'trans': sequence['transl'].detach().cpu().numpy(),
+ }
+ with open(out_path / f'sample_{idx}_smplx.npz', 'wb') as f:
+ np.savez(f, **data_dict)
+
+ abs_path = out_path.absolute()
+ print(f'[Done] Results are at [{abs_path}]')
+
+if __name__ == '__main__':
+ optim_args = tyro.cli(OptimArgs)
+ # TRY NOT TO MODIFY: seeding
+ random.seed(optim_args.seed)
+ np.random.seed(optim_args.seed)
+ torch.manual_seed(optim_args.seed)
+ torch.set_default_dtype(torch.float32)
+ torch.backends.cudnn.deterministic = optim_args.torch_deterministic
+ device = torch.device(optim_args.device if torch.cuda.is_available() else "cpu")
+ optim_args.device = device
+
+ denoiser_args, denoiser_model, vae_args, vae_model = load_mld(optim_args.denoiser_checkpoint, device)
+ denoiser_checkpoint = Path(optim_args.denoiser_checkpoint)
+ save_dir = denoiser_checkpoint.parent / denoiser_checkpoint.name.split('.')[0] / 'optim'
+ save_dir.mkdir(parents=True, exist_ok=True)
+ optim_args.save_dir = save_dir
+
+ diffusion_args = denoiser_args.diffusion_args
+ diffusion_args.respacing = optim_args.respacing
+ print('diffusion_args:', asdict(diffusion_args))
+ diffusion = create_gaussian_diffusion(diffusion_args)
+
+ # load initial seed dataset
+ seq_path = Path(optim_args.optim_input)
+ dataset = SinglePrimitiveDataset(cfg_path=vae_args.data_args.cfg_path, # cfg path from model checkpoint
+ dataset_path=vae_args.data_args.data_dir, # dataset path from model checkpoint
+ sequence_path=seq_path,
+ body_type=vae_args.data_args.body_type,
+ batch_size=optim_args.batch_size,
+ device=device,
+ enforce_gender='male',
+ enforce_zero_beta=1,
+ )
+ future_length = dataset.future_length
+ history_length = dataset.history_length
+ primitive_length = history_length + future_length
+ primitive_utility = dataset.primitive_utility
+ print('body type:', primitive_utility.body_type)
+
+ with open(seq_path, 'rb') as f:
+ input_sequence = pickle.load(f)
+ seq_length = input_sequence['transl'].shape[0]
+ num_rollout = int(np.ceil((seq_length - 1) / future_length)) if optim_args.seed_type == 'repeat' else int(np.ceil((seq_length - history_length) / future_length))
+ print(f'seq_length: {seq_length}, num_rollout: {num_rollout}')
+ text_prompt = input_sequence['texts'][0] if optim_args.text_prompt is None else optim_args.text_prompt
+ text_prompt = f"{text_prompt}*{num_rollout}"
+
+ body_pose = torch.tensor(input_sequence['body_pose'], dtype=torch.float32)
+ body_pose = transforms.axis_angle_to_matrix(body_pose.reshape(-1, 3)).reshape(-1, 21, 3, 3).unsqueeze(
+ 0) # [1, T, 21, 3, 3]
+ global_orient = torch.tensor(input_sequence['global_orient'], dtype=torch.float32)
+ global_orient = transforms.axis_angle_to_matrix(global_orient.reshape(-1, 3)).reshape(-1, 3, 3).unsqueeze(
+ 0) # [1, T, 3, 3]
+ transl = torch.tensor(input_sequence['transl'], dtype=torch.float32).unsqueeze(0) # [1, T, 3]
+ betas = torch.tensor(input_sequence['betas'],
+ dtype=torch.float32) if not dataset.enforce_zero_beta else torch.zeros(10, dtype=torch.float32)
+ betas = betas.expand(1, seq_length, 10) # [1, T, 10]
+ # transl[:, 1:-1] = transl[:, 0]
+ # body_pose[:, 1:-1] = body_pose[:, 0]
+ # global_orient[:, 1:-1] = global_orient[:, 0]
+ seq_dict = {
+ 'gender': dataset.enforce_gender,
+ 'betas': betas,
+ 'transl': transl,
+ 'body_pose': body_pose,
+ 'global_orient': global_orient,
+ 'transf_rotmat': torch.eye(3).unsqueeze(0),
+ 'transf_transl': torch.zeros(1, 1, 3),
+ }
+ seq_dict = tensor_dict_to_device(seq_dict, device)
+ _, _, canonicalized_primitive_dict = primitive_utility.canonicalize(seq_dict)
+ body_model = primitive_utility.get_smpl_model(dataset.enforce_gender)
+ joints = body_model(return_verts=False,
+ betas=canonicalized_primitive_dict['betas'][0],
+ body_pose=canonicalized_primitive_dict['body_pose'][0],
+ global_orient=canonicalized_primitive_dict['global_orient'][0],
+ transl=canonicalized_primitive_dict['transl'][0]
+ ).joints[:, :22, :] # [T, 22, 3]
+ canonicalized_primitive_dict['joints'] = joints.unsqueeze(0) # [1, T, 22, 3]
+ goal_joints = joints[[-1]].expand(optim_args.batch_size, -1, -1) # [B, 22, 3]
+ joints_mask = torch.ones(22, dtype=torch.bool, device=device)
+
+ optimize(text_prompt, canonicalized_primitive_dict, goal_joints, joints_mask, denoiser_args, denoiser_model, vae_args, vae_model, diffusion, dataset, optim_args)
+
+
+
diff --git a/genmo/__init__.py b/genmo/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/genmo/callbacks/autoresume_callback.py b/genmo/callbacks/autoresume_callback.py
new file mode 100644
index 0000000000000000000000000000000000000000..de035a5a681b317a4b74803bc7d0eb1deda14776
--- /dev/null
+++ b/genmo/callbacks/autoresume_callback.py
@@ -0,0 +1,140 @@
+import os
+from typing import Any
+
+from pytorch_lightning import Callback, LightningModule, Trainer
+
+try:
+ import sys
+
+ sys.path.append(os.environ.get("SUBMIT_SCRIPTS", "."))
+ from userlib.auto_resume import AutoResume
+except ModuleNotFoundError:
+ AutoResume = None
+
+
+class AutoResumeCallback(Callback):
+ def __init__(self, version=None) -> None:
+ if AutoResume is not None:
+ AutoResume.init()
+ self.version = version
+ self.last_epoch_checkpoint = None
+
+ def _dump_current_checkpoint(self, trainer):
+ cp = trainer.checkpoint_callback
+ try:
+ checkpoint_dict = trainer._checkpoint_connector.dump_checkpoint(
+ cp.save_weights_only
+ )
+ except ValueError:
+ # This can happen for 16 precision for the first iterations
+ # when the GradScaler is still adapting and skipping steps
+ checkpoint_dict = None
+ return checkpoint_dict
+
+ def _save_checkpoint(self, trainer, checkpoint, filepath):
+ """Save a checkpoint to the memory."""
+
+ trainer.strategy.save_checkpoint(
+ checkpoint,
+ filepath,
+ )
+ # trainer.strategy.barrier("Trainer.save_checkpoint")
+ os.chmod(filepath, 0o755)
+
+ def _save_last_checkpoints(self, trainer):
+ cp = trainer.checkpoint_callback
+ # monitor_candidates = cp._monitor_candidates(trainer)
+
+ # Save last epoch. This is the one we will use for the autoresume
+ filepath_epoch = cp.output_dir / "last.ckpt"
+ if self.last_epoch_checkpoint is not None:
+ self._save_checkpoint(trainer, self.last_epoch_checkpoint, filepath_epoch)
+ else:
+ filepath_epoch = None
+
+ # Save last step just in case
+ # filepath_step = cp.output_dir / 'last_step.ckpt'
+ # last_step_checkpoint = self._dump_current_checkpoint(trainer)
+ # if last_step_checkpoint is not None:
+ # self._save_checkpoint(trainer, last_step_checkpoint, filepath_step)
+
+ # save top k
+ # self._save_topk_checkpoint(trainer)
+ return filepath_epoch
+
+ def _check_autoresume(self, trainer, pl_module):
+ if AutoResume is not None and AutoResume.termination_requested():
+ if trainer.global_rank == 0:
+ checkpoint = self._save_last_checkpoints(trainer)
+ details = {
+ "checkpoint": checkpoint,
+ "version": str(self.version),
+ }
+ message = f"[Auto Resume] Terminateing. checkpoint: {checkpoint} version: {details['version']}"
+ print(message, flush=True)
+ AutoResume.request_resume(details, message=message)
+ trainer.should_stop = True
+ trainer.limit_val_batches = 0
+ else:
+ print(f"[Auto Resume] Rank {trainer.global_rank} exiting.", flush=True)
+
+ if hasattr(pl_module, "cleanup_for_autoresume"):
+ pl_module.cleanup_for_autoresume()
+
+ def on_train_epoch_end(
+ self,
+ trainer: Trainer,
+ pl_module: LightningModule,
+ ) -> None:
+ # only for rank 0
+ if trainer.global_rank == 0:
+ # save the last epoch checkpoint in memory
+ # this is the one we will dump for the "last" checkpoint
+ checkpoint_dict = self._dump_current_checkpoint(trainer)
+ self.last_epoch_checkpoint = checkpoint_dict
+
+ def on_train_batch_end(
+ self,
+ trainer: Trainer,
+ pl_module: LightningModule,
+ outputs: Any,
+ batch: Any,
+ batch_idx: int,
+ ) -> None:
+ self._check_autoresume(trainer, pl_module)
+
+ def on_validation_batch_end(
+ self,
+ trainer: Trainer,
+ pl_module: LightningModule,
+ outputs: Any,
+ batch: Any,
+ batch_idx: int,
+ dataloader_idx: int = 0,
+ ) -> None:
+ pass
+ # self._check_autoresume(trainer, pl_module)
+
+ def on_test_batch_end(
+ self,
+ trainer: Trainer,
+ pl_module: LightningModule,
+ outputs: Any,
+ batch: Any,
+ batch_idx: int,
+ dataloader_idx: int = 0,
+ ) -> None:
+ pass
+ # self._check_autoresume(trainer, pl_module)
+
+ def on_predict_batch_end(
+ self,
+ trainer: Trainer,
+ pl_module: LightningModule,
+ outputs: Any,
+ batch: Any,
+ batch_idx: int,
+ dataloader_idx: int = 0,
+ ) -> None:
+ pass
+ # self._check_autoresume(trainer, pl_module)
diff --git a/genmo/callbacks/metric/metric_emdb.py b/genmo/callbacks/metric/metric_emdb.py
new file mode 100644
index 0000000000000000000000000000000000000000..830d4c7b46146388a607a807a807c8ff358eaecb
--- /dev/null
+++ b/genmo/callbacks/metric/metric_emdb.py
@@ -0,0 +1,196 @@
+import numpy as np
+import pytorch_lightning as pl
+import torch
+from einops import einsum
+
+from genmo.utils.eval_utils import as_np_array, compute_camcoord_metrics
+from genmo.utils.gather import all_gather
+from genmo.utils.geo_transform import apply_T_on_points
+from genmo.utils.pylogger import Log
+from genmo.utils.vis_utils import visualize_smpl_scene
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+class MetricMocap(pl.Callback):
+ def __init__(self, vis_every_n_val=10):
+ super().__init__()
+ self.vis_every_n_val = vis_every_n_val
+ self.num_val = 0
+ # vid->result
+ self.metric_aggregator = {
+ "pa_mpjpe": {},
+ "mpjpe": {},
+ "pve": {},
+ "accel": {},
+ }
+
+ # SMPLX and SMPL
+ self.smplx = make_smplx("supermotion_EVAL3DPW")
+ self.smpl = {
+ "male": make_smplx("smpl", gender="male"),
+ "female": make_smplx("smpl", gender="female"),
+ }
+ self.J_regressor = torch.load(
+ "third_party/GVHMR/inputs/checkpoints/body_models/smpl_3dpw14_J_regressor_sparse.pt"
+ ).to_dense()
+ self.J_regressor24 = torch.load(
+ "third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt"
+ )
+ self.smplx2smpl = torch.load(
+ "third_party/GVHMR/inputs/checkpoints/body_models/smplx2smpl_sparse.pt"
+ )
+ self.faces_smplx = self.smplx.faces
+ self.faces_smpl = self.smpl["male"].faces
+ self.img_h = self.img_w = 256
+
+ # The metrics are calculated similarly for val/test/predict
+ self.on_test_batch_end = self.on_validation_batch_end = (
+ self.on_predict_batch_end
+ )
+
+ # Only validation record the metrics with logger
+ self.on_test_epoch_end = self.on_validation_epoch_end = (
+ self.on_predict_epoch_end
+ )
+ self.on_test_epoch_start = self.on_validation_epoch_start = (
+ self.on_predict_epoch_start
+ )
+
+ # ================== Batch-based Computation ================== #
+ def on_predict_batch_end(
+ self, trainer, pl_module, outputs, batch, batch_idx, dataloader_idx=0
+ ):
+ """The behaviour is the same for val/test/predict"""
+ assert batch["B"] == 1
+ dataset_id = batch["meta"][0]["dataset_id"]
+ if dataset_id != "3DPW":
+ return
+
+ # Move to cuda if not
+ self.smplx = self.smplx.cuda()
+ for g in ["male", "female"]:
+ self.smpl[g] = self.smpl[g].cuda()
+ self.J_regressor = self.J_regressor.cuda()
+ self.J_regressor24 = self.J_regressor24.cuda()
+ self.smplx2smpl = self.smplx2smpl.cuda()
+
+ vid = batch["meta"][0]["vid"]
+ # seq_length = batch["length"][0].item()
+ gender = batch["gender"][0]
+ T_w2c = batch["gt_T_w2c"][0]
+ mask = batch["mask"]["valid"][0]
+
+ # Groundtruth (cam)
+ target_w_params = {k: v[0] for k, v in batch["smpl_params"].items()}
+ target_w_output = self.smpl[gender](**target_w_params)
+ target_w_verts = target_w_output.vertices
+ target_c_verts = apply_T_on_points(target_w_verts, T_w2c)
+ target_c_j3d = torch.matmul(self.J_regressor, target_c_verts)
+ target_c_j3d24 = torch.matmul(self.J_regressor24, target_c_verts)
+ offset = target_c_j3d24[..., [1, 2], :].mean(-2, keepdim=True) # (L, 1, 3)
+ target_cr_j3d24 = target_c_j3d24 - offset
+
+ # + Prediction -> Metric
+ smpl_out = self.smplx(**outputs["pred_smpl_params_incam"])
+ pred_c_verts = torch.stack(
+ [torch.matmul(self.smplx2smpl, v_) for v_ in smpl_out.vertices]
+ )
+ pred_c_j3d = einsum(self.J_regressor, pred_c_verts, "j v, l v i -> l j i")
+ pred_c_j3d24 = einsum(self.J_regressor24, pred_c_verts, "j v, l v i -> l j i")
+ offset = pred_c_j3d24[..., [1, 2], :].mean(-2, keepdim=True) # (L, 1, 3)
+ pred_cr_j3d24 = pred_c_j3d24 - offset
+ del smpl_out # Prevent OOM
+
+ if trainer.global_rank == 0 and self.num_val % self.vis_every_n_val == 0:
+ wandb_dict = visualize_smpl_scene(
+ "vis_3dpw_incam",
+ batch_idx,
+ vid,
+ pred_cr_j3d24,
+ target_cr_j3d24,
+ transform_mode="local",
+ )
+ self.wandb_html_dict.update(wandb_dict)
+
+ # Metric of current sequence
+ batch_eval = {
+ "pred_j3d": pred_c_j3d,
+ "target_j3d": target_c_j3d,
+ "pred_verts": pred_c_verts,
+ "target_verts": target_c_verts,
+ }
+ camcoord_metrics = compute_camcoord_metrics(
+ batch_eval, mask=mask, pelvis_idxs=[2, 3]
+ )
+ for k in camcoord_metrics:
+ self.metric_aggregator[k][vid] = as_np_array(camcoord_metrics[k])
+ # print(f"{vid} {k}: {camcoord_metrics[k].mean()}")
+
+ def on_predict_epoch_start(self, trainer, pl_module):
+ self.wandb_html_dict = {}
+
+ # ================== Epoch Summary ================== #
+ def on_predict_epoch_end(self, trainer, pl_module):
+ self.num_val += 1
+ pl_module.logger.log_metrics(self.wandb_html_dict)
+
+ """Without logger"""
+ local_rank, _ = trainer.local_rank, trainer.world_size
+ monitor_metric = "pa_mpjpe"
+
+ # Reduce metric_aggregator across all processes
+ metric_keys = list(self.metric_aggregator.keys())
+ with torch.inference_mode(False): # allow in-place operation of all_gather
+ metric_aggregator_gathered = all_gather(
+ self.metric_aggregator
+ ) # list of dict
+ for metric_key in metric_keys:
+ for d in metric_aggregator_gathered:
+ self.metric_aggregator[metric_key].update(d[metric_key])
+
+ total = len(self.metric_aggregator[monitor_metric])
+ Log.info(f"{total} sequences evaluated in {self.__class__.__name__}")
+ if total == 0:
+ return
+
+ # print monitored metric per sequence
+ mm_per_seq = {
+ k: v.mean() for k, v in self.metric_aggregator[monitor_metric].items()
+ }
+ if len(mm_per_seq) > 0:
+ sorted_mm_per_seq = sorted(
+ mm_per_seq.items(), key=lambda x: x[1], reverse=True
+ )
+ n_worst = 5 if trainer.state.stage == "validate" else len(sorted_mm_per_seq)
+ if local_rank == 0:
+ Log.info(
+ f"monitored metric {monitor_metric} per sequence\n"
+ + "\n".join(
+ [f"{m:5.1f} : {s}" for s, m in sorted_mm_per_seq[:n_worst]]
+ )
+ + "\n------"
+ )
+
+ # average over all batches
+ metrics_avg = {
+ k: np.concatenate(list(v.values())).mean()
+ for k, v in self.metric_aggregator.items()
+ }
+ if local_rank == 0:
+ Log.info(
+ "[Metrics] 3DPW:\n"
+ + "\n".join(f"{k}: {v:.1f}" for k, v in metrics_avg.items())
+ + "\n------"
+ )
+
+ # save to logger if available
+ if pl_module.logger is not None:
+ cur_epoch = pl_module.current_epoch
+ for k, v in metrics_avg.items():
+ pl_module.logger.log_metrics(
+ {f"val_metric_3DPW/{k}": v}, step=cur_epoch
+ )
+
+ # reset
+ for k in self.metric_aggregator:
+ self.metric_aggregator[k] = {}
diff --git a/genmo/callbacks/metric/metric_unity.py b/genmo/callbacks/metric/metric_unity.py
new file mode 100644
index 0000000000000000000000000000000000000000..5897b56524b070ec7927d2d47f634ab4a7c0a6ab
--- /dev/null
+++ b/genmo/callbacks/metric/metric_unity.py
@@ -0,0 +1,146 @@
+import numpy as np
+import pytorch_lightning as pl
+import torch
+from einops import einsum
+
+from genmo.utils.eval_utils import (
+ as_np_array,
+ compute_camcoord_metrics,
+)
+from genmo.utils.gather import all_gather
+from genmo.utils.geo_transform import apply_T_on_points
+from genmo.utils.pylogger import Log
+from genmo.utils.vis_utils import visualize_smpl_scene
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+class MetricUnity(pl.Callback):
+ def __init__(self, vis_every_n_val=10):
+ super().__init__()
+ self.vis_every_n_val = vis_every_n_val
+ self.num_val = 0
+
+ self.metric_aggregator = {
+ "pa_mpjpe": {},
+ "mpjpe": {},
+ "pve": {},
+ "accel": {},
+ }
+
+ # SMPL Models
+ self.smplx = make_smplx("supermotion")
+ self.smpl_model = {
+ "male": make_smplx("smpl", gender="male"),
+ "female": make_smplx("smpl", gender="female"),
+ "neutral": make_smplx("smpl", gender="neutral"),
+ }
+
+ # Regressors
+ self.J_regressor = torch.load(
+ "./third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt"
+ )
+ self.smplx2smpl = torch.load(
+ "./third_party/GVHMR/inputs/checkpoints/body_models/smplx2smpl_sparse.pt"
+ )
+
+ # Bind methods for all evaluation stages
+ self.on_test_batch_end = self.on_validation_batch_end = self.on_predict_batch_end
+ self.on_test_epoch_end = self.on_validation_epoch_end = self.on_predict_epoch_end
+ self.on_test_epoch_start = self.on_validation_epoch_start = self.on_predict_epoch_start
+
+ def on_predict_batch_end(self, trainer, pl_module, outputs, batch, batch_idx, dataloader_idx=0):
+ # Filter for Unity dataset
+ dataset_id = batch["meta"][0].get("dataset_id", "")
+ if dataset_id != "Unity":
+ return
+
+ # Move models to device
+ device = pl_module.device
+ self.smplx = self.smplx.to(device)
+ for g in self.smpl_model:
+ self.smpl_model[g] = self.smpl_model[g].to(device)
+ self.J_regressor = self.J_regressor.to(device)
+ self.smplx2smpl = self.smplx2smpl.to(device)
+
+ vid = batch["meta"][0]["vid"]
+
+ # Handle gender (Default to neutral if missing/unknown)
+ gender = batch.get("gender", ["neutral"])
+ if isinstance(gender, (list, tuple)):
+ gender = gender[0]
+ if gender not in ["male", "female", "neutral"]:
+ gender = "neutral"
+
+ # Get GT data
+ T_w2c = batch["gt_T_w2c"][0]
+ target_w_params = {k: v[0] for k, v in batch["smpl_params"].items()}
+
+ # Compute GT vertices/joints
+ target_w_output = self.smpl_model[gender](**target_w_params)
+ target_w_verts = target_w_output.vertices
+ target_w_j3d = torch.matmul(self.J_regressor, target_w_verts)
+
+ # Transform GT to Camera Frame
+ target_c_verts = apply_T_on_points(target_w_verts, T_w2c)
+ target_c_j3d = apply_T_on_points(target_w_j3d, T_w2c)
+
+ # Get Predictions (In Camera Frame)
+ pred_smpl_params_incam = outputs["pred_smpl_params_incam"]
+ smpl_out = self.smplx(**pred_smpl_params_incam)
+
+ # Convert SMPLX to SMPL topology
+ pred_c_verts = torch.stack(
+ [torch.matmul(self.smplx2smpl, v_) for v_ in smpl_out.vertices]
+ )
+ pred_c_j3d = einsum(self.J_regressor, pred_c_verts, "j v, l v i -> l j i")
+
+ # Visualization
+ if trainer.global_rank == 0 and self.num_val % self.vis_every_n_val == 0:
+ # Root relative for visualization
+ offset_pred = pred_c_j3d[..., [1, 2], :].mean(-2, keepdim=True)
+ offset_target = target_c_j3d[..., [1, 2], :].mean(-2, keepdim=True)
+
+ vis_dict = visualize_smpl_scene(
+ "vis_unity_incam",
+ batch_idx,
+ vid,
+ pred_c_j3d - offset_pred,
+ target_c_j3d - offset_target,
+ transform_mode="local",
+ )
+ self.vis_artifacts.update(vis_dict)
+
+ # Compute Metrics
+ batch_eval = {
+ "pred_j3d": pred_c_j3d,
+ "target_j3d": target_c_j3d,
+ "pred_verts": pred_c_verts,
+ "target_verts": target_c_verts,
+ }
+ mask = batch["mask"]["valid"][0] if "mask" in batch and "valid" in batch["mask"] else None
+ metrics = compute_camcoord_metrics(batch_eval, mask=mask)
+
+ for k in metrics:
+ self.metric_aggregator[k][vid] = as_np_array(metrics[k])
+
+ def on_predict_epoch_start(self, trainer, pl_module):
+ self.vis_artifacts = {}
+ for k in self.metric_aggregator:
+ self.metric_aggregator[k] = {}
+
+ def on_predict_epoch_end(self, trainer, pl_module):
+ self.num_val += 1
+
+ # Gather and Log
+ with torch.inference_mode(False):
+ gathered = all_gather(self.metric_aggregator)
+ for k in self.metric_aggregator:
+ for d in gathered:
+ self.metric_aggregator[k].update(d[k])
+
+ metrics_avg = {k: np.concatenate(list(v.values())).mean() for k, v in self.metric_aggregator.items() if v}
+
+ if pl_module.logger is not None:
+ for k, v in metrics_avg.items():
+ pl_module.log(f"val_metric_Unity/{k}", v, sync_dist=False)
+ # Keep metric namespace Unity-only for this fine-tune setup.
diff --git a/genmo/callbacks/prog_bar.py b/genmo/callbacks/prog_bar.py
new file mode 100644
index 0000000000000000000000000000000000000000..1a73e6a8e5b000a66b6e4c83cdb4f54c342fa7ea
--- /dev/null
+++ b/genmo/callbacks/prog_bar.py
@@ -0,0 +1,460 @@
+from collections import OrderedDict, deque
+from datetime import datetime, timedelta
+from numbers import Number
+from time import time
+from typing import Any, Dict, Union
+
+import pytorch_lightning as pl
+import torch
+from pytorch_lightning.callbacks.progress import ProgressBar
+from pytorch_lightning.callbacks.progress.tqdm_progress import Tqdm, TQDMProgressBar
+from pytorch_lightning.utilities import rank_zero_only
+
+from genmo.utils.pylogger import Log
+
+# ========== Helper functions ========== #
+
+
+def format_num(n):
+ f = "{0:.3g}".format(n).replace("+0", "+").replace("-0", "-")
+ n = str(n)
+ return f if len(f) < len(n) else n
+
+
+def convert_kwargs_to_str(**kwargs):
+ # Sort in alphabetical order to be more deterministic
+ postfix = OrderedDict([])
+ for key in sorted(kwargs.keys()):
+ new_key = key.split("/")[-1]
+ postfix[new_key] = kwargs[key]
+ # Preprocess stats according to datatype
+ for key in postfix.keys():
+ # Number: limit the length of the string
+ if isinstance(postfix[key], Number):
+ postfix[key] = format_num(postfix[key])
+ # Else for any other type, try to get the string conversion
+ elif not isinstance(postfix[key], str):
+ postfix[key] = str(postfix[key])
+ # Else if it's a string, don't need to preprocess anything
+ # Stitch together to get the final postfix
+ postfix = ", ".join(key + "=" + postfix[key].strip() for key in postfix.keys())
+ return postfix
+
+
+def convert_t_to_str(t):
+ """Convert time in second to string in format hour:minute:second.
+ If hour is 0, don't show it. Always show minute and second.
+ """
+ t_str = timedelta(seconds=t) # e.g. 0:00:00.704186
+ t_str = str(t_str).split(".")[0] # e.g. 0:00:00
+ if t_str[:2] == "0:":
+ t_str = t_str[2:]
+ return t_str
+
+
+class MyTQDMProgressBar(TQDMProgressBar, pl.Callback):
+ def init_train_tqdm(self):
+ bar = Tqdm(
+ desc="Training", # this will be overwritten anyway
+ bar_format="{desc}{percentage:3.0f}%[{bar:10}][{n_fmt}/{total_fmt}, {elapsed}→{remaining},{rate_fmt}]{postfix}",
+ position=(2 * self.process_position),
+ disable=self.is_disabled,
+ leave=False,
+ smoothing=0,
+ dynamic_ncols=False,
+ )
+ return bar
+
+ @rank_zero_only
+ def on_train_batch_end(self, trainer, pl_module, outputs, batch, batch_idx):
+ # this function also updates the main progress bar
+ super().on_train_batch_end(trainer, pl_module, outputs, batch, batch_idx)
+ # in this function, we only set the postfix of the main progress bar
+ n = batch_idx + 1
+ if self._should_update(n, self.train_progress_bar.total):
+ # Set post-fix string
+ # 1. maximum GPU usage
+ max_mem = torch.cuda.max_memory_allocated() / 1024.0 / 1024.0 / 1024.0
+ post_fix_str = f"maxGPU={max_mem:.1f}GB"
+
+ # 2. training metrics
+ training_metrics = self.get_metrics(trainer, pl_module)
+ training_metrics.pop("v_num", None)
+ post_fix_str += ", " + convert_kwargs_to_str(**training_metrics)
+
+ # extra message if applicable
+ if "message" in outputs:
+ post_fix_str += ", " + outputs["message"]
+
+ self.train_progress_bar.set_postfix_str(post_fix_str)
+
+
+class ProgressReporter(ProgressBar, pl.Callback):
+ def __init__(
+ self,
+ log_every_percent: float = 0.1, # report interval
+ exp_name=None, # if None, use pl_module.exp_name or "Unnamed Experiment"
+ data_name=None, # if None, use pl_module.exp_name or "Unknown Data"
+ **kwargs,
+ ):
+ super().__init__()
+ self.enable = True
+ # 1. Store experiment meta data.
+ self.log_every_percent = log_every_percent
+ self.exp_name = exp_name
+ self.data_name = data_name
+ self.batch_time_queue = deque(maxlen=5)
+ self.start_prompt = "🚀"
+ self.finish_prompt = "✅"
+ # 2. Utils for evaluation
+ self.n_finished = 0
+ self.time_train_epoch_start = time()
+
+ def disable(self):
+ self.enable = False
+
+ def setup(
+ self, trainer: pl.Trainer, pl_module: pl.LightningModule, stage: str
+ ) -> None:
+ # Connect to the trainer object.
+ super().setup(trainer, pl_module, stage)
+ self.stage = stage
+ self.time_exp_start = time()
+ self.epoch_exp_start = trainer.current_epoch
+
+ if self.exp_name is None:
+ if hasattr(pl_module, "exp_name"):
+ self.exp_name = pl_module.exp_name
+ else:
+ self.exp_name = "Unnamed Experiment"
+ if self.data_name is None:
+ if hasattr(pl_module, "data_name"):
+ self.data_name = pl_module.data_name
+ else:
+ self.data_name = "Unknown Data"
+
+ def print(self, *args: Any, **kwargs: Any) -> None:
+ print(*args)
+
+ def get_metrics(
+ self, trainer: pl.Trainer, pl_module: pl.LightningModule
+ ) -> Dict[str, Union[str, float]]:
+ """Get metrics from trainer for progress bar."""
+ items = super().get_metrics(trainer, pl_module)
+ items.pop("v_num", None)
+ return items
+
+ def _should_update(self, n_finished: int, total: int) -> bool:
+ """
+ Rule: Log every `log_every_percent` percent, or the last batch.
+ """
+ log_interval = max(int(total * self.log_every_percent), 1)
+ able = n_finished % log_interval == 0 or n_finished == total
+ if log_interval > 10:
+ able = able or n_finished in [5, 10] # always log
+ able = able and self.enable
+ return able
+
+ @rank_zero_only
+ def on_train_epoch_start(self, trainer: "pl.Trainer", *_: Any) -> None:
+ self.print("=" * 80)
+ Log.info(
+ f"{self.start_prompt}[FIT][Epoch {trainer.current_epoch}] Data: {self.data_name} Experiment: {self.exp_name}"
+ )
+ self.time_train_epoch_start = time()
+
+ @rank_zero_only
+ def on_train_batch_end(self, trainer, pl_module, outputs, batch, batch_idx):
+ super().on_train_batch_end(
+ trainer, pl_module, outputs, batch, batch_idx
+ ) # don't forget this :)
+ total = self.total_train_batches
+
+ # Speed
+ n_finished = batch_idx + 1
+ percent = 100 * n_finished / total
+ time_current = time()
+ self.batch_time_queue.append(time_current)
+ time_elapsed = time_current - self.time_train_epoch_start # second
+ time_remaining = time_elapsed * (total - n_finished) / n_finished # second
+ if len(self.batch_time_queue) == 1: # cannot compute speed
+ speed = 1 / time_elapsed
+ else:
+ speed = (len(self.batch_time_queue) - 1) / (
+ self.batch_time_queue[-1] - self.batch_time_queue[0]
+ )
+
+ # Skip if not update
+ if not self._should_update(n_finished, total):
+ return
+
+ # ===== Set Prefix string ===== #
+ # General
+ desc = "[Train]"
+
+ # Speed: Get elapsed time and estimated remaining time
+ time_elapsed_str = convert_t_to_str(time_elapsed)
+ time_remaining_str = convert_t_to_str(time_remaining)
+ speed_str = f"{speed:.2f}it/s" if speed > 1 else f"{1 / speed:.1f}s/it"
+ n_digit = len(str(total))
+ desc_speed = f"[{n_finished:{n_digit}d}/{total}={percent:3.0f}%, {time_elapsed_str} → {time_remaining_str}, {speed_str}]"
+
+ # ===== Set postfix string ===== #
+ # 1. maximum GPU usage
+ max_mem = torch.cuda.max_memory_allocated() / 1024.0 / 1024.0 / 1024.0
+ post_fix_str = f"maxGPU={max_mem:.1f}GB"
+
+ # 2. training step metrics
+ train_metrics = self.get_metrics(trainer, pl_module)
+ train_metrics = {
+ k: v
+ for k, v in train_metrics.items()
+ if ("train" in k and "epoch" not in k)
+ }
+ post_fix_str += ", " + convert_kwargs_to_str(**train_metrics)
+
+ # extra message if applicable
+ if "message" in outputs:
+ post_fix_str += ", " + outputs["message"]
+ post_fix_str = f"[{post_fix_str}]"
+
+ # ===== Output ===== #
+ bar_output = f"{desc}{desc_speed}{post_fix_str}"
+ self.print(bar_output)
+
+ @rank_zero_only
+ def on_train_epoch_end(
+ self, trainer: pl.Trainer, pl_module: pl.LightningModule
+ ) -> None:
+ super().on_train_epoch_end(trainer, pl_module)
+
+ # Clear
+ self.batch_time_queue.clear()
+
+ # Estimate Epoch time
+ n_finished = trainer.current_epoch + 1 - self.epoch_exp_start
+ n_to_finish = trainer.max_epochs - trainer.current_epoch - 1
+ time_current = time()
+ time_elapsed = time_current - self.time_exp_start
+ time_remaining = time_elapsed * n_to_finish / n_finished
+ time_elapsed_str = convert_t_to_str(time_elapsed)
+ time_remaining_str = convert_t_to_str(time_remaining)
+
+ # Metrics
+ # training epoch metrics
+ train_metrics = self.get_metrics(trainer, pl_module)
+ train_metrics = {
+ k: v for k, v in train_metrics.items() if ("train" in k and "epoch" in k)
+ }
+ train_metrics_str = convert_kwargs_to_str(**train_metrics)
+
+ Log.info(
+ f"{self.finish_prompt}[FIT][Epoch {trainer.current_epoch}] finished! {time_elapsed_str}→{time_remaining_str} | {train_metrics_str}"
+ )
+
+ # ===== Validation/Test/Prediction ===== #
+ @rank_zero_only
+ def on_validation_epoch_start(self, trainer, pl_module):
+ self.time_val_epoch_start = time()
+
+ @rank_zero_only
+ def on_validation_batch_end(
+ self, trainer, pl_module, outputs, batch, batch_idx, dataloader_idx=0
+ ):
+ self.n_finished += 1
+ n_finished = self.n_finished
+ total = self.total_val_batches
+ if not self._should_update(n_finished, total):
+ return
+
+ # General
+ desc = "[Val]"
+
+ # Speed
+ percent = 100 * n_finished / total
+ time_current = time()
+ time_elapsed = time_current - self.time_val_epoch_start # second
+ time_remaining = time_elapsed * (total - n_finished) / n_finished # second
+ time_elapsed_str = convert_t_to_str(time_elapsed)
+ time_remaining_str = convert_t_to_str(time_remaining)
+ desc_speed = f"[{n_finished}/{total} ={percent:3.0f}%, {time_elapsed_str}→{time_remaining_str}]"
+
+ # Output
+ bar_output = f"{desc} {desc_speed}"
+ self.print(bar_output)
+
+ def on_validation_epoch_end(
+ self, trainer: pl.Trainer, pl_module: pl.LightningModule
+ ) -> None:
+ # Reset
+ self.n_finished = 0
+
+
+class EmojiProgressReporter(ProgressBar, pl.Callback):
+ def __init__(
+ self,
+ refresh_rate_batch: Union[
+ int, None
+ ] = 1, # report interval of batch, set None to disable it
+ refresh_rate_epoch: int = 1, # report interval of epoch
+ **kwargs,
+ ):
+ super().__init__()
+ self.enable = True
+ # Store experiment meta data.
+ self.refresh_rate_batch = refresh_rate_batch
+ self.refresh_rate_epoch = refresh_rate_epoch
+
+ # Style of the progress bar.
+ self.title_prompt = "📝"
+ self.prog_prompt = "🚀"
+ self.timer_prompt = "⌛️"
+ self.metric_prompt = "📌"
+ self.finish_prompt = "✅"
+
+ def disable(self):
+ self.enable = False
+
+ def setup(self, trainer: pl.Trainer, pl_module: pl.LightningModule, stage: str):
+ # Connect to the trainer object.
+ super().setup(trainer, pl_module, stage)
+ self.stage = stage
+ self.time_start_batch = None
+ self.time_start_epoch = None
+ if hasattr(pl_module, "exp_name"):
+ self.exp_name = pl_module.exp_name
+ else:
+ self.exp_name = "Unnamed Experiment"
+ Log.warn(
+ "Experiment name not found, please set it to `pl_module.exp_name`!"
+ )
+
+ def print(self, *args: Any, **kwargs: Any):
+ print(*args)
+
+ def get_metrics(
+ self, trainer: pl.Trainer, pl_module: pl.LightningModule
+ ) -> Dict[str, Union[str, float]]:
+ """Get metrics from trainer for progress bar."""
+ items = super().get_metrics(trainer, pl_module)
+ items.pop("v_num", None)
+ return dict(sorted(items.items()))
+
+ def _should_log_batch(self, n: int) -> bool:
+ # Disable batch log.
+ if self.refresh_rate_batch is None:
+ return False
+ # Log at the first & last batch, and every `self.refresh_rate_batch` batches.
+ able = n % self.refresh_rate_batch == 0 or n == self.total_train_batches - 1
+ able = able and self.enable
+ return able
+
+ def _should_log_epoch(self, n: int) -> bool:
+ # Log at the first & last epoch, and every `self.refresh_rate_epoch` epochs.
+ able = n % self.refresh_rate_epoch == 0 or n == self.trainer.max_epochs - 1
+ able = able and self.enable
+ return able
+
+ def timestamp_delta_to_str(self, timestamp_delta: float):
+ """Convert delta timestamp to string."""
+ time_rest = timedelta(seconds=timestamp_delta)
+ hours, remainder = divmod(time_rest.seconds, 3600)
+ minutes, seconds = divmod(remainder, 60)
+ time_str = ""
+
+ # Check if the time is valid. Note that, if `hours` is visible, then `minutes` must be visible.
+ if hours <= 0:
+ hours = None
+ if minutes <= 0:
+ minutes = None
+ if seconds <= 0:
+ seconds = None
+
+ time_str += f"{hours}h " if hours is not None else ""
+ time_str += f"{minutes}m " if minutes is not None else ""
+ time_str += f"{seconds}s" if seconds is not None else ""
+ return time_str
+
+ @rank_zero_only
+ def on_train_batch_start(
+ self,
+ trainer: pl.Trainer,
+ pl_module: pl.LightningModule,
+ batch: Any,
+ batch_idx: int,
+ ):
+ super().on_train_batch_start(trainer, pl_module, batch, batch_idx)
+ # Initialize some meta data.
+ if self.time_start_batch is None:
+ self.time_start_batch = datetime.now().timestamp()
+
+ @rank_zero_only
+ def on_train_batch_end(self, trainer, pl_module, outputs, batch, batch_idx):
+ super().on_train_batch_end(
+ trainer, pl_module, outputs, batch, batch_idx
+ ) # don't forget this :)
+ # Get some meta data.
+ epoch_idx = trainer.current_epoch
+ percent = 100 * (batch_idx + 1) / (self.total_train_batches + 1)
+ metrics = self.get_metrics(trainer, pl_module)
+
+ # Current time.
+ time_cur_stamp = datetime.now().timestamp()
+ time_cur_str = datetime.fromtimestamp(time_cur_stamp).strftime("%m-%d %H:%M:%S")
+ # Rest time.
+ time_rest_stamp = (
+ (time_cur_stamp - self.time_start_batch) * (100 - percent) / percent
+ )
+ time_rest_str = self.timestamp_delta_to_str(time_rest_stamp)
+
+ if not self._should_log_batch(batch_idx):
+ return
+
+ # Print the logs.
+ self.print(
+ f"{self.title_prompt} [{self.stage.upper()}] Exp: {self.exp_name}..."
+ )
+ self.print(
+ f"{self.prog_prompt} Ep {epoch_idx}: {int(percent):02d}% <= [{batch_idx}/{self.total_train_batches}]"
+ )
+ self.print(
+ f"{self.timer_prompt} Time: {time_cur_str} | Ep Rest: {time_rest_str}"
+ )
+ for k, v in metrics.items():
+ self.print(f"{self.metric_prompt} {k}: {v}")
+ self.print("") # Add a blank line.
+
+ def on_train_epoch_start(self, trainer: pl.Trainer, pl_module: pl.LightningModule):
+ super().on_train_epoch_start(trainer, pl_module)
+ # Initialize some meta data.
+ self.time_start_batch = None
+ if self.time_start_epoch is None:
+ self.time_start_epoch = datetime.now().timestamp()
+
+ @rank_zero_only
+ def on_train_epoch_end(self, trainer: pl.Trainer, pl_module: pl.LightningModule):
+ super().on_train_epoch_end(trainer, pl_module)
+ # Get some meta data.
+ epoch_idx = trainer.current_epoch
+ percent = 100 * (epoch_idx + 1) / (self.trainer.max_epochs + 1)
+ metrics = self.get_metrics(trainer, pl_module)
+
+ # Current time.
+ time_cur = datetime.now().timestamp()
+ time_str = datetime.fromtimestamp(time_cur).strftime("%m-%d %H: %M:%S")
+ # Rest time.
+ time_rest_stamp = (time_cur - self.time_start_epoch) * (100 - percent) / percent
+ time_rest_str = self.timestamp_delta_to_str(time_rest_stamp)
+
+ if not self._should_log_batch(epoch_idx):
+ return
+
+ # Print the logs.
+ self.print(">> >> >> >>")
+ self.print(f"{self.title_prompt} [{self.stage.upper()}] Exp: {self.exp_name}")
+ self.print(f"{self.finish_prompt} Ep {epoch_idx} finished!")
+ self.print(f"{self.timer_prompt} Time: {time_str} | Rest: {time_rest_str}")
+ for k, v in metrics.items():
+ self.print(f"{self.metric_prompt} {k}: {v}")
+ self.print("<< << << <<")
+ self.print("") # Add a blank line.
diff --git a/genmo/callbacks/simple_ckpt_saver.py b/genmo/callbacks/simple_ckpt_saver.py
new file mode 100644
index 0000000000000000000000000000000000000000..f98a2d818041cce53078496c603c10c1bdaf5875
--- /dev/null
+++ b/genmo/callbacks/simple_ckpt_saver.py
@@ -0,0 +1,111 @@
+import os
+from copy import deepcopy
+from pathlib import Path
+from typing import Dict
+
+import pytorch_lightning as pl
+import torch
+from pytorch_lightning.callbacks.checkpoint import Checkpoint
+from pytorch_lightning.utilities import rank_zero_only
+from torch import Tensor
+
+from genmo.utils.pylogger import Log
+
+
+class SimpleCkptSaver(Checkpoint):
+ """
+ This callback runs at the end of each training epoch.
+ Check {every_n_epochs} and save at most {save_top_k} model if it is time.
+ """
+
+ def __init__(
+ self,
+ output_dir,
+ filename="e{epoch:03d}-s{step:06d}.ckpt",
+ save_top_k=1,
+ every_n_epochs=None,
+ every_n_steps=None,
+ save_last=None,
+ save_weights_only=False,
+ ):
+ super().__init__()
+ self.output_dir = Path(output_dir)
+ self.filename = filename
+ self.save_top_k = save_top_k
+ self.every_n_epochs = every_n_epochs
+ self.every_n_steps = every_n_steps
+ self.save_last = save_last
+ self.save_weights_only = save_weights_only
+
+ # Setup output dir
+ if rank_zero_only.rank == 0:
+ self.output_dir.mkdir(parents=True, exist_ok=True)
+ Log.info(f"[Simple Ckpt Saver]: Save to `{self.output_dir}'")
+
+ def _monitor_candidates(self, trainer: "pl.Trainer") -> Dict[str, Tensor]:
+ monitor_candidates = deepcopy(trainer.callback_metrics)
+ # cast to int if necessary because `self.log("epoch", 123)` will convert it to float. if it's not a tensor
+ # or does not exist we overwrite it as it's likely an error
+ epoch = monitor_candidates.get("epoch")
+ monitor_candidates["epoch"] = (
+ epoch.int()
+ if isinstance(epoch, Tensor)
+ else torch.tensor(trainer.current_epoch)
+ )
+ step = monitor_candidates.get("step")
+ monitor_candidates["step"] = (
+ step.int()
+ if isinstance(step, Tensor)
+ else torch.tensor(trainer.global_step)
+ )
+ return monitor_candidates
+
+ def _save_last_checkpoint(self, trainer, pl_module, monitor_candidates) -> None:
+ lastpath = self.output_dir / "last.ckpt"
+ checkpoint = trainer._checkpoint_connector.dump_checkpoint()
+ trainer.strategy.save_checkpoint(checkpoint, lastpath)
+ os.chmod(lastpath, 0o755)
+
+ @rank_zero_only
+ def on_train_epoch_end(self, trainer, pl_module):
+ """Save a checkpoint at the end of the training epoch."""
+ if (
+ self.every_n_epochs is not None
+ and (trainer.current_epoch + 1) % self.every_n_epochs == 0
+ ):
+ if self.save_top_k == 0:
+ return
+
+ # Save cureent checkpoint
+ filepath = self.output_dir / self.filename.format(
+ epoch=trainer.current_epoch, step=trainer.global_step
+ )
+ lastpath = self.output_dir / "last.ckpt"
+ checkpoint = trainer._checkpoint_connector.dump_checkpoint()
+ trainer.strategy.save_checkpoint(checkpoint, filepath)
+ trainer.strategy.save_checkpoint(checkpoint, lastpath)
+ os.chmod(filepath, 0o755)
+ os.chmod(lastpath, 0o755)
+
+ @rank_zero_only
+ def on_train_batch_end(
+ self, trainer, pl_module, outputs, batch, batch_idx: int
+ ) -> None:
+ """Save a checkpoint at the end of the training epoch."""
+ if (
+ self.every_n_steps is not None
+ and trainer.global_step % self.every_n_steps == 0
+ ):
+ if self.save_top_k == 0:
+ return
+
+ # Save cureent checkpoint
+ filepath = self.output_dir / "s{step:06d}.ckpt".format(
+ step=trainer.global_step
+ )
+ lastpath = self.output_dir / "last.ckpt"
+ checkpoint = trainer._checkpoint_connector.dump_checkpoint()
+ trainer.strategy.save_checkpoint(checkpoint, filepath)
+ trainer.strategy.save_checkpoint(checkpoint, lastpath)
+ os.chmod(filepath, 0o755)
+ os.chmod(lastpath, 0o755)
diff --git a/genmo/callbacks/train_speed_timer.py b/genmo/callbacks/train_speed_timer.py
new file mode 100644
index 0000000000000000000000000000000000000000..1f623590f0d5cfb92940856ecd4d0808fb4956c4
--- /dev/null
+++ b/genmo/callbacks/train_speed_timer.py
@@ -0,0 +1,79 @@
+from collections import deque
+from time import time
+
+import pytorch_lightning as pl
+from pytorch_lightning.utilities import rank_zero_only
+
+
+class TrainSpeedTimer(pl.Callback):
+ def __init__(self, N_avg=5):
+ """
+ This callback times the training speed (averge over recent 5 iterations)
+ 1. Data waiting time: this should be small, otherwise the data loading should be improved
+ 2. Single batch time: this is the time for one batch of training (excluding data waiting)
+ """
+ super().__init__()
+ self.last_batch_end = None
+ self.this_batch_start = None
+
+ # time queues for averaging
+ self.data_waiting_time_queue = deque(maxlen=N_avg)
+ self.single_batch_time_queue = deque(maxlen=N_avg)
+
+ @rank_zero_only
+ def on_train_batch_start(self, trainer, pl_module, batch, batch_idx):
+ """Count the time of data waiting"""
+ if self.last_batch_end is not None:
+ # This should be small, otherwise the data loading should be improved
+ data_waiting = time() - self.last_batch_end
+
+ # Average the time
+ self.data_waiting_time_queue.append(data_waiting)
+ average_time = sum(self.data_waiting_time_queue) / len(
+ self.data_waiting_time_queue
+ )
+
+ # Log to prog-bar
+ pl_module.log(
+ "train_timer/data_waiting",
+ average_time,
+ on_step=True,
+ on_epoch=False,
+ prog_bar=True,
+ logger=True,
+ )
+
+ self.this_batch_start = time()
+
+ @rank_zero_only
+ def on_train_batch_end(self, trainer, pl_module, outputs, batch, batch_idx):
+ # Effective training time elapsed (excluding data waiting)
+ single_batch = time() - self.this_batch_start
+
+ # Average the time
+ self.single_batch_time_queue.append(single_batch)
+ average_time = sum(self.single_batch_time_queue) / len(
+ self.single_batch_time_queue
+ )
+
+ # Log iter time
+ pl_module.log(
+ "train_timer/single_batch",
+ average_time,
+ on_step=True,
+ on_epoch=False,
+ prog_bar=False,
+ logger=True,
+ )
+
+ # Set timer for counting data waiting
+ self.last_batch_end = time()
+
+ @rank_zero_only
+ def on_train_epoch_end(self, trainer, pl_module):
+ # Reset the timer
+ self.last_batch_end = None
+ self.this_batch_start = None
+ # Clear the queue
+ self.data_waiting_time_queue.clear()
+ self.single_batch_time_queue.clear()
diff --git a/genmo/callbacks/vis/vis_music.py b/genmo/callbacks/vis/vis_music.py
new file mode 100644
index 0000000000000000000000000000000000000000..c440f29538ceb04a0789c6d0349cd9bf3354ae92
--- /dev/null
+++ b/genmo/callbacks/vis/vis_music.py
@@ -0,0 +1,199 @@
+import os
+
+import hydra
+import pytorch_lightning as pl
+import torch
+from moviepy.audio.AudioClip import AudioArrayClip
+
+from genmo.utils.vis_utils import visualize_smpl_scene_mp4
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+class VisMusic(pl.Callback):
+ def __init__(
+ self,
+ vis_every_n_val=1,
+ save_feats=False,
+ save_dir=None,
+ dataset_part_ind=-1,
+ endecoder=None,
+ trial_ind=0,
+ text_len=120,
+ ):
+ super().__init__()
+ self.vis_every_n_val = vis_every_n_val
+ self.num_val = 0
+ self.save_feats = save_feats
+ self.trial_ind = trial_ind
+ self.save_dir = save_dir
+ self.text_len = text_len
+ self.dataset_part_ind = dataset_part_ind
+ if endecoder is not None:
+ self.endecoder = hydra.utils.instantiate(endecoder).cuda()
+ # vid->result
+
+ # SMPL
+ self.smplx_model = {
+ "male": make_smplx("supermotion_smpl24"),
+ "female": make_smplx("supermotion_smpl24"),
+ "neutral": make_smplx("supermotion_smpl24"),
+ }
+ self.smplx = make_smplx("supermotion")
+ self.smplx2smpl = torch.load(
+ "third_party/GVHMR/inputs/checkpoints/body_models/smplx2smpl_sparse.pt"
+ )
+
+ self.J_regressor = torch.load(
+ "third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt"
+ )
+ self.faces_smpl = make_smplx("smpl").faces
+ self.faces_smplx = self.smplx_model["neutral"].faces
+
+ # The metrics are calculated similarly for val/test/predict
+ self.on_test_batch_end = self.on_validation_batch_end = (
+ self.on_predict_batch_end
+ )
+
+ # Only validation record the metrics with logger
+ self.on_test_epoch_end = self.on_validation_epoch_end = (
+ self.on_predict_epoch_end
+ )
+
+ # Only validation record the metrics with logger
+ self.on_test_epoch_start = self.on_validation_epoch_start = (
+ self.on_predict_epoch_start
+ )
+
+ # ================== Batch-based Computation ================== #
+ def on_predict_batch_end(
+ self, trainer, pl_module, outputs, batch, batch_idx, dataloader_idx=0
+ ):
+ """The behaviour is the same for val/test/predict"""
+ assert batch["B"] == 1
+ data_name = batch["meta"][0]["data_name"]
+ vid = batch["meta"][0]["vid"]
+
+ if data_name not in ["aist++"]:
+ return
+
+ # Move to cuda if not
+ for g in ["male", "female", "neutral"]:
+ self.smplx_model[g] = self.smplx_model[g].cuda()
+ self.smplx = self.smplx.cuda()
+ self.J_regressor = self.J_regressor.cuda()
+ self.smplx2smpl = self.smplx2smpl.cuda()
+
+ # print(batch_idx)
+ # os.makedirs('out/motions', exist_ok=True)
+ # torch.save(outputs, f'out/motions/outputs_{batch_idx}.pt')
+ # if 'multi_text_data' in batch['meta'][0]:
+ # multi_text_data = batch['meta'][0]['multi_text_data']
+ # for i in range(len(multi_text_data['vid'])):
+ # print(multi_text_data['vid'][i], multi_text_data['caption'][i])
+
+ seq_length = batch["length"][0].item()
+ gender = "neutral"
+ smpl_key = (
+ "2d_pred_smpl_params_global"
+ if data_name == "motion-x++2d"
+ else "pred_smpl_params_global"
+ )
+
+ # Groundtruth (world, cam)
+ if data_name == "aist++":
+ target_w_params = {k: v[0] for k, v in batch["smpl_params_w"].items()}
+ target_w_j3d = self.smplx_model[gender](**target_w_params)
+ offset = batch["smpl_params_w"]["transl"][0, :, None] - target_w_j3d[:, [0]]
+ target_w_j3d = target_w_j3d + offset
+ # target_w_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in target_w_output.vertices])
+ # target_w_j3d = torch.matmul(self.J_regressor, target_w_verts)
+
+ os.makedirs("out/music_pred", exist_ok=True)
+ torch.save(outputs, f"out/music_pred/{batch_idx}-{vid}.pt")
+ # 2. ay
+ pred_smpl_params_global = outputs[smpl_key]
+ pred_ay_j3d = self.smplx_model["neutral"](**pred_smpl_params_global)
+ pred_smplx = self.smplx(**pred_smpl_params_global)
+ pred_ay_verts = torch.stack(
+ [torch.matmul(self.smplx2smpl, v_) for v_ in pred_smplx.vertices]
+ )
+
+ # pred_ay_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in smpl_out.vertices])
+ # pred_ay_j3d = einsum(self.J_regressor, pred_ay_verts, "j v, l v i -> l j i")
+ music_array = batch["music_array"][0].cpu().numpy()
+ music_fps = int(batch["music_fps"][0])
+ music_clip = AudioArrayClip(music_array, music_fps)
+ if self.save_feats:
+ encoder_inputs = {
+ "smpl_params_w": {
+ k: v.unsqueeze(0) for k, v in outputs[smpl_key].items()
+ },
+ }
+ feats = self.endecoder.encode_humanml3d(encoder_inputs)
+ self.feats_arr.append(feats)
+ else:
+ pred_res = {
+ "batch_idx": batch_idx,
+ "vid": vid,
+ "pred_ay_verts": pred_ay_verts,
+ "target_w_j3d": target_w_j3d,
+ "music_array": music_array,
+ "music_fps": music_fps,
+ }
+ torch.save(pred_res, f"out/music_pred/{batch_idx}-{vid}.pt")
+ # Visualize
+ if trainer.global_rank == 0 and self.num_val % self.vis_every_n_val == 0:
+ wandb_dict = visualize_smpl_scene_mp4(
+ f"vis_text_global_{data_name}",
+ batch_idx,
+ vid,
+ pred_ay_verts,
+ target_w_j3d,
+ transform_mode="global",
+ faces_smpl=self.faces_smpl,
+ J_regressor=self.J_regressor,
+ audio_clip=music_clip,
+ )
+ self.wandb_html_dict.update(wandb_dict)
+ return
+
+ def on_predict_epoch_start(self, trainer, pl_module):
+ self.wandb_html_dict = {}
+ if self.save_feats:
+ self.feats_arr = []
+ self.text_arr = []
+ print(
+ f"start generating text-to-motion features which will be saved at {self.save_dir}\n"
+ )
+ print(
+ "#### saving a dump feature first to check the correctness of the path #### \n"
+ )
+ dump_feats = torch.randn(1, 1, 1, 1)
+ os.makedirs(self.save_dir, exist_ok=True)
+ torch.save(dump_feats, self.save_dir + "/dump.pt")
+ print(f"dump feature saved to {self.save_dir}/dump.pt\n")
+
+ # ================== Epoch Summary ================== #
+ def on_predict_epoch_end(self, trainer, pl_module):
+ self.num_val += 1
+ if len(self.wandb_html_dict) > 0:
+ pl_module.logger.log_metrics(self.wandb_html_dict)
+ if self.save_feats:
+ feats_arr = torch.cat(self.feats_arr, dim=0).cpu()
+ results = {
+ "feats": feats_arr,
+ "text": self.text_arr,
+ }
+ os.makedirs(self.save_dir, exist_ok=True)
+ if self.dataset_part_ind >= 0:
+ fname = (
+ self.save_dir
+ + f"/new_feats_part{self.dataset_part_ind}_len{self.text_len}_{self.trial_ind}.pt"
+ )
+ else:
+ fname = (
+ self.save_dir + f"/new_feats_len{self.text_len}_{self.trial_ind}.pt"
+ )
+ torch.save(results, fname)
+ os.chmod(fname, 0o755)
+ print(f"text-to-motion features saved to {fname}")
diff --git a/genmo/callbacks/vis/vis_speech.py b/genmo/callbacks/vis/vis_speech.py
new file mode 100644
index 0000000000000000000000000000000000000000..6a545ee9695ce6a6016032149b8b79e38399e8e1
--- /dev/null
+++ b/genmo/callbacks/vis/vis_speech.py
@@ -0,0 +1,189 @@
+import os
+
+import hydra
+import pytorch_lightning as pl
+import torch
+from moviepy.audio.AudioClip import AudioArrayClip
+
+from genmo.utils.vis_utils import visualize_smpl_scene_mp4
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+class VisSpeech(pl.Callback):
+ def __init__(
+ self,
+ vis_every_n_val=1,
+ save_feats=False,
+ save_dir=None,
+ dataset_part_ind=-1,
+ endecoder=None,
+ trial_ind=0,
+ text_len=120,
+ ):
+ super().__init__()
+ self.vis_every_n_val = vis_every_n_val
+ self.num_val = 0
+ self.save_feats = save_feats
+ self.trial_ind = trial_ind
+ self.save_dir = save_dir
+ self.text_len = text_len
+ self.dataset_part_ind = dataset_part_ind
+ if endecoder is not None:
+ self.endecoder = hydra.utils.instantiate(endecoder).cuda()
+ # vid->result
+
+ # SMPL
+ self.smplx_model = {
+ "male": make_smplx("supermotion_smpl24"),
+ "female": make_smplx("supermotion_smpl24"),
+ "neutral": make_smplx("supermotion_smpl24"),
+ }
+ self.smplx = make_smplx("supermotion")
+ self.smplx2smpl = torch.load(
+ "third_party/GVHMR/inputs/checkpoints/body_models/smplx2smpl_sparse.pt"
+ )
+
+ self.J_regressor = torch.load(
+ "third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt"
+ )
+ self.faces_smpl = make_smplx("smpl").faces
+ self.faces_smplx = self.smplx_model["neutral"].faces
+
+ # The metrics are calculated similarly for val/test/predict
+ self.on_test_batch_end = self.on_validation_batch_end = (
+ self.on_predict_batch_end
+ )
+
+ # Only validation record the metrics with logger
+ self.on_test_epoch_end = self.on_validation_epoch_end = (
+ self.on_predict_epoch_end
+ )
+
+ # Only validation record the metrics with logger
+ self.on_test_epoch_start = self.on_validation_epoch_start = (
+ self.on_predict_epoch_start
+ )
+
+ # ================== Batch-based Computation ================== #
+ def on_predict_batch_end(
+ self, trainer, pl_module, outputs, batch, batch_idx, dataloader_idx=0
+ ):
+ """The behaviour is the same for val/test/predict"""
+ assert batch["B"] == 1
+ data_name = batch["meta"][0]["data_name"]
+ vid = batch["meta"][0]["vid"]
+
+ if data_name not in ["beat2"]:
+ return
+
+ # Move to cuda if not
+ for g in ["male", "female", "neutral"]:
+ self.smplx_model[g] = self.smplx_model[g].cuda()
+ self.smplx = self.smplx.cuda()
+ self.J_regressor = self.J_regressor.cuda()
+ self.smplx2smpl = self.smplx2smpl.cuda()
+
+ # print(batch_idx)
+ # os.makedirs('out/motions', exist_ok=True)
+ # torch.save(outputs, f'out/motions/outputs_{batch_idx}.pt')
+ # if 'multi_text_data' in batch['meta'][0]:
+ # multi_text_data = batch['meta'][0]['multi_text_data']
+ # for i in range(len(multi_text_data['vid'])):
+ # print(multi_text_data['vid'][i], multi_text_data['caption'][i])
+
+ # seq_length = batch["length"][0].item()
+ gender = "neutral"
+ smpl_key = (
+ "2d_pred_smpl_params_global"
+ if data_name == "motion-x++2d"
+ else "pred_smpl_params_global"
+ )
+
+ # Groundtruth (world, cam)
+ if data_name == "beat2":
+ target_w_params = {k: v[0] for k, v in batch["smpl_params_w"].items()}
+ target_w_j3d = self.smplx_model[gender](**target_w_params)
+ offset = batch["smpl_params_w"]["transl"][0, :, None] - target_w_j3d[:, [0]]
+ target_w_j3d = target_w_j3d + offset
+ # target_w_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in target_w_output.vertices])
+ # target_w_j3d = torch.matmul(self.J_regressor, target_w_verts)
+
+ # 2. ay
+ pred_smpl_params_global = outputs[smpl_key]
+ # pred_ay_j3d = self.smplx_model["neutral"](**pred_smpl_params_global)
+ pred_smplx = self.smplx(**pred_smpl_params_global)
+ pred_ay_verts = torch.stack(
+ [torch.matmul(self.smplx2smpl, v_) for v_ in pred_smplx.vertices]
+ )
+
+ # pred_ay_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in smpl_out.vertices])
+ # pred_ay_j3d = einsum(self.J_regressor, pred_ay_verts, "j v, l v i -> l j i")
+ audio_array = batch["audio_array"][0].cpu().numpy()
+ audio_fps = 18000
+ audio_clip = AudioArrayClip(audio_array[:, None].repeat(2, 1), audio_fps)
+ if self.save_feats:
+ encoder_inputs = {
+ "smpl_params_w": {
+ k: v.unsqueeze(0) for k, v in outputs[smpl_key].items()
+ },
+ }
+ feats = self.endecoder.encode_humanml3d(encoder_inputs)
+ self.feats_arr.append(feats)
+ # self.text_arr.append(text)
+ else:
+ # Visualize
+ if trainer.global_rank == 0 and self.num_val % self.vis_every_n_val == 0:
+ wandb_dict = visualize_smpl_scene_mp4(
+ f"vis_text_global_{data_name}",
+ batch_idx,
+ vid,
+ pred_ay_verts,
+ target_w_j3d,
+ transform_mode="global",
+ faces_smpl=self.faces_smpl,
+ J_regressor=self.J_regressor,
+ audio_clip=audio_clip,
+ )
+ self.wandb_html_dict.update(wandb_dict)
+ return
+
+ def on_predict_epoch_start(self, trainer, pl_module):
+ self.wandb_html_dict = {}
+ if self.save_feats:
+ self.feats_arr = []
+ self.text_arr = []
+ print(
+ f"start generating text-to-motion features which will be saved at {self.save_dir}\n"
+ )
+ print(
+ "#### saving a dump feature first to check the correctness of the path #### \n"
+ )
+ dump_feats = torch.randn(1, 1, 1, 1)
+ os.makedirs(self.save_dir, exist_ok=True)
+ torch.save(dump_feats, self.save_dir + "/dump.pt")
+ print(f"dump feature saved to {self.save_dir}/dump.pt\n")
+
+ # ================== Epoch Summary ================== #
+ def on_predict_epoch_end(self, trainer, pl_module):
+ self.num_val += 1
+ if len(self.wandb_html_dict) > 0:
+ pl_module.logger.log_metrics(self.wandb_html_dict)
+ if self.save_feats:
+ feats_arr = torch.cat(self.feats_arr, dim=0).cpu()
+ results = {
+ "feats": feats_arr,
+ "text": self.text_arr,
+ }
+ os.makedirs(self.save_dir, exist_ok=True)
+ if self.dataset_part_ind >= 0:
+ fname = (
+ self.save_dir
+ + f"/new_feats_part{self.dataset_part_ind}_len{self.text_len}_{self.trial_ind}.pt"
+ )
+ else:
+ fname = (
+ self.save_dir + f"/new_feats_len{self.text_len}_{self.trial_ind}.pt"
+ )
+ torch.save(results, fname)
+ os.chmod(fname, 0o755)
+ print(f"text-to-motion features saved to {fname}")
diff --git a/genmo/callbacks/vis/vis_text.py b/genmo/callbacks/vis/vis_text.py
new file mode 100644
index 0000000000000000000000000000000000000000..8b36f856a97f4a0e4a6792dacbd2be41f1258425
--- /dev/null
+++ b/genmo/callbacks/vis/vis_text.py
@@ -0,0 +1,189 @@
+import os
+
+import hydra
+import pytorch_lightning as pl
+import torch
+from einops import einsum
+
+from genmo.utils.vis_utils import (
+ visualize_intermediate_smplmesh_scene_img,
+ visualize_smpl_scene,
+)
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+class VisText(pl.Callback):
+ def __init__(
+ self,
+ vis_every_n_val=1,
+ save_feats=False,
+ save_dir=None,
+ dataset_part_ind=-1,
+ endecoder=None,
+ trial_ind=0,
+ text_len=120,
+ ):
+ super().__init__()
+ self.vis_every_n_val = vis_every_n_val
+ self.num_val = 0
+ self.save_feats = save_feats
+ self.trial_ind = trial_ind
+ self.save_dir = save_dir
+ self.text_len = text_len
+ self.dataset_part_ind = dataset_part_ind
+ if endecoder is not None:
+ self.endecoder = hydra.utils.instantiate(endecoder).cuda()
+ # vid->result
+
+ # SMPL
+ self.smplx_model = {
+ "neutral": make_smplx("supermotion_smpl24"),
+ "male": make_smplx("supermotion_smpl24", gender="male"),
+ "female": make_smplx("supermotion_smpl24", gender="female"),
+ }
+ self.smplx = make_smplx("supermotion")
+ self.smplx2smpl = torch.load(
+ "third_party/GVHMR/inputs/checkpoints/body_models/smplx2smpl_sparse.pt"
+ )
+
+ self.J_regressor = torch.load(
+ "third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt"
+ )
+ self.faces_smpl = make_smplx("smpl").faces
+ self.faces_smplx = self.smplx_model["neutral"].faces
+
+ # The metrics are calculated similarly for val/test/predict
+ self.on_test_batch_end = self.on_validation_batch_end = (
+ self.on_predict_batch_end
+ )
+
+ # Only validation record the metrics with logger
+ self.on_test_epoch_end = self.on_validation_epoch_end = (
+ self.on_predict_epoch_end
+ )
+
+ # Only validation record the metrics with logger
+ self.on_test_epoch_start = self.on_validation_epoch_start = (
+ self.on_predict_epoch_start
+ )
+
+ # ================== Batch-based Computation ================== #
+ def on_predict_batch_end(
+ self, trainer, pl_module, outputs, batch, batch_idx, dataloader_idx=0
+ ):
+ """The behaviour is the same for val/test/predict"""
+ mode = batch["meta"][0].get("mode", None)
+ if mode != "default":
+ return
+ assert batch["B"] == 1
+ dataset_id = batch["meta"][0]["dataset_id"]
+ if dataset_id not in ["humanml3d", "motion-x++2d"]:
+ return
+
+ # Move to cuda if not
+ for g in ["male", "female", "neutral"]:
+ self.smplx_model[g] = self.smplx_model[g].cuda()
+ self.smplx = self.smplx.cuda()
+ self.J_regressor = self.J_regressor.cuda()
+ self.smplx2smpl = self.smplx2smpl.cuda()
+
+ # print(batch_idx)
+ # os.makedirs('out/motions', exist_ok=True)
+ # torch.save(outputs, f'out/motions/outputs_{batch_idx}.pt')
+ # if 'multi_text_data' in batch['meta'][0]:
+ # multi_text_data = batch['meta'][0]['multi_text_data']
+ # for i in range(len(multi_text_data['vid'])):
+ # print(multi_text_data['vid'][i], multi_text_data['caption'][i])
+
+ text = batch["caption"][0]
+ vid = text.replace(" ", "_").replace(".", "_").replace(",", "_")
+ # seq_length = batch["length"][0].item()
+ gender = "neutral"
+ smpl_key = (
+ "2d_pred_smpl_params_global"
+ if dataset_id == "motion-x++2d"
+ else "pred_smpl_params_global"
+ )
+
+ # Groundtruth (world, cam)
+ if dataset_id == "humanml3d":
+ target_w_params = {k: v[0] for k, v in batch["smpl_params_w"].items()}
+ target_w_j3d = self.smplx_model[gender](**target_w_params)
+ offset = batch["smpl_params_w"]["transl"][0, :, None] - target_w_j3d[:, [0]]
+ target_w_j3d = target_w_j3d + offset
+ # target_w_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in target_w_output.vertices])
+ # target_w_j3d = torch.matmul(self.J_regressor, target_w_verts)
+ else:
+ target_w_j3d = None
+
+ # 2. ay
+ pred_smpl_params_global = outputs[smpl_key]
+ pred_ay_j3d = self.smplx_model["neutral"](**pred_smpl_params_global)
+ floor_hight = pred_ay_j3d[:, :, 2].min()
+ pred_ay_j3d[:, :, 2] -= floor_hight
+ # pred_ay_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in smpl_out.vertices])
+ # pred_ay_j3d = einsum(self.J_regressor, pred_ay_verts, "j v, l v i -> l j i")
+
+ if self.save_feats:
+ encoder_inputs = {
+ "smpl_params_w": {
+ k: v.unsqueeze(0) for k, v in outputs[smpl_key].items()
+ },
+ }
+ feats = self.endecoder.encode_humanml3d(encoder_inputs)
+ self.feats_arr.append(feats)
+ self.text_arr.append(text)
+ else:
+ # Visualize
+ if trainer.global_rank == 0 and self.num_val % self.vis_every_n_val == 0:
+ wandb_dict = visualize_smpl_scene(
+ f"vis_text_global_{dataset_id}",
+ batch_idx,
+ vid,
+ pred_ay_j3d,
+ target_w_j3d,
+ transform_mode="global",
+ )
+ self.wandb_html_dict.update(wandb_dict)
+ return
+
+ def on_predict_epoch_start(self, trainer, pl_module):
+ self.wandb_html_dict = {}
+ if self.save_feats:
+ self.feats_arr = []
+ self.text_arr = []
+ print(
+ f"start generating text-to-motion features which will be saved at {self.save_dir}\n"
+ )
+ print(
+ "#### saving a dump feature first to check the correctness of the path #### \n"
+ )
+ dump_feats = torch.randn(1, 1, 1, 1)
+ os.makedirs(self.save_dir, exist_ok=True)
+ torch.save(dump_feats, self.save_dir + "/dump.pt")
+ print(f"dump feature saved to {self.save_dir}/dump.pt\n")
+
+ # ================== Epoch Summary ================== #
+ def on_predict_epoch_end(self, trainer, pl_module):
+ self.num_val += 1
+ if len(self.wandb_html_dict) > 0:
+ pl_module.logger.log_metrics(self.wandb_html_dict)
+ if self.save_feats:
+ feats_arr = torch.cat(self.feats_arr, dim=0).cpu()
+ results = {
+ "feats": feats_arr,
+ "text": self.text_arr,
+ }
+ os.makedirs(self.save_dir, exist_ok=True)
+ if self.dataset_part_ind >= 0:
+ fname = (
+ self.save_dir
+ + f"/new_feats_part{self.dataset_part_ind}_len{self.text_len}_{self.trial_ind}.pt"
+ )
+ else:
+ fname = (
+ self.save_dir + f"/new_feats_len{self.text_len}_{self.trial_ind}.pt"
+ )
+ torch.save(results, fname)
+ os.chmod(fname, 0o755)
+ print(f"text-to-motion features saved to {fname}")
diff --git a/genmo/callbacks/vis/vis_unity_val.py b/genmo/callbacks/vis/vis_unity_val.py
new file mode 100644
index 0000000000000000000000000000000000000000..001b2b63b6843141b497c6ac53d52f457bba8da3
--- /dev/null
+++ b/genmo/callbacks/vis/vis_unity_val.py
@@ -0,0 +1,840 @@
+from __future__ import annotations
+
+from pathlib import Path
+
+import cv2
+import numpy as np
+import pytorch_lightning as pl
+import torch
+import torch.nn.functional as F
+
+from genmo.utils.geo_transform import apply_T_on_points, compute_T_ayfz2ay
+from genmo.utils.pylogger import Log
+from genmo.utils.rotation_conversions import (
+ axis_angle_to_matrix,
+ matrix_to_axis_angle,
+ matrix_to_euler_angles,
+)
+from genmo.utils.video_io_utils import get_writer
+from genmo.utils.vis.renderer import (
+ Renderer,
+ get_global_cameras_static,
+ get_ground_params_from_points,
+)
+from genmo.pipeline.genmo_pipeline import get_smpl_params_w_Rt_v2
+from third_party.GVHMR.hmr4d.model.gvhmr.utils.postprocess import pp_static_joint, process_ik
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import get_local_transl_vel
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import create_camera_sensor
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+class VisUnityVal(pl.Callback):
+ def __init__(
+ self,
+ enabled: bool = False,
+ every_n_epochs: int = 1,
+ num_batches: int = 1,
+ num_frames: int = 30,
+ render_incam: bool = True,
+ render_global: bool = True,
+ pad_incam_canvas: bool = False,
+ incam_background: str = "images",
+ batch_select: str = "first",
+ batch_select_seed: int = 123,
+ use_gt_betas_for_pred: bool = True,
+ global_root_relative: bool = False,
+ postprocess_global: bool = True,
+ crf: int = 23,
+ save_dir: str = "vis",
+ pred_color=(176, 100, 244),
+ gt_color=(0, 255, 0),
+ ):
+ super().__init__()
+ self.enabled = enabled
+ self.every_n_epochs = every_n_epochs
+ self.num_batches = num_batches
+ self.num_frames = num_frames
+ self.render_incam = render_incam
+ self.render_global = render_global
+ self.pad_incam_canvas = bool(pad_incam_canvas)
+ self.incam_background = str(incam_background or "images").strip().lower()
+ self.batch_select = str(batch_select or "first").strip().lower()
+ self.batch_select_seed = int(batch_select_seed)
+ self.use_gt_betas_for_pred = use_gt_betas_for_pred
+ self.global_root_relative = global_root_relative
+ self.postprocess_global = postprocess_global
+ self.crf = crf
+ self.save_dir = save_dir
+ self.pred_color = pred_color
+ self.gt_color = gt_color
+
+ self._smplx = None
+ self._smplx2smpl = None
+ self._faces = None
+ self._J_regressor = None
+ self._selected_batch_idxs_by_loader = {}
+ self._seen_batch_count_by_loader = {}
+
+ def on_validation_epoch_start(self, trainer, pl_module):
+ self._selected_batch_idxs_by_loader = {}
+ self._seen_batch_count_by_loader = {}
+ if not self.enabled:
+ return
+ if trainer.global_rank != 0:
+ return
+ if self.every_n_epochs is not None and (trainer.current_epoch % self.every_n_epochs) != 0:
+ return
+
+ # Try to deterministically select which batches to render for each val dataloader.
+ try:
+ num_val_batches = getattr(trainer, "num_val_batches", None)
+ if num_val_batches is None:
+ return
+ if isinstance(num_val_batches, int):
+ num_val_batches = [num_val_batches]
+ for dl_idx, n in enumerate(list(num_val_batches)):
+ n = int(n)
+ if n <= 0:
+ self._selected_batch_idxs_by_loader[dl_idx] = set()
+ continue
+ k = min(int(self.num_batches), n)
+ if self.batch_select == "random":
+ rng = np.random.default_rng(int(self.batch_select_seed) + int(trainer.current_epoch) * 1000 + int(dl_idx))
+ chosen = rng.choice(np.arange(n, dtype=np.int64), size=k, replace=False)
+ self._selected_batch_idxs_by_loader[dl_idx] = set(int(x) for x in chosen.tolist())
+ else:
+ self._selected_batch_idxs_by_loader[dl_idx] = set(range(k))
+ except Exception:
+ # Fallback: keep legacy behavior (first N batches).
+ self._selected_batch_idxs_by_loader = {}
+
+ def _lazy_init_models(self, device: torch.device):
+ if self._smplx is None:
+ self._smplx = make_smplx("supermotion").to(device)
+
+ smplx2smpl_path = (
+ Path("third_party/GVHMR/hmr4d/utils/body_model/smplx2smpl_sparse.pt")
+ )
+ if smplx2smpl_path.exists():
+ self._smplx2smpl = torch.load(smplx2smpl_path).to(device)
+ self._faces = make_smplx("smpl").faces
+ else:
+ self._faces = self._smplx.faces
+
+ j_reg_path = Path(
+ "third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt"
+ )
+ if j_reg_path.exists():
+ self._J_regressor = torch.load(j_reg_path).to(device)
+
+ def _select_time(self, tensor: torch.Tensor, frame_idxs) -> torch.Tensor:
+ if not isinstance(tensor, torch.Tensor):
+ raise TypeError(f"Expected tensor, got {type(tensor)}")
+ frame_idxs_t = torch.as_tensor(frame_idxs, dtype=torch.long, device=tensor.device)
+
+ # Common shapes:
+ # - (L, D) / (L, 3, 3) ...
+ # - (B, L, D) with B=1
+ # - (B, D) with B=1 (e.g., betas)
+ if tensor.ndim >= 3 and tensor.shape[0] == 1:
+ tensor = tensor[0]
+ if tensor.ndim == 2 and tensor.shape[0] == 1:
+ tensor = tensor.repeat(len(frame_idxs), 1)
+ if tensor.ndim >= 2 and tensor.shape[0] >= int(frame_idxs_t.max()) + 1:
+ return tensor.index_select(0, frame_idxs_t)
+ # Fallback: if it's static (no time dim), repeat across frames.
+ if tensor.ndim == 1:
+ return tensor.unsqueeze(0).repeat(len(frame_idxs), 1)
+ return tensor
+
+ def _verts_from_params(self, params: dict, frame_idxs, device: torch.device):
+ params = {k: self._select_time(v.to(device), frame_idxs) for k, v in params.items()}
+ out = self._smplx(**params)
+ verts = out.vertices if hasattr(out, "vertices") else out[0].vertices
+ if self._smplx2smpl is not None:
+ verts = torch.stack([torch.matmul(self._smplx2smpl, v) for v in verts])
+ return verts
+
+ def _safe_vid(self, vid: str) -> str:
+ return vid.replace("/", "_").replace(" ", "_")
+
+ def on_validation_batch_end(
+ self, trainer, pl_module, outputs, batch, batch_idx, dataloader_idx: int = 0
+ ):
+ if not self.enabled:
+ return
+ if trainer.global_rank != 0:
+ return
+ if self.every_n_epochs is not None and (trainer.current_epoch % self.every_n_epochs) != 0:
+ return
+ dl_i = int(dataloader_idx)
+ local_idx = int(self._seen_batch_count_by_loader.get(dl_i, 0))
+ self._seen_batch_count_by_loader[dl_i] = local_idx + 1
+
+ selected = self._selected_batch_idxs_by_loader.get(dl_i, None)
+ if selected is None:
+ # Fallback: legacy behavior.
+ if batch_idx >= self.num_batches:
+ return
+ else:
+ # Use loader-local index (CombinedLoader may provide a global `batch_idx`).
+ if local_idx not in selected:
+ return
+
+ if outputs is None or "pred_smpl_params_incam" not in outputs:
+ Log.warning("[VisUnityVal] Missing `pred_smpl_params_incam` in outputs; skipping.")
+ return
+
+ meta_render = None
+ if "meta_render" in batch and isinstance(batch["meta_render"], list) and batch["meta_render"]:
+ meta_render = batch["meta_render"][0]
+ # NOTE: Do not depend on image/video I/O for validation visualization; render on black.
+
+ vid = batch["meta"][0].get("vid", f"b{batch_idx:03d}")
+ vid = self._safe_vid(str(vid))
+
+ # Pick frames to render (within the already-sliced/padded window).
+ L = int(batch["K_fullimg"].shape[1]) if "K_fullimg" in batch else 0
+ if L <= 0:
+ return
+ num_frames = min(self.num_frames, L)
+ frame_idxs = np.linspace(0, L - 1, num_frames).round().astype(int)
+
+ device = pl_module.device
+ self._lazy_init_models(device)
+
+ # Render on black; infer output size from principal point (usually near W/2, H/2).
+ K = batch["K_fullimg"][0, 0].to(device)
+ try:
+ cx = float(K[0, 2].detach().cpu().item())
+ cy = float(K[1, 2].detach().cpu().item())
+ width = max(64, int(round(cx * 2.0)))
+ height = max(64, int(round(cy * 2.0)))
+ except Exception:
+ width, height = 1280, 720
+
+ # Params
+ gt_params_c = {k: v[0] for k, v in batch["smpl_params_c"].items()}
+ pred_params_incam = outputs["pred_smpl_params_incam"]
+ if self.use_gt_betas_for_pred and "betas" in gt_params_c:
+ pred_params_incam = dict(pred_params_incam)
+ pred_params_incam["betas"] = gt_params_c["betas"]
+
+ # Build GT global params in the same convention as `pred_smpl_params_global` (GV0 / inference frame).
+ # This prevents misleading "global breaks" visuals when the dataset `smpl_params_w` convention differs.
+ gt_params_w_infer = None
+ try:
+ if (
+ "smpl_params_w" in batch
+ and "R_c2gv" in batch
+ and "cam_angvel" in batch
+ and "smpl_params_c" in batch
+ ):
+ smpl_w = batch["smpl_params_w"] # (B, L, ...)
+ R_c2gv = batch["R_c2gv"] # (B, L, 3, 3)
+ cam_angvel = batch["cam_angvel"] # (B, L, 6)
+ smpl_c = batch["smpl_params_c"]
+
+ go_c_gt = smpl_c["global_orient"] # (B, L, 3)
+ R_c = axis_angle_to_matrix(go_c_gt) # (B, L, 3, 3)
+ go_gv_gt = matrix_to_axis_angle(R_c2gv @ R_c) # (B, L, 3)
+
+ local_tv_gt = get_local_transl_vel(
+ smpl_w["transl"], smpl_w["global_orient"]
+ ) # (B, L, 3)
+ root_gt = get_smpl_params_w_Rt_v2(
+ global_orient_gv=go_gv_gt,
+ local_transl_vel=local_tv_gt,
+ global_orient_c=go_c_gt,
+ cam_angvel=cam_angvel,
+ )
+
+ gt_params_w_infer = {
+ "body_pose": smpl_c["body_pose"],
+ "betas": smpl_c["betas"],
+ "global_orient": root_gt["global_orient"],
+ "transl": root_gt["transl"],
+ }
+ except Exception as e:
+ Log.warning(
+ f"[VisUnityVal] Failed to compute GT global params in inference frame; "
+ f"falling back to raw `smpl_params_w`. err={e}"
+ )
+
+ with torch.inference_mode():
+ pred_verts_incam = self._verts_from_params(pred_params_incam, frame_idxs, device).float()
+ gt_verts_incam = self._verts_from_params(gt_params_c, frame_idxs, device).float()
+
+ orig_width, orig_height = int(width), int(height)
+
+ # Optional: pad the rendering canvas so off-screen GT/pred motion is visible.
+ # Disabled by default because it changes output video resolution.
+ pad_left = pad_right = pad_top = pad_bottom = 0
+ K_pad = K
+ if self.pad_incam_canvas:
+ try:
+ K_pad = K.clone()
+ # Compute 2D extents from both GT+pred verts for the rendered frames.
+ # Use a z-threshold for stability (same as reproj thresholds elsewhere).
+ z_thr = 0.3
+
+ def _uv_extents(verts_f: torch.Tensor, K_f: torch.Tensor):
+ v = verts_f
+ m = v[:, 2] > z_thr
+ if not bool(m.any().item()):
+ return None
+ v = v[m]
+ uv = (v[:, :2] / v[:, 2:3].clamp_min(1e-6)) * K_f.diag()[:2] + K_f[:2, 2]
+ umin = float(uv[:, 0].min().item())
+ umax = float(uv[:, 0].max().item())
+ vmin = float(uv[:, 1].min().item())
+ vmax = float(uv[:, 1].max().item())
+ return umin, vmin, umax, vmax
+
+ umins, vmins, umaxs, vmaxs = [], [], [], []
+ for i, fi in enumerate(frame_idxs):
+ try:
+ K_fi = batch["K_fullimg"][0, int(fi)].to(device)
+ except Exception:
+ K_fi = K
+ for verts in (gt_verts_incam[i], pred_verts_incam[i]):
+ ext = _uv_extents(verts, K_fi)
+ if ext is None:
+ continue
+ umin, vmin, umax, vmax = ext
+ umins.append(umin)
+ vmins.append(vmin)
+ umaxs.append(umax)
+ vmaxs.append(vmax)
+
+ if umins and vmins and umaxs and vmaxs:
+ margin = 16
+ min_u = min(umins)
+ min_v = min(vmins)
+ max_u = max(umaxs)
+ max_v = max(vmaxs)
+ pad_left = max(0, int(np.ceil(-(min_u - margin))))
+ pad_top = max(0, int(np.ceil(-(min_v - margin))))
+ pad_right = max(0, int(np.ceil((max_u + margin) - (width - 1))))
+ pad_bottom = max(0, int(np.ceil((max_v + margin) - (height - 1))))
+
+ if pad_left or pad_top or pad_right or pad_bottom:
+ K_pad[0, 2] = K_pad[0, 2] + float(pad_left)
+ K_pad[1, 2] = K_pad[1, 2] + float(pad_top)
+ width = int(width + pad_left + pad_right)
+ height = int(height + pad_top + pad_bottom)
+
+ # Many video encoders require even dimensions.
+ if width % 2 == 1:
+ width += 1
+ pad_right += 1
+ if height % 2 == 1:
+ height += 1
+ pad_bottom += 1
+ else:
+ K_pad = K
+ except Exception:
+ K_pad = K
+
+ renderer_incam = Renderer(width, height, device=device, faces=self._faces, K=K_pad)
+ # Make the overlay look "flat colored" (no Phong shading).
+ try:
+ from pytorch3d.renderer import AmbientLights
+
+ renderer_incam.lights = AmbientLights(device=device)
+ renderer_incam.create_renderer()
+ except Exception:
+ pass
+
+ # Numeric incam sanity: report both (a) original canvas (dataset "out of frame") and
+ # (b) padded canvas (what the visualization will render).
+ try:
+ def _proj_stats(
+ pts_c: torch.Tensor, K_3x3: torch.Tensor, w: int, h: int
+ ):
+ # pts_c: (V, 3) in camera coords
+ z = pts_c[:, 2]
+ valid_z = z > 0.3
+ pts = pts_c[valid_z]
+ if pts.numel() == 0:
+ return None
+ xy = pts[:, :2] / pts[:, 2:3].clamp_min(1e-6)
+ uv = (xy * K_3x3.diag()[:2]) + K_3x3[:2, 2]
+ u = uv[:, 0]
+ v = uv[:, 1]
+ umin, umax = float(u.min().item()), float(u.max().item())
+ vmin, vmax = float(v.min().item()), float(v.max().item())
+ oof = ((u < 0) | (u >= w) | (v < 0) | (v >= h)).float().mean()
+ return (umin, vmin, umax, vmax, float(oof.item()))
+
+ fi0 = 0
+ fil = int(len(frame_idxs) - 1)
+ K0 = batch["K_fullimg"][0, int(frame_idxs[fi0])].to(device)
+ Kl = batch["K_fullimg"][0, int(frame_idxs[fil])].to(device)
+ gt0 = _proj_stats(gt_verts_incam[fi0], K0, orig_width, orig_height)
+ pr0 = _proj_stats(pred_verts_incam[fi0], K0, orig_width, orig_height)
+ gtl = _proj_stats(gt_verts_incam[fil], Kl, orig_width, orig_height)
+ prl = _proj_stats(pred_verts_incam[fil], Kl, orig_width, orig_height)
+ Log.info(f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} incam_proj_bbox_oof f0 gt={gt0} pred={pr0}")
+ Log.info(f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} incam_proj_bbox_oof fl gt={gtl} pred={prl}")
+ if pad_left or pad_top or pad_right or pad_bottom:
+ try:
+ K0p = K0.clone()
+ Klp = Kl.clone()
+ K0p[0, 2] = K0p[0, 2] + float(pad_left)
+ K0p[1, 2] = K0p[1, 2] + float(pad_top)
+ Klp[0, 2] = Klp[0, 2] + float(pad_left)
+ Klp[1, 2] = Klp[1, 2] + float(pad_top)
+ gt0p = _proj_stats(gt_verts_incam[fi0], K0p, width, height)
+ pr0p = _proj_stats(pred_verts_incam[fi0], K0p, width, height)
+ gtlp = _proj_stats(gt_verts_incam[fil], Klp, width, height)
+ prlp = _proj_stats(pred_verts_incam[fil], Klp, width, height)
+ Log.info(
+ f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} incam_proj_bbox_oof_padded "
+ f"f0 gt={gt0p} pred={pr0p}"
+ )
+ Log.info(
+ f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} incam_proj_bbox_oof_padded "
+ f"fl gt={gtlp} pred={prlp}"
+ )
+ except Exception:
+ pass
+ Log.info(
+ f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} incam_canvas_pad "
+ f"l/t/r/b={pad_left}/{pad_top}/{pad_right}/{pad_bottom} new_size={width}x{height}"
+ )
+ except Exception as e:
+ Log.warning(f"[VisUnityVal] Failed incam projection sanity. err={e}")
+
+ # Numeric incam sanity: GT vs pred camera translation (in meters).
+ try:
+ if (
+ isinstance(pred_params_incam, dict)
+ and isinstance(gt_params_c, dict)
+ and "transl" in pred_params_incam
+ and "transl" in gt_params_c
+ ):
+ gt_tr_c = self._select_time(gt_params_c["transl"].to(device), frame_idxs).float()
+ pr_tr_c = self._select_time(pred_params_incam["transl"].to(device), frame_idxs).float()
+ tr_err = (pr_tr_c - gt_tr_c).norm(dim=-1)
+ Log.info(
+ f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} incam_transl_err_m "
+ f"mean={float(tr_err.mean().item()):.3f} max={float(tr_err.max().item()):.3f} "
+ f"gt_z_mean={float(gt_tr_c[:,2].mean().item()):.3f} pred_z_mean={float(pr_tr_c[:,2].mean().item()):.3f}"
+ )
+ # Oscillation debug: frame-to-frame variability of the error (pred - gt).
+ try:
+ d_tr = (pr_tr_c - gt_tr_c)
+ d_tr_std = d_tr.std(dim=0)
+ d_tr_norm_std = float(d_tr.norm(dim=-1).std().item())
+ Log.info(
+ f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} incam_delta_transl_std_m "
+ f"xyz=({float(d_tr_std[0].item()):.4f},{float(d_tr_std[1].item()):.4f},{float(d_tr_std[2].item()):.4f}) "
+ f"norm_std={d_tr_norm_std:.4f}"
+ )
+ except Exception:
+ pass
+ except Exception as e:
+ Log.warning(f"[VisUnityVal] Failed incam transl sanity. err={e}")
+
+ out_dir = Path(self.save_dir)
+ out_dir.mkdir(parents=True, exist_ok=True)
+
+ if self.render_incam:
+ # Background frames: prefer extracted images from preprocessing (derived from the original mp4).
+ # `UnityDataset` puts these paths under `meta_render.img_paths` when available.
+ img_paths = None
+ frame_ids = None
+ video_path = None
+ try:
+ if (
+ isinstance(batch.get("meta_render", None), list)
+ and len(batch["meta_render"]) > 0
+ and isinstance(batch["meta_render"][0], dict)
+ ):
+ img_paths = batch["meta_render"][0].get("img_paths", None)
+ frame_ids = batch["meta_render"][0].get("frame_ids", None)
+ video_path = batch["meta_render"][0].get("video_path", None)
+ except Exception:
+ img_paths = None
+ frame_ids = None
+ video_path = None
+
+ incam_path = out_dir / f"e{trainer.current_epoch:03d}_{vid}_incam.mp4"
+ writer = get_writer(str(incam_path), fps=30, crf=int(self.crf))
+ try:
+ cap = None
+ if self.incam_background == "video" and isinstance(video_path, str) and video_path:
+ cap = cv2.VideoCapture(video_path)
+ for i, fi in enumerate(frame_idxs):
+ # Unity intrinsics can change slightly per-frame; keep rendering aligned.
+ try:
+ K_fi = batch["K_fullimg"][0, int(fi)].to(device)
+ if pad_left or pad_top:
+ K_fi = K_fi.clone()
+ K_fi[0, 2] = K_fi[0, 2] + float(pad_left)
+ K_fi[1, 2] = K_fi[1, 2] + float(pad_top)
+ renderer_incam.set_intrinsic(K_fi)
+ except Exception:
+ pass
+ frame = np.zeros((height, width, 3), dtype=np.uint8) # RGB black
+
+ # Preferred: background from original mp4 (raw dataset).
+ # Fallback: background from extracted images (if present).
+ if self.incam_background != "black":
+ try:
+ bg_rgb = None
+
+ if cap is not None and cap.isOpened():
+ f_id = int(fi)
+ if isinstance(frame_ids, list) and len(frame_ids) > int(fi):
+ try:
+ f_id = int(frame_ids[int(fi)])
+ except Exception:
+ f_id = int(fi)
+ cap.set(cv2.CAP_PROP_POS_FRAMES, float(f_id))
+ ok, bg_bgr = cap.read()
+ if ok and bg_bgr is not None:
+ bg_rgb = bg_bgr[:, :, ::-1]
+
+ if bg_rgb is None and img_paths is not None:
+ p = img_paths[int(fi)]
+ if isinstance(p, str) and p:
+ bg_bgr = cv2.imread(p, cv2.IMREAD_COLOR)
+ if bg_bgr is not None:
+ bg_rgb = bg_bgr[:, :, ::-1]
+
+ if bg_rgb is not None:
+ if bg_rgb.shape[0] != orig_height or bg_rgb.shape[1] != orig_width:
+ bg_rgb = cv2.resize(
+ bg_rgb,
+ (orig_width, orig_height),
+ interpolation=cv2.INTER_LINEAR,
+ )
+ if (pad_left or pad_top or pad_right or pad_bottom) and (
+ bg_rgb.shape[0] == orig_height and bg_rgb.shape[1] == orig_width
+ ):
+ frame[
+ pad_top : pad_top + orig_height,
+ pad_left : pad_left + orig_width,
+ :,
+ ] = bg_rgb
+ elif bg_rgb.shape[0] == height and bg_rgb.shape[1] == width:
+ frame = bg_rgb
+ else:
+ frame = cv2.resize(
+ bg_rgb,
+ (width, height),
+ interpolation=cv2.INTER_LINEAR,
+ )
+ except Exception:
+ pass
+ img = renderer_incam.render_mesh(gt_verts_incam[i], frame, colors=self.gt_color)
+ img = renderer_incam.render_mesh(pred_verts_incam[i], img, colors=self.pred_color)
+ writer.write_frame(img.astype(np.uint8))
+ finally:
+ try:
+ if cap is not None:
+ cap.release()
+ except Exception:
+ pass
+ writer.close()
+
+ if self.render_global:
+ # Follow `scripts/demo/infer_video.py`:
+ # - Use GT world params (`smpl_params_w`) and model-produced global params (`pred_smpl_params_global`)
+ # - Ground + face-Z align using GT joints from the first frame
+
+ if "smpl_params_w" not in batch:
+ Log.warning("[VisUnityVal] Missing `smpl_params_w`; skipping global rendering.")
+ return
+ if "pred_smpl_params_global" not in outputs:
+ Log.warning("[VisUnityVal] Missing `pred_smpl_params_global`; skipping global rendering.")
+ return
+
+ if gt_params_w_infer is not None:
+ gt_params_w = {k: v[0] for k, v in gt_params_w_infer.items()}
+ else:
+ gt_params_w = {k: v[0] for k, v in batch["smpl_params_w"].items()}
+ with torch.inference_mode():
+ gt_verts_world = self._verts_from_params(gt_params_w, frame_idxs, device).float()
+ pred_params_global = outputs["pred_smpl_params_global"]
+ if self.use_gt_betas_for_pred and "betas" in gt_params_w:
+ pred_params_global = dict(pred_params_global)
+ pred_params_global["betas"] = gt_params_w["betas"]
+
+ # In `GENMO.validation_step`, postprocess is disabled by default (only enabled in test).
+ # Without postprocess, `pred_smpl_params_global.transl` can drift wildly.
+ if (
+ self.postprocess_global
+ and (
+ "static_conf_logits" in outputs
+ or (
+ "model_output" in outputs
+ and isinstance(outputs["model_output"], dict)
+ and "static_conf_logits" in outputs["model_output"]
+ )
+ )
+ and isinstance(getattr(pl_module, "pipeline", None), object)
+ ):
+ try:
+ # `pp_static_joint` / `process_ik` expect (B, L, ...), but this callback's
+ # `pred_params_global` is typically unbatched (L, ...). Meanwhile
+ # `outputs["static_conf_logits"]` is batched (B, L, J). Normalize shapes here.
+ b0 = 0
+ L = None
+ for k in ("transl", "body_pose", "global_orient", "betas"):
+ v = pred_params_global.get(k, None)
+ if isinstance(v, torch.Tensor) and v.ndim >= 2:
+ L = int(v.shape[0])
+ break
+ if L is None:
+ raise RuntimeError("Cannot infer sequence length L for postprocess inputs.")
+
+ def _to_batched_seq(x: torch.Tensor, L: int) -> torch.Tensor:
+ if x.ndim == 0:
+ return x[None, None]
+ # Already batched: (B, L, ...)
+ if x.ndim >= 3 and int(x.shape[1]) == L:
+ return x[b0 : b0 + 1]
+ # Unbatched sequence: (L, ...)
+ if x.ndim >= 2 and int(x.shape[0]) == L:
+ return x[None]
+ # Per-seq constants: (D,) (e.g., betas10)
+ if x.ndim == 1 and int(x.shape[0]) in (10, 16):
+ return x[None, None, :].expand(1, L, -1)
+ raise RuntimeError(f"Unexpected tensor shape for postprocess: {tuple(x.shape)} (L={L})")
+
+ pp_out = dict(outputs)
+ pp_out["pred_smpl_params_global"] = {
+ k: _to_batched_seq(v, L) if isinstance(v, torch.Tensor) else v
+ for k, v in pred_params_global.items()
+ }
+ static_conf_logits = outputs.get("static_conf_logits", None)
+ if static_conf_logits is None and isinstance(outputs.get("model_output", None), dict):
+ static_conf_logits = outputs["model_output"].get("static_conf_logits", None)
+ if not isinstance(static_conf_logits, torch.Tensor):
+ raise RuntimeError("static_conf_logits missing or not a Tensor.")
+ # static_conf_logits is (B, L, J)
+ pp_out["static_conf_logits"] = static_conf_logits[b0 : b0 + 1]
+
+ # Keep postprocess behavior consistent with training/inference settings.
+ try:
+ if not bool(getattr(pl_module.pipeline, "args", {}).get("pp_ground", True)):
+ pp_out["disable_pp_ground"] = True
+ except Exception:
+ pass
+ pp_out["pred_smpl_params_global"]["transl"] = pp_static_joint(
+ pp_out, pl_module.pipeline.endecoder
+ )
+ body_pose = process_ik(pp_out, pl_module.pipeline.endecoder)
+ pred_params_global = dict(pred_params_global)
+ pred_params_global["transl"] = pp_out["pred_smpl_params_global"][
+ "transl"
+ ][0]
+ pred_params_global["body_pose"] = body_pose[0, :, :]
+ except Exception as e:
+ try:
+ shp = {}
+ if "static_conf_logits" in outputs and isinstance(outputs["static_conf_logits"], torch.Tensor):
+ shp["static_conf_logits"] = tuple(outputs["static_conf_logits"].shape)
+ if "pred_smpl_params_global" in outputs and isinstance(outputs["pred_smpl_params_global"], dict):
+ for k in ("transl", "global_orient", "body_pose", "betas"):
+ v = outputs["pred_smpl_params_global"].get(k, None)
+ if isinstance(v, torch.Tensor):
+ shp[f"pred_smpl_params_global.{k}"] = tuple(v.shape)
+ if isinstance(pred_params_global, dict):
+ for k in ("transl", "global_orient", "body_pose", "betas"):
+ v = pred_params_global.get(k, None)
+ if isinstance(v, torch.Tensor):
+ shp[f"render_pred_params_global.{k}"] = tuple(v.shape)
+ shp_str = str(shp)
+ except Exception:
+ shp_str = ""
+ Log.warning(
+ f"[VisUnityVal] Global postprocess failed; using raw outputs. err={e} shapes={shp_str}"
+ )
+ pred_verts_world = self._verts_from_params(
+ pred_params_global, frame_idxs, device
+ ).float()
+
+ # Numeric global sanity: transl drift (GT vs pred) in the same rendered frame.
+ try:
+ if isinstance(pred_params_global, dict) and "transl" in pred_params_global and "transl" in gt_params_w:
+ gt_tr = self._select_time(gt_params_w["transl"].to(device), frame_idxs).float()
+ pr_tr = self._select_time(pred_params_global["transl"].to(device), frame_idxs).float()
+ tr_err = (pr_tr - gt_tr).norm(dim=-1) # (F,)
+ Log.info(
+ f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} global_transl_err_m "
+ f"mean={float(tr_err.mean().item()):.3f} max={float(tr_err.max().item()):.3f}"
+ )
+ except Exception as e:
+ Log.warning(f"[VisUnityVal] Failed global transl sanity. err={e}")
+
+ # Alignment:
+ # - Always align XZ using GT root (so trajectories are comparable).
+ # - Ground GT and pred independently (prevents double-grounding when pred was already
+ # grounded by postprocess).
+ if self._J_regressor is not None:
+ root0 = torch.matmul(self._J_regressor, gt_verts_world[0])[0]
+ offset_xz = root0.clone()
+ else:
+ offset_xz = gt_verts_world[0].mean(0)
+ offset_xz[1] = 0.0
+ gt_verts_world = gt_verts_world - offset_xz
+ pred_verts_world = pred_verts_world - offset_xz
+
+ gt_min_y = gt_verts_world[..., 1].min()
+ pred_min_y = pred_verts_world[..., 1].min()
+ gt_verts_world[..., 1] = gt_verts_world[..., 1] - gt_min_y
+ pred_verts_world[..., 1] = pred_verts_world[..., 1] - pred_min_y
+
+ if self._J_regressor is not None:
+ joints0 = torch.matmul(self._J_regressor, gt_verts_world[0])
+ T_ay2ayfz = compute_T_ayfz2ay(joints0[None], inverse=True)
+ gt_verts_world = apply_T_on_points(gt_verts_world, T_ay2ayfz)
+ pred_verts_world = apply_T_on_points(pred_verts_world, T_ay2ayfz)
+
+ # Log first-frame vertical root offset (pred vs GT) for debugging.
+ try:
+ if self._J_regressor is not None:
+ gt_root_y0 = float(torch.matmul(self._J_regressor, gt_verts_world[0])[0, 1].item())
+ pred_root_y0 = float(torch.matmul(self._J_regressor, pred_verts_world[0])[0, 1].item())
+ else:
+ gt_root_y0 = float(gt_verts_world[0].mean(0)[1].item())
+ pred_root_y0 = float(pred_verts_world[0].mean(0)[1].item())
+ Log.info(
+ f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} root_y0: "
+ f"gt={gt_root_y0:+.4f} pred={pred_root_y0:+.4f} delta(pred-gt)={pred_root_y0-gt_root_y0:+.4f}"
+ )
+ except Exception as e:
+ Log.warning(f"[VisUnityVal] Failed to log root height delta. err={e}")
+
+ # Log first-frame global orientation (axis-angle + Euler) mismatch.
+ try:
+ fi0 = int(frame_idxs[0]) if len(frame_idxs) > 0 else 0
+ gt_go0 = gt_params_w.get("global_orient")
+ pred_go0 = pred_params_global.get("global_orient") if "pred_params_global" in locals() else None
+ if isinstance(gt_go0, torch.Tensor) and isinstance(pred_go0, torch.Tensor):
+ gt_go0 = gt_go0[fi0].to(device)
+ pred_go0 = pred_go0[fi0].to(device)
+
+ R_gt = axis_angle_to_matrix(gt_go0[None]).to(device)
+ R_pred = axis_angle_to_matrix(pred_go0[None]).to(device)
+ R_rel = R_pred @ R_gt.transpose(-1, -2)
+
+ e_gt = matrix_to_euler_angles(R_gt, "YXZ")[0] * (180.0 / float(np.pi))
+ e_pred = matrix_to_euler_angles(R_pred, "YXZ")[0] * (180.0 / float(np.pi))
+ e_rel = matrix_to_euler_angles(R_rel, "YXZ")[0] * (180.0 / float(np.pi))
+
+ def _wrap_deg(x: torch.Tensor) -> torch.Tensor:
+ return (x + 180.0) % 360.0 - 180.0
+
+ e_gt = _wrap_deg(e_gt)
+ e_pred = _wrap_deg(e_pred)
+ e_rel = _wrap_deg(e_rel)
+
+ Log.info(
+ f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} global_orient0_aa(gt)={gt_go0.detach().cpu().numpy()} "
+ f"global_orient0_aa(pred)={pred_go0.detach().cpu().numpy()}"
+ )
+ # If we rendered GT in inference-frame (GV0), also log the raw dataset world-orient for comparison.
+ try:
+ if gt_params_w_infer is not None and "smpl_params_w" in batch:
+ raw_go = batch["smpl_params_w"]["global_orient"][0, fi0].to(device)
+ Log.info(
+ f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} global_orient0_aa(raw_gt_world)={raw_go.detach().cpu().numpy()}"
+ )
+ except Exception:
+ pass
+ Log.info(
+ f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} global_orient0_yxz_deg "
+ f"gt=({e_gt[0]:+.2f},{e_gt[1]:+.2f},{e_gt[2]:+.2f}) "
+ f"pred=({e_pred[0]:+.2f},{e_pred[1]:+.2f},{e_pred[2]:+.2f}) "
+ f"pred_vs_gt=({e_rel[0]:+.2f},{e_rel[1]:+.2f},{e_rel[2]:+.2f})"
+ )
+ except Exception as e:
+ Log.warning(f"[VisUnityVal] Failed to log global_orient mismatch. err={e}")
+
+ # Log first-frame yaw mismatch (pred vs GT) in the rendered/global frame.
+ # Uses the same facing heuristic as `compute_T_ayfz2ay`: hips + shoulders define left-right axis.
+ try:
+ if self._J_regressor is not None:
+ gt_j0 = torch.matmul(self._J_regressor, gt_verts_world[0]) # (J, 3)
+ pred_j0 = torch.matmul(self._J_regressor, pred_verts_world[0])
+ else:
+ gt_j0 = None
+ pred_j0 = None
+
+ def face_z_xz(j0: torch.Tensor):
+ RL_xz_h = j0[1, [0, 2]] - j0[2, [0, 2]]
+ RL_xz_s = j0[16, [0, 2]] - j0[17, [0, 2]]
+ RL_xz = RL_xz_h + RL_xz_s
+ if float(RL_xz.pow(2).sum().item()) < 1e-8:
+ return None
+ x_dir_xz = F.normalize(RL_xz[None], p=2, dim=-1)[0]
+ z_dir_xz = torch.stack([-x_dir_xz[1], x_dir_xz[0]], dim=0) # rotate +90deg in XZ
+ z_dir_xz = F.normalize(z_dir_xz[None], p=2, dim=-1)[0]
+ return z_dir_xz
+
+ if gt_j0 is not None and pred_j0 is not None:
+ gt_fwd = face_z_xz(gt_j0)
+ pred_fwd = face_z_xz(pred_j0)
+ if gt_fwd is not None and pred_fwd is not None:
+ dot = torch.clamp((gt_fwd * pred_fwd).sum(), -1.0, 1.0)
+ cross = gt_fwd[0] * pred_fwd[1] - gt_fwd[1] * pred_fwd[0]
+ ang = torch.atan2(cross, dot) # signed radians
+ ang_deg = float((ang * (180.0 / float(np.pi))).item())
+ Log.info(
+ f"[VisUnityVal] e{trainer.current_epoch:03d}_{vid} yaw0_deg(pred_vs_gt)={ang_deg:+.2f}"
+ )
+ except Exception as e:
+ Log.warning(f"[VisUnityVal] Failed to log yaw mismatch. err={e}")
+
+ # Optional: remove global translation (useful for spotting pose issues without
+ # the distraction of root drift / foot sliding in the source motion).
+ if self.global_root_relative and self._J_regressor is not None:
+ gt_joints = torch.einsum("jv,fvi->fji", self._J_regressor, gt_verts_world)
+ pred_joints = torch.einsum(
+ "jv,fvi->fji", self._J_regressor, pred_verts_world
+ )
+ gt_root = gt_joints[:, 0] # (F, 3)
+ pred_root = pred_joints[:, 0]
+ gt_verts_world = gt_verts_world - gt_root[:, None, :]
+ pred_verts_world = pred_verts_world - pred_root[:, None, :]
+
+ global_path = out_dir / f"e{trainer.current_epoch:03d}_{vid}_global.mp4"
+ writer = get_writer(str(global_path), fps=30, crf=int(self.crf))
+ try:
+ _, _, K_global = create_camera_sensor(width, height, 24)
+ renderer_global = Renderer(width, height, device=device, faces=self._faces, K=K_global, bin_size=0)
+ try:
+ from pytorch3d.renderer import AmbientLights
+
+ renderer_global.lights = AmbientLights(device=device)
+ renderer_global.create_renderer()
+ except Exception:
+ pass
+ global_R, global_T, global_lights = get_global_cameras_static(
+ gt_verts_world.detach().cpu(), beta=2.0, cam_height_degree=20, target_center_height=1.0
+ )
+ scale, cx, cz = get_ground_params_from_points(
+ gt_verts_world[:, 0].detach().cpu(), gt_verts_world.detach().cpu()
+ )
+ renderer_global.set_ground(scale * 1.5, cx, cz)
+
+ pred_color = torch.tensor(self.pred_color, device=device).float() / 255.0
+ gt_color = torch.tensor(self.gt_color, device=device).float() / 255.0
+ colors = torch.stack([gt_color, pred_color], dim=0)
+
+ for i in range(gt_verts_world.shape[0]):
+ cameras = renderer_global.create_camera(global_R[i], global_T[i])
+ img = renderer_global.render_with_ground(
+ torch.stack([gt_verts_world[i], pred_verts_world[i]], dim=0),
+ colors,
+ cameras,
+ global_lights,
+ )
+ writer.write_frame(img.astype(np.uint8))
+ finally:
+ writer.close()
diff --git a/genmo/datamodule/mocap_trainX_testY.py b/genmo/datamodule/mocap_trainX_testY.py
new file mode 100644
index 0000000000000000000000000000000000000000..e7565b64724105fe7ef7163314f7c3039cbe8015
--- /dev/null
+++ b/genmo/datamodule/mocap_trainX_testY.py
@@ -0,0 +1,255 @@
+import resource
+from functools import partial
+
+import pytorch_lightning as pl
+import torch
+from hydra.utils import instantiate
+from numpy.random import choice
+from omegaconf import DictConfig
+from pytorch_lightning.utilities.combined_loader import CombinedLoader
+from torch.utils.data import ConcatDataset, DataLoader, Subset, default_collate
+
+from genmo.utils.pylogger import Log
+
+rlimit = resource.getrlimit(resource.RLIMIT_NOFILE)
+resource.setrlimit(resource.RLIMIT_NOFILE, (4096, rlimit[1]))
+
+
+def collate_fn(batch, mode, collate_cfg=None):
+ """Handle meta and Add batch size to the return dict
+ Args:
+ batch: list of dict, each dict is a data point
+ collate_cfg: configuration for collation
+ """
+ # Assume all keys in the batch are the same
+ return_dict = {"B": len(batch)}
+
+ length = collate_cfg.max_motion_frames
+ if mode in ["val", "test"] and "K_fullimg" in batch[0]:
+ length = batch[0]["K_fullimg"].shape[0]
+
+ # Get a superset of all keys from all batch items
+ mandatory_keys = [
+ "has_text",
+ # "has_audio",
+ # "has_music",
+ "caption",
+ "text_embed",
+ "music_embed",
+ "music_array",
+ "music_fps",
+ "music_beats",
+ "audio_array",
+ "audio_fps",
+ "use_det_kp",
+ ]
+ keys = set(mandatory_keys)
+ for item in batch:
+ keys.update(item.keys())
+ keys = sorted(keys)
+
+ for k in keys:
+ if k.startswith("meta"): # data information, do not batch
+ return_dict[k] = [d[k] for d in batch]
+ elif k == "multi_text_embed":
+ # Get max length across batch
+ max_len = max(d[k].shape[0] for d in batch if k in d)
+ padded_tensors = []
+ for d in batch:
+ if k in d:
+ padded = torch.cat(
+ [
+ d[k],
+ torch.zeros(max_len - d[k].shape[0], *d[k].shape[1:]).to(
+ d[k]
+ ),
+ ],
+ dim=0,
+ )
+ padded_tensors.append(padded)
+ else:
+ # Handle case where key is missing in this batch item
+ continue
+ if padded_tensors:
+ return_dict[k] = default_collate(padded_tensors)
+ else:
+ vals = []
+ for d in batch:
+ if k not in d:
+ if k in collate_cfg.default_feature_val:
+ val = collate_cfg.default_feature_val[k]
+ elif k in collate_cfg.default_frame_feature_dim:
+ length_multiplier = (
+ collate_cfg.default_seq_feature_length_multiplier.get(k, 1)
+ )
+ val = torch.zeros(
+ length * length_multiplier,
+ *collate_cfg.default_frame_feature_dim[k],
+ dtype=eval(
+ collate_cfg.default_feature_type.get(k, "torch.float32")
+ ),
+ )
+ elif k in collate_cfg.default_seq_feature_dim:
+ val = torch.zeros(
+ *collate_cfg.default_seq_feature_dim[k],
+ dtype=eval(
+ collate_cfg.default_feature_type.get(k, "torch.float32")
+ ),
+ )
+ else:
+ raise ValueError(f"Key {k} not found in collate_cfg")
+ vals.append(val)
+ else:
+ vals.append(d[k])
+ return_dict[k] = default_collate(vals)
+
+ return return_dict
+
+
+class DataModule(pl.LightningDataModule):
+ def __init__(
+ self,
+ dataset_opts: DictConfig,
+ loader_opts: DictConfig,
+ limit_each_trainset=None,
+ train_subset_ratio=None,
+ train_2d_only=False,
+ collate_cfg: DictConfig = None,
+ ):
+ """This is a general datamodule that can be used for any dataset.
+ Train uses ConcatDataset
+ Val and Test use CombinedLoader, sequential, completely consumes ecah iterable sequentially, and returns a triplet (data, idx, iterable_idx)
+
+ Args:
+ dataset_opts: the target of the dataset. e.g. dataset_opts.train = {_target_: ..., limit_size: None}
+ loader_opts: the options for the dataset
+ limit_each_trainset: limit the size of each dataset, None means no limit, useful for debugging
+ """
+ super().__init__()
+ self.loader_opts = loader_opts
+ self.limit_each_trainset = limit_each_trainset
+ self.train_subset_ratio = train_subset_ratio
+ self.train_2d_only = train_2d_only
+ self.collate_cfg = collate_cfg
+ # Train uses concat dataset
+ if "train" in dataset_opts:
+ assert "train" in self.loader_opts, "train not in loader_opts"
+ split_opts = dataset_opts.get("train")
+ assert isinstance(split_opts, DictConfig), (
+ "split_opts should be a dict for each dataset"
+ )
+ dataset = []
+ dataset_num = len(split_opts)
+ for idx, (k, v) in enumerate(split_opts.items()):
+ if v is None:
+ continue
+ try:
+ dataset_i = instantiate(v)
+ except Exception as e:
+ Log.warning(f"[Train Dataset] Skipping {k} due to error: {e}")
+ continue
+ if self.limit_each_trainset:
+ dataset_i = Subset(
+ dataset_i, choice(len(dataset_i), self.limit_each_trainset)
+ )
+ if self.train_subset_ratio is not None:
+ dataset_i = Subset(
+ dataset_i,
+ choice(
+ len(dataset_i),
+ int(len(dataset_i) * self.train_subset_ratio),
+ ),
+ )
+ dataset.append(dataset_i)
+ Log.info(
+ f"[Train Dataset][{idx + 1}/{dataset_num}]: name={k}, size={len(dataset[-1])}, {v._target_}"
+ )
+ if not dataset:
+ raise ValueError("No training datasets were successfully loaded!")
+ dataset = ConcatDataset(dataset)
+ self.trainset = dataset
+ Log.info(f"[Train Dataset][All]: ConcatDataset size={len(dataset)}")
+ Log.info("")
+
+ # Val and Test use sequential dataset
+ for split in ("val", "test"):
+ if split not in dataset_opts:
+ continue
+ assert split in self.loader_opts, f"split={split} not in loader_opts"
+ split_opts = dataset_opts.get(split)
+ assert isinstance(split_opts, DictConfig), (
+ "split_opts should be a dict for each dataset"
+ )
+ dataset = []
+ dataset_num = len(split_opts)
+ for idx, (k, v) in enumerate(split_opts.items()):
+ if v is None:
+ continue
+ try:
+ ds = instantiate(v)
+ dataset.append(ds)
+ dataset_type = "Val Dataset" if split == "val" else "Test Dataset"
+ Log.info(
+ f"[{dataset_type}][{idx + 1}/{dataset_num}]: name={k}, size={len(ds)}, {v._target_}"
+ )
+ except Exception as e:
+ Log.warning(f"[{split}] Skipping {k} due to error: {e}")
+ setattr(self, f"{split}sets", dataset)
+ Log.info("")
+
+ def train_dataloader(self):
+ if hasattr(self, "trainset"):
+ return DataLoader(
+ self.trainset,
+ shuffle=True,
+ num_workers=self.loader_opts.train.num_workers,
+ persistent_workers=True and self.loader_opts.train.num_workers > 0,
+ batch_size=self.loader_opts.train.batch_size,
+ drop_last=True,
+ collate_fn=partial(
+ collate_fn, mode="train", collate_cfg=self.collate_cfg
+ ),
+ )
+ else:
+ return super().train_dataloader()
+
+ def val_dataloader(self):
+ if hasattr(self, "valsets"):
+ loaders = []
+ for valset in self.valsets:
+ loaders.append(
+ DataLoader(
+ valset,
+ shuffle=False,
+ num_workers=self.loader_opts.val.num_workers,
+ persistent_workers=True
+ and self.loader_opts.val.num_workers > 0,
+ batch_size=self.loader_opts.val.batch_size,
+ collate_fn=partial(
+ collate_fn, mode="val", collate_cfg=self.collate_cfg
+ ),
+ )
+ )
+ return CombinedLoader(loaders, mode="sequential")
+ else:
+ return None
+
+ def test_dataloader(self):
+ if hasattr(self, "testsets"):
+ loaders = []
+ for testset in self.testsets:
+ loaders.append(
+ DataLoader(
+ testset,
+ shuffle=False,
+ num_workers=self.loader_opts.test.num_workers,
+ persistent_workers=False,
+ batch_size=self.loader_opts.test.batch_size,
+ collate_fn=partial(
+ collate_fn, mode="test", collate_cfg=self.collate_cfg
+ ),
+ )
+ )
+ return CombinedLoader(loaders, mode="sequential")
+ else:
+ return super().test_dataloader()
diff --git a/genmo/datasets/aistplusplus/aistplusplus.py b/genmo/datasets/aistplusplus/aistplusplus.py
new file mode 100644
index 0000000000000000000000000000000000000000..85d910c3a6d2a9e4f989c5c4fb156a7b86537197
--- /dev/null
+++ b/genmo/datasets/aistplusplus/aistplusplus.py
@@ -0,0 +1,331 @@
+import os
+from pathlib import Path
+
+import numpy as np
+import torch
+from moviepy.editor import AudioFileClip
+from torch.utils import data
+
+from genmo.utils.geo_transform import (
+ compute_cam_angvel,
+ compute_cam_tvel,
+ get_bbx_xys_from_xyxy,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import (
+ get_valid_mask,
+ repeat_to_max_len,
+ repeat_to_max_len_dict,
+)
+from genmo.utils.pylogger import Log
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import get_c_rootparam, get_R_c2gv
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+class AISTPlusPlusSmplDataset(data.Dataset):
+ def __init__(
+ self,
+ root="inputs/AIST++",
+ split="train",
+ motion_frames=120,
+ lazy_load=False,
+ eval_gen_only=True,
+ feat_version="v1",
+ ):
+ super().__init__()
+ # Path
+ self.root = Path(root)
+
+ # Setting
+ self.motion_frames = motion_frames
+ self.lazy_load = lazy_load
+ self.split = split
+ self.eval_gen_only = eval_gen_only
+ self.feat_version = feat_version
+ self._load_dataset()
+ self._get_idx2meta()
+
+ def _load_dataset(self):
+ # smplpose
+ tic = Log.time()
+ fn = self.root / "annot_aist_30fps.pt"
+ self.smpl_model = make_smplx("supermotion")
+ Log.info(f"[AIST++ {self.feat_version}] Loading from {fn} ...")
+ self.motion_files = torch.load(fn)
+ self.split_set = torch.load(self.root / f"{self.split}.pt")
+ # Dict of {
+ # "smpl_params_glob": {'body_pose', 'global_orient', 'transl', 'betas'}, FxC
+ # "cam_Rt": tensor(F, 3),
+ # "cam_K": tensor(1, 10),
+ # }
+ self.seqs = list(self.motion_files.keys())
+ Log.info(
+ f"[AIST++ {self.feat_version}] {len(self.seqs)} sequences. Elapsed: {Log.time() - tic:.2f}s"
+ )
+
+ def _get_idx2meta(self):
+ # We expect to see the entire sequence during one epoch,
+ # so each sequence will be sampled max(SeqLength // MotionFrames, 1) times
+ seq_lengths = []
+ self.idx2meta = []
+ for vid in self.motion_files:
+ if vid not in self.split_set:
+ continue
+ seq_length = self.motion_files[vid]["bbox_xyxy"].shape[0]
+ seq_lengths.append(seq_length)
+ self.idx2meta.extend([vid])
+ hours = sum(seq_lengths) / 30 / 3600
+ Log.info(
+ f"[AIST++] has {hours:.1f} hours motion -> Resampled to {len(self.idx2meta)} samples."
+ )
+
+ def __len__(self):
+ return len(self.idx2meta)
+
+ def _load_data(self, idx):
+ sampled_motion = {}
+ vid = self.idx2meta[idx]
+ motion = self.motion_files[vid]
+
+ music_feat = torch.load(
+ self.root / f"musicfeat_{self.feat_version}/{vid}_musicfeat_fps30.pt"
+ )
+ seq_length = min(music_feat.shape[0], motion["bbox_xyxy"].shape[0])
+ sampled_motion["vid"] = vid
+
+ # Random select a subset
+ target_length = self.motion_frames
+ if target_length > seq_length: # this should not happen
+ start = 0
+ length = seq_length
+ Log.info(
+ f"[AIST++] ({idx}) target length < sequence length: {target_length} <= {seq_length}"
+ )
+ elif self.split in ["train", "minitrain"]:
+ start = np.random.randint(0, seq_length - target_length)
+ length = target_length
+ else:
+ start = 0
+ length = seq_length
+ end = start + length
+ sampled_motion["length"] = length
+ sampled_motion["start_end"] = (start, end)
+
+ music_beats = torch.load(self.root / f"musicfeat/{vid}_musicfeat_fps30.pt")[
+ ..., 53
+ ]
+ sampled_motion["music_beats"] = torch.from_numpy(music_beats[start:end]).float()
+
+ # Select motion subset
+ # body_pose, global_orient, transl, betas
+ sampled_motion["smpl_params_glob"] = {
+ "body_pose": torch.from_numpy(
+ motion["smpl_pose_global"][start:end][:, 3:66]
+ ).float(),
+ "betas": torch.zeros((length, 10)).float(),
+ "global_orient": torch.from_numpy(
+ motion["smpl_pose_global"][start:end][:, :3]
+ ).float(),
+ "transl": torch.from_numpy(motion["smpl_trans_global"][start:end]).float(),
+ }
+
+ sampled_motion["smpl_params_cam"] = {
+ "body_pose": torch.from_numpy(
+ motion["smpl_pose"][start:end][:, 3:66]
+ ).float(),
+ "betas": torch.zeros((length, 10)).float(),
+ "global_orient": torch.from_numpy(
+ motion["smpl_pose"][start:end][:, :3]
+ ).float(),
+ "transl": torch.from_numpy(motion["smpl_trans"][start:end]).float(),
+ }
+
+ # Image as feature
+ # vitfeat = torch.load(self.root / f"vitfeat/{vid}_vitfeat.pt")
+ # vitfeat = vitfeat[::2] # 60fps -> 30fps
+ # sampled_motion["f_imgseq"] = vitfeat[start:end].float() # (L, 1024)
+ sampled_motion["f_imgseq"] = torch.zeros((length, 1024)).float()
+
+ bbx_xys = get_bbx_xys_from_xyxy(
+ torch.from_numpy(motion["bbox_xyxy"][start:end]), base_enlarge=1.2
+ )
+ sampled_motion["bbx_xys"] = bbx_xys.float()
+ sampled_motion["K_fullimg"] = motion["intrinsics"]
+ # sampled_motion["kp2d"] = self.vitpose[vid][start:end].float() # (L, 17, 3)
+ # vitpose = torch.load(self.root / f"vitpose/{vid}_vitpose.pt")[0][::2] # 60fps -> 30fps
+ # sampled_motion["kp2d"] = vitpose[start:end].float() # (L, 17, 3)
+ sampled_motion["kp2d"] = torch.zeros((length, 17, 3)).float()
+
+ # Camera
+ sampled_motion["T_w2c"] = motion["T_w2c"] # (4, 4)
+
+ sampled_motion["music_embed"] = torch.from_numpy(
+ music_feat[start:end]
+ ).float() # (L, 1024)
+
+ # load audio
+ if self.split in ["train", "minitrain"]:
+ music_fps = 30
+ music_array = torch.zeros((length, 1024)).float()
+ else:
+ music_array = torch.load(os.path.join(self.root, f"audio_array/{vid}.pt"))
+ music_array = torch.from_numpy(music_array).float()
+ music = AudioFileClip(os.path.join(self.root, f"audio/{vid}.mp3"))
+ music_fps = music.fps
+ start_audio = int(start * music_fps / 30)
+ end_audio = int(end * music_fps / 30)
+ music_array = music_array[start_audio:end_audio]
+ # music_fps = 30
+
+ sampled_motion["music_array"] = music_array
+ sampled_motion["music_fps"] = music_fps
+ sampled_motion["height"] = motion["height"]
+ sampled_motion["width"] = motion["width"]
+ return sampled_motion
+
+ def _process_data(self, data, idx):
+ length = data["length"]
+
+ # SMPL params in world
+ smpl_params_w = data["smpl_params_glob"] # in az
+ old_smpl_params_c = data["smpl_params_cam"]
+ music_fps = data["music_fps"]
+
+ # SMPL params in cam
+ T_w2c = data["T_w2c"] # (4, 4)
+ offset = self.smpl_model.get_skeleton(smpl_params_w["betas"][0])[0] # (3)
+ global_orient_c, transl_c = get_c_rootparam(
+ smpl_params_w["global_orient"],
+ smpl_params_w["transl"],
+ T_w2c,
+ offset,
+ )
+ assert (
+ old_smpl_params_c["global_orient"] - global_orient_c
+ ).abs().max() < 1e-4, (
+ (old_smpl_params_c["global_orient"] - global_orient_c).abs().max(),
+ data["vid"],
+ data["start_end"],
+ )
+ assert (old_smpl_params_c["transl"] - transl_c).abs().max() < 1e-4, (
+ (old_smpl_params_c["transl"] - transl_c).abs().max(),
+ data["vid"],
+ data["start_end"],
+ )
+
+ smpl_params_c = {
+ "body_pose": smpl_params_w["body_pose"].clone(), # (F, 63)
+ "betas": smpl_params_w["betas"].clone(), # (F, 10)
+ "global_orient": global_orient_c, # (F, 3)
+ "transl": transl_c, # (F, 3)
+ }
+
+ # World params
+ gravity_vec = torch.tensor([0, 0, -1]).float() # (3), AIST++ is az
+ T_w2c = T_w2c.repeat(length, 1, 1) # (F, 4, 4)
+ R_c2gv = get_R_c2gv(
+ T_w2c[..., :3, :3], axis_gravity_in_w=gravity_vec
+ ) # (F, 3, 3)
+
+ # Image
+ bbx_xys = data["bbx_xys"] # (F, 3)
+ K_fullimg = data["K_fullimg"].repeat(length, 1, 1) # (F, 3, 3)
+ f_imgseq = data["f_imgseq"] # (F, 1024)
+
+ normed_T_w2c = normalize_T_w2c(T_w2c)
+
+ cam_angvel = compute_cam_angvel(
+ normed_T_w2c[:, :3, :3]
+ ) # (F, 6) slightly different from WHAM
+ cam_tvel = compute_cam_tvel(normed_T_w2c[:, :3, 3]) # (F, 3)
+ assert cam_tvel.sum() == 0, cam_tvel
+
+ # Returns: do not forget to make it batchable! (last lines)
+ max_len = self.motion_frames if self.split in ["train", "minitrain"] else length
+ return_data = {
+ "meta": {
+ "data_name": "aist++",
+ "dataset_id": "aist++",
+ "idx": idx,
+ "vid": data["vid"],
+ "height": data["height"],
+ "width": data["width"],
+ "eval_gen_only": self.eval_gen_only,
+ },
+ "length": length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w,
+ "R_c2gv": R_c2gv, # (F, 3, 3)
+ "gravity_vec": gravity_vec, # (3)
+ "bbx_xys": bbx_xys, # (F, 3)
+ "K_fullimg": K_fullimg, # (F, 3, 3)
+ "f_imgseq": f_imgseq, # (F, D)
+ "kp2d": data["kp2d"], # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "cam_tvel": cam_tvel, # (F, 3)
+ "noisy_cam_tvel": cam_tvel, # (F, 3)
+ "T_w2c": normed_T_w2c, # (F, 4, 4)
+ "music_embed": data["music_embed"], # (F, C)
+ "music_array": data["music_array"], # (F / 30 * audio_fps, C)
+ "music_fps": music_fps,
+ "music_beats": data["music_beats"], # (F,)
+ # "has_music": True,
+ "mask": {
+ "valid": get_valid_mask(length, length),
+ "has_img_mask": get_valid_mask(length, 0),
+ "has_2d_mask": get_valid_mask(length, length),
+ "has_cam_mask": get_valid_mask(length, length),
+ "has_audio_mask": get_valid_mask(length, 0),
+ "has_music_mask": get_valid_mask(length, length),
+ "2d_only": False,
+ "vitpose": False,
+ "bbx_xys": False,
+ "f_imgseq": False,
+ "spv_incam_only": False,
+ "invalid_contact": True,
+ },
+ }
+
+ # Batchable
+ if self.split in ["train", "minitrain"]:
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(
+ return_data["smpl_params_c"], max_len
+ )
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(
+ return_data["smpl_params_w"], max_len
+ )
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["bbx_xys"] = repeat_to_max_len(return_data["bbx_xys"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(
+ return_data["K_fullimg"], max_len
+ )
+ return_data["f_imgseq"] = repeat_to_max_len(
+ return_data["f_imgseq"], max_len
+ )
+ return_data["music_embed"] = repeat_to_max_len(
+ return_data["music_embed"], max_len
+ )
+ return_data["music_array"] = repeat_to_max_len(
+ return_data["music_array"], int(max_len / 30 * music_fps)
+ )
+ return_data["music_beats"] = repeat_to_max_len(
+ return_data["music_beats"], max_len
+ )
+
+ return_data["kp2d"] = repeat_to_max_len(return_data["kp2d"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(
+ return_data["cam_angvel"], max_len
+ )
+ return_data["cam_tvel"] = repeat_to_max_len(
+ return_data["cam_tvel"], max_len
+ )
+ return_data["noisy_cam_tvel"] = repeat_to_max_len(
+ return_data["noisy_cam_tvel"], max_len
+ )
+ return_data["T_w2c"] = repeat_to_max_len(return_data["T_w2c"], max_len)
+ return return_data
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data, idx)
+ return data
diff --git a/genmo/datasets/beat2/beat2.py b/genmo/datasets/beat2/beat2.py
new file mode 100644
index 0000000000000000000000000000000000000000..2b0b9894f35268f1849ce2828ccc891085f1f2cd
--- /dev/null
+++ b/genmo/datasets/beat2/beat2.py
@@ -0,0 +1,313 @@
+from pathlib import Path
+
+import librosa
+import numpy as np
+import torch
+from torch.utils import data
+
+from genmo.datasets.pure_motion.cam_traj_utils import CameraAugmentorV11
+from genmo.utils.geo_transform import (
+ compute_cam_angvel,
+ compute_cam_tvel,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import (
+ get_valid_mask,
+ repeat_to_max_len,
+ repeat_to_max_len_dict,
+)
+from genmo.utils.pylogger import Log
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import create_camera_sensor
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import get_c_rootparam, get_R_c2gv
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+from third_party.GVHMR.hmr4d.utils.wis3d_utils import add_motion_as_lines, make_wis3d
+
+
+class BEAT2SmplDataset(data.Dataset):
+ def __init__(
+ self,
+ root="inputs/BEAT2",
+ split="train",
+ motion_frames=120,
+ cam_augmentation="static",
+ lazy_load=False,
+ ):
+ super().__init__()
+ self.root = Path(root)
+ self.split = split
+ self.motion_frames = motion_frames
+ self.lazy_load = lazy_load
+
+ self.audio_fps = 18000
+ self.video_fps = 30
+
+ self.cam_augmentation = cam_augmentation
+
+ self.smplx = make_smplx("supermotion")
+
+ self.smplx_neutral = make_smplx(type="supermotion_smpl24")
+ self.smplx_dict = {
+ "male": self.smplx_neutral,
+ "female": self.smplx_neutral,
+ "neutral": self.smplx_neutral,
+ }
+
+ self._load_dataset()
+ self._get_idx2meta()
+
+ def _load_dataset(self):
+ # smplpose
+ tic = Log.time()
+ fn = self.root / "all_splits.pth"
+ self.smpl_model = make_smplx("supermotion")
+ Log.info(f"[BEAT2] Loading from {fn} ...")
+ self.motion_files = torch.load(fn)[self.split]
+ Log.info(
+ f"[BEAT2] {len(self.motion_files)} sequences. Elapsed: {Log.time() - tic:.2f}s"
+ )
+
+ def _get_idx2meta(self):
+ seq_lengths = []
+ self.idx2meta = []
+ for item in self.motion_files:
+ seq_length = item["length"]
+ # vid = item["video_id"]
+ seq_lengths.append(seq_length)
+ self.idx2meta.append(item)
+ hours = sum(seq_lengths) / 30 / 3600
+ Log.info(
+ f"[BEAT2] has {hours:.1f} hours motion -> Resampled to {len(self.idx2meta)} samples."
+ )
+
+ def __len__(self):
+ return len(self.idx2meta)
+
+ def _load_data(self, idx):
+ sampled_motion = {}
+ item = self.idx2meta[idx]
+ vid = item["video_id"]
+ subset = item["subset"]
+ motion = np.load(self.root / f"{subset}/smplxflame_30/{vid}.npz")
+ seq_length = motion["poses"].shape[0]
+
+ audio_array, audio_fps = librosa.load(
+ self.root / f"{subset}/wave16k/{vid}.wav", sr=self.audio_fps
+ )
+ audio_array = audio_array[: seq_length * int(self.audio_fps / self.video_fps)]
+ audio_array = torch.from_numpy(audio_array).float()
+
+ # Random select a subset
+ target_length = self.motion_frames
+ if target_length > seq_length:
+ start = 0
+ length = seq_length
+ Log.info(
+ f"[BEAT2] ({idx}) target length < sequence length: {target_length} <= {seq_length}"
+ )
+ elif self.split in ["train", "minitrain"]:
+ start = np.random.randint(0, seq_length - target_length)
+ length = target_length
+ else:
+ start = 0
+ length = seq_length
+ end = start + length
+ sampled_motion["vid"] = vid
+ sampled_motion["length"] = length
+ sampled_motion["start_end"] = (start, end)
+
+ # Select motion subset
+ audio_array = audio_array[
+ start * int(self.audio_fps / self.video_fps) : end
+ * int(self.audio_fps / self.video_fps)
+ ]
+ fb_poses = motion["poses"]
+ fb_betas = motion["betas"]
+ transl = motion["trans"]
+
+ body_pose = fb_poses[start:end, 3:66]
+ global_orient = fb_poses[start:end, :3]
+ betas = fb_betas[None].repeat(length, axis=0)[:, :10]
+ transl = transl[start:end]
+
+ sampled_motion["smpl_params_glob"] = {
+ "body_pose": torch.from_numpy(body_pose).float(),
+ "betas": torch.from_numpy(betas).float(),
+ "global_orient": torch.from_numpy(global_orient).float(),
+ "transl": torch.from_numpy(transl).float(),
+ }
+ sampled_motion["gender"] = motion["gender"]
+
+ sampled_motion["f_imgseq"] = torch.zeros((length, 1024)).float()
+ sampled_motion["kp2d"] = torch.zeros((length, 17, 3)).float()
+
+ sampled_motion["audio_array"] = audio_array
+ sampled_motion["audio_fps"] = self.audio_fps
+
+ return sampled_motion
+
+ def _process_data(self, data, idx):
+ length = data["length"]
+ gender = str(data["gender"])
+
+ # SMPL params in world
+ smpl_params_w = data["smpl_params_glob"]
+ audio_fps = data["audio_fps"]
+
+ if self.cam_augmentation == "v11":
+ N = 10
+ smpl_layer = self.smplx_dict[gender]
+ w_j3d = smpl_layer(
+ smpl_params_w["body_pose"][::N],
+ smpl_params_w["betas"][::N],
+ smpl_params_w["global_orient"][::N],
+ None,
+ )
+ w_j3d = (
+ w_j3d.repeat_interleave(N, dim=0)[:length]
+ + smpl_params_w["transl"][:, None]
+ ) # (F, 24, 3)
+
+ width, height, K_fullimg = create_camera_sensor(1000, 1000, 43.3) # WHAM
+ # focal_length = K_fullimg[0, 0]
+ wham_cam_augmentor = CameraAugmentorV11()
+ T_w2c = wham_cam_augmentor(w_j3d, length) # (F, 4, 4)
+ elif self.cam_augmentation == "static":
+ # interleave repeat to original length (faster)
+ N = 10
+ smpl_layer = self.smplx_dict[gender]
+ w_j3d = smpl_layer(
+ smpl_params_w["body_pose"][::N],
+ smpl_params_w["betas"][::N],
+ smpl_params_w["global_orient"][::N],
+ None,
+ )
+ w_j3d = (
+ w_j3d.repeat_interleave(N, dim=0)[:length]
+ + smpl_params_w["transl"][:, None]
+ ) # (F, 24, 3)
+
+ if False:
+ wis3d = make_wis3d(name="debug_amass")
+ add_motion_as_lines(w_j3d, wis3d, "w_j3d")
+
+ width, height, K_fullimg = create_camera_sensor(1000, 1000, 43.3) # WHAM
+ # focal_length = K_fullimg[0, 0]
+ wham_cam_augmentor = CameraAugmentorV11()
+ T_w2c = wham_cam_augmentor(w_j3d, length, camera_type="static") # (F, 4, 4)
+ else:
+ raise NotImplementedError
+
+ T_c2w = T_w2c.inverse()
+ noisy_T_c2w = T_c2w.clone()
+ # R_c2w = as_identity(T_c2w[:, :3, :3])
+ t_c2w = T_c2w[:, :3, 3]
+ rand_scale = min(max(0.1, torch.randn(1) + 3), 10)
+ noisy_t_c2w = t_c2w / rand_scale
+ noisy_T_c2w[:, :3, 3] = noisy_t_c2w
+ noisy_T_w2c = noisy_T_c2w.inverse()
+ del noisy_T_c2w
+
+ normed_noisy_T_w2c = normalize_T_w2c(noisy_T_w2c)
+ normed_T_w2c = normalize_T_w2c(T_w2c)
+ # normed_R_w2c = as_identity(normed_T_w2c[:, :3, :3])
+ # normed_t_w2c = normed_T_w2c[:, :3, 3]
+ # normed_noisy_R_w2c = as_identity(normed_noisy_T_w2c[:, :3, :3])
+ # normed_noisy_t_w2c = normed_noisy_T_w2c[:, :3, 3]
+ del noisy_T_w2c
+
+ # SMPL params in cam
+ offset = self.smplx.get_skeleton(smpl_params_w["betas"][0])[0] # (3)
+ global_orient_c, transl_c = get_c_rootparam(
+ smpl_params_w["global_orient"],
+ smpl_params_w["transl"],
+ T_w2c,
+ offset,
+ )
+ smpl_params_c = {
+ "body_pose": smpl_params_w["body_pose"].clone(), # (F, 63)
+ "betas": smpl_params_w["betas"].clone(), # (F, 10)
+ "global_orient": global_orient_c, # (F, 3)
+ "transl": transl_c, # (F, 3)
+ }
+
+ # World params
+ gravity_vec = torch.tensor([0, -1, 0], dtype=torch.float32) # (3), BEDLAM is ay
+ R_c2gv = get_R_c2gv(T_w2c[:, :3, :3], gravity_vec) # (F, 3, 3)
+
+ # Image
+ K_fullimg = K_fullimg.repeat(length, 1, 1) # (F, 3, 3)
+ cam_angvel = compute_cam_angvel(normed_T_w2c[:, :3, :3]) # (F, 6)
+ cam_tvel = compute_cam_tvel(normed_T_w2c[:, :3, 3]) # (F, 3)
+ noisy_cam_tvel = compute_cam_tvel(normed_noisy_T_w2c[:, :3, 3]) # (F, 3)
+
+ # Returns: do not forget to make it batchable! (last lines)
+ # NOTE: bbx_xys and f_imgseq will be added later
+ max_len = length
+ return_data = {
+ "meta": {
+ "data_name": "beat2",
+ "idx": idx,
+ "vid": data["vid"],
+ "eval_gen_only": True,
+ },
+ "length": length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w,
+ "R_c2gv": R_c2gv, # (F, 3, 3)
+ "gravity_vec": gravity_vec, # (3)
+ "bbx_xys": torch.zeros((length, 3)), # (F, 3) # NOTE: a placeholder
+ "K_fullimg": K_fullimg, # (F, 3, 3)
+ "f_imgseq": torch.zeros((length, 1024)), # (F, D) # NOTE: a placeholder
+ "kp2d": torch.zeros(length, 17, 3), # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "cam_tvel": cam_tvel, # (F, 3),
+ "noisy_cam_tvel": noisy_cam_tvel, # (F, 3),
+ "T_w2c": normed_T_w2c,
+ "audio_array": data["audio_array"],
+ "audio_fps": audio_fps,
+ # "has_audio": True,
+ "mask": {
+ "valid": get_valid_mask(length, length),
+ "has_img_mask": get_valid_mask(length, 0),
+ "has_2d_mask": get_valid_mask(length, length),
+ "has_cam_mask": get_valid_mask(length, length),
+ "has_audio_mask": get_valid_mask(length, length),
+ "has_music_mask": get_valid_mask(length, 0),
+ "2d_only": False,
+ "vitpose": False,
+ "bbx_xys": False,
+ "f_imgseq": False,
+ "spv_incam_only": False,
+ "invalid_contact": False,
+ },
+ }
+
+ # Batchable
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(
+ return_data["smpl_params_c"], max_len
+ )
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(
+ return_data["smpl_params_w"], max_len
+ )
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["bbx_xys"] = repeat_to_max_len(return_data["bbx_xys"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(return_data["K_fullimg"], max_len)
+ return_data["f_imgseq"] = repeat_to_max_len(return_data["f_imgseq"], max_len)
+ return_data["audio_array"] = repeat_to_max_len(
+ return_data["audio_array"], int(max_len * self.audio_fps / self.video_fps)
+ )
+ return_data["kp2d"] = repeat_to_max_len(return_data["kp2d"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(
+ return_data["cam_angvel"], max_len
+ )
+ return_data["cam_tvel"] = repeat_to_max_len(return_data["cam_tvel"], max_len)
+ return_data["noisy_cam_tvel"] = repeat_to_max_len(
+ return_data["noisy_cam_tvel"], max_len
+ )
+ return_data["T_w2c"] = repeat_to_max_len(return_data["T_w2c"], max_len)
+ return return_data
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data, idx)
+ return data
diff --git a/genmo/datasets/bedlam/bedlam.py b/genmo/datasets/bedlam/bedlam.py
new file mode 100644
index 0000000000000000000000000000000000000000..d1d8a68a2d44ddeb9d713b499d783823c46814df
--- /dev/null
+++ b/genmo/datasets/bedlam/bedlam.py
@@ -0,0 +1,257 @@
+from pathlib import Path
+from time import time
+
+import numpy as np
+import torch
+
+import genmo.utils.matrix as matrix
+from genmo.datasets.bedlam.utils import mid2featname, mid2vname
+from genmo.datasets.imgfeat_motion.base_dataset import ImgfeatMotionDatasetBase
+from genmo.utils.geo_transform import (
+ apply_T_on_points,
+ as_identity,
+ compute_cam_angvel,
+ compute_cam_tvel,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import (
+ get_valid_mask,
+ repeat_to_max_len,
+ repeat_to_max_len_dict,
+)
+from genmo.utils.pylogger import Log
+from genmo.utils.rotation_conversions import axis_angle_to_matrix, matrix_to_axis_angle
+from genmo.utils.video_io_utils import read_video_np, save_video
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import (
+ get_c_rootparam,
+ get_R_c2gv,
+ get_T_w2c_from_wcparams,
+)
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+from third_party.GVHMR.hmr4d.utils.wis3d_utils import add_motion_as_lines, make_wis3d
+
+
+class BedlamDatasetV2(ImgfeatMotionDatasetBase):
+ """mid_to_valid_range and features are newly generated."""
+
+ MIDINDEX_TO_LOAD = {
+ "all60": ("mid_to_valid_range_all60.pt", "imgfeats/bedlam_all60"),
+ "maxspan60": ("mid_to_valid_range_maxspan60.pt", "imgfeats/bedlam_maxspan60"),
+ }
+
+ def __init__(
+ self,
+ mid_indices=["all60", "maxspan60"],
+ lazy_load=True, # Load from disk when needed
+ random1024=False, # Faster loading for debugging
+ ):
+ self.root = Path("inputs/BEDLAM/hmr4d_support")
+ self.min_motion_frames = 60
+ self.max_motion_frames = 120
+ self.lazy_load = lazy_load
+ self.random1024 = random1024
+
+ # speficify mid_index to handle
+ if not isinstance(mid_indices, list):
+ mid_indices = [mid_indices]
+ self.mid_indices = mid_indices
+ assert all([m in self.MIDINDEX_TO_LOAD for m in mid_indices])
+
+ super().__init__()
+
+ def _load_dataset(self):
+ Log.info(f"[BEDLAM] Loading from {self.root}")
+ tic = time()
+ # Load mid to valid range
+ self.mid_to_valid_range = {}
+ self.mid_to_imgfeat_dir = {}
+ for m in self.mid_indices:
+ fn, feat_dir = self.MIDINDEX_TO_LOAD[m]
+ mid_to_valid_range_ = torch.load(self.root / fn)
+ self.mid_to_valid_range.update(mid_to_valid_range_)
+ self.mid_to_imgfeat_dir.update(
+ {mid: self.root / feat_dir for mid in mid_to_valid_range_}
+ )
+
+ # Load motionfiles
+ Log.info(f"[BEDLAM] Start loading motion files")
+ if self.random1024: # Debug, faster loading
+ try:
+ Log.info(f"[BEDLAM] Loading 1024 samples for debugging ...")
+ self.motion_files = torch.load(self.root / "smplpose_v2_random1024.pth")
+ except:
+ Log.info(f"[BEDLAM] Not found, saving 1024 samples to disk ...")
+ self.motion_files = torch.load(self.root / "smplpose_v2.pth")
+ keys = list(self.motion_files.keys())
+ keys = np.random.choice(keys, 1024, replace=False)
+ self.motion_files = {k: self.motion_files[k] for k in keys}
+ torch.save(self.motion_files, self.root / "smplpose_v2_random1024.pth")
+ self.mid_to_valid_range = {
+ k: v
+ for k, v in self.mid_to_valid_range.items()
+ if k in self.motion_files
+ }
+ else:
+ self.motion_files = torch.load(self.root / "smplpose_v2.pth")
+ Log.info(f"[BEDLAM] Motion files loaded. Elapsed: {time() - tic:.2f}s")
+
+ def _get_idx2meta(self):
+ # sum_frame = sum([e-s for s, e in self.mid_to_valid_range.values()])
+ self.idx2meta = list(self.mid_to_valid_range.keys())
+ Log.info(f"[BEDLAM] {len(self.idx2meta)} sequences. ")
+
+ def _load_data(self, idx):
+ mid = self.idx2meta[idx]
+ # neutral smplx : "pose": (F, 63), "trans": (F, 3), "beta": (10),
+ # and : "skeleton": (J, 3)
+ data = self.motion_files[mid].copy()
+
+ # Random select a subset
+ range1, range2 = self.mid_to_valid_range[mid] # [range1, range2)
+ mlength = range2 - range1
+ min_motion_len = self.min_motion_frames
+ max_motion_len = self.max_motion_frames
+
+ if mlength < min_motion_len: # the minimal mlength is 30 when generating data
+ start = range1
+ length = mlength
+ else:
+ effect_max_motion_len = min(max_motion_len, mlength)
+ length = np.random.randint(
+ min_motion_len, effect_max_motion_len + 1
+ ) # [low, high)
+ start = np.random.randint(range1, range2 - length + 1)
+ end = start + length
+ data["start_end"] = (start, end)
+ data["length"] = length
+
+ # Update data to a subset
+ for k, v in data.items():
+ if isinstance(v, torch.Tensor) and len(v.shape) > 1 and k != "skeleton":
+ data[k] = v[start:end]
+
+ # Load img(as feature) : {mid -> 'features', 'bbx_xys', 'img_wh', 'start_end'}
+ imgfeat_dir = self.mid_to_imgfeat_dir[mid]
+ f_img_dict = torch.load(imgfeat_dir / mid2featname(mid))
+
+ # remap (start, end)
+ start_mapped = start - f_img_dict["start_end"][0]
+ end_mapped = end - f_img_dict["start_end"][0]
+
+ data["f_imgseq"] = f_img_dict["features"][
+ start_mapped:end_mapped
+ ].float() # (L, 1024)
+ data["bbx_xys"] = f_img_dict["bbx_xys"][
+ start_mapped:end_mapped
+ ].float() # (L, 4)
+ data["img_wh"] = f_img_dict["img_wh"] # (2)
+ data["kp2d"] = torch.zeros(
+ (end - start), 17, 3
+ ) # (L, 17, 3) # do not provide kp2d
+
+ return data
+
+ def _process_data(self, data, idx):
+ length = data["length"]
+
+ # SMPL params in cam
+ body_pose = data["pose"][:, 3:] # (F, 63)
+ betas = data["beta"].repeat(length, 1) # (F, 10)
+ global_orient = data["global_orient_incam"] # (F, 3)
+ transl = (
+ data["trans_incam"] + data["cam_ext"][:, :3, 3]
+ ) # (F, 3), bedlam convention
+ smpl_params_c = {
+ "body_pose": body_pose,
+ "betas": betas,
+ "transl": transl,
+ "global_orient": global_orient,
+ }
+
+ # SMPL params in world
+ global_orient_w = data["pose"][:, :3] # (F, 3)
+ transl_w = data["trans"] # (F, 3)
+ smpl_params_w = {
+ "body_pose": body_pose,
+ "betas": betas,
+ "transl": transl_w,
+ "global_orient": global_orient_w,
+ }
+
+ gravity_vec = torch.tensor([0, -1, 0], dtype=torch.float32) # (3), BEDLAM is ay
+ T_w2c = get_T_w2c_from_wcparams(
+ global_orient_w=global_orient_w,
+ transl_w=transl_w,
+ global_orient_c=global_orient,
+ transl_c=transl,
+ offset=data["skeleton"][0],
+ ) # (F, 4, 4)
+ R_c2gv = get_R_c2gv(T_w2c[:, :3, :3], gravity_vec) # (F, 3, 3)
+
+ # cam_angvel (slightly different from WHAM)
+ normed_T_w2c = normalize_T_w2c(T_w2c)
+ noisy_normed_T_w2c = normed_T_w2c.clone()
+ noisy_t_w2c = noisy_normed_T_w2c[:, :3, 3]
+ rand_scale = min(max(0.1, torch.randn(1) + 3), 10)
+ noisy_t_w2c = noisy_t_w2c / rand_scale
+ noisy_normed_T_w2c[:, :3, 3] = noisy_t_w2c
+
+ cam_angvel = compute_cam_angvel(normed_T_w2c[:, :3, :3]) # (F, 6)
+ cam_tvel = compute_cam_tvel(normed_T_w2c[:, :3, 3]) # (F, 3)
+ noisy_cam_tvel = compute_cam_tvel(noisy_normed_T_w2c[:, :3, 3]) # (F, 3)
+
+ # Returns: do not forget to make it batchable! (last lines)
+ max_len = self.max_motion_frames
+ return_data = {
+ "meta": {"data_name": "bedlam", "idx": idx},
+ "length": length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w,
+ "R_c2gv": R_c2gv, # (F, 3, 3)
+ "gravity_vec": gravity_vec, # (3)
+ "bbx_xys": data["bbx_xys"], # (F, 3)
+ "K_fullimg": data["cam_int"], # (F, 3, 3)
+ "f_imgseq": data["f_imgseq"], # (F, D)
+ "kp2d": data["kp2d"], # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "cam_tvel": cam_tvel, # (F, 3)
+ "noisy_cam_tvel": noisy_cam_tvel, # (F, 3)
+ "T_w2c": normed_T_w2c, # (F, 4, 4)
+ "mask": {
+ "valid": get_valid_mask(max_len, length),
+ "has_img_mask": get_valid_mask(max_len, length),
+ "has_2d_mask": get_valid_mask(max_len, length),
+ "has_cam_mask": get_valid_mask(max_len, length),
+ "has_audio_mask": get_valid_mask(max_len, 0),
+ "has_music_mask": get_valid_mask(max_len, 0),
+ "2d_only": False,
+ "vitpose": False,
+ "bbx_xys": True,
+ "f_imgseq": True,
+ "spv_incam_only": False,
+ "invalid_contact": False,
+ },
+ }
+
+ # Batchable
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(
+ return_data["smpl_params_c"], max_len
+ )
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(
+ return_data["smpl_params_w"], max_len
+ )
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["bbx_xys"] = repeat_to_max_len(return_data["bbx_xys"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(return_data["K_fullimg"], max_len)
+ return_data["f_imgseq"] = repeat_to_max_len(return_data["f_imgseq"], max_len)
+ return_data["kp2d"] = repeat_to_max_len(return_data["kp2d"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(
+ return_data["cam_angvel"], max_len
+ )
+ return_data["cam_tvel"] = repeat_to_max_len(return_data["cam_tvel"], max_len)
+ return_data["noisy_cam_tvel"] = repeat_to_max_len(
+ return_data["noisy_cam_tvel"], max_len
+ )
+ return_data["T_w2c"] = repeat_to_max_len(return_data["T_w2c"], max_len)
+
+ return return_data
diff --git a/genmo/datasets/bedlam/utils.py b/genmo/datasets/bedlam/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..3db6ce49c2131ecc5bd60c55e8aa95a5bcd3d0ca
--- /dev/null
+++ b/genmo/datasets/bedlam/utils.py
@@ -0,0 +1,40 @@
+from pathlib import Path
+
+import numpy as np
+import torch
+
+resource_dir = Path(__file__).parent / "resource"
+
+
+def mid2vname(mid):
+ """vname = {scene}/{seq}, Note that it ends with .mp4"""
+ # mid example: "inputs/bedlam/bedlam_download/20221011_1_250_batch01hand_closeup_suburb_a/mp4/seq_000001.mp4-rp_emma_posed_008"
+ # -> vname: 20221011_1_250_batch01hand_closeup_suburb_a/seq_000001.mp4
+ scene = mid.split("/")[-3]
+ seq = mid.split("/")[-1].split("-")[0]
+ vname = f"{scene}/{seq}"
+ return vname
+
+
+def mid2featname(mid):
+ """featname = {scene}/{seqsubj}, Note that it ends with .pt (extra)"""
+ # mid example: "inputs/bedlam/bedlam_download/20221011_1_250_batch01hand_closeup_suburb_a/mp4/seq_000001.mp4-rp_emma_posed_008"
+ # -> featname: 20221011_1_250_batch01hand_closeup_suburb_a/seq_000001.mp4-rp_emma_posed_008.pt
+ scene = mid.split("/")[-3]
+ seqsubj = mid.split("/")[-1]
+ featname = f"{scene}/{seqsubj}.pt"
+ return featname
+
+
+def featname2mid(featname):
+ """reverse func of mid2featname, Note that it removes .pt (extra)"""
+ # featname example: 20221011_1_250_batch01hand_closeup_suburb_a/seq_000001.mp4-rp_emma_posed_008.pt
+ # -> mid: inputs/bedlam/bedlam_download/20221011_1_250_batch01hand_closeup_suburb_a/mp4/seq_000001.mp4-rp_emma_posed_008
+ scene = featname.split("/")[0]
+ seqsubj = featname.split("/")[1].strip(".pt")
+ mid = f"inputs/bedlam/bedlam_download/{scene}/mp4/{seqsubj}"
+ return mid
+
+
+def load_vname2lwh():
+ return torch.load(resource_dir / "vname2lwh.pt")
diff --git a/genmo/datasets/emdb/emdb_motion_test.py b/genmo/datasets/emdb/emdb_motion_test.py
new file mode 100644
index 0000000000000000000000000000000000000000..37f4ad12bb9ca1e9ae11cd451acee3219ace0b7d
--- /dev/null
+++ b/genmo/datasets/emdb/emdb_motion_test.py
@@ -0,0 +1,252 @@
+from pathlib import Path
+
+import torch
+from torch.utils import data
+
+from genmo.utils.geo_transform import (
+ compute_cam_angvel,
+ compute_cam_tvel,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import get_valid_mask
+from genmo.utils.pylogger import Log
+from genmo.utils.rotation_conversions import matrix_to_axis_angle, quaternion_to_matrix
+from third_party.GVHMR.hmr4d.utils.geo.flip_utils import flip_kp2d_coco17
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import estimate_K, resize_K
+
+from .utils import EMDB1_NAMES, EMDB2_NAMES
+
+VID_PRESETS = {1: EMDB1_NAMES, 2: EMDB2_NAMES}
+
+
+def as_identity(R):
+ is_I = matrix_to_axis_angle(R).norm(dim=-1) < 1e-5
+ R[is_I] = torch.eye(3)[None].expand(is_I.sum(), -1, -1).to(R)
+ return R
+
+
+class EmdbSmplFullSeqDataset(data.Dataset):
+ def __init__(self, split=1, flip_test=False):
+ """
+ split: 1 for EMDB-1, 2 for EMDB-2
+ flip_test: if True, extra flip data will be returned
+ """
+ super().__init__()
+ self.dataset_name = "EMDB"
+ self.split = split
+ self.dataset_id = f"EMDB_{split}"
+ Log.info(f"[{self.dataset_name}] Full sequence, split={split}")
+
+ # Load evaluation protocol from WHAM labels
+ tic = Log.time()
+ self.emdb_dir = Path("inputs/EMDB/hmr4d_support")
+ # 'name', 'gender', 'smpl_params', 'mask', 'K_fullimg', 'T_w2c', 'bbx_xys', 'kp2d', 'features'
+ self.labels = torch.load(self.emdb_dir / "emdb_vit_v4.pt")
+ self.cam_traj = torch.load(
+ self.emdb_dir / "emdb_dpvo_traj.pt"
+ ) # estimated with DPVO
+
+ self.vimo_labels = torch.load(self.emdb_dir / "emdb_vimo.pt")
+ self.droid_cam_traj = torch.load(
+ self.emdb_dir / "emdb_slam_traj.pt"
+ ) # estimated with SLAM
+
+ # Setup dataset index
+ self.idx2meta = []
+ for vid in VID_PRESETS[split]:
+ seq_length = len(self.labels[vid]["mask"])
+ self.idx2meta.append((vid, 0, seq_length)) # start=0, end=seq_length
+ Log.info(
+ f"[{self.dataset_name}] {len(self.idx2meta)} sequences. Elapsed: {Log.time() - tic:.2f}s"
+ )
+
+ # If flip_test is enabled, we will return extra data for flipped test
+ self.flip_test = flip_test
+ if self.flip_test:
+ Log.info(f"[{self.dataset_name}] Flip test enabled")
+
+ def __len__(self):
+ return len(self.idx2meta)
+
+ def _load_data(self, idx):
+ data = {}
+
+ # [vid, start, end]
+ vid, start, end = self.idx2meta[idx]
+ length = end - start
+ meta = {
+ "dataset_id": self.dataset_id,
+ "vid": vid,
+ "vid-start-end": (start, end),
+ }
+ data.update({"meta": meta, "length": length})
+
+ label = self.labels[vid]
+ vimo_label = self.vimo_labels[vid]
+ droid_label = self.droid_cam_traj[vid]
+
+ # smpl_params in world
+ gender = label["gender"]
+ smpl_params = label["smpl_params"]
+ mask = label["mask"]
+ mask_dict = {
+ "valid": mask,
+ "has_img_mask": get_valid_mask(length, length),
+ "has_2d_mask": get_valid_mask(length, length),
+ "has_cam_mask": get_valid_mask(length, length),
+ "has_audio_mask": get_valid_mask(length, 0),
+ "has_music_mask": get_valid_mask(length, 0),
+ # "2d_only": False,
+ }
+ data.update({"smpl_params": smpl_params, "gender": gender, "mask": mask_dict})
+ vimo_smpl_params = {
+ "pred_cam": vimo_label["vimo_params"]["pred_cam"],
+ "pred_pose": vimo_label["vimo_params"]["pred_pose"],
+ "pred_shape": vimo_label["vimo_params"]["pred_shape"],
+ "pred_trans_c": vimo_label["vimo_params"]["pred_trans"],
+ }
+
+ data.update({"vimo_smpl_params": vimo_smpl_params})
+
+ # camera
+ # load droid slam
+ R_c2w = torch.from_numpy(droid_label["pred_cam_R"]).float()
+ t_c2w = torch.from_numpy(droid_label["pred_cam_T"]).float()
+ scales = torch.from_numpy(droid_label["all_scales"]).float()
+ mean_scale = droid_label["scale"]
+ T_c2w = torch.eye(4)[None].repeat(length, 1, 1).to(R_c2w)
+ T_c2w[:, :3, :3] = R_c2w
+ T_c2w[:, :3, 3] = t_c2w
+ T_w2c = T_c2w.inverse()
+
+ # K_fullimg = label["K_fullimg"] # We use estimated K
+ width_height = (1440, 1920) if vid != "P0_09_outdoor_walk" else (720, 960)
+ K_fullimg = estimate_K(*width_height)
+ # T_w2c = label["T_w2c"] # use GT camera trajectory
+ gt_T_w2c = label["T_w2c"]
+ data.update(
+ {
+ "K_fullimg": K_fullimg,
+ "T_w2c": T_w2c,
+ "scales": scales,
+ "mean_scale": mean_scale,
+ "gt_T_w2c": gt_T_w2c,
+ }
+ )
+
+ if "vimo_params_flip" in vimo_label:
+ flipped_trans_c = vimo_label["vimo_params_flip"]["pred_trans"]
+ orig_trans_c = data["vimo_smpl_params"]["pred_trans_c"]
+ tz = flipped_trans_c[..., 2]
+ tx = flipped_trans_c[..., 0]
+ focal = K_fullimg[0, 0]
+ cx = K_fullimg[0, 2]
+ width = width_height[0]
+
+ flipped_tx = tz * (width - 1 - 2 * cx) / focal - tx
+ avg_trans_c = torch.zeros_like(flipped_trans_c)
+ avg_trans_c[..., 0] = (flipped_tx + orig_trans_c[..., 0]) / 2
+ avg_trans_c[..., 0] = orig_trans_c[..., 0]
+ avg_trans_c[..., 1] = (flipped_trans_c[..., 1] + orig_trans_c[..., 1]) / 2
+ avg_trans_c[..., 2] = (tz + orig_trans_c[..., 2]) / 2
+ data["vimo_smpl_params"]["pred_trans_c"] = avg_trans_c
+
+ # R_w2c -> cam_angvel
+ use_DPVO = False
+ if use_DPVO:
+ traj = self.cam_traj[data["meta"]["vid"]] # (L, 7)
+ R_w2c = quaternion_to_matrix(traj[:, [6, 3, 4, 5]]).mT # (L, 3, 3)
+ t_c2w = traj[:, :3]
+ else: # GT
+ # L = data["T_w2c"].shape[0]
+ norm_T_w2c = normalize_T_w2c(data["T_w2c"])
+
+ R_w2c = norm_T_w2c[:, :3, :3]
+ t_w2c = norm_T_w2c[:, :3, 3]
+
+ data["cam_angvel"] = compute_cam_angvel(R_w2c) # (L, 6)
+ data["cam_tvel"] = compute_cam_tvel(t_w2c) # (L, 3)
+ data["R_w2c"] = R_w2c
+
+ # image bbx, features
+ bbx_xys = label["bbx_xys"]
+ f_imgseq = label["features"]
+ kp2d = label["kp2d"]
+ data.update({"bbx_xys": bbx_xys, "f_imgseq": f_imgseq, "kp2d": kp2d})
+
+ # to render a video
+ video_path = self.emdb_dir / f"videos/{vid}.mp4"
+ frame_id = torch.where(mask)[0].long()
+ resize_factor = 0.5
+ width_height_render = torch.tensor(width_height) * resize_factor
+ K_render = resize_K(K_fullimg, resize_factor)
+ bbx_xys_render = bbx_xys * resize_factor
+ data["meta_render"] = {
+ "split": self.split,
+ "name": vid,
+ "video_path": str(video_path),
+ "resize_factor": resize_factor,
+ "frame_id": frame_id,
+ "width_height": width_height_render.int(),
+ "K": K_render,
+ "bbx_xys": bbx_xys_render,
+ "R_cam_type": "DPVO" if use_DPVO else "GtGyro",
+ }
+
+ # if enable flip_test
+ if self.flip_test:
+ imgfeat_dir = self.emdb_dir / "imgfeats/emdb_flip"
+ f_img_dict = torch.load(imgfeat_dir / f"{vid}.pt")
+
+ flipped_bbx_xys = f_img_dict["bbx_xys"].float() # (L, 3)
+ flipped_features = f_img_dict["features"].float() # (L, 1024)
+ width = width_height[0]
+ flipped_kp2d = flip_kp2d_coco17(kp2d, width) # (L, 17, 3)
+
+ R_flip_x = torch.tensor([[1, 0, 0], [0, -1, 0], [0, 0, -1]]).float()
+ flipped_R_w2c = R_flip_x @ R_w2c.clone()
+ flipped_t_w2c = (R_flip_x @ t_w2c.clone()[..., None])[..., 0]
+ flipped_T_w2c = torch.eye(4)[None].repeat(length, 1, 1).to(flipped_R_w2c)
+ flipped_T_w2c[:, :3, :3] = flipped_R_w2c
+ flipped_T_w2c[:, :3, 3] = flipped_t_w2c
+
+ data_flip = {
+ "bbx_xys": flipped_bbx_xys,
+ "f_imgseq": flipped_features,
+ "kp2d": flipped_kp2d,
+ "cam_angvel": compute_cam_angvel(flipped_R_w2c),
+ "cam_tvel": compute_cam_tvel(flipped_t_w2c),
+ "R_w2c": flipped_R_w2c,
+ }
+ flipped_trans_c = vimo_label["vimo_params_flip"]["pred_trans"]
+ flipped_trans_c[..., 2] = avg_trans_c[..., 2]
+ vimo_smpl_params_flip = {
+ "pred_cam": vimo_label["vimo_params_flip"]["pred_cam"],
+ "pred_pose": vimo_label["vimo_params_flip"]["pred_pose"],
+ "pred_shape": vimo_label["vimo_params_flip"]["pred_shape"],
+ "pred_trans_c": flipped_trans_c,
+ }
+ data_flip["vimo_smpl_params"] = vimo_smpl_params_flip
+
+ flipped_K_fullimg = K_fullimg.clone()
+ data_flip.update(
+ {
+ "K_fullimg": flipped_K_fullimg,
+ "T_w2c": flipped_T_w2c,
+ "scales": scales,
+ "mean_scale": mean_scale,
+ }
+ )
+ data["flip_test"] = data_flip
+
+ return data
+
+ def _process_data(self, data):
+ length = data["length"]
+ data["K_fullimg"] = data["K_fullimg"][None].repeat(length, 1, 1)
+ return data
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data)
+ return data
diff --git a/genmo/datasets/emdb/utils.py b/genmo/datasets/emdb/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..e6512b5db832b7d30375e6d07cede2d1fc3a3f52
--- /dev/null
+++ b/genmo/datasets/emdb/utils.py
@@ -0,0 +1,129 @@
+import pickle
+from pathlib import Path
+
+import numpy as np
+import torch
+from tqdm import tqdm
+
+from genmo.utils.geo_transform import convert_lurb_to_bbx_xys
+from genmo.utils.video_io_utils import get_video_lwh
+
+
+def name_to_subfolder(name):
+ return f"{name[:2]}/{name[3:]}"
+
+
+def name_to_local_pkl_path(name):
+ return f"{name_to_subfolder(name)}/{name}_data.pkl"
+
+
+def load_raw_pkl(fp):
+ annot = pickle.load(open(fp, "rb"))
+ annot["subfolder"] = name_to_subfolder(annot["name"])
+ return annot
+
+
+def load_pkl(fp):
+ annot = pickle.load(open(fp, "rb"))
+ # ['gender', 'name', 'emdb1', 'emdb2', 'n_frames', 'good_frames_mask', 'camera', 'smpl', 'kp2d', 'bboxes', 'subfolder']
+ data = {}
+
+ F = annot["n_frames"]
+ smpl_params = {
+ "body_pose": annot["smpl"]["poses_body"], # (F, 69)
+ "betas": annot["smpl"]["betas"][None].repeat(F, axis=0), # (F, 10)
+ "global_orient": annot["smpl"]["poses_root"], # (F, 3)
+ "transl": annot["smpl"]["trans"], # (F, 3)
+ }
+ smpl_params = {k: torch.from_numpy(v).float() for k, v in smpl_params.items()}
+
+ data["name"] = annot["name"]
+ data["gender"] = annot["gender"]
+ data["smpl_params"] = smpl_params
+ data["mask"] = torch.from_numpy(annot["good_frames_mask"]).bool() # (L,)
+ data["K_fullimg"] = torch.from_numpy(
+ annot["camera"]["intrinsics"]
+ ).float() # (3, 3)
+ data["T_w2c"] = torch.from_numpy(annot["camera"]["extrinsics"]).float() # (L, 4, 4)
+ bbx_lurb = torch.from_numpy(annot["bboxes"]["bboxes"]).float()
+ data["bbx_xys"] = convert_lurb_to_bbx_xys(bbx_lurb) # (L, 3)
+
+ return data
+
+
+EMDB1_LIST = [
+ "P1/14_outdoor_climb/P1_14_outdoor_climb_data.pkl",
+ "P2/23_outdoor_hug_tree/P2_23_outdoor_hug_tree_data.pkl",
+ "P3/31_outdoor_workout/P3_31_outdoor_workout_data.pkl",
+ "P3/32_outdoor_soccer_warmup_a/P3_32_outdoor_soccer_warmup_a_data.pkl",
+ "P3/33_outdoor_soccer_warmup_b/P3_33_outdoor_soccer_warmup_b_data.pkl",
+ "P5/42_indoor_dancing/P5_42_indoor_dancing_data.pkl",
+ "P5/44_indoor_rom/P5_44_indoor_rom_data.pkl",
+ "P6/49_outdoor_big_stairs_down/P6_49_outdoor_big_stairs_down_data.pkl", # DUPLICATE
+ "P6/50_outdoor_workout/P6_50_outdoor_workout_data.pkl",
+ "P6/51_outdoor_dancing/P6_51_outdoor_dancing_data.pkl",
+ "P7/57_outdoor_rock_chair/P7_57_outdoor_rock_chair_data.pkl", # DUPLICATE
+ "P7/59_outdoor_rom/P7_59_outdoor_rom_data.pkl",
+ "P7/60_outdoor_workout/P7_60_outdoor_workout_data.pkl",
+ "P8/64_outdoor_skateboard/P8_64_outdoor_skateboard_data.pkl", # DUPLICATE
+ "P8/68_outdoor_handstand/P8_68_outdoor_handstand_data.pkl",
+ "P8/69_outdoor_cartwheel/P8_69_outdoor_cartwheel_data.pkl",
+ "P9/76_outdoor_sitting/P9_76_outdoor_sitting_data.pkl",
+]
+EMDB1_NAMES = ["_".join(p.split("/")[:2]) for p in EMDB1_LIST]
+
+
+EMDB2_LIST = [
+ "P0/09_outdoor_walk/P0_09_outdoor_walk_data.pkl",
+ "P2/19_indoor_walk_off_mvs/P2_19_indoor_walk_off_mvs_data.pkl",
+ "P2/20_outdoor_walk/P2_20_outdoor_walk_data.pkl",
+ "P2/24_outdoor_long_walk/P2_24_outdoor_long_walk_data.pkl",
+ "P3/27_indoor_walk_off_mvs/P3_27_indoor_walk_off_mvs_data.pkl",
+ "P3/28_outdoor_walk_lunges/P3_28_outdoor_walk_lunges_data.pkl",
+ "P3/29_outdoor_stairs_up/P3_29_outdoor_stairs_up_data.pkl",
+ "P3/30_outdoor_stairs_down/P3_30_outdoor_stairs_down_data.pkl",
+ "P4/35_indoor_walk/P4_35_indoor_walk_data.pkl",
+ "P4/36_outdoor_long_walk/P4_36_outdoor_long_walk_data.pkl",
+ "P4/37_outdoor_run_circle/P4_37_outdoor_run_circle_data.pkl",
+ "P5/40_indoor_walk_big_circle/P5_40_indoor_walk_big_circle_data.pkl",
+ "P6/48_outdoor_walk_downhill/P6_48_outdoor_walk_downhill_data.pkl",
+ "P6/49_outdoor_big_stairs_down/P6_49_outdoor_big_stairs_down_data.pkl", # DUPLICATE
+ "P7/55_outdoor_walk/P7_55_outdoor_walk_data.pkl",
+ "P7/56_outdoor_stairs_up_down/P7_56_outdoor_stairs_up_down_data.pkl",
+ "P7/57_outdoor_rock_chair/P7_57_outdoor_rock_chair_data.pkl", # DUPLICATE
+ "P7/58_outdoor_parcours/P7_58_outdoor_parcours_data.pkl",
+ "P7/61_outdoor_sit_lie_walk/P7_61_outdoor_sit_lie_walk_data.pkl",
+ "P8/64_outdoor_skateboard/P8_64_outdoor_skateboard_data.pkl", # DUPLICATE
+ "P8/65_outdoor_walk_straight/P8_65_outdoor_walk_straight_data.pkl",
+ "P9/77_outdoor_stairs_up/P9_77_outdoor_stairs_up_data.pkl",
+ "P9/78_outdoor_stairs_up_down/P9_78_outdoor_stairs_up_down_data.pkl",
+ "P9/79_outdoor_walk_rectangle/P9_79_outdoor_walk_rectangle_data.pkl",
+ "P9/80_outdoor_walk_big_circle/P9_80_outdoor_walk_big_circle_data.pkl",
+]
+EMDB2_NAMES = ["_".join(p.split("/")[:2]) for p in EMDB2_LIST]
+EMDB_NAMES = list(sorted(set(EMDB1_NAMES + EMDB2_NAMES)))
+
+
+def _check_annot(emdb_raw_dir=Path("inputs/EMDB/EMDB")):
+ for pkl_local_path in set(EMDB1_LIST + EMDB2_LIST):
+ annot = load_raw_pkl(emdb_raw_dir / pkl_local_path)
+ if any(
+ (annot["bboxes"]["invalid_idxs"] != np.where(~annot["good_frames_mask"])[0])
+ ):
+ print(annot["name"])
+
+
+def _check_length(
+ emdb_raw_dir=Path("inputs/EMDB/EMDB"),
+ emdb_hmr4d_support_dir=Path("inputs/EMDB/hmr4d_support"),
+):
+ lengths = []
+ for local_pkl_path in tqdm(set(EMDB1_LIST + EMDB2_LIST)):
+ data = load_pkl(emdb_raw_dir / local_pkl_path)
+ video_path = emdb_hmr4d_support_dir / "videos" / f"{data['name']}.mp4"
+ length, width, height = get_video_lwh(video_path)
+ lengths.append(length)
+ print(sorted(lengths))
+
+ video_ram = length[-1] * (width / 4) * (height / 4) * 3 / 1e6
+ print(f"Video RAM for {lengths[-1]} x {width} x {height}: {video_ram:.2f} MB")
diff --git a/genmo/datasets/h36m/h36m.py b/genmo/datasets/h36m/h36m.py
new file mode 100644
index 0000000000000000000000000000000000000000..dd455eecafaef49080485ebedae30c17ef29de06
--- /dev/null
+++ b/genmo/datasets/h36m/h36m.py
@@ -0,0 +1,226 @@
+from pathlib import Path
+
+import numpy as np
+import torch
+
+from genmo.datasets.imgfeat_motion.base_dataset import ImgfeatMotionDatasetBase
+from genmo.utils.geo_transform import (
+ compute_cam_angvel,
+ compute_cam_tvel,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import (
+ get_valid_mask,
+ repeat_to_max_len,
+ repeat_to_max_len_dict,
+)
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import (
+ get_c_rootparam,
+ get_R_c2gv,
+)
+from third_party.GVHMR.hmr4d.utils.pylogger import Log
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+class H36mSmplDataset(ImgfeatMotionDatasetBase):
+ def __init__(
+ self,
+ root="inputs/H36M/hmr4d_support",
+ original_coord="az",
+ motion_frames=120, # H36M's videos are 25fps and very long
+ lazy_load=False,
+ ):
+ # Path
+ self.root = Path(root)
+
+ # Coord
+ self.original_coord = original_coord
+
+ # Setting
+ self.motion_frames = motion_frames
+ self.lazy_load = lazy_load
+
+ super().__init__()
+
+ def _load_dataset(self):
+ # smplpose
+ tic = Log.time()
+ fn = self.root / "smplxpose_v1.pt"
+ self.smpl_model = make_smplx("supermotion")
+ Log.info(f"[H36M] Loading from {fn} ...")
+ self.motion_files = torch.load(fn)
+ # Dict of {
+ # "smpl_params_glob": {'body_pose', 'global_orient', 'transl', 'betas'}, FxC
+ # "cam_Rt": tensor(F, 3),
+ # "cam_K": tensor(1, 10),
+ # }
+ self.seqs = list(self.motion_files.keys())
+ Log.info(f"[H36M] {len(self.seqs)} sequences. Elapsed: {Log.time() - tic:.2f}s")
+
+ # img(as feature)
+ # vid -> (features, vid, meta {bbx_xys, K_fullimg})
+ if not self.lazy_load:
+ tic = Log.time()
+ fn = self.root / "vitfeat_h36m.pt"
+ Log.info(f"[H36M] Fully Loading to RAM ViT-Feat: {fn}")
+ self.f_img_dicts = torch.load(fn)
+ Log.info(f"[H36M] Finished. Elapsed: {Log.time() - tic:.2f}s")
+ else:
+ raise NotImplementedError # "Check BEDLAM-SMPL for lazy_load"
+
+ def _get_idx2meta(self):
+ # We expect to see the entire sequence during one epoch,
+ # so each sequence will be sampled max(SeqLength // MotionFrames, 1) times
+ seq_lengths = []
+ self.idx2meta = []
+ for vid in self.f_img_dicts:
+ seq_length = self.f_img_dicts[vid]["bbx_xys"].shape[0]
+ num_samples = max(seq_length // self.motion_frames, 1)
+ seq_lengths.append(seq_length)
+ self.idx2meta.extend([vid] * num_samples)
+ hours = sum(seq_lengths) / 25 / 3600
+ Log.info(
+ f"[H36M] has {hours:.1f} hours motion -> Resampled to {len(self.idx2meta)} samples."
+ )
+
+ def _load_data(self, idx):
+ sampled_motion = {}
+ vid = self.idx2meta[idx]
+ motion = self.motion_files[vid]
+ seq_length = self.f_img_dicts[vid]["bbx_xys"].shape[
+ 0
+ ] # this is a better choice
+ sampled_motion["vid"] = vid
+
+ # Random select a subset
+ target_length = self.motion_frames
+ if target_length > seq_length: # this should not happen
+ start = 0
+ length = seq_length
+ Log.info(
+ f"[H36M] ({idx}) target length < sequence length: {target_length} <= {seq_length}"
+ )
+ else:
+ start = np.random.randint(0, seq_length - target_length)
+ length = target_length
+ end = start + length
+ sampled_motion["length"] = length
+ sampled_motion["start_end"] = (start, end)
+
+ # Select motion subset
+ # body_pose, global_orient, transl, betas
+ sampled_motion["smpl_params_global"] = {
+ k: v[start:end] for k, v in motion["smpl_params_glob"].items()
+ }
+
+ # Image as feature
+ f_img_dict = self.f_img_dicts[vid]
+ sampled_motion["f_imgseq"] = f_img_dict["features"][
+ start:end
+ ].float() # (L, 1024)
+ sampled_motion["bbx_xys"] = f_img_dict["bbx_xys"][start:end]
+ sampled_motion["K_fullimg"] = f_img_dict["K_fullimg"]
+ # sampled_motion["kp2d"] = self.vitpose[vid][start:end].float() # (L, 17, 3)
+ sampled_motion["kp2d"] = torch.zeros((end - start), 17, 3) # (L, 17, 3)
+
+ # Camera
+ sampled_motion["T_w2c"] = motion["cam_Rt"] # (4, 4)
+
+ return sampled_motion
+
+ def _process_data(self, data, idx):
+ length = data["length"]
+
+ # SMPL params in world
+ smpl_params_w = data["smpl_params_global"].copy() # in az
+
+ # SMPL params in cam
+ T_w2c = data["T_w2c"] # (4, 4)
+ offset = self.smpl_model.get_skeleton(smpl_params_w["betas"][0])[0] # (3)
+ global_orient_c, transl_c = get_c_rootparam(
+ smpl_params_w["global_orient"],
+ smpl_params_w["transl"],
+ T_w2c,
+ offset,
+ )
+ smpl_params_c = {
+ "body_pose": smpl_params_w["body_pose"].clone(), # (F, 63)
+ "betas": smpl_params_w["betas"].clone(), # (F, 10)
+ "global_orient": global_orient_c, # (F, 3)
+ "transl": transl_c, # (F, 3)
+ }
+
+ # World params
+ gravity_vec = torch.tensor([0, 0, -1]).float() # (3), H36M is az
+ T_w2c = T_w2c.repeat(length, 1, 1) # (F, 4, 4)
+ R_c2gv = get_R_c2gv(
+ T_w2c[..., :3, :3], axis_gravity_in_w=gravity_vec
+ ) # (F, 3, 3)
+
+ # Image
+ bbx_xys = data["bbx_xys"] # (F, 3)
+ K_fullimg = data["K_fullimg"].repeat(length, 1, 1) # (F, 3, 3)
+ f_imgseq = data["f_imgseq"] # (F, 1024)
+
+ normed_T_w2c = normalize_T_w2c(T_w2c)
+
+ cam_angvel = compute_cam_angvel(
+ normed_T_w2c[:, :3, :3]
+ ) # (F, 6) slightly different from WHAM
+ cam_tvel = compute_cam_tvel(normed_T_w2c[:, :3, 3]) # (F, 3)
+ assert cam_tvel.sum() == 0, cam_tvel
+
+ # Returns: do not forget to make it batchable! (last lines)
+ max_len = self.motion_frames
+ return_data = {
+ "meta": {"data_name": "h36m", "idx": idx, "vid": data["vid"]},
+ "length": length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w,
+ "R_c2gv": R_c2gv, # (F, 3, 3)
+ "gravity_vec": gravity_vec, # (3)
+ "bbx_xys": bbx_xys, # (F, 3)
+ "K_fullimg": K_fullimg, # (F, 3, 3)
+ "f_imgseq": f_imgseq, # (F, D)
+ "kp2d": data["kp2d"], # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "cam_tvel": cam_tvel, # (F, 3)
+ "noisy_cam_tvel": cam_tvel, # (F, 3)
+ "T_w2c": normed_T_w2c, # (F, 4, 4)
+ "mask": {
+ "valid": get_valid_mask(max_len, length),
+ "has_img_mask": get_valid_mask(max_len, length),
+ "has_2d_mask": get_valid_mask(max_len, length),
+ "has_cam_mask": get_valid_mask(max_len, length),
+ "has_audio_mask": get_valid_mask(max_len, 0),
+ "has_music_mask": get_valid_mask(max_len, 0),
+ "2d_only": False,
+ "vitpose": False,
+ "bbx_xys": True,
+ "f_imgseq": True,
+ "spv_incam_only": False,
+ "invalid_contact": False,
+ },
+ }
+
+ # Batchable
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(
+ return_data["smpl_params_c"], max_len
+ )
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(
+ return_data["smpl_params_w"], max_len
+ )
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["bbx_xys"] = repeat_to_max_len(return_data["bbx_xys"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(return_data["K_fullimg"], max_len)
+ return_data["f_imgseq"] = repeat_to_max_len(return_data["f_imgseq"], max_len)
+ return_data["kp2d"] = repeat_to_max_len(return_data["kp2d"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(
+ return_data["cam_angvel"], max_len
+ )
+ return_data["cam_tvel"] = repeat_to_max_len(return_data["cam_tvel"], max_len)
+ return_data["noisy_cam_tvel"] = repeat_to_max_len(
+ return_data["noisy_cam_tvel"], max_len
+ )
+ return_data["T_w2c"] = repeat_to_max_len(return_data["T_w2c"], max_len)
+ return return_data
diff --git a/genmo/datasets/imgfeat_motion/base_dataset.py b/genmo/datasets/imgfeat_motion/base_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..c4847ec9dd061713b069fcd7d30072a3e92e0953
--- /dev/null
+++ b/genmo/datasets/imgfeat_motion/base_dataset.py
@@ -0,0 +1,28 @@
+from torch.utils import data
+
+
+class ImgfeatMotionDatasetBase(data.Dataset):
+ def __init__(self):
+ super().__init__()
+ self._load_dataset()
+ self._get_idx2meta() # -> Set self.idx2meta
+
+ def __len__(self):
+ return len(self.idx2meta)
+
+ def _load_dataset(self):
+ raise NotImplementedError
+
+ def _get_idx2meta(self):
+ raise NotImplementedError
+
+ def _load_data(self, idx):
+ raise NotImplementedError
+
+ def _process_data(self, data, idx):
+ raise NotImplementedError
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data, idx)
+ return data
diff --git a/genmo/datasets/pure_motion/amass.py b/genmo/datasets/pure_motion/amass.py
new file mode 100644
index 0000000000000000000000000000000000000000..89485beae4263c69e08ce1d86f7edafff972acba
--- /dev/null
+++ b/genmo/datasets/pure_motion/amass.py
@@ -0,0 +1,126 @@
+from pathlib import Path
+
+import numpy as np
+import torch
+
+from genmo.utils.pylogger import Log
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import get_tgtcoord_rootparam
+
+from .base_dataset import BaseDataset
+from .utils import interpolate_smpl_params
+
+
+class AmassDataset(BaseDataset):
+ def __init__(
+ self,
+ motion_frames=120,
+ l_factor=1.5, # speed augmentation
+ skip_moyo=True, # not contained in the ICCV19 released version
+ cam_augmentation="v11",
+ random1024=False, # DEBUG
+ limit_size=None,
+ ):
+ self.root = Path("inputs/AMASS/hmr4d_support")
+ self.motion_frames = motion_frames
+ self.l_factor = l_factor
+ self.random1024 = random1024
+ self.skip_moyo = skip_moyo
+ self.dataset_name = "AMASS"
+ super().__init__(cam_augmentation, limit_size)
+
+ def _load_dataset(self):
+ filename = self.root / "smplxpose_v2.pth"
+ Log.info(f"[{self.dataset_name}] Loading from {filename} ...")
+ tic = Log.time()
+ if self.random1024: # Debug, faster loading
+ try:
+ Log.info(
+ f"[{self.dataset_name}] Loading 1024 samples for debugging ..."
+ )
+ self.motion_files = torch.load(
+ self.root / "smplxpose_v2_random1024.pth"
+ )
+ except Exception:
+ Log.info(
+ f"[{self.dataset_name}] Not found! Saving 1024 samples for debugging ..."
+ )
+ self.motion_files = torch.load(filename)
+ keys = list(self.motion_files.keys())
+ keys = np.random.choice(keys, 1024, replace=False)
+ self.motion_files = {k: self.motion_files[k] for k in keys}
+ torch.save(self.motion_files, self.root / "smplxpose_v2_random1024.pth")
+ else:
+ self.motion_files = torch.load(filename)
+ self.seqs = list(self.motion_files.keys())
+ Log.info(
+ f"[{self.dataset_name}] {len(self.seqs)} sequences. Elapsed: {Log.time() - tic:.2f}s"
+ )
+
+ def _get_idx2meta(self):
+ # We expect to see the entire sequence during one epoch,
+ # so each sequence will be sampled max(SeqLength // MotionFrames, 1) times
+ seq_lengths = []
+ self.idx2meta = []
+
+ # Skip too-long idle-prefix
+ motion_start_id = {}
+ for vid in self.motion_files:
+ if self.skip_moyo and "moyo_smplxn" in vid:
+ continue
+ seq_length = self.motion_files[vid]["pose"].shape[0]
+ start_id = motion_start_id[vid] if vid in motion_start_id else 0
+ seq_length = seq_length - start_id
+ if seq_length < 25: # Skip clips that are too short
+ continue
+ num_samples = max(seq_length // self.motion_frames, 1)
+ seq_lengths.append(seq_length)
+ self.idx2meta.extend([(vid, start_id)] * num_samples)
+ hours = sum(seq_lengths) / 30 / 3600
+ Log.info(
+ f"[{self.dataset_name}] has {hours:.1f} hours motion -> Resampled to {len(self.idx2meta)} samples."
+ )
+
+ def _load_data(self, idx):
+ """
+ - Load original data
+ - Augmentation: speed-augmentation to L frames
+ """
+ # Load original data
+ mid, start_id = self.idx2meta[idx]
+ raw_data = self.motion_files[mid]
+ raw_len = raw_data["pose"].shape[0] - start_id
+ data = {
+ "body_pose": raw_data["pose"][start_id:, 3:], # (F, 63)
+ "betas": raw_data["beta"].repeat(raw_len, 1), # (10)
+ "global_orient": raw_data["pose"][start_id:, :3], # (F, 3)
+ "transl": raw_data["trans"][start_id:], # (F, 3)
+ }
+
+ # Get {tgt_len} frames from data
+ # Random select a subset with speed augmentation [start, end)
+ tgt_len = self.motion_frames
+ raw_subset_len = np.random.randint(
+ int(tgt_len / self.l_factor), int(tgt_len * self.l_factor)
+ )
+ if raw_subset_len <= raw_len:
+ start = np.random.randint(0, raw_len - raw_subset_len + 1)
+ end = start + raw_subset_len
+ else: # interpolation will use all possible frames (results in a slow motion)
+ start = 0
+ end = raw_len
+ data = {k: v[start:end] for k, v in data.items()}
+
+ # Interpolation (vec + r6d)
+ data_interpolated = interpolate_smpl_params(data, tgt_len)
+
+ # AZ -> AY
+ data_interpolated["global_orient"], data_interpolated["transl"], _ = (
+ get_tgtcoord_rootparam(
+ data_interpolated["global_orient"],
+ data_interpolated["transl"],
+ tsf="az->ay",
+ )
+ )
+
+ data_interpolated["data_name"] = "amass"
+ return data_interpolated
diff --git a/genmo/datasets/pure_motion/base_dataset.py b/genmo/datasets/pure_motion/base_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..3dfc16b73f634ecaf12890746d4bbf25ce463986
--- /dev/null
+++ b/genmo/datasets/pure_motion/base_dataset.py
@@ -0,0 +1,222 @@
+import torch
+from torch.utils.data import Dataset
+
+from genmo.utils.geo_transform import (
+ compute_cam_angvel,
+ compute_cam_tvel,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import (
+ get_valid_mask,
+ repeat_to_max_len,
+ repeat_to_max_len_dict,
+)
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import create_camera_sensor
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import get_c_rootparam, get_R_c2gv
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+from third_party.GVHMR.hmr4d.utils.wis3d_utils import add_motion_as_lines, make_wis3d
+
+from .cam_traj_utils import CameraAugmentorV11
+from .utils import augment_betas, rotate_around_axis
+
+
+class BaseDataset(Dataset):
+ def __init__(self, cam_augmentation, limit_size=None):
+ super().__init__()
+ self.cam_augmentation = cam_augmentation
+ self.limit_size = limit_size
+ self.smplx = make_smplx("supermotion")
+ self.smplx_lite = make_smplx("supermotion_smpl24")
+
+ self._load_dataset()
+ self._get_idx2meta()
+
+ def _load_dataset(self):
+ raise NotImplementedError("_load_dataset is not implemented")
+
+ def _get_idx2meta(self):
+ self.idx2meta = None
+ raise NotImplementedError("_get_idx2meta is not implemented")
+
+ def __len__(self):
+ if self.limit_size is not None:
+ return min(self.limit_size, len(self.idx2meta))
+ return len(self.idx2meta)
+
+ def _load_data(self, idx):
+ raise NotImplementedError("_load_data is not implemented")
+
+ def _process_data(self, data, idx):
+ """
+ Args:
+ data: dict {
+ "body_pose": (F, 63),
+ "betas": (F, 10),
+ "global_orient": (F, 3), in the AY coordinates
+ "transl": (F, 3), in the AY coordinates
+ }
+ """
+ data_name = data["data_name"]
+ length = data["body_pose"].shape[0]
+ # Augmentation: betas, SMPL (gravity-axis)
+ body_pose = data["body_pose"]
+ betas = augment_betas(data["betas"], std=0.1)
+ global_orient_w, transl_w = rotate_around_axis(
+ data["global_orient"], data["transl"], axis="y"
+ )
+ del data
+
+ # SMPL_params in world
+ smpl_params_w = {
+ "body_pose": body_pose, # (F, 63)
+ "betas": betas, # (F, 10)
+ "global_orient": global_orient_w, # (F, 3)
+ "transl": transl_w, # (F, 3)
+ }
+
+ # Camera trajectory augmentation
+ if self.cam_augmentation == "v11":
+ # interleave repeat to original length (faster)
+ N = 10
+ w_j3d = self.smplx_lite(
+ smpl_params_w["body_pose"][::N],
+ smpl_params_w["betas"][::N],
+ smpl_params_w["global_orient"][::N],
+ None,
+ )
+ w_j3d = (
+ w_j3d.repeat_interleave(N, dim=0) + smpl_params_w["transl"][:, None]
+ ) # (F, 24, 3)
+
+ if False:
+ wis3d = make_wis3d(name="debug_amass")
+ add_motion_as_lines(w_j3d, wis3d, "w_j3d")
+
+ width, height, K_fullimg = create_camera_sensor(1000, 1000, 43.3) # WHAM
+ # focal_length = K_fullimg[0, 0]
+ wham_cam_augmentor = CameraAugmentorV11()
+ T_w2c = wham_cam_augmentor(w_j3d, length) # (F, 4, 4)
+ elif self.cam_augmentation == "static":
+ # interleave repeat to original length (faster)
+ N = 10
+ w_j3d = self.smplx_lite(
+ smpl_params_w["body_pose"][::N],
+ smpl_params_w["betas"][::N],
+ smpl_params_w["global_orient"][::N],
+ None,
+ )
+ w_j3d = (
+ w_j3d.repeat_interleave(N, dim=0) + smpl_params_w["transl"][:, None]
+ ) # (F, 24, 3)
+
+ if False:
+ wis3d = make_wis3d(name="debug_amass")
+ add_motion_as_lines(w_j3d, wis3d, "w_j3d")
+
+ width, height, K_fullimg = create_camera_sensor(1000, 1000, 43.3) # WHAM
+ # focal_length = K_fullimg[0, 0]
+ wham_cam_augmentor = CameraAugmentorV11()
+ T_w2c = wham_cam_augmentor(w_j3d, length, camera_type="static") # (F, 4, 4)
+ else:
+ raise NotImplementedError
+
+ T_c2w = T_w2c.inverse()
+ noisy_T_c2w = T_c2w.clone()
+ # R_c2w = as_identity(T_c2w[:, :3, :3])
+ t_c2w = T_c2w[:, :3, 3]
+ rand_scale = min(max(0.1, torch.randn(1) + 3), 10)
+ noisy_t_c2w = t_c2w / rand_scale
+ noisy_T_c2w[:, :3, 3] = noisy_t_c2w
+ noisy_T_w2c = noisy_T_c2w.inverse()
+ del noisy_T_c2w
+
+ normed_noisy_T_w2c = normalize_T_w2c(noisy_T_w2c)
+ normed_T_w2c = normalize_T_w2c(T_w2c)
+ # normed_R_w2c = as_identity(normed_T_w2c[:, :3, :3])
+ # normed_t_w2c = normed_T_w2c[:, :3, 3]
+ # normed_noisy_R_w2c = as_identity(normed_noisy_T_w2c[:, :3, :3])
+ # normed_noisy_t_w2c = normed_noisy_T_w2c[:, :3, 3]
+ del noisy_T_w2c
+
+ # SMPL params in cam
+ offset = self.smplx.get_skeleton(smpl_params_w["betas"][0])[0] # (3)
+ global_orient_c, transl_c = get_c_rootparam(
+ smpl_params_w["global_orient"],
+ smpl_params_w["transl"],
+ T_w2c,
+ offset,
+ )
+ smpl_params_c = {
+ "body_pose": smpl_params_w["body_pose"].clone(), # (F, 63)
+ "betas": smpl_params_w["betas"].clone(), # (F, 10)
+ "global_orient": global_orient_c, # (F, 3)
+ "transl": transl_c, # (F, 3)
+ }
+
+ # World params
+ gravity_vec = torch.tensor([0, -1, 0], dtype=torch.float32) # (3), BEDLAM is ay
+ R_c2gv = get_R_c2gv(T_w2c[:, :3, :3], gravity_vec) # (F, 3, 3)
+
+ # Image
+ K_fullimg = K_fullimg.repeat(length, 1, 1) # (F, 3, 3)
+ cam_angvel = compute_cam_angvel(normed_T_w2c[:, :3, :3]) # (F, 6)
+ cam_tvel = compute_cam_tvel(normed_T_w2c[:, :3, 3]) # (F, 3)
+ noisy_cam_tvel = compute_cam_tvel(normed_noisy_T_w2c[:, :3, 3]) # (F, 3)
+
+ # Returns: do not forget to make it batchable! (last lines)
+ # NOTE: bbx_xys and f_imgseq will be added later
+ max_len = length
+ return_data = {
+ "meta": {"data_name": data_name, "idx": idx, "T_w2c": T_w2c},
+ "length": length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w,
+ "R_c2gv": R_c2gv, # (F, 3, 3)
+ "gravity_vec": gravity_vec, # (3)
+ "bbx_xys": torch.zeros((length, 3)), # (F, 3) # NOTE: a placeholder
+ "K_fullimg": K_fullimg, # (F, 3, 3)
+ "f_imgseq": torch.zeros((length, 1024)), # (F, D) # NOTE: a placeholder
+ "kp2d": torch.zeros(length, 17, 3), # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "cam_tvel": cam_tvel, # (F, 3),
+ "noisy_cam_tvel": noisy_cam_tvel, # (F, 3),
+ "T_w2c": normed_T_w2c,
+ "mask": {
+ "valid": get_valid_mask(length, length),
+ "has_img_mask": get_valid_mask(length, 0),
+ "has_2d_mask": get_valid_mask(length, length),
+ "has_cam_mask": get_valid_mask(length, length),
+ "has_audio_mask": get_valid_mask(length, 0),
+ "has_music_mask": get_valid_mask(length, 0),
+ "2d_only": False,
+ "vitpose": False,
+ "bbx_xys": False,
+ "f_imgseq": False,
+ "spv_incam_only": False,
+ "invalid_contact": False,
+ },
+ }
+
+ # Batchable
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(
+ return_data["smpl_params_c"], max_len
+ )
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(
+ return_data["smpl_params_w"], max_len
+ )
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(return_data["K_fullimg"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(
+ return_data["cam_angvel"], max_len
+ )
+ return_data["cam_tvel"] = repeat_to_max_len(return_data["cam_tvel"], max_len)
+ return_data["noisy_cam_tvel"] = repeat_to_max_len(
+ return_data["noisy_cam_tvel"], max_len
+ )
+ return_data["T_w2c"] = repeat_to_max_len(return_data["T_w2c"], max_len)
+ return return_data
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data, idx)
+ return data
diff --git a/genmo/datasets/pure_motion/cam_traj_utils.py b/genmo/datasets/pure_motion/cam_traj_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..244e51d2778b6fa06a2280e60c7c2dd1e638f4ac
--- /dev/null
+++ b/genmo/datasets/pure_motion/cam_traj_utils.py
@@ -0,0 +1,460 @@
+import numpy as np
+import torch
+from numpy.random import rand, randn
+
+import genmo.utils.matrix as matrix
+from genmo.utils.geo_transform import transform_mat
+from genmo.utils.rotation_conversions import (
+ axis_angle_to_matrix,
+ matrix_to_rotation_6d,
+ rotation_6d_to_matrix,
+)
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import create_camera_sensor
+from third_party.GVHMR.hmr4d.utils.geo.transforms import axis_rotate_to_matrix
+
+halfpi = np.pi / 2
+R_y_upsidedown = torch.tensor([[-1, 0, 0], [0, -1, 0], [0, 0, 1]]).float()
+
+
+def noisy_interpolation(x, length, step_noise_perc=0.2):
+ """Non-linear interpolation with noise, although with noise, the jittery is very small
+ Args:
+ x: (2, C)
+ length: scalar
+ step_noise_perc: [x0, x1 +-(step_noise_perc * step), x2], where step = x1-x0
+ """
+ assert x.shape[0] == 2 and len(x.shape) == 2
+ dim = x.shape[-1]
+ output = np.zeros((length, dim))
+
+ # Use linsapce(0, 1) +- noise as reference
+ linspace = np.repeat(np.linspace(0, 1, length)[None], dim, axis=0) # (D, L)
+ noise = (linspace[0, 1] - linspace[0, 0]) * step_noise_perc
+ space_noise = np.random.uniform(-noise, noise, (dim, length - 2)) # (D, L-2)
+ linspace[:, 1:-1] = linspace[:, 1:-1] + space_noise
+
+ # Do 1d interp
+ for i in range(dim):
+ output[:, i] = np.interp(linspace[i], np.array([0.0, 1.0]), x[:, i])
+ return output
+
+
+def noisy_impluse_interpolation(data1, data2, step_noise_perc=0.2):
+ """Non-linear interpolation of impluse with noise"""
+
+ dim = data1.shape[-1]
+ L = data1.shape[0]
+
+ linspace1 = np.stack([np.linspace(0, 1, L // 2) for _ in range(dim)])
+ linspace2 = np.stack([np.linspace(0, 1, L // 2)[::-1] for _ in range(dim)])
+ linspace = np.concatenate([linspace1, linspace2], axis=-1)
+ noise = (linspace[0, 1] - linspace[0, 0]) * step_noise_perc
+ space_noise = np.stack(
+ [np.random.uniform(-noise, noise, L - 2) for _ in range(dim)]
+ )
+
+ linspace[:, 1:-1] = linspace[:, 1:-1] + space_noise
+ linspace = linspace.T
+ output = data1 * (1 - linspace) + data2 * linspace
+ return output
+
+
+def create_camera(w_root, cfg):
+ """Create static camera pose
+ Args:
+ w_root: (3,), y-up coordinates
+ Returns:
+ R_w2c: (3, 3)
+ t_w2c: (3)
+ """
+ # Parse
+ pitch_std = cfg["pitch_std"]
+ pitch_mean = cfg["pitch_mean"]
+ roll_std = cfg["roll_std"]
+ tz_range1_prob = cfg["tz_range1_prob"]
+ tz_range1 = cfg["tz_range1"]
+ tz_range2 = cfg["tz_range2"]
+ f = cfg["f"]
+ w = cfg["w"]
+
+ # algo
+ yaw = rand() * 2 * np.pi # Look at any direction in xz-plane
+ pitch = np.clip(randn() * pitch_std + pitch_mean, -halfpi, halfpi)
+ roll = np.clip(randn() * roll_std, -halfpi, halfpi) # Normal-dist
+
+ # Note we use OpenCV's camera system by first applying R_y_upsidedown
+ yaw_rm = axis_rotate_to_matrix(yaw, axis="y")
+ pitch_rm = axis_rotate_to_matrix(pitch, axis="x")
+ roll_rm = axis_rotate_to_matrix(roll, axis="z")
+ R_w2c = (roll_rm @ pitch_rm @ yaw_rm @ R_y_upsidedown).squeeze(0) # (3, 3)
+
+ # Place people in the scene
+ if rand() < tz_range1_prob:
+ tz = rand() * (tz_range1[1] - tz_range1[0]) + tz_range1[0]
+ max_dist_in_fov = (w / 2) / f * tz
+ tx = (rand() * 2 - 1) * 0.7 * max_dist_in_fov
+ ty = (rand() * 2 - 1) * 0.5 * max_dist_in_fov
+
+ else:
+ tz = rand() * (tz_range2[1] - tz_range2[0]) + tz_range2[0]
+ max_dist_in_fov = (w / 2) / f * tz
+ max_dist_in_fov *= 0.9 # add a threshold
+ tx = torch.randn(1) * 1.6
+ tx = torch.clamp(tx, -max_dist_in_fov, max_dist_in_fov)
+ ty = torch.randn(1) * 0.8
+ ty = torch.clamp(ty, -max_dist_in_fov, max_dist_in_fov)
+
+ dist = torch.tensor([tx, ty, tz], dtype=torch.float)
+ t_w2c = dist - torch.matmul(R_w2c, w_root)
+
+ return R_w2c, t_w2c
+
+
+def create_rotation_move(R, length, r_xyz_w_std=[np.pi / 8, np.pi / 4, np.pi / 8]):
+ """Create rotational move for the camera
+ Args:
+ R: (3, 3)
+ Return:
+ R_move: (L, 3, 3)
+ """
+ # Create final camera pose
+ assert len(R.size()) == 2
+ r_xyz = (2 * rand(3) - 1) * r_xyz_w_std
+ Rf = R @ axis_angle_to_matrix(torch.from_numpy(r_xyz).float())
+
+ # Inbetweening two poses
+ Rs = torch.stack((R, Rf)) # (2, 3, 3)
+ rs = matrix_to_rotation_6d(Rs).numpy() # (2, 6)
+ rs_move = noisy_interpolation(rs, length) # (L, 6)
+ R_move = rotation_6d_to_matrix(torch.from_numpy(rs_move).float())
+
+ return R_move
+
+
+def create_translation_move(R_w2c, t_w2c, length, t_xyz_w_std=[1.0, 0.25, 1.0]):
+ """Create translational move for the camera
+ Args:
+ R_w2c: (3, 3),
+ t_w2c: (3,),
+ """
+ # Create subject final displacement
+ subj_start_final = np.array([[0, 0, 0], randn(3) * t_xyz_w_std])
+ subj_move = noisy_interpolation(subj_start_final, length)
+ subj_move = torch.from_numpy(subj_move).float() # (L, 3)
+
+ # Equal to camera move
+ t_move = t_w2c + torch.einsum("ij,lj->li", R_w2c, subj_move)
+
+ return t_move
+
+
+class CameraAugmentorV11:
+ cfg_create_camera = {
+ "pitch_mean": np.pi / 36,
+ "pitch_std": np.pi / 8,
+ "roll_std": np.pi / 24,
+ "tz_range1_prob": 0.4,
+ "tz_range1": [1.0, 6.0], # uniform sample
+ "tz_range2": [4.0, 12.0],
+ "tx_scale": 0.7,
+ "ty_scale": 0.3,
+ }
+
+ # r_xyz_w_std = [np.pi / 8, np.pi / 4, np.pi / 8] # in world coords
+ r_xyz_w_std = [np.pi / 6, np.pi / 3, np.pi / 6] # in world coords
+ t_xyz_w_std = [1.0, 0.25, 1.0] # in world coords
+ r_xyz_w_std_half = [x / 2 for x in r_xyz_w_std]
+ t_xyz_w_std_half = [x / 2 for x in t_xyz_w_std]
+
+ t_factor = 1.0
+ tz_bias_factor = 1.0
+
+ rotx_impluse_noise = np.pi / 36
+ roty_impluse_noise = np.pi / 36
+ rotz_impluse_noise = np.pi / 36
+ rot_impluse_n = 1
+
+ tx_step_noise = 0.0025
+ ty_step_noise = 0.0025
+ tz_step_noise = 0.0025
+
+ tx_impluse_noise = 0.15
+ ty_impluse_noise = 0.15
+ tz_impluse_noise = 0.15
+ t_impluse_n = 1
+
+ # === Postprocess === #
+ height_max = 4.0
+ height_min = -2.0 # -1.5 -> -2.0 allow look upside
+ tz_post_min = 0.5
+
+ def __init__(self):
+ self.w = 1000
+ self.f = create_camera_sensor(1000, 1000, 24)[2][0, 0] # use 24mm camera
+ self.half_fov_tol = (self.w / 2) / self.f
+
+ def create_rotation_track(
+ self, cam_mat, root, rx_factor=1.0, ry_factor=1.0, rz_factor=1.0
+ ):
+ """Create rotational move for the camera with rotating human"""
+ human_mat = matrix.get_TRS(matrix.identity_mat()[None, :3, :3], root)
+ cam2human_mat = matrix.get_mat_BtoA(human_mat, cam_mat)
+ R = matrix.get_rotation(cam2human_mat)
+
+ # Create final camera pose
+ yaw = np.random.normal(scale=ry_factor)
+ pitch = np.random.normal(scale=rx_factor)
+ roll = np.random.normal(scale=rz_factor)
+
+ yaw_rm = axis_angle_to_matrix(torch.tensor([0, yaw, 0]).float())
+ pitch_rm = axis_angle_to_matrix(torch.tensor([pitch, 0, 0]).float())
+ roll_rm = axis_angle_to_matrix(torch.tensor([0, 0, roll]).float())
+ Rf = roll_rm @ pitch_rm @ yaw_rm @ R[0]
+
+ # Inbetweening two poses
+ Rs = torch.stack((R[0], Rf))
+ rs = matrix_to_rotation_6d(Rs).numpy()
+ rs_move = noisy_interpolation(rs, self.l)
+ R_move = rotation_6d_to_matrix(torch.from_numpy(rs_move).float())
+ R_move = torch.inverse(R_move)
+ return R_move
+
+ def create_translation_track(self, cam_mat, root, t_factor=1.0, tz_bias_factor=0.0):
+ """Create translational move for the camera with tracking human"""
+ delta_T0 = matrix.get_position(cam_mat)[0] - root[0]
+ T_new = matrix.get_position(cam_mat)
+
+ tz_bias = (
+ delta_T0.norm(dim=-1)
+ * tz_bias_factor
+ * np.clip(1 + np.random.normal(scale=0.1), 0.67, 1.5)
+ )
+
+ T_new[1:] = root[1:] + delta_T0
+ cam_mat = matrix.get_TRS(matrix.get_rotation(cam_mat), T_new)
+ w2c = torch.inverse(cam_mat)
+ T_new = matrix.get_position(w2c)
+
+ # Create final camera position
+ tx = np.random.normal(scale=t_factor)
+ ty = np.random.normal(scale=t_factor)
+ tz = np.random.normal(scale=t_factor) + tz_bias
+ Ts = np.array([[0, 0, 0], [tx, ty, tz]])
+
+ T_move = noisy_interpolation(Ts, self.l)
+ T_move = torch.from_numpy(T_move).float()
+ return T_move + T_new
+
+ def add_stepnoise(self, R, T):
+ w2c = matrix.get_TRS(R, T)
+ cam_mat = torch.inverse(w2c)
+ R_new = matrix.get_rotation(cam_mat)
+ T_new = matrix.get_position(cam_mat)
+
+ L = R_new.shape[0]
+ window = 10
+
+ def add_impulse_rot(R_new):
+ N = np.random.randint(1, self.rot_impluse_n + 1)
+ rx = np.random.normal(scale=self.rotx_impluse_noise, size=N)
+ ry = np.random.normal(scale=self.roty_impluse_noise, size=N)
+ rz = np.random.normal(scale=self.rotz_impluse_noise, size=N)
+ R_impluse_noise = axis_angle_to_matrix(
+ torch.from_numpy(np.array([rx, ry, rz])).float().transpose(0, 1)
+ )
+ R_noise = R_new.clone()
+ last_i = 0
+ for i in range(N):
+ n_i = np.random.randint(last_i + window, L - (N - i) * window * 2)
+
+ # make impluse smooth
+ window_R = R_noise[n_i - window : n_i + window].clone()
+ window_r = matrix_to_rotation_6d(window_R).numpy()
+ impluse_R = R_impluse_noise[i] @ window_R[window]
+ window_impluse_R = window_R.clone()
+ window_impluse_R[:] = impluse_R[None]
+ window_impluse_r = matrix_to_rotation_6d(window_impluse_R).numpy()
+
+ window_new_r = noisy_impluse_interpolation(window_r, window_impluse_r)
+ window_new_R = rotation_6d_to_matrix(
+ torch.from_numpy(window_new_r).float()
+ )
+ R_noise[n_i - window : n_i + window] = window_new_R
+ last_i = n_i
+ R_new = R_noise
+ return R_new
+
+ def add_impulse_t(T_new):
+ N = np.random.randint(1, self.t_impluse_n + 1)
+ tx = np.random.normal(scale=self.tx_impluse_noise, size=N)
+ ty = np.random.normal(scale=self.ty_impluse_noise, size=N)
+ tz = np.random.normal(scale=self.tz_impluse_noise, size=N)
+ T_impluse_noise = (
+ torch.from_numpy(np.array([tx, ty, tz])).float().transpose(0, 1)
+ )
+ T_noise = T_new.clone()
+ last_i = 0
+ for i in range(N):
+ n_i = np.random.randint(last_i + window, L - N * window * 2)
+
+ # make impluse smooth
+ window_T = T_noise[n_i - window : n_i + window].clone()
+ window_impluse_T = window_T.clone()
+ window_impluse_T += T_impluse_noise[i : i + 1]
+ window_impluse_T = window_impluse_T.numpy()
+ window_T = window_T.numpy()
+
+ window_new_T = noisy_impluse_interpolation(window_T, window_impluse_T)
+ window_new_T = torch.from_numpy(window_new_T).float()
+ T_noise[n_i - window : n_i + window] = window_new_T
+ last_i = n_i
+ T_new = T_noise
+ return T_new
+
+ impulse_type_prob = {
+ "t": 0.2,
+ "r": 0.2,
+ "both": 0.1,
+ "pass": 0.5,
+ }
+ impulse_type = np.random.choice(
+ list(impulse_type_prob.keys()), p=list(impulse_type_prob.values())
+ )
+ if impulse_type == "t":
+ # impluse translation only
+ T_new = add_impulse_t(T_new)
+ elif impulse_type == "r":
+ # impluse rotation only
+ R_new = add_impulse_rot(R_new)
+ elif impulse_type == "both":
+ # impluse rotation and translation
+ R_new = add_impulse_rot(R_new)
+ T_new = add_impulse_t(T_new)
+ else:
+ assert impulse_type == "pass"
+
+ cam_mat_new = matrix.get_TRS(R_new, T_new)
+ w2c_new = torch.inverse(cam_mat_new)
+ R_new = matrix.get_rotation(w2c_new)
+ T_new = matrix.get_position(w2c_new)
+ tx = np.random.normal(scale=self.tx_step_noise, size=L)
+ ty = np.random.normal(scale=self.ty_step_noise, size=L)
+ tz = np.random.normal(scale=self.tz_step_noise, size=L)
+ T_new = T_new + torch.from_numpy(np.array([tx, ty, tz])).float().transpose(0, 1)
+
+ return R_new, T_new
+
+ def __call__(self, w_j3d, length=120, camera_type=None):
+ """
+ Args:
+ w_j3d: (L, J, 3)
+ length: scalar
+ """
+ # Check
+ self.l = length
+ assert w_j3d.size(0) == self.l, "currently, only support fixed length"
+
+ # Setup
+ w_j3d = w_j3d.clone()
+ w_root = w_j3d[:, 0] # (L, 3)
+
+ # Simulate a static camera pose
+ cfg_camera0 = {**self.cfg_create_camera, "w": self.w, "f": self.f}
+ R0_w2c, t0_w2c = create_camera(w_root[0], cfg_camera0) # (3, 3) and (3,)
+
+ # Move camera
+ camera_type_prob = {
+ "random": 0.25,
+ "track": 0.15,
+ "trackrotate": 0.10,
+ "trackpush": 0.05,
+ "trackpull": 0.05,
+ "static": 0.4,
+ }
+ if camera_type is None:
+ camera_type = np.random.choice(
+ list(camera_type_prob.keys()), p=list(camera_type_prob.values())
+ )
+ if camera_type == "random": # random move + add noise on cam
+ R_w2c = create_rotation_move(R0_w2c, length, self.r_xyz_w_std)
+ t_w2c = create_translation_move(R0_w2c, t0_w2c, length, self.t_xyz_w_std)
+ R_w2c, t_w2c = self.add_stepnoise(R_w2c, t_w2c)
+
+ elif camera_type == "track": # track human
+ R_w2c = create_rotation_move(R0_w2c, length, self.r_xyz_w_std_half)
+ cam_mat = torch.inverse(transform_mat(R0_w2c, t0_w2c)).repeat(
+ length, 1, 1
+ ) # (F, 4, 4)
+ t_w2c = self.create_translation_track(cam_mat, w_root, 0.5)
+ R_w2c, t_w2c = self.add_stepnoise(R_w2c, t_w2c)
+
+ elif camera_type == "trackrotate": # track human and rotate
+ cam_mat = torch.inverse(transform_mat(R0_w2c, t0_w2c)).repeat(
+ length, 1, 1
+ ) # (F, 4, 4)
+ t_w2c = self.create_translation_track(cam_mat, w_root, 0.5)
+ cam_mat = matrix.get_TRS(matrix.get_rotation(cam_mat), t_w2c)
+ R_w2c = self.create_rotation_track(
+ cam_mat, w_root, np.pi / 16, np.pi, np.pi / 16
+ )
+ R_w2c, t_w2c = self.add_stepnoise(R_w2c, t_w2c)
+
+ elif camera_type == "trackpush": # track human and push close to human
+ R_w2c = create_rotation_move(R0_w2c, length, self.r_xyz_w_std_half)
+ # [1/tz_bias_factor, 1] * dist
+ cam_mat = torch.inverse(transform_mat(R0_w2c, t0_w2c)).repeat(
+ length, 1, 1
+ ) # (F, 4, 4)
+ t_w2c = self.create_translation_track(
+ cam_mat, w_root, 0.5, (1.0 / (1 + self.tz_bias_factor) - 1)
+ )
+ R_w2c, t_w2c = self.add_stepnoise(R_w2c, t_w2c)
+
+ elif camera_type == "trackpull": # track human and pull far from human
+ R_w2c = create_rotation_move(R0_w2c, length, self.r_xyz_w_std_half)
+ # [1, (tz_bias_factor + 1)] * dist
+ cam_mat = torch.inverse(transform_mat(R0_w2c, t0_w2c)).repeat(
+ length, 1, 1
+ ) # (F, 4, 4)
+ t_w2c = self.create_translation_track(
+ cam_mat, w_root, 0.5, self.tz_bias_factor
+ )
+ R_w2c, t_w2c = self.add_stepnoise(R_w2c, t_w2c)
+
+ else:
+ assert camera_type == "static"
+ R_w2c = R0_w2c.repeat(length, 1, 1) # (F, 3, 3)
+ t_w2c = t0_w2c.repeat(length, 1) # (F, 3)
+
+ # Recompute t_w2c for better camera height
+ # cam_w = torch.einsum("lji,lj->li", R_w2c, -t_w2c) # (L, 3), camera center in world: cam_w = - R_w2c^t_w2c @ t
+ # height = cam_w[..., 1] - w_root[:, 1]
+ # height = torch.clamp(height, self.height_min, self.height_max)
+ # new_pos = cam_w.clone()
+ # new_pos[:, 1] = w_root[:, 1] + height
+ # t_w2c = torch.einsum("lij,lj->li", R_w2c, -new_pos) # (L, 3), new t = -R_w2c @ cam_w
+
+ # Recompute t_w2c for better depth and FoV
+ c_j3d = torch.einsum("lij,lkj->lki", R_w2c, w_j3d) + t_w2c[:, None] # (L, J, 3)
+ delta = torch.zeros_like(t_w2c) # (L, 3) this will be later added to t_w2c
+ # - If the person is too close to the camera, push away the person in the z direction
+ c_j3d_min = c_j3d[..., 2].min() # scalar
+ if c_j3d_min < self.tz_post_min:
+ push_away = self.tz_post_min - c_j3d_min
+ delta[..., 2] += push_away
+ c_j3d[..., 2] += push_away
+ # - If the person is not in the FoV, push away the person in the z direction
+ c_root = c_j3d[:, 0] # (L, 3)
+ half_fov = torch.div(c_root[:, :2], c_root[:, 2:]).abs() # (L, 2), [x/z, y/z]
+ if half_fov.max() > self.half_fov_tol:
+ max_idx1, max_idx2 = torch.where(torch.max(half_fov) == half_fov)
+ max_idx1, max_idx2 = max_idx1[0], max_idx2[0]
+ z_trg = (
+ c_root[max_idx1, max_idx2].abs() / self.half_fov_tol
+ ) # extreme fitted z in the fov
+ push_away = z_trg - c_root[max_idx1, 2]
+ delta[..., 2] += push_away
+ t_w2c += delta
+
+ T_w2c = transform_mat(R_w2c, t_w2c) # (F, 4, 4)
+ return T_w2c
diff --git a/genmo/datasets/pure_motion/humanml3d.py b/genmo/datasets/pure_motion/humanml3d.py
new file mode 100644
index 0000000000000000000000000000000000000000..26dc45c4bc074f9bf58c23b59bde761f6e87315e
--- /dev/null
+++ b/genmo/datasets/pure_motion/humanml3d.py
@@ -0,0 +1,536 @@
+from pathlib import Path
+
+import numpy as np
+import torch
+
+from genmo.datasets.pure_motion.base_dataset import BaseDataset
+from genmo.datasets.pure_motion.cam_traj_utils import CameraAugmentorV11
+from genmo.datasets.pure_motion.utils import (
+ augment_betas,
+ interpolate_smpl_params,
+ pad_data,
+ rotate_around_axis,
+)
+from genmo.utils.geo_transform import (
+ compute_cam_angvel,
+ compute_cam_tvel,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import (
+ get_valid_mask,
+ repeat_to_max_len,
+ repeat_to_max_len_dict,
+)
+from genmo.utils.pylogger import Log
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import create_camera_sensor
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import (
+ get_c_rootparam,
+ get_R_c2gv,
+ get_tgtcoord_rootparam,
+)
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+from third_party.GVHMR.hmr4d.utils.wis3d_utils import add_motion_as_lines, make_wis3d
+
+
+class Humanml3dDataset(BaseDataset):
+ def __init__(
+ self,
+ mode="default",
+ motion_frames=120,
+ l_factor=1.5, # speed augmentation
+ skip_moyo=True, # not contained in the ICCV19 released version
+ cam_augmentation="v11",
+ split="train",
+ random1024=False, # DEBUG
+ limit_size=None,
+ no_subsample=False,
+ max_text_len=50,
+ part_ind=-1,
+ num_parts=-1,
+ eval_gen_only=False,
+ use_random_subset=False,
+ random_subset_size=32,
+ random_subset_seed=7,
+ use_multi_text=False,
+ num_multi_text=3,
+ multi_text_vid=None,
+ motion_start_mode="first",
+ enable_speed_aug=False,
+ eval_seed=None,
+ discard_last_frame=False,
+ ):
+ self.root = Path("inputs/HumanML3D_SMPL/hmr4d_support")
+ if split == "train":
+ self.text_embed_file = Path(
+ "inputs/HumanML3D_SMPL_ye/t5_embeddings_v1_half/all_text_embed.pth"
+ ) # TODO: USE THE STANDARD PATH
+ else:
+ self.text_embed_file = Path(
+ "inputs/HumanML3D_SMPL_ye/t5_embeddings_v1_half/test_text_embed.pth"
+ )
+ if split == "test":
+ no_subsample = True
+ self.mode = mode
+ self.motion_frames = motion_frames
+ self.l_factor = l_factor
+ self.random1024 = random1024
+ self.skip_moyo = skip_moyo
+ self.dataset_name = "HumanML3D"
+ self.smplx_neutral = make_smplx(type="supermotion_smpl24")
+ self.smplx_dict = {
+ "male": self.smplx_neutral,
+ "female": self.smplx_neutral,
+ "neutral": self.smplx_neutral,
+ }
+ self.max_text_len = max_text_len
+ self.split = split
+ self.no_subsample = no_subsample
+ self.num_parts = num_parts
+ self.part_ind = part_ind
+ self.eval_gen_only = eval_gen_only
+ self.use_random_subset = use_random_subset
+ self.random_subset_seed = random_subset_seed
+ self.random_subset_size = random_subset_size
+ self.use_multi_text = use_multi_text
+ self.num_multi_text = num_multi_text
+ self.multi_text_vid = multi_text_vid
+ if self.use_multi_text:
+ self.num_multi_text = len(self.multi_text_vid)
+ self.vid_to_idx = {}
+ self.motion_start_mode = motion_start_mode
+ self.enable_speed_aug = enable_speed_aug
+ self.eval_seed = eval_seed
+ self.discard_last_frame = discard_last_frame
+ super().__init__(cam_augmentation, limit_size)
+ if self.use_multi_text:
+ for i, (vid, _) in enumerate(self.idx2meta):
+ self.vid_to_idx[vid] = i
+ return
+
+ def _load_dataset(self):
+ filename = self.root / f"humanml3d_smplhpose_{self.split}.pth"
+ Log.info(f"[{self.dataset_name}] Loading from {filename} ...")
+ tic = Log.time()
+ if self.random1024: # Debug, faster loading
+ try:
+ Log.info(
+ f"[{self.dataset_name}] Loading 1024 samples for debugging ..."
+ )
+ self.motion_files = torch.load(
+ self.root / "smplxpose_v2_random1024.pth"
+ )
+ except Exception:
+ Log.info(
+ f"[{self.dataset_name}] Not found! Saving 1024 samples for debugging ..."
+ )
+ self.motion_files = torch.load(filename)
+ keys = list(self.motion_files.keys())
+ keys = np.random.choice(keys, 1024, replace=False)
+ self.motion_files = {k: self.motion_files[k] for k in keys}
+ torch.save(
+ self.motion_files, self.root / "humanml3d_smplhpose_random1024.pth"
+ )
+ else:
+ self.motion_files = torch.load(filename)
+ self.text_embed_dict = torch.load(self.text_embed_file)
+ self.seqs = list(self.motion_files.keys())
+ Log.info(
+ f"[{self.dataset_name}] {len(self.seqs)} sequences. Elapsed: {Log.time() - tic:.2f}s"
+ )
+
+ def _get_idx2meta(self):
+ # We expect to see the entire sequence during one epoch,
+ # so each sequence will be sampled max(SeqLength // MotionFrames, 1) times
+ seq_lengths = []
+ self.idx2meta = []
+
+ # Skip too-long idle-prefix
+ motion_start_id = {}
+ for vid in self.motion_files:
+ # test_data = self.motion_files[vid]['text_data']
+ # print(vid, test_data[0]['caption'])
+ seq_length = self.motion_files[vid]["pose"].shape[0]
+ start_id = motion_start_id[vid] if vid in motion_start_id else 0
+ seq_length = seq_length - start_id
+ if (
+ seq_length < 25 and not self.no_subsample
+ ): # Skip clips that are too short
+ continue
+ num_samples = max(seq_length // self.motion_frames, 1)
+ if self.use_random_subset or self.no_subsample:
+ num_samples = 1
+ seq_lengths.append(seq_length)
+ self.idx2meta.extend([(vid, start_id)] * num_samples)
+ assert start_id == 0, f"start_id is not 0 for {vid}"
+
+ if self.num_parts > 0:
+ part_size = len(self.idx2meta) // self.num_parts
+ start_idx = self.part_ind * part_size
+ end_idx = (self.part_ind + 1) * part_size
+ self.idx2meta = self.idx2meta[start_idx:end_idx]
+ seq_lengths = seq_lengths[start_idx:end_idx]
+
+ if self.use_random_subset:
+ self.rng = np.random.RandomState(self.random_subset_seed)
+ shuffle_ind = np.arange(len(self.idx2meta))
+ self.rng.shuffle(shuffle_ind)
+ self.idx2meta = [
+ self.idx2meta[i] for i in shuffle_ind[: self.random_subset_size]
+ ]
+ seq_lengths = [
+ seq_lengths[i] for i in shuffle_ind[: self.random_subset_size]
+ ]
+ else:
+ self.rng = np.random
+ hours = sum(seq_lengths) / 30 / 3600
+ Log.info(
+ f"[{self.dataset_name}] has {hours:.1f} hours motion -> Resampled to {len(self.idx2meta)} samples."
+ )
+
+ def _load_data(self, idx):
+ """
+ - Load original data
+ - Augmentation: speed-augmentation to L frames
+ """
+ # Load original data
+ mid, start_id = self.idx2meta[idx]
+ raw_data = self.motion_files[mid]
+ text_embed_data = self.text_embed_dict[mid].float()
+
+ raw_len = raw_data["pose"].shape[0] - start_id
+ offset = 1 if self.discard_last_frame else 0
+ raw_len -= offset
+ data = {
+ "body_pose": raw_data["pose"][start_id : start_id + raw_len, 3:], # (F, 63)
+ "betas": raw_data["beta"].repeat(raw_len, 1), # (10)
+ "global_orient": raw_data["pose"][
+ start_id : start_id + raw_len, :3
+ ], # (F, 3)
+ "transl": raw_data["trans"][start_id : start_id + raw_len], # (F, 3)
+ }
+
+ # Get {tgt_len} frames from data
+ # Random select a subset with speed augmentation [start, end)
+ tgt_len = self.motion_frames
+ if self.enable_speed_aug:
+ raw_subset_len = self.rng.randint(
+ int(tgt_len / self.l_factor), int(tgt_len * self.l_factor)
+ )
+ else:
+ raw_subset_len = tgt_len
+ raw_subset_len = min(raw_subset_len, raw_len)
+ if raw_subset_len <= raw_len:
+ start = self.rng.randint(0, raw_len - raw_subset_len + 1)
+ end = start + raw_subset_len
+ else: # interpolation will use all possible frames (results in a slow motion)
+ start = 0
+ end = raw_len
+ data = {k: v[start:end] for k, v in data.items()}
+
+ # Interpolation (vec + r6d)
+ if self.enable_speed_aug:
+ data_interpolated = interpolate_smpl_params(data, tgt_len)
+ valid_length = tgt_len
+ else:
+ data_interpolated = data
+ if raw_subset_len < tgt_len:
+ data_interpolated = pad_data(data, tgt_len)
+ valid_length = raw_subset_len
+
+ # text_data = self.rng.choice(raw_data["text_data"])
+ if self.use_random_subset or self.eval_gen_only:
+ text_ind = 0
+ else:
+ text_ind = self.rng.randint(0, len(raw_data["text_data"]))
+ text_data = raw_data["text_data"][text_ind]
+ text_embed = text_embed_data[text_ind]
+
+ caption, tokens = text_data["caption"], text_data["tokens"]
+
+ if len(tokens) < self.max_text_len:
+ # pad with "unk"
+ tokens = ["sos/OTHER"] + tokens + ["eos/OTHER"]
+ sent_len = len(tokens)
+ tokens = tokens + ["unk/OTHER"] * (self.max_text_len + 2 - sent_len)
+ else:
+ # crop
+ tokens = tokens[: self.max_text_len]
+ tokens = ["sos/OTHER"] + tokens + ["eos/OTHER"]
+ sent_len = len(tokens)
+
+ # AZ -> AY
+ data_interpolated["global_orient"], data_interpolated["transl"], _ = (
+ get_tgtcoord_rootparam(
+ data_interpolated["global_orient"],
+ data_interpolated["transl"],
+ tsf="az->ay",
+ )
+ )
+
+ data_interpolated["data_name"] = "humanml3d"
+ data_interpolated["mid"] = mid
+ data_interpolated["text_ind"] = text_ind
+ data_interpolated["gender"] = raw_data["gender"]
+ data_interpolated["caption"] = caption
+ data_interpolated["text_embed"] = text_embed
+ data_interpolated["valid_length"] = valid_length
+
+ return data_interpolated
+
+ def _process_data(self, data, idx):
+ """
+ Args:
+ data: dict {
+ "body_pose": (F, 63),
+ "betas": (F, 10),
+ "global_orient": (F, 3), in the AY coordinates
+ "transl": (F, 3), in the AY coordinates
+ }
+ """
+ if self.motion_start_mode == "sample":
+ mlength = data["body_pose"].shape[0]
+ length = min(self.motion_frames, mlength)
+ start = np.random.randint(0, max(mlength - length + 1, 1))
+ for k, v in data.items():
+ if v in ["body_pose", "betas", "global_orient", "transl"]:
+ data[k] = v[start:]
+
+ data_name = data["data_name"]
+ mid = data["mid"]
+ text_ind = data["text_ind"]
+ length = data["body_pose"].shape[0]
+ valid_length = data["valid_length"]
+ # Augmentation: betas, SMPL (gravity-axis)
+ gender = str(data["gender"])
+ body_pose = data["body_pose"]
+ betas = augment_betas(data["betas"], std=0.1)
+ global_orient_w, transl_w = rotate_around_axis(
+ data["global_orient"], data["transl"], axis="y"
+ )
+ caption = data["caption"]
+ text_embed = data["text_embed"]
+
+ del data
+
+ # SMPL_params in world
+ smpl_params_w = {
+ "body_pose": body_pose, # (F, 63)
+ "betas": betas, # (F, 10)
+ "global_orient": global_orient_w, # (F, 3)
+ "transl": transl_w, # (F, 3)
+ }
+
+ # Camera trajectory augmentation
+ if self.cam_augmentation == "v11":
+ # interleave repeat to original length (faster)
+ N = 10
+ smpl_layer = self.smplx_dict[gender]
+ w_j3d = smpl_layer(
+ smpl_params_w["body_pose"][::N],
+ smpl_params_w["betas"][::N],
+ smpl_params_w["global_orient"][::N],
+ None,
+ )
+ w_j3d = (
+ w_j3d.repeat_interleave(N, dim=0)[:length]
+ + smpl_params_w["transl"][:, None]
+ ) # (F, 24, 3)
+
+ if False:
+ wis3d = make_wis3d(name="debug_amass")
+ add_motion_as_lines(w_j3d, wis3d, "w_j3d")
+
+ width, height, K_fullimg = create_camera_sensor(1000, 1000, 43.3) # WHAM
+ # focal_length = K_fullimg[0, 0]
+ wham_cam_augmentor = CameraAugmentorV11()
+ T_w2c = wham_cam_augmentor(w_j3d, length) # (F, 4, 4)
+ elif self.cam_augmentation == "static":
+ # interleave repeat to original length (faster)
+ N = 10
+ smpl_layer = self.smplx_dict[gender]
+ w_j3d = smpl_layer(
+ smpl_params_w["body_pose"][::N],
+ smpl_params_w["betas"][::N],
+ smpl_params_w["global_orient"][::N],
+ None,
+ )
+ w_j3d = (
+ w_j3d.repeat_interleave(N, dim=0)[:length]
+ + smpl_params_w["transl"][:, None]
+ ) # (F, 24, 3)
+
+ if False:
+ wis3d = make_wis3d(name="debug_amass")
+ add_motion_as_lines(w_j3d, wis3d, "w_j3d")
+
+ width, height, K_fullimg = create_camera_sensor(1000, 1000, 43.3) # WHAM
+ # focal_length = K_fullimg[0, 0]
+ wham_cam_augmentor = CameraAugmentorV11()
+ T_w2c = wham_cam_augmentor(w_j3d, length, camera_type="static") # (F, 4, 4)
+
+ else:
+ raise NotImplementedError
+
+ T_c2w = T_w2c.inverse()
+ noisy_T_c2w = T_c2w.clone()
+ # R_c2w = as_identity(T_c2w[:, :3, :3])
+ t_c2w = T_c2w[:, :3, 3]
+ rand_scale = min(max(0.1, torch.randn(1) + 3), 10)
+ noisy_t_c2w = t_c2w / rand_scale
+ noisy_T_c2w[:, :3, 3] = noisy_t_c2w
+ noisy_T_w2c = noisy_T_c2w.inverse()
+ del noisy_T_c2w
+
+ normed_noisy_T_w2c = normalize_T_w2c(noisy_T_w2c)
+ normed_T_w2c = normalize_T_w2c(T_w2c)
+ # normed_R_w2c = as_identity(normed_T_w2c[:, :3, :3])
+ # normed_t_w2c = normed_T_w2c[:, :3, 3]
+ # normed_noisy_R_w2c = as_identity(normed_noisy_T_w2c[:, :3, :3])
+ # normed_noisy_t_w2c = normed_noisy_T_w2c[:, :3, 3]
+ del noisy_T_w2c
+
+ # SMPL params in cam
+ offset = self.smplx.get_skeleton(smpl_params_w["betas"][0])[0] # (3)
+ global_orient_c, transl_c = get_c_rootparam(
+ smpl_params_w["global_orient"],
+ smpl_params_w["transl"],
+ T_w2c,
+ offset,
+ )
+ smpl_params_c = {
+ "body_pose": smpl_params_w["body_pose"].clone(), # (F, 63)
+ "betas": smpl_params_w["betas"].clone(), # (F, 10)
+ "global_orient": global_orient_c, # (F, 3)
+ "transl": transl_c, # (F, 3)
+ }
+
+ # World params
+ gravity_vec = torch.tensor([0, -1, 0], dtype=torch.float32) # (3), BEDLAM is ay
+ R_c2gv = get_R_c2gv(T_w2c[:, :3, :3], gravity_vec) # (F, 3, 3)
+
+ # Image
+ K_fullimg = K_fullimg.repeat(length, 1, 1) # (F, 3, 3)
+ cam_angvel = compute_cam_angvel(normed_T_w2c[:, :3, :3]) # (F, 6)
+ cam_tvel = compute_cam_tvel(normed_T_w2c[:, :3, 3]) # (F, 3)
+ noisy_cam_tvel = compute_cam_tvel(normed_noisy_T_w2c[:, :3, 3]) # (F, 3)
+
+ # Returns: do not forget to make it batchable! (last lines)
+ # NOTE: bbx_xys and f_imgseq will be added later
+ max_len = length
+ if self.eval_gen_only:
+ valid_length = length
+
+ return_data = {
+ "meta": {
+ "data_name": data_name,
+ "dataset_id": "humanml3d",
+ "idx": idx,
+ "T_w2c": T_w2c,
+ "eval_gen_only": self.eval_gen_only,
+ "mid": mid,
+ "text_ind": text_ind,
+ "mode": self.mode,
+ "eval_seed": self.eval_seed,
+ },
+ "length": valid_length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w,
+ "R_c2gv": R_c2gv, # (F, 3, 3)
+ "gravity_vec": gravity_vec, # (3)
+ "bbx_xys": torch.zeros((length, 3)), # (F, 3) # NOTE: a placeholder
+ "K_fullimg": K_fullimg, # (F, 3, 3)
+ "f_imgseq": torch.zeros((length, 1024)), # (F, D) # NOTE: a placeholder
+ "kp2d": torch.zeros(length, 17, 3), # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "cam_tvel": cam_tvel, # (F, 3),
+ "noisy_cam_tvel": noisy_cam_tvel, # (F, 3),
+ "T_w2c": normed_T_w2c,
+ "caption": caption,
+ "has_text": caption != "",
+ "text_embed": text_embed,
+ "mask": {
+ "valid": get_valid_mask(length, valid_length),
+ "has_img_mask": get_valid_mask(length, 0),
+ "has_2d_mask": get_valid_mask(length, valid_length),
+ "has_cam_mask": get_valid_mask(length, valid_length),
+ "has_audio_mask": get_valid_mask(length, 0),
+ "has_music_mask": get_valid_mask(length, 0),
+ "2d_only": False,
+ "vitpose": False,
+ "bbx_xys": False,
+ "f_imgseq": False,
+ "spv_incam_only": False,
+ "invalid_contact": False,
+ },
+ }
+
+ # Batchable
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(
+ return_data["smpl_params_c"], max_len
+ )
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(
+ return_data["smpl_params_w"], max_len
+ )
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(return_data["K_fullimg"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(
+ return_data["cam_angvel"], max_len
+ )
+ return_data["cam_tvel"] = repeat_to_max_len(return_data["cam_tvel"], max_len)
+ return_data["noisy_cam_tvel"] = repeat_to_max_len(
+ return_data["noisy_cam_tvel"], max_len
+ )
+ return_data["T_w2c"] = repeat_to_max_len(return_data["T_w2c"], max_len)
+ return return_data
+
+ def __getitem__(self, idx):
+ if self.multi_text_vid is not None:
+ idx = self.vid_to_idx[self.multi_text_vid[0]]
+ data = self._load_data(idx)
+ data = self._process_data(data, idx)
+ if self.use_multi_text:
+ all_data = [data]
+ for i in range(1, self.num_multi_text):
+ if self.multi_text_vid is not None:
+ new_idx = self.vid_to_idx[self.multi_text_vid[i]]
+ else:
+ new_idx = self.rng.randint(0, len(self))
+ data_i = self._load_data(new_idx)
+ data_i = self._process_data(data_i, new_idx)
+ all_data.append(data_i)
+ multi_text_data = {
+ "vid": [],
+ "caption": [],
+ "text_ind": [],
+ "text_embed": [],
+ "window_start": [],
+ "window_end": [],
+ }
+ window_stride = 1 / self.num_multi_text
+ for i, data_i in enumerate(all_data):
+ multi_text_data["vid"].append(data_i["meta"]["mid"])
+ multi_text_data["caption"].append(data_i["caption"])
+ multi_text_data["text_ind"].append(data_i["meta"]["text_ind"])
+ multi_text_data["text_embed"].append(data_i["text_embed"])
+ window_start = i * window_stride
+ window_end = (i + 1) * window_stride
+ # window_start = i * 100/720
+ # window_end = (i + 1) * 100/720
+ # if i == len(all_data) - 1:
+ # window_end = 1
+ multi_text_data["window_start"].append(window_start)
+ multi_text_data["window_end"].append(window_end)
+ # multi_text_data["window_end"][0] = 120/600
+ # multi_text_data["window_end"][1] = 300/600
+ # multi_text_data["window_start"][1] = 120/600
+ # multi_text_data["window_start"][2] = 300/600
+ multi_text_data["text_embed"] = torch.stack(multi_text_data["text_embed"])
+ multi_text_data["window_start"] = torch.tensor(
+ multi_text_data["window_start"]
+ )
+ multi_text_data["window_end"] = torch.tensor(multi_text_data["window_end"])
+ print("vid & captions:")
+ print(multi_text_data["vid"])
+ print(multi_text_data["caption"])
+ data["meta"]["multi_text_data"] = multi_text_data
+ return data
diff --git a/genmo/datasets/pure_motion/utils.py b/genmo/datasets/pure_motion/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..ac2c65bffdbf84bc8fff6c49a9ac4e3109950369
--- /dev/null
+++ b/genmo/datasets/pure_motion/utils.py
@@ -0,0 +1,90 @@
+import torch
+import torch.nn.functional as F
+from einops import rearrange
+
+from genmo.utils.rotation_conversions import (
+ axis_angle_to_matrix,
+ matrix_to_axis_angle,
+ matrix_to_rotation_6d,
+ rotation_6d_to_matrix,
+)
+
+
+def aa_to_r6d(x):
+ return matrix_to_rotation_6d(axis_angle_to_matrix(x))
+
+
+def r6d_to_aa(x):
+ return matrix_to_axis_angle(rotation_6d_to_matrix(x))
+
+
+def interpolate_smpl_params(smpl_params, tgt_len):
+ """
+ smpl_params['body_pose'] (L, 63)
+ tgt_len: L->L'
+ """
+ betas = smpl_params["betas"]
+ body_pose = smpl_params["body_pose"]
+ global_orient = smpl_params["global_orient"] # (L, 3)
+ transl = smpl_params["transl"] # (L, 3)
+
+ # Interpolate
+ body_pose = rearrange(aa_to_r6d(body_pose.reshape(-1, 21, 3)), "l j c -> c j l")
+ body_pose = F.interpolate(body_pose, tgt_len, mode="linear", align_corners=True)
+ body_pose = r6d_to_aa(rearrange(body_pose, "c j l -> l j c")).reshape(-1, 63)
+
+ # although this should be the same as above, we do it for consistency
+ betas = rearrange(betas, "l c -> c 1 l")
+ betas = F.interpolate(betas, tgt_len, mode="linear", align_corners=True)
+ betas = rearrange(betas, "c 1 l -> l c")
+
+ global_orient = rearrange(
+ aa_to_r6d(global_orient.reshape(-1, 1, 3)), "l j c -> c j l"
+ )
+ global_orient = F.interpolate(
+ global_orient, tgt_len, mode="linear", align_corners=True
+ )
+ global_orient = r6d_to_aa(rearrange(global_orient, "c j l -> l j c")).reshape(-1, 3)
+
+ transl = rearrange(transl, "l c -> c 1 l")
+ transl = F.interpolate(transl, tgt_len, mode="linear", align_corners=True)
+ transl = rearrange(transl, "c 1 l -> l c")
+
+ return {
+ "body_pose": body_pose,
+ "betas": betas,
+ "global_orient": global_orient,
+ "transl": transl,
+ }
+
+
+def pad_data(smpl_params, tgt_len):
+ for key in smpl_params.keys():
+ smpl_params[key] = torch.cat(
+ [
+ smpl_params[key],
+ torch.zeros(
+ tgt_len - smpl_params[key].shape[0], *smpl_params[key].shape[1:]
+ ).to(smpl_params[key]),
+ ],
+ dim=0,
+ )
+ return smpl_params
+
+
+def rotate_around_axis(global_orient, transl, axis="y"):
+ """Global coordinate augmentation. Random rotation around y-axis"""
+ angle = torch.rand(1) * 2 * torch.pi
+ if axis == "y":
+ aa = torch.tensor([0.0, angle, 0.0]).float().unsqueeze(0)
+ rmat = axis_angle_to_matrix(aa)
+
+ global_orient = matrix_to_axis_angle(rmat @ axis_angle_to_matrix(global_orient))
+ transl = (rmat.squeeze(0) @ transl.T).T
+ return global_orient, transl
+
+
+def augment_betas(betas, std=0.1):
+ noise = torch.normal(mean=torch.zeros(10), std=torch.ones(10) * std)
+ betas_aug = betas + noise[None]
+ return betas_aug
diff --git a/genmo/datasets/rich/resource/cam2params.pt b/genmo/datasets/rich/resource/cam2params.pt
new file mode 100644
index 0000000000000000000000000000000000000000..ed6da4575844b39ae17ef3118421c9166bfc3814
--- /dev/null
+++ b/genmo/datasets/rich/resource/cam2params.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e260ae6f8f65e8e0a56af2a1130181ec663f3a382147984272111aec74f7c260
+size 26211
diff --git a/genmo/datasets/rich/resource/seqname2imgrange.json b/genmo/datasets/rich/resource/seqname2imgrange.json
new file mode 100644
index 0000000000000000000000000000000000000000..1abefb9dd5d1e69478620c8ffe10e8de29be8196
--- /dev/null
+++ b/genmo/datasets/rich/resource/seqname2imgrange.json
@@ -0,0 +1 @@
+{"ParkingLot1_002_burpee3": [1, 351], "ParkingLot1_002_overfence1": [1, 268], "ParkingLot1_002_overfence2": [1, 270], "ParkingLot1_002_stretching1": [1, 327], "ParkingLot1_002_pushup1": [1, 220], "ParkingLot1_004_pushup2": [1, 347], "ParkingLot1_004_burpeejump1": [1, 296], "ParkingLot1_004_eating1": [1, 522], "ParkingLot1_004_takingphotos1": [1, 593], "ParkingLot1_004_phonetalk1": [1, 724], "ParkingLot1_005_burpeejump2": [1, 270], "ParkingLot1_005_overfence1": [1, 301], "ParkingLot1_005_pushup2": [1, 262], "ParkingLot1_005_pushup3": [1, 243], "ParkingLot1_004_005_greetingchattingeating1": [275, 849], "ParkingLot1_007_overfence2": [1, 263], "ParkingLot1_007_eating1": [1, 426], "ParkingLot1_007_eating2": [1, 498], "ParkingLot2_008_phonetalk1": [171, 1215], "ParkingLot2_008_burpeejump1": [78, 505], "ParkingLot2_008_overfence1": [161, 459], "ParkingLot2_008_pushup1": [165, 459], "ParkingLot2_008_pushup2": [107, 719], "ParkingLot2_008_overfence2": [138, 632], "ParkingLot2_008_overfence3": [100, 661], "ParkingLot2_008_eating1": [180, 1332], "ParkingLot2_014_pushup2": [80, 420], "ParkingLot2_014_burpeejump1": [50, 348], "ParkingLot2_014_burpeejump2": [50, 248], "ParkingLot2_014_phonetalk2": [121, 1141], "ParkingLot2_014_takingphotos2": [91, 906], "ParkingLot2_014_overfence3": [40, 502], "ParkingLot2_015_overfence1": [170, 692], "ParkingLot2_015_burpeejump2": [344, 678], "ParkingLot2_015_pushup1": [190, 817], "ParkingLot2_015_eating2": [31, 835], "ParkingLot2_016_burpeejump2": [100, 793], "ParkingLot2_016_overfence2": [100, 720], "ParkingLot2_016_pushup1": [61, 680], "ParkingLot2_016_pushup2": [100, 570], "ParkingLot2_016_stretching1": [100, 691], "Pavallion_000_yoga2": [1, 1643], "Pavallion_000_plankjack": [1, 900], "Pavallion_000_phonesiteat": [1, 1157], "Pavallion_000_sidebalancerun": [1, 1091], "Pavallion_002_plankjack": [110, 699], "Pavallion_002_phonesiteat": [1, 1030], "Pavallion_003_plankjack": [1, 764], "Pavallion_003_phonesiteat": [75, 838], "Pavallion_003_sidebalancerun": [1, 942], "Pavallion_006_phonesiteat": [130, 841], "Pavallion_006_sidebalancerun": [1, 798], "Pavallion_006_plankjack": [1, 615], "Pavallion_013_phonesiteat": [1, 1254], "Pavallion_013_plankjack": [1, 641], "Pavallion_013_yoga2": [1, 884], "Pavallion_003_018_tossball": [230, 949], "LectureHall_018_wipingchairs1": [1, 1166], "LectureHall_018_wipingspray1": [1, 904], "LectureHall_020_wipingtable1": [1, 897], "BBQ_001_juggle": [0, 297], "BBQ_001_guitar": [0, 381], "ParkingLot1_002_stretching2": [240, 240], "ParkingLot1_002_burpee1": [1, 286], "ParkingLot1_002_burpee2": [1, 203], "ParkingLot1_004_pushup1": [1, 354], "ParkingLot1_004_eating2": [1, 516], "ParkingLot1_004_phonetalk2": [1, 960], "ParkingLot1_004_takingphotos2": [1, 571], "ParkingLot1_004_stretching2": [1, 399], "ParkingLot1_005_overfence2": [1, 298], "ParkingLot1_005_pushup1": [1, 476], "ParkingLot1_005_burpeejump1": [1, 252], "ParkingLot1_007_burpee2": [1, 349], "ParkingLot2_008_eating2": [160, 1100], "ParkingLot2_008_burpeejump2": [129, 492], "ParkingLot2_014_overfence1": [95, 547], "ParkingLot2_014_eating2": [101, 986], "ParkingLot2_016_phonetalk5": [170, 1259], "Pavallion_002_sidebalancerun": [1, 655], "Pavallion_013_sidebalancerun": [1, 810], "Pavallion_018_sidebalancerun": [1, 873], "LectureHall_018_wipingtable1": [1, 1280], "LectureHall_020_wipingchairs1": [1, 1163], "LectureHall_003_wipingchairs1": [1, 724], "Pavallion_000_yoga1": [1, 1757], "Pavallion_002_yoga1": [1, 613], "Pavallion_003_yoga1": [1, 792], "Pavallion_006_yoga1": [1, 930], "Pavallion_018_yoga1": [1, 880], "ParkingLot2_017_burpeejump2": [118, 612], "ParkingLot2_017_burpeejump1": [40, 817], "ParkingLot2_017_overfence1": [110, 661], "ParkingLot2_017_overfence2": [90, 944], "ParkingLot2_017_eating1": [97, 895], "ParkingLot2_017_pushup1": [191, 719], "ParkingLot2_017_pushup2": [74, 811], "ParkingLot2_009_burpeejump1": [200, 1085], "ParkingLot2_009_burpeejump2": [150, 399], "ParkingLot2_009_overfence1": [140, 601], "ParkingLot2_009_overfence2": [150, 559], "LectureHall_009_sidebalancerun1": [1, 673], "LectureHall_010_plankjack1": [1, 532], "LectureHall_010_sidebalancerun1": [1, 919], "LectureHall_021_plankjack1": [1, 507], "LectureHall_021_sidebalancerun1": [1, 855], "LectureHall_019_wipingchairs1": [1, 978], "LectureHall_009_021_reparingprojector1": [1, 499], "ParkingLot2_009_spray1": [145, 1242], "ParkingLot2_009_impro1": [100, 990], "ParkingLot2_009_impro2": [100, 1140], "ParkingLot2_009_impro5": [100, 649], "Gym_010_pushup1": [1, 475], "Gym_010_pushup2": [1, 407], "Gym_011_pushup1": [1, 346], "Gym_011_pushup2": [1, 540], "Gym_011_burpee2": [1, 479], "Gym_012_pushup2": [1, 291], "Gym_010_mountainclimber1": [0, 0], "Gym_010_mountainclimber2": [1, 471], "Gym_013_dips1": [1, 503], "Gym_013_dips2": [1, 333], "Gym_013_dips3": [1, 502], "Gym_013_lunge1": [1, 690], "Gym_013_lunge2": [1, 834], "Gym_013_pushup1": [1, 861], "Gym_013_pushup2": [1, 477], "Gym_013_burpee4": [1, 320], "Gym_010_lunge1": [1, 337], "Gym_010_lunge2": [1, 312], "Gym_010_dips1": [1, 572], "Gym_010_dips2": [1, 603], "Gym_010_cooking1": [1, 779], "Gym_011_cooking1": [1, 1141], "Gym_011_cooking2": [1, 1145], "Gym_011_dips1": [1, 494], "Gym_011_dips4": [1, 495], "Gym_011_dips3": [1, 320], "Gym_011_dips2": [1, 382], "Gym_012_lunge1": [1, 225], "Gym_012_lunge2": [1, 318], "Gym_012_cooking2": [1, 993]}
diff --git a/genmo/datasets/rich/resource/test.txt b/genmo/datasets/rich/resource/test.txt
new file mode 100644
index 0000000000000000000000000000000000000000..5a5176a0cfbac48948ce047e78a962ccf9f4492f
--- /dev/null
+++ b/genmo/datasets/rich/resource/test.txt
@@ -0,0 +1,54 @@
+sequence_name capture_name scan_name id moving_cam gender scene action/scene-interaction subjects view_id
+ParkingLot2_017_burpeejump2 ParkingLot2 scan_camcoord 017 V female V V X 0,2,3
+ParkingLot2_017_burpeejump1 ParkingLot2 scan_camcoord 017 V female V V X 0,1,5
+ParkingLot2_017_overfence1 ParkingLot2 scan_camcoord 017 V female V V X 0,3,4
+ParkingLot2_017_overfence2 ParkingLot2 scan_camcoord 017 V female V V X 0,1,4
+ParkingLot2_017_eating1 ParkingLot2 scan_camcoord 017 V female V V X 0,2,4
+ParkingLot2_017_pushup1 ParkingLot2 scan_camcoord 017 X female V V X 0,1,4,5
+ParkingLot2_017_pushup2 ParkingLot2 scan_camcoord 017 V female V V X 0,4,5
+ParkingLot2_009_burpeejump1 ParkingLot2 scan_camcoord 009 X female V V X 0,1,2,3
+ParkingLot2_009_burpeejump2 ParkingLot2 scan_camcoord 009 X female V V X 0,2,3,4
+ParkingLot2_009_overfence1 ParkingLot2 scan_camcoord 009 X female V V X 0,3,4,5
+ParkingLot2_009_overfence2 ParkingLot2 scan_camcoord 009 X female V V X 0,1,4,5
+LectureHall_009_sidebalancerun1 LectureHall scan_yoga_scene_camcoord 009 X female V V X 0,1,4,5
+LectureHall_010_plankjack1 LectureHall scan_yoga_scene_camcoord 010 X female V V X 0,2,4,6
+LectureHall_010_sidebalancerun1 LectureHall scan_yoga_scene_camcoord 010 X female V V X 0,1,2,4
+LectureHall_021_plankjack1 LectureHall scan_yoga_scene_camcoord 021 X female V V X 0,3,5,6
+LectureHall_021_sidebalancerun1 LectureHall scan_yoga_scene_camcoord 021 X female V V X 0,4,5,6
+LectureHall_019_wipingchairs1 LectureHall scan_chair_scene_camcoord 019 X female V V X 0,1,2,3
+LectureHall_009_021_reparingprojector1 LectureHall scan_yoga_scene_camcoord 009 X female V X X 0,3,4,5
+LectureHall_009_021_reparingprojector1 LectureHall scan_yoga_scene_camcoord 021 X female V X X 0,3,4,5
+ParkingLot2_009_spray1 ParkingLot2 scan_camcoord 009 X female V X X 0,1,2,3
+ParkingLot2_009_impro1 ParkingLot2 scan_camcoord 009 X female V X X 0,2,3,4
+ParkingLot2_009_impro2 ParkingLot2 scan_camcoord 009 X female V X X 0,3,4,5
+ParkingLot2_009_impro5 ParkingLot2 scan_camcoord 009 X female V X X 0,2,4,5
+Gym_010_pushup1 Gym scan_camcoord 010 X female X V X 3,4,5,6
+Gym_010_pushup2 Gym scan_camcoord 010 X female X V X 2,3,4,5
+Gym_011_pushup1 Gym scan_camcoord 011 X male X V X 2,3,4,5
+Gym_011_pushup2 Gym scan_camcoord 011 X male X V X 2,3,4,5
+Gym_011_burpee2 Gym scan_camcoord 011 X male X V X 2,3,4,5
+Gym_012_pushup2 Gym scan_camcoord 012 X female X V X 3,4,5,6
+Gym_010_mountainclimber1 Gym scan_camcoord 010 X female X V X 3,4,5,6
+Gym_010_mountainclimber2 Gym scan_camcoord 010 X female X V X 3,4,5,6
+Gym_013_dips1 Gym scan_camcoord 013 X female X X V 0,3,4,5
+Gym_013_dips2 Gym scan_camcoord 013 X female X X V 1,2,4,5
+Gym_013_dips3 Gym scan_camcoord 013 X female X X V 1,2,4,5
+Gym_013_lunge1 Gym scan_camcoord 013 X female X X V 1,4,5,6
+Gym_013_lunge2 Gym scan_camcoord 013 X female X X V 0,4,5,6
+Gym_013_pushup1 Gym scan_camcoord 013 X female X V V 0,3,4,5
+Gym_013_pushup2 Gym scan_camcoord 013 X female X V V 1,2,4,5
+Gym_013_burpee4 Gym scan_camcoord 013 X female X V V 0,4,5,6
+Gym_010_lunge1 Gym scan_camcoord 010 X female X X X 1,4,5,6
+Gym_010_lunge2 Gym scan_camcoord 010 X female X X X 0,2,4,5
+Gym_010_dips1 Gym scan_camcoord 010 X female X X X 0,4,5,6
+Gym_010_dips2 Gym scan_camcoord 010 X female X X X 1,2,4,5
+Gym_010_cooking1 Gym scan_table_camcoord 010 X female X X X 1,3,4,5
+Gym_011_cooking1 Gym scan_table_camcoord 011 V male X X X 4,5,6
+Gym_011_cooking2 Gym scan_table_camcoord 011 V male X X X 2,4,5
+Gym_011_dips1 Gym scan_camcoord 011 X male X X X 1,3,4,5
+Gym_011_dips4 Gym scan_camcoord 011 X male X X X 0,2,4,5
+Gym_011_dips3 Gym scan_camcoord 011 X male X X X 0,3,4,5
+Gym_011_dips2 Gym scan_camcoord 011 X male X X X 1,3,4,5
+Gym_012_lunge1 Gym scan_camcoord 012 X female X X X 0,3,4,5
+Gym_012_lunge2 Gym scan_camcoord 012 X female X X X 0,4,5,6
+Gym_012_cooking2 Gym scan_table_camcoord 012 V female X X X 3,4,5
diff --git a/genmo/datasets/rich/resource/train.txt b/genmo/datasets/rich/resource/train.txt
new file mode 100644
index 0000000000000000000000000000000000000000..fcde59dd90cc4ff5293abaced341edc2e2c7df58
--- /dev/null
+++ b/genmo/datasets/rich/resource/train.txt
@@ -0,0 +1,65 @@
+sequence_name capture_name scan_name id moving_cam gender view_id
+ParkingLot1_002_burpee3 ParkingLot1 scan_camcoord 002 X male 0,1,2,3,4,5,6,7
+ParkingLot1_002_overfence1 ParkingLot1 scan_camcoord 002 X male 0,1,2,3,4,5,6,7
+ParkingLot1_002_overfence2 ParkingLot1 scan_camcoord 002 X male 0,1,2,3,4,5,6,7
+ParkingLot1_002_stretching1 ParkingLot1 scan_camcoord 002 X male 0,1,2,3,4,5,6,7
+ParkingLot1_002_pushup1 ParkingLot1 scan_camcoord 002 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_pushup2 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_burpeejump1 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_eating1 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_takingphotos1 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_phonetalk1 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_005_burpeejump2 ParkingLot1 scan_camcoord 005 X male 0,1,2,3,4,5,6,7
+ParkingLot1_005_overfence1 ParkingLot1 scan_camcoord 005 X male 0,1,2,3,4,5,6,7
+ParkingLot1_005_pushup2 ParkingLot1 scan_camcoord 005 X male 0,1,2,3,4,5,6,7
+ParkingLot1_005_pushup3 ParkingLot1 scan_camcoord 005 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_005_greetingchattingeating1 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_005_greetingchattingeating1 ParkingLot1 scan_camcoord 005 X male 0,1,2,3,4,5,6,7
+ParkingLot1_007_overfence2 ParkingLot1 scan_camcoord 007 X male 0,1,2,3,4,5,6,7
+ParkingLot1_007_eating1 ParkingLot1 scan_camcoord 007 X male 0,1,2,3,4,5,6,7
+ParkingLot1_007_eating2 ParkingLot1 scan_camcoord 007 X male 0,1,2,3,4,5,6,7
+ParkingLot2_008_phonetalk1 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_burpeejump1 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_overfence1 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_pushup1 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_pushup2 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_overfence2 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_overfence3 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_eating1 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_014_pushup2 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_014_burpeejump1 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_014_burpeejump2 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_014_phonetalk2 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_014_takingphotos2 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_014_overfence3 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_015_overfence1 ParkingLot2 scan_camcoord 015 X male 0,1,2,3,4,5
+ParkingLot2_015_burpeejump2 ParkingLot2 scan_camcoord 015 X male 0,1,2,3,4,5
+ParkingLot2_015_pushup1 ParkingLot2 scan_camcoord 015 X male 0,1,2,3,4,5
+ParkingLot2_015_eating2 ParkingLot2 scan_camcoord 015 X male 0,1,2,3,4,5
+ParkingLot2_016_burpeejump2 ParkingLot2 scan_camcoord 016 V female 0,1,2,3,4,5
+ParkingLot2_016_overfence2 ParkingLot2 scan_camcoord 016 V female 0,1,2,3,4,5
+ParkingLot2_016_pushup1 ParkingLot2 scan_camcoord 016 V female 0,1,2,3,4,5
+ParkingLot2_016_pushup2 ParkingLot2 scan_camcoord 016 V female 0,1,2,3,4,5
+ParkingLot2_016_stretching1 ParkingLot2 scan_camcoord 016 V female 0,1,2,3,4,5
+Pavallion_000_yoga2 Pavallion scan_camcoord 000 X male 0,1,2,3,4,5,6
+Pavallion_000_plankjack Pavallion scan_camcoord 000 X male 0,1,2,3,4,5,6
+Pavallion_000_phonesiteat Pavallion scan_camcoord 000 X male 0,1,3,4,6
+Pavallion_000_sidebalancerun Pavallion scan_camcoord 000 X male 0,1,2,3,4,5,6
+Pavallion_002_plankjack Pavallion scan_camcoord 002 V male 0,1,2,3,4,5,6
+Pavallion_002_phonesiteat Pavallion scan_camcoord 002 V male 0,1,3,4,6
+Pavallion_003_plankjack Pavallion scan_camcoord 003 V male 0,1,2,3,4,5,6
+Pavallion_003_phonesiteat Pavallion scan_camcoord 003 V male 0,1,3,4,6
+Pavallion_003_sidebalancerun Pavallion scan_camcoord 003 V male 0,1,2,3,4,5,6
+Pavallion_006_phonesiteat Pavallion scan_camcoord 006 V male 0,1,3,4,6
+Pavallion_006_sidebalancerun Pavallion scan_camcoord 006 V male 0,1,2,3,4,5,6
+Pavallion_006_plankjack Pavallion scan_camcoord 006 V male 0,1,2,3,4,5,6
+Pavallion_013_phonesiteat Pavallion scan_camcoord 013 X female 0,1,3,4,6
+Pavallion_013_plankjack Pavallion scan_camcoord 013 X female 0,1,2,3,4,5,6
+Pavallion_013_yoga2 Pavallion scan_camcoord 013 V female 0,1,2,3,4,5,6
+Pavallion_003_018_tossball Pavallion scan_camcoord 003 X male 0,1,2,3,4,5,6
+Pavallion_003_018_tossball Pavallion scan_camcoord 018 X female 0,1,2,3,4,5,6
+LectureHall_018_wipingchairs1 LectureHall scan_chair_scene_camcoord 018 X female 0,1,2,3,4,5,6
+LectureHall_018_wipingspray1 LectureHall scan_chair_scene_camcoord 018 X female 2,3,4
+LectureHall_020_wipingtable1 LectureHall scan_chair_scene_camcoord 020 X male 0,2,4,5,6
+BBQ_001_juggle BBQ scan_camcoord 001 X male 0,1,2,3,4,5,6,7
+BBQ_001_guitar BBQ scan_camcoord 001 X male 0,1,2,3,4,5,6,7
diff --git a/genmo/datasets/rich/resource/val.txt b/genmo/datasets/rich/resource/val.txt
new file mode 100644
index 0000000000000000000000000000000000000000..7a3dbe9506f50b2d8609c175c572980e5cc73e1f
--- /dev/null
+++ b/genmo/datasets/rich/resource/val.txt
@@ -0,0 +1,29 @@
+sequence_name capture_name scan_name id moving_cam gender scene action/scene-interaction subjects view_id
+ParkingLot1_002_stretching2 ParkingLot1 scan_camcoord 002 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_002_burpee1 ParkingLot1 scan_camcoord 002 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_002_burpee2 ParkingLot1 scan_camcoord 002 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_004_pushup1 ParkingLot1 scan_camcoord 004 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_004_eating2 ParkingLot1 scan_camcoord 004 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_004_phonetalk2 ParkingLot1 scan_camcoord 004 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_004_takingphotos2 ParkingLot1 scan_camcoord 004 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_004_stretching2 ParkingLot1 scan_camcoord 004 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_005_overfence2 ParkingLot1 scan_camcoord 005 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_005_pushup1 ParkingLot1 scan_camcoord 005 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_005_burpeejump1 ParkingLot1 scan_camcoord 005 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_007_burpee2 ParkingLot1 scan_camcoord 007 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot2_008_eating2 ParkingLot2 scan_camcoord 008 V male V V V 0,1,2,3,4,5
+ParkingLot2_008_burpeejump2 ParkingLot2 scan_camcoord 008 V male V V V 0,1,2,3,4,5
+ParkingLot2_014_overfence1 ParkingLot2 scan_camcoord 014 X male V V V 0,1,2,3,4,5
+ParkingLot2_014_eating2 ParkingLot2 scan_camcoord 014 X male V V V 0,1,2,3,4,5
+ParkingLot2_016_phonetalk5 ParkingLot2 scan_camcoord 016 V female V V V 0,1,2,3,4,5
+Pavallion_002_sidebalancerun Pavallion scan_camcoord 002 V male V V V 0,1,2,3,4,5,6
+Pavallion_013_sidebalancerun Pavallion scan_camcoord 013 X female V V V 0,1,2,3,4,5,6
+Pavallion_018_sidebalancerun Pavallion scan_camcoord 018 V female V V V 0,1,2,3,4,5,6
+LectureHall_018_wipingtable1 LectureHall scan_chair_scene_camcoord 018 X female V V V 0,2,4,5,6
+LectureHall_020_wipingchairs1 LectureHall scan_chair_scene_camcoord 020 X male V V V 0,1,2,3,4,5,6
+LectureHall_003_wipingchairs1 LectureHall scan_chair_scene_camcoord 003 X male V V V 0,1,2,3,4,5,6
+Pavallion_000_yoga1 Pavallion scan_camcoord 000 X male V X V 0,1,2,3,4,5,6
+Pavallion_002_yoga1 Pavallion scan_camcoord 002 V male V X V 0,1,2,3,4,5,6
+Pavallion_003_yoga1 Pavallion scan_camcoord 003 V male V X V 0,1,2,3,4,5,6
+Pavallion_006_yoga1 Pavallion scan_camcoord 006 V male V X V 0,1,2,3,4,5,6
+Pavallion_018_yoga1 Pavallion scan_camcoord 018 V female V X V 0,1,2,3,4,5,6
diff --git a/genmo/datasets/rich/resource/w2az_sahmr.json b/genmo/datasets/rich/resource/w2az_sahmr.json
new file mode 100644
index 0000000000000000000000000000000000000000..a94fc4c861e5172377a79087372a9cb6a3c954c0
--- /dev/null
+++ b/genmo/datasets/rich/resource/w2az_sahmr.json
@@ -0,0 +1 @@
+{"BBQ_scan_camcoord": [[0.9989829107564298, 0.03367618890797693, -0.029984301180211045, 0.0008183751635392625], [0.03414262169451401, -0.1305975871406019, 0.9908473906797644, -0.005059823133706893], [0.02945208652127451, -0.9908633531086326, -0.13161455111748036, 1.4054905296083466], [0.0, 0.0, 0.0, 1.0]], "Gym_scan_camcoord": [[0.9932599733260449, -0.07628732032461205, 0.0872632233306122, -0.047601130084306706], [-0.10233962102690007, -0.22374853741942266, 0.9692590953768503, -0.04091804681182174], [-0.05441716049582774, -0.9716567484252654, -0.23004768176013274, 1.537911791136788], [0.0, 0.0, 0.0, 1.0]], "Gym_scan_table_camcoord": [[0.9974451989415423, -0.06250743213795668, 0.03458172980064169, 0.02231858470834599], [-0.04804912583358893, -0.22882402250236075, 0.972281259838159, 0.039081886755815726], [-0.05286167435026744, -0.9714588965331274, -0.2312428501197992, 1.5421821446346522], [0.0, 0.0, 0.0, 1.0]], "LectureHall_scan_chair_scene_camcoord": [[0.9992930513998263, 0.030087515976743376, -0.0225419343977731, 0.001998908749589632], [0.030705594681969043, -0.30721111058653017, 0.9511458878570781, -0.025811963513866963], [0.021692484396004613, -0.9511656401040444, -0.307917783192506, 2.060346184503773], [0.0, 0.0, 0.0, 1.0]], "LectureHall_scan_yoga_scene_camcoord": [[0.9993358324246812, 0.03030060260429296, -0.020242715082476024, -0.003510046042036605], [0.028600729415016745, -0.3079667078507395, 0.9509671419836329, -0.01748548118379142], [0.022580795137075255, -0.9509144968594153, -0.3086287856852993, 2.0424701474796567], [0.0, 0.0, 0.0, 1.0]], "ParkingLot1_scan_camcoord": [[0.9989627324729327, -0.03724260727951709, 0.02620013994738054, 0.0070941466745699025], [-0.03091587075252664, -0.13228243926883107, 0.9907298144280939, -0.0274920377236923], [-0.03343154297742938, -0.9905121627037764, -0.13329661462331338, 1.3859200914120975], [0.0, 0.0, 0.0, 1.0]], "ParkingLot2_scan_camcoord": [[0.9989532636786039, -0.04044665659892979, 0.021364572447267097, 0.01646827411554571], [-0.026687287930043047, -0.13600581518076985, 0.9903485279940424, 0.030197722289598695], [-0.03715058073335097, -0.9898820567153364, -0.13694286452455984, 1.4372015171546513], [0.0, 0.0, 0.0, 1.0]], "Pavallion_scan_camcoord": [[0.9971864096076799, 0.05693557331723671, -0.048760690979605295, 0.0012478238054067193], [0.05746407703876882, -0.16289761936471214, 0.9849681443861059, -0.006002953831755452], [0.04813672552068054, -0.9849988355812122, -0.16571104235928033, 1.7638454838942128], [0.0, 0.0, 0.0, 1.0]]}
diff --git a/genmo/datasets/rich/rich_motion_test.py b/genmo/datasets/rich/rich_motion_test.py
new file mode 100644
index 0000000000000000000000000000000000000000..697546d14dcc60f6ae9d3149a0d550140da61dbb
--- /dev/null
+++ b/genmo/datasets/rich/rich_motion_test.py
@@ -0,0 +1,226 @@
+from pathlib import Path
+
+import torch
+from torch.utils import data
+
+from genmo.utils.geo_transform import (
+ compute_cam_angvel,
+ compute_cam_tvel,
+ normalize_T_w2c,
+ transform_mat,
+)
+from genmo.utils.net_utils import get_valid_mask
+from genmo.utils.pylogger import Log
+from genmo.utils.rotation_conversions import axis_angle_to_matrix
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import resize_K
+
+from .rich_utils import (
+ get_cam2params,
+ get_cam_key_wham_vid,
+ get_w2az_sahmr,
+ parse_seqname_info,
+)
+
+VID_PRESETS = {
+ "easytohard": [
+ "test/Gym_013_burpee4/cam_06",
+ "test/Gym_011_pushup1/cam_02",
+ "test/LectureHall_019_wipingchairs1/cam_03",
+ "test/ParkingLot2_009_overfence1/cam_04",
+ "test/LectureHall_021_sidebalancerun1/cam_00",
+ "test/Gym_010_dips2/cam_05",
+ ],
+}
+
+
+class RichSmplFullSeqDataset(data.Dataset):
+ def __init__(self, vid_presets=None):
+ """
+ Args:
+ vid_presets is a key in VID_PRESETS
+ """
+ super().__init__()
+ self.dataset_name = "RICH"
+ self.dataset_id = "RICH"
+ Log.info(f"[{self.dataset_name}] Full sequence, Test")
+ tic = Log.time()
+
+ # Load evaluation protocol from WHAM labels
+ self.rich_dir = Path("inputs/RICH/hmr4d_support")
+ self.labels = torch.load(self.rich_dir / "rich_test_labels.pt")
+ self.preproc_data = torch.load(self.rich_dir / "rich_test_preproc.pt")
+ vids = select_subset(self.labels, vid_presets)
+
+ self.vimo_labels = torch.load(self.rich_dir / "rich_test_vimo_preproc.pt")
+ # Setup dataset index
+ self.idx2meta = []
+ for vid in vids:
+ seq_length = len(self.labels[vid]["frame_id"])
+ self.idx2meta.append((vid, 0, seq_length)) # start=0, end=seq_length
+ # print(sum([end - start for _, _, start, end in self.idx2meta]))
+
+ # Prepare ground truth motion in ay-coordinate
+ self.w2az = (
+ get_w2az_sahmr()
+ ) # scan_name -> T_w2az, w-coordinate refers to cam-1-coordinate
+ self.cam2params = get_cam2params() # cam_key -> (T_w2c, K)
+ seqname_info = parse_seqname_info(
+ skip_multi_persons=True
+ ) # {k: (scan_name, subject_id, gender, cam_ids)}
+ self.seqname_to_scanname = {k: v[0] for k, v in seqname_info.items()}
+
+ Log.info(
+ f"[RICH] {len(self.idx2meta)} sequences. Elapsed: {Log.time() - tic:.2f}s"
+ )
+
+ def __len__(self):
+ return len(self.idx2meta)
+
+ def _load_data(self, idx):
+ data = {}
+
+ # [start, end), when loading data from labels
+ vid, start, end = self.idx2meta[idx]
+ label = self.labels[vid]
+ preproc_data = self.preproc_data[vid]
+
+ vimo_label = self.vimo_labels[vid]
+
+ length = end - start
+ meta = {"dataset_id": "RICH", "vid": vid, "vid-start-end": (start, end)}
+ data.update({"meta": meta, "length": length})
+
+ # SMPLX
+ data.update(
+ {"gt_smpl_params": label["gt_smplx_params"], "gender": label["gender"]}
+ )
+ vimo_smpl_params = {
+ "pred_cam": vimo_label["vimo_params"]["pred_cam"],
+ "pred_pose": vimo_label["vimo_params"]["pred_pose"],
+ "pred_shape": vimo_label["vimo_params"]["pred_shape"],
+ "pred_trans_c": vimo_label["vimo_params"]["pred_trans"],
+ }
+ data.update({"vimo_smpl_params": vimo_smpl_params})
+ data["vimo_label"] = vimo_label
+ # camera
+ cam_key = get_cam_key_wham_vid(vid)
+ scan_name = self.seqname_to_scanname[vid.split("/")[1]]
+ T_w2c, K = self.cam2params[cam_key] # (4, 4) (3, 3)
+ T_w2az = self.w2az[scan_name]
+ data.update({"T_w2c": T_w2c, "T_w2az": T_w2az, "K": K})
+
+ # image features
+ data.update(
+ {
+ "f_imgseq": preproc_data["f_imgseq"],
+ "bbx_xys": preproc_data["bbx_xys"],
+ "img_wh": preproc_data["img_wh"],
+ "kp2d": preproc_data["kp2d"],
+ }
+ )
+
+ # to render a video
+ video_path = self.rich_dir / "video" / vid / "video.mp4"
+ frame_id = label["frame_id"] # (F,)
+ width, height = data["img_wh"] / 4 # Video saved has been downsampled 1/4
+ K_render = resize_K(K, 0.25)
+ bbx_xys_render = data["bbx_xys"] / 4
+ data["meta_render"] = {
+ "name": vid.replace("/", "@"),
+ "video_path": str(video_path),
+ "frame_id": frame_id,
+ "width_height": (width, height),
+ "K": K_render,
+ "bbx_xys": bbx_xys_render,
+ }
+
+ return data
+
+ def _process_data(self, data):
+ # T_w2az is pre-computed by using floor clue. az2zy uses a rotation along x-axis.
+ R_az2ay = axis_angle_to_matrix(
+ torch.tensor([1.0, 0.0, 0.0]) * -torch.pi / 2
+ ) # (3, 3)
+ T_w2ay = (
+ transform_mat(R_az2ay, R_az2ay.new([0, 0, 0])) @ data["T_w2az"]
+ ) # (4, 4)
+
+ vimo_label = data["vimo_label"]
+
+ # process img feature with xys
+ length = data["length"]
+ f_imgseq = data["f_imgseq"] # (F, 1024)
+ normed_T_w2c = normalize_T_w2c(data["T_w2c"])
+ R_w2c = normed_T_w2c[:, :3, :3].repeat(length, 1, 1) # (L, 4, 4)
+ t_w2c = normed_T_w2c[:, :3, 3].repeat(length, 1) # (L, 3)
+ cam_angvel = compute_cam_angvel(R_w2c) # (L, 6)
+ cam_tvel = compute_cam_tvel(t_w2c) # (L, 3)
+
+ K_fullimg = data["K"][None].expand(length, -1, -1) # (L, 3, 3)
+
+ scales = torch.ones(length)
+ mean_scale = 1.0
+ # Return
+ return_data = {
+ # --- not batched
+ "task": "CAP-Seq",
+ "meta": data["meta"],
+ "meta_render": data["meta_render"],
+ # --- we test on single sequence, so set kv manually
+ "length": length,
+ "f_imgseq": f_imgseq,
+ "cam_angvel": cam_angvel,
+ "cam_tvel": cam_tvel,
+ "R_w2c": R_w2c,
+ "bbx_xys": data["bbx_xys"], # (F, 3)
+ "K_fullimg": K_fullimg, # (L, 3, 3)
+ "kp2d": data["kp2d"], # (F, 17, 3)
+ # --- dataset specific
+ "model": "smplx",
+ "gender": data["gender"],
+ "gt_smpl_params": data["gt_smpl_params"],
+ "T_w2ay": T_w2ay, # (4, 4)
+ "T_w2c": data["T_w2c"], # (4, 4)
+ "scales": scales,
+ "mean_scale": mean_scale,
+ "vimo_smpl_params": data["vimo_smpl_params"],
+ "mask": {
+ "valid": get_valid_mask(length, length),
+ "has_img_mask": get_valid_mask(length, length),
+ "has_2d_mask": get_valid_mask(length, length),
+ "has_cam_mask": get_valid_mask(length, length),
+ "has_audio_mask": get_valid_mask(length, 0),
+ "has_music_mask": get_valid_mask(length, 0),
+ },
+ }
+
+ if "vimo_params_flip" in vimo_label:
+ flipped_trans_c = vimo_label["vimo_params_flip"]["pred_trans"]
+ orig_trans_c = data["vimo_smpl_params"]["pred_trans_c"]
+ tz = flipped_trans_c[..., 2]
+ tx = flipped_trans_c[..., 0]
+ focal = K_fullimg[0, 0, 0]
+ cx = K_fullimg[0, 0, 2]
+ width = data["meta_render"]["width_height"][0]
+
+ flipped_tx = tz * (width - 1 - 2 * cx) / focal - tx
+ avg_trans_c = torch.zeros_like(flipped_trans_c)
+ avg_trans_c[..., 0] = (flipped_tx + orig_trans_c[..., 0]) / 2
+ avg_trans_c[..., 0] = orig_trans_c[..., 0]
+ avg_trans_c[..., 1] = (flipped_trans_c[..., 1] + orig_trans_c[..., 1]) / 2
+ avg_trans_c[..., 2] = (tz + orig_trans_c[..., 2]) / 2
+ return_data["vimo_smpl_params"]["pred_trans_c"] = avg_trans_c
+
+ return return_data
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data)
+ return data
+
+
+def select_subset(labels, vid_presets):
+ vids = list(labels.keys())
+ if vid_presets is not None: # Use a subset of the videos
+ vids = VID_PRESETS[vid_presets]
+ return vids
diff --git a/genmo/datasets/rich/rich_utils.py b/genmo/datasets/rich/rich_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..49f51ccafe05301b6df207338a0c0e868436db42
--- /dev/null
+++ b/genmo/datasets/rich/rich_utils.py
@@ -0,0 +1,421 @@
+import json
+from pathlib import Path
+
+import cv2
+import numpy as np
+import torch
+
+from genmo.utils.geo_transform import apply_T_on_points, project_p2d
+
+# ----- Meta sample utils ----- #
+
+
+def sample_idx2meta(idx2meta, sample_interval):
+ """
+ 1. remove frames that < 45
+ 2. sample frames by sample_interval
+ 3. sorted
+ """
+ idx2meta = [
+ v
+ for k, v in idx2meta.items()
+ if int(v["frame_name"]) > 45
+ and (int(v["frame_name"]) + int(v["cam_id"])) % sample_interval == 0
+ ]
+ idx2meta = sorted(idx2meta, key=lambda meta: meta["img_key"])
+ return idx2meta
+
+
+def remove_bbx_invisible_frame(idx2meta, img2gtbbx):
+ raw_img_lu = np.array([0.0, 0.0])
+ raw_img_rb_type1 = np.array([4112.0, 3008.0]) - 1 # horizontal
+ raw_img_rb_type2 = np.array([3008.0, 4112.0]) - 1 # vertical
+
+ idx2meta_new = []
+ for meta in idx2meta:
+ gtbbx_center = np.array(
+ [
+ img2gtbbx[meta["img_key"]][[0, 2]].mean(),
+ img2gtbbx[meta["img_key"]][[1, 3]].mean(),
+ ]
+ )
+ if (gtbbx_center < raw_img_lu).any():
+ continue
+ raw_img_rb = (
+ raw_img_rb_type1
+ if meta["cam_key"] not in ["Pavallion_3", "Pavallion_5"]
+ else raw_img_rb_type2
+ )
+ if (gtbbx_center > raw_img_rb).any():
+ continue
+ idx2meta_new.append(meta)
+ return idx2meta_new
+
+
+def remove_extra_rules(idx2meta):
+ multi_person_seqs = ["LectureHall_009_021_reparingprojector1"]
+ idx2meta = [meta for meta in idx2meta if meta["seq_name"] not in multi_person_seqs]
+ return idx2meta
+
+
+# ----- Image utils ----- #
+
+
+def compute_bbx(dataset, data):
+ """
+ Use gt_smplh_params to compute bbx (w.r.t. original image resolution)
+ Args:
+ dataset: rich_pose.RichPose
+ data: dict
+
+ # This function need extra scripts to run
+ from hmr4d.utils.smplx_utils import make_smplx
+ self.smplh_male = make_smplx("rich-smplh", gender="male")
+ self.smplh_female = make_smplx("rich-smplh", gender="female")
+ self.smplh = {
+ "male": self.smplh_male,
+ "female": self.smplh_female,
+ }
+ """
+ gender = data["meta"]["gender"]
+ smplh_params = {k: v.reshape(1, -1) for k, v in data["gt_smplh_params"].items()}
+ smplh_opt = dataset.smplh[gender](**smplh_params)
+ verts_3d_w = smplh_opt.vertices
+ T_w2c, K = data["T_w2c"], data["K"]
+ verts_3d_c = apply_T_on_points(verts_3d_w, T_w2c[None])
+ verts_2d = project_p2d(verts_3d_c, K[None])[0]
+ min_2d = verts_2d.T.min(-1)[0]
+ max_2d = verts_2d.T.max(-1)[0]
+ bbx = torch.stack([min_2d, max_2d]).reshape(-1).numpy()
+ return bbx
+
+
+def get_2d(dataset, data):
+ gender = data["meta"]["gender"]
+ smplh_params = {k: v.reshape(1, -1) for k, v in data["gt_smplh_params"].items()}
+ smplh_opt = dataset.smplh[gender](**smplh_params)
+ joints_3d_w = smplh_opt.joints
+ T_w2c, K = data["T_w2c"], data["K"]
+ joints_3d_c = apply_T_on_points(joints_3d_w, T_w2c[None])
+ joints_2d = project_p2d(joints_3d_c, K[None])[0]
+ conf = torch.ones((73, 1))
+ keypoints = torch.cat([joints_2d, conf], dim=1)
+ return keypoints
+
+
+def squared_crop_and_resize(dataset, img, bbx_lurb, dst_size=224, state=None):
+ if state is not None:
+ np.random.set_state(state)
+ center_rand = dataset.BBX_CENTER * (np.random.random(2) * 2 - 1)
+ center_x = (bbx_lurb[0] + bbx_lurb[2]) / 2 + center_rand[0]
+ center_y = (bbx_lurb[1] + bbx_lurb[3]) / 2 + center_rand[1]
+ ori_half_size = max(bbx_lurb[2] - bbx_lurb[0], bbx_lurb[3] - bbx_lurb[1]) / 2
+ ori_half_size *= 1 + 0.15 + dataset.BBX_ZOOM * np.random.random() # zoom
+
+ src = np.array(
+ [
+ [center_x - ori_half_size, center_y - ori_half_size],
+ [center_x + ori_half_size, center_y - ori_half_size],
+ [center_x, center_y],
+ ],
+ dtype=np.float32,
+ )
+ dst = np.array(
+ [[0, 0], [dst_size - 1, 0], [dst_size / 2 - 0.5, dst_size / 2 - 0.5]],
+ dtype=np.float32,
+ )
+
+ A = cv2.getAffineTransform(src, dst)
+ img_crop = cv2.warpAffine(img, A, (dst_size, dst_size), flags=cv2.INTER_LINEAR)
+ bbx_new = np.array(
+ [
+ center_x - ori_half_size,
+ center_y - ori_half_size,
+ center_x + ori_half_size,
+ center_y + ori_half_size,
+ ],
+ dtype=bbx_lurb.dtype,
+ )
+ return img_crop, bbx_new, A
+
+
+# Augment bbx
+def get_augmented_square_bbx(
+ bbx_lurb, per_shift=0.1, per_zoomout=0.2, base_zoomout=0.15, state=None
+):
+ """
+ Args:
+ per_shift: in percent, maximum random shift
+ per_zoomout: in percent, maximum random zoom
+ """
+ if state is not None:
+ np.random.set_state(state)
+ maxsize_bbx = max(bbx_lurb[2] - bbx_lurb[0], bbx_lurb[3] - bbx_lurb[1])
+ # shift of center
+ shift = maxsize_bbx * per_shift * (np.random.random(2) * 2 - 1)
+ center_x = (bbx_lurb[0] + bbx_lurb[2]) / 2 + shift[0]
+ center_y = (bbx_lurb[1] + bbx_lurb[3]) / 2 + shift[1]
+ # zoomout of half-size
+ halfsize_bbx = maxsize_bbx / 2
+ halfsize_bbx *= 1 + base_zoomout + per_zoomout * np.random.random()
+
+ bbx_lurb = np.array(
+ [
+ center_x - halfsize_bbx,
+ center_y - halfsize_bbx,
+ center_x + halfsize_bbx,
+ center_y + halfsize_bbx,
+ ]
+ )
+ return bbx_lurb
+
+
+def get_squared_bbx_region_and_resize(frames, bbx_xys, dst_size=224):
+ """
+ Args:
+ frames: (F, H, W, 3)
+ bbx_xys: (F, 3), xys
+ """
+ frames_np = frames.numpy() if isinstance(frames, torch.Tensor) else frames
+ bbx_xys = (
+ bbx_xys if isinstance(bbx_xys, torch.Tensor) else torch.tensor(bbx_xys)
+ ) # use tensor
+ srcs = torch.stack(
+ [
+ torch.stack(
+ [bbx_xys[:, 0] - bbx_xys[:, 2] / 2, bbx_xys[:, 1] - bbx_xys[:, 2] / 2],
+ dim=-1,
+ ),
+ torch.stack(
+ [bbx_xys[:, 0] + bbx_xys[:, 2] / 2, bbx_xys[:, 1] - bbx_xys[:, 2] / 2],
+ dim=-1,
+ ),
+ bbx_xys[:, :2],
+ ],
+ dim=1,
+ ) # (F, 3, 2)
+ dst = np.array(
+ [[0, 0], [dst_size - 1, 0], [dst_size / 2 - 0.5, dst_size / 2 - 0.5]],
+ dtype=np.float32,
+ )
+ As = np.stack([cv2.getAffineTransform(src, dst) for src in srcs.numpy()])
+
+ img_crops = np.stack(
+ [
+ cv2.warpAffine(
+ frames_np[i], As[i], (dst_size, dst_size), flags=cv2.INTER_LINEAR
+ )
+ for i in range(len(As))
+ ]
+ )
+ img_crops = torch.from_numpy(img_crops)
+ As = torch.from_numpy(As)
+ return img_crops, As
+
+
+# ----- Camera utils ----- #
+
+
+def extract_cam_xml(xml_path="", dtype=torch.float32):
+ import xml.etree.ElementTree as ET
+
+ tree = ET.parse(xml_path)
+
+ extrinsics_mat = [float(s) for s in tree.find("./CameraMatrix/data").text.split()]
+ intrinsics_mat = [float(s) for s in tree.find("./Intrinsics/data").text.split()]
+ distortion_vec = [float(s) for s in tree.find("./Distortion/data").text.split()]
+
+ return {
+ "ext_mat": torch.tensor(extrinsics_mat).float(),
+ "int_mat": torch.tensor(intrinsics_mat).float(),
+ "dis_vec": torch.tensor(distortion_vec).float(),
+ }
+
+
+def get_cam2params(scene_info_root=None):
+ """
+ Args:
+ scene_info_root: this could be repalced by path to scan_calibration
+ """
+ if scene_info_root is not None:
+ cam_params = {}
+ cam_xml_files = Path(scene_info_root).glob("*/calibration/*.xml")
+ for cam_xml_file in cam_xml_files:
+ cam_param = extract_cam_xml(cam_xml_file)
+ T_w2c = cam_param["ext_mat"].reshape(3, 4)
+ T_w2c = torch.cat([T_w2c, torch.tensor([[0, 0, 0, 1.0]])], dim=0) # (4, 4)
+ K = cam_param["int_mat"].reshape(3, 3)
+ cap_name = cam_xml_file.parts[-3]
+ cam_id = int(cam_xml_file.stem)
+ cam_key = f"{cap_name}_{cam_id}"
+ cam_params[cam_key] = (T_w2c, K)
+ else:
+ cam_params = torch.load(Path(__file__).parent / "resource/cam2params.pt")
+ return cam_params
+
+
+# ----- Parse Raw Resource ----- #
+
+
+def get_w2az_sahmr():
+ """
+ Returns:
+ w2az_sahmr: dict, {scan_name: Tw2az}, Tw2az is a tensor of (4,4)
+ """
+ fn = Path(__file__).parent / "resource/w2az_sahmr.json"
+ with open(fn, "r") as f:
+ kvs = json.load(f).items()
+ w2az_sahmr = {k: torch.tensor(v) for k, v in kvs}
+ return w2az_sahmr
+
+
+def has_multi_persons(seq_name):
+ """
+ Args:
+ seq_name: e.g. LectureHall_009_021_reparingprojector1
+ """
+ return len(seq_name.split("_")) != 3
+
+
+def parse_seqname_info(skip_multi_persons=True):
+ """
+ This function will skip multi-person sequences.
+ Returns:
+ sname_to_info: scan_name, subject_id, gender, cam_ids
+ """
+ fns = [
+ Path(__file__).parent / f"resource/{split}.txt"
+ for split in ["train", "val", "test"]
+ ]
+ # Train / Val&Test Header:
+ # sequence_name capture_name scan_name id moving_cam gender view_id
+ # sequence_name capture_name scan_name id moving_cam gender scene action/scene-interaction subjects view_id
+ sname_to_info = {}
+ for fn in fns:
+ with open(fn, "r") as f:
+ for line in f.readlines()[1:]:
+ raw_values = line.strip().split()
+ seq_name = raw_values[0]
+ if skip_multi_persons and has_multi_persons(seq_name):
+ continue
+ scan_name = f"{raw_values[1]}_{raw_values[2]}"
+ subject_id = int(raw_values[3])
+ gender = raw_values[5]
+ cam_ids = [int(c) for c in raw_values[-1].split(",")]
+ sname_to_info[seq_name] = (scan_name, subject_id, gender, cam_ids)
+ return sname_to_info
+
+
+def get_seqnames_of_split(splits=["train"], skip_multi_persons=True):
+ if not isinstance(splits, list):
+ splits = [splits]
+ fns = [Path(__file__).parent / f"resource/{split}.txt" for split in splits]
+ seqnames = []
+ for fn in fns:
+ with open(fn, "r") as f:
+ for line in f.readlines()[1:]:
+ seq_name = line.strip().split()[0]
+ if skip_multi_persons and has_multi_persons(seq_name):
+ continue
+ seqnames.append(seq_name)
+ return seqnames
+
+
+def get_seqname_to_imgrange():
+ """Each sequence has a different range of image ids."""
+ from tqdm import tqdm
+
+ split_seqnames = {
+ split: get_seqnames_of_split(split) for split in ["train", "val", "test"]
+ }
+ seqname_to_imgrange = {}
+ for split in ["train", "val", "test"]:
+ for seqname in tqdm(split_seqnames[split]):
+ img_root = (
+ Path("inputs/RICH") / "images_ds4" / split
+ ) # compressed (not original)
+ img_dir = img_root / seqname
+ img_names = sorted([n.name for n in img_dir.glob("**/*.jpeg")])
+ if len(img_names) == 0:
+ img_range = (0, 0)
+ else:
+ img_range = (
+ int(img_names[0].split("_")[0]),
+ int(img_names[-1].split("_")[0]),
+ )
+ seqname_to_imgrange[seqname] = img_range
+ return seqname_to_imgrange
+
+
+# ----- Compose keys ----- #
+
+
+def get_img_key(seq_name, cam_id, f_id):
+ assert len(seq_name.split("_")) == 3
+ subject_id = int(seq_name.split("_")[1])
+ return f"{seq_name}_{int(cam_id)}_{int(f_id):05d}_{subject_id}"
+
+
+def get_seq_cam_fn(img_root, seq_name, cam_id):
+ """
+ Args:
+ img_root: "inputs/RICH/images_ds4/train"
+ """
+ img_root = Path(img_root)
+ cam_id = int(cam_id)
+ return str(img_root / f"{seq_name}/cam_{cam_id:02d}")
+
+
+def get_img_fn(img_root, seq_name, cam_id, f_id):
+ """
+ Args:
+ img_root: "inputs/RICH/images_ds4/train"
+ """
+ img_root = Path(img_root)
+ cam_id = int(cam_id)
+ f_id = int(f_id)
+ return str(
+ img_root / f"{seq_name}/cam_{cam_id:02d}" / f"{f_id:05d}_{cam_id:02d}.jpeg"
+ )
+
+
+# ----- WHAM ----- #
+
+
+def get_cam_key_wham_vid(vid):
+ _, sname, cname = vid.split("/")
+ scene = sname.split("_")[0]
+ cid = int(cname.split("_")[1])
+ cam_key = f"{scene}_{cid}"
+ return cam_key
+
+
+def get_K_wham_vid(vid):
+ cam_key = get_cam_key_wham_vid(vid)
+ cam2params = get_cam2params()
+ K = cam2params[cam_key][1]
+ return K
+
+
+class RichVid2Tc2az:
+ def __init__(self) -> None:
+ self.w2az = get_w2az_sahmr() # scan_name: tensor 4,4
+ seqname_info = parse_seqname_info(
+ skip_multi_persons=True
+ ) # {k: (scan_name, subject_id, gender, cam_ids)}
+ self.seqname_to_scanname = {k: v[0] for k, v in seqname_info.items()}
+ self.cam2params = get_cam2params() # cam_key -> (T_w2c, K)
+
+ def __call__(self, vid):
+ cam_key = get_cam_key_wham_vid(vid)
+ scan_name = self.seqname_to_scanname[vid.split("/")[1]]
+ T_w2c, K = self.cam2params[cam_key] # (4, 4) (3, 3)
+ T_w2az = self.w2az[scan_name]
+ T_c2az = T_w2az @ T_w2c.inverse()
+ return T_c2az
+
+ def get_T_w2az(self, vid):
+ # cam_key = get_cam_key_wham_vid(vid)
+ scan_name = self.seqname_to_scanname[vid.split("/")[1]]
+ T_w2az = self.w2az[scan_name]
+ return T_w2az
diff --git a/genmo/datasets/threedpw/threedpw_motion_test.py b/genmo/datasets/threedpw/threedpw_motion_test.py
new file mode 100644
index 0000000000000000000000000000000000000000..18daeabdc0fef5231452fb8246736bfab968ca83
--- /dev/null
+++ b/genmo/datasets/threedpw/threedpw_motion_test.py
@@ -0,0 +1,244 @@
+from pathlib import Path
+
+import torch
+from torch.utils import data
+
+from genmo.utils.geo_transform import (
+ compute_cam_angvel,
+ compute_cam_tvel,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import get_valid_mask
+from genmo.utils.pylogger import Log
+from third_party.GVHMR.hmr4d.utils.geo.flip_utils import flip_kp2d_coco17
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import estimate_K, resize_K
+
+VID_HARD = []
+# VID_HARD = ["downtown_bar_00_1"]
+
+
+class ThreedpwSmplFullSeqDataset(data.Dataset):
+ def __init__(self, flip_test=False, skip_invalid=False):
+ super().__init__()
+ self.dataset_name = "3DPW"
+ self.skip_invalid = skip_invalid
+ Log.info(f"[{self.dataset_name}] Full sequence")
+
+ # Load evaluation protocol from WHAM labels
+ self.threedpw_dir = Path("inputs/3DPW/hmr4d_support")
+ # ['vname', 'K_fullimg', 'T_w2c', 'smpl_params', 'gender', 'mask_raw', 'mask_wham', 'img_wh']
+ self.labels = torch.load(self.threedpw_dir / "test_3dpw_gt_labels.pt")
+ self.vid2bbx = torch.load(self.threedpw_dir / "preproc_test_bbx.pt")
+ self.vid2kp2d = torch.load(self.threedpw_dir / "preproc_test_kp2d_v0.pt")
+
+ self.vimo_labels = torch.load(self.threedpw_dir / "test_3dpw_vimo_labels.pt")
+ self.droid_cam_traj = torch.load(self.threedpw_dir / "3dpw_test_slam_traj.pt")
+ # Setup dataset index
+ self.idx2meta = list(self.labels)
+ if len(VID_HARD) > 0: # Pick subsets for fast testing
+ self.idx2meta = VID_HARD
+ Log.info(f"[{self.dataset_name}] {len(self.idx2meta)} sequences.")
+
+ # If flip_test is enabled, we will return extra data for flipped test
+ self.flip_test = flip_test
+ if self.flip_test:
+ Log.info(f"[{self.dataset_name}] Flip test enabled")
+
+ def __len__(self):
+ return len(self.idx2meta)
+
+ def _load_data(self, idx):
+ data = {}
+ vid = self.idx2meta[idx]
+ meta = {"dataset_id": self.dataset_name, "vid": vid}
+ data.update({"meta": meta})
+
+ # Add useful data
+ label = self.labels[vid]
+ vimo_label = self.vimo_labels[vid]
+ droid_label = self.droid_cam_traj[vid]
+
+ mask = label["mask_wham"]
+ width_height = label["img_wh"]
+ vimo_smpl_params = {
+ "pred_cam": vimo_label["vimo_params"]["pred_cam"],
+ "pred_pose": vimo_label["vimo_params"]["pred_pose"],
+ "pred_shape": vimo_label["vimo_params"]["pred_shape"],
+ "pred_trans_c": vimo_label["vimo_params"]["pred_trans"],
+ }
+
+ length = len(mask)
+ data.update(
+ {
+ "length": length, # F
+ "smpl_params": label["smpl_params"], # world
+ "gender": label["gender"], # str
+ # "T_w2c": label["T_w2c"], # (F, 4, 4)
+ "mask": {
+ "valid": mask, # (F)
+ "has_img_mask": get_valid_mask(length, length),
+ "has_2d_mask": get_valid_mask(length, length),
+ "has_cam_mask": get_valid_mask(length, length),
+ "has_audio_mask": get_valid_mask(length, 0),
+ "has_music_mask": get_valid_mask(length, 0),
+ "2d_only": False,
+ },
+ }
+ )
+
+ gt_T_w2c = label["T_w2c"]
+ data.update({"vimo_smpl_params": vimo_smpl_params})
+
+ # camera
+ # load droid slam
+ R_c2w = torch.from_numpy(droid_label["pred_cam_R"]).float()
+ t_c2w = torch.from_numpy(droid_label["pred_cam_T"]).float()
+ scales = torch.from_numpy(droid_label["all_scales"]).float()
+ mean_scale = droid_label["scale"]
+ T_c2w = torch.eye(4)[None].repeat(length, 1, 1).to(R_c2w)
+ T_c2w[:, :3, :3] = R_c2w
+ T_c2w[:, :3, 3] = t_c2w
+ T_w2c = T_c2w.inverse()
+
+ K_fullimg = label["K_fullimg"] # (3, 3)
+ if False:
+ K_fullimg = estimate_K(*width_height)
+ data["K_fullimg"] = K_fullimg
+ data.update(
+ {
+ "T_w2c": T_w2c,
+ "scales": scales,
+ "mean_scale": mean_scale,
+ "gt_T_w2c": gt_T_w2c,
+ }
+ )
+
+ if "vimo_params_flip" in vimo_label:
+ flipped_trans_c = vimo_label["vimo_params_flip"]["pred_trans"]
+ orig_trans_c = data["vimo_smpl_params"]["pred_trans_c"]
+ tz = flipped_trans_c[..., 2]
+ tx = flipped_trans_c[..., 0]
+ focal = K_fullimg[0, 0]
+ cx = K_fullimg[0, 2]
+ width = width_height[0]
+
+ flipped_tx = tz * (width - 1 - 2 * cx) / focal - tx
+ avg_trans_c = torch.zeros_like(flipped_trans_c)
+ avg_trans_c[..., 0] = (flipped_tx + orig_trans_c[..., 0]) / 2
+ avg_trans_c[..., 0] = orig_trans_c[..., 0]
+ avg_trans_c[..., 1] = (flipped_trans_c[..., 1] + orig_trans_c[..., 1]) / 2
+ avg_trans_c[..., 2] = (tz + orig_trans_c[..., 2]) / 2
+ data["vimo_smpl_params"]["pred_trans_c"] = avg_trans_c
+
+ # Preprocessed: bbx, kp2d, image as feature
+ bbx_xys = self.vid2bbx[vid]["bbx_xys"] # (F, 3)
+ kp2d = self.vid2kp2d[vid] # (F, 17, 3)
+ norm_T_w2c = normalize_T_w2c(data["T_w2c"])
+ cam_angvel = compute_cam_angvel(norm_T_w2c[:, :3, :3]) # (L, 6)
+ cam_tvel = compute_cam_tvel(norm_T_w2c[:, :3, 3]) # (L, 3)
+ data.update(
+ {
+ "bbx_xys": bbx_xys,
+ "kp2d": kp2d,
+ "cam_angvel": cam_angvel,
+ "cam_tvel": cam_tvel,
+ }
+ )
+ data["R_w2c"] = norm_T_w2c[:, :3, :3]
+
+ imgfeat_dir = self.threedpw_dir / "imgfeats/3dpw_test"
+ f_img_dict = torch.load(imgfeat_dir / f"{vid}.pt")
+ f_imgseq = f_img_dict["features"].float()
+ data["f_imgseq"] = f_imgseq # (F, 1024)
+
+ # to render a video
+ vname = label["vname"]
+ video_path = self.threedpw_dir / f"videos/{vname}.mp4"
+ frame_id = torch.where(mask)[0].long()
+ ds = 0.5
+ K_render = resize_K(K_fullimg, ds)
+ bbx_xys_render = bbx_xys * ds
+ kp2d_render = kp2d.clone()
+ kp2d_render[..., :2] *= ds
+ data["meta_render"] = {
+ "name": vid,
+ "video_path": str(video_path),
+ "ds": ds,
+ "frame_id": frame_id,
+ "K": K_render,
+ "bbx_xys": bbx_xys_render,
+ "kp2d": kp2d_render,
+ }
+
+ if self.flip_test:
+ imgfeat_dir = self.threedpw_dir / "imgfeats/3dpw_test_flip"
+ f_img_dict = torch.load(imgfeat_dir / f"{vid}.pt")
+ flipped_bbx_xys = f_img_dict["bbx_xys"].float() # (L, 3)
+ flipped_features = f_img_dict["features"].float() # (L, 1024)
+ flipped_kp2d = flip_kp2d_coco17(kp2d, width_height[0]) # (L, 17, 3)
+
+ R_flip_x = torch.tensor([[1, 0, 0], [0, -1, 0], [0, 0, -1]]).float()
+ flipped_R_w2c = R_flip_x @ norm_T_w2c[:, :3, :3].clone()
+ flipped_t_w2c = (R_flip_x @ norm_T_w2c[:, :3, 3:].clone())[..., 0]
+ flipped_T_w2c = torch.eye(4)[None].repeat(length, 1, 1).to(flipped_R_w2c)
+ flipped_T_w2c[:, :3, :3] = flipped_R_w2c
+ flipped_T_w2c[:, :3, 3] = flipped_t_w2c
+
+ data_flip = {
+ "bbx_xys": flipped_bbx_xys,
+ "f_imgseq": flipped_features,
+ "kp2d": flipped_kp2d,
+ "cam_angvel": compute_cam_angvel(flipped_R_w2c),
+ "cam_tvel": compute_cam_tvel(flipped_t_w2c),
+ "R_w2c": flipped_R_w2c,
+ }
+ flipped_trans_c = vimo_label["vimo_params_flip"]["pred_trans"]
+ flipped_trans_c[..., 2] = avg_trans_c[..., 2]
+ vimo_smpl_params_flip = {
+ "pred_cam": vimo_label["vimo_params_flip"]["pred_cam"],
+ "pred_pose": vimo_label["vimo_params_flip"]["pred_pose"],
+ "pred_shape": vimo_label["vimo_params_flip"]["pred_shape"],
+ "pred_trans_c": flipped_trans_c,
+ }
+ data_flip["vimo_smpl_params"] = vimo_smpl_params_flip
+
+ flipped_K_fullimg = K_fullimg.clone()
+ data_flip.update(
+ {
+ "K_fullimg": flipped_K_fullimg,
+ "T_w2c": flipped_T_w2c,
+ "scales": scales,
+ "mean_scale": mean_scale,
+ }
+ )
+ data["flip_test"] = data_flip
+ return data
+
+ def _process_data(self, data):
+ length = data["length"]
+ data["K_fullimg"] = data["K_fullimg"][None].repeat(length, 1, 1)
+
+ if self.skip_invalid: # Drop all invalid frames
+ mask = data["mask"].clone()
+ data["length"] = sum(mask)
+ data["smpl_params"] = {
+ k: v[mask].clone() for k, v in data["smpl_params"].items()
+ }
+ data["T_w2c"] = data["T_w2c"][mask].clone()
+ data["mask"] = data["mask"][mask].clone()
+ data["K_fullimg"] = data["K_fullimg"][mask].clone()
+ data["bbx_xys"] = data["bbx_xys"][mask].clone()
+ data["kp2d"] = data["kp2d"][mask].clone()
+ data["cam_angvel"] = data["cam_angvel"][mask].clone()
+ data["cam_tvel"] = data["cam_tvel"][mask].clone()
+ data["f_imgseq"] = data["f_imgseq"][mask].clone()
+ data["flip_test"] = {
+ k: v[mask].clone() for k, v in data["flip_test"].items()
+ }
+
+ return data
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data)
+ return data
diff --git a/genmo/datasets/threedpw/threedpw_motion_train.py b/genmo/datasets/threedpw/threedpw_motion_train.py
new file mode 100644
index 0000000000000000000000000000000000000000..95df3261509645d5d2c50b220b5ca09fd2a11e5a
--- /dev/null
+++ b/genmo/datasets/threedpw/threedpw_motion_train.py
@@ -0,0 +1,185 @@
+# from torch.utils import data
+from pathlib import Path
+
+import numpy as np
+import torch
+
+from genmo.datasets.imgfeat_motion.base_dataset import ImgfeatMotionDatasetBase
+from genmo.utils.geo_transform import (
+ compute_cam_angvel,
+ compute_cam_tvel,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import (
+ get_valid_mask,
+ repeat_to_max_len,
+ repeat_to_max_len_dict,
+)
+from genmo.utils.pylogger import Log
+
+
+class ThreedpwSmplDataset(ImgfeatMotionDatasetBase):
+ def __init__(self):
+ # Path
+ self.hmr4d_support_dir = Path("inputs/3DPW/hmr4d_support")
+ self.dataset_name = "3DPW"
+
+ # Setting
+ self.min_motion_frames = 60
+ self.max_motion_frames = 120
+ super().__init__()
+
+ def _load_dataset(self):
+ self.train_labels = torch.load(
+ self.hmr4d_support_dir / "train_3dpw_gt_labels.pt"
+ )
+ self.refit_smplx = torch.load(self.hmr4d_support_dir / "train_refit_smplx.pt")
+ if True: # Remove clips that have obvious error
+ update_list = {
+ "courtyard_basketball_00_1": [(0, 300), (340, 468)],
+ "courtyard_laceShoe_00_0": [(0, 620), (780, 931)],
+ "courtyard_rangeOfMotions_00_1": [(0, 370), (410, 601)],
+ "courtyard_shakeHands_00_1": [(0, 100), (120, 391)],
+ }
+ for k, v in update_list.items():
+ self.refit_smplx[k]["valid_range_list"] = v
+
+ self.f_img_folder = self.hmr4d_support_dir / "imgfeats/3dpw_train_smplx_refit"
+ Log.info(f"[{self.dataset_name}] Train")
+
+ def _get_idx2meta(self):
+ # We expect to see the entire sequence during one epoch,
+ # so each sequence will be sampled max(SeqLength // MotionFrames, 1) times
+ seq_lengths = []
+ self.idx2meta = []
+ for vid in self.refit_smplx:
+ valid_range_list = self.refit_smplx[vid]["valid_range_list"]
+ for start, end in valid_range_list:
+ seq_length = end - start
+ num_samples = max(seq_length // self.max_motion_frames, 1)
+ seq_lengths.append(seq_length)
+ self.idx2meta.extend([(vid, start, end)] * num_samples)
+ minutes = sum(seq_lengths) / 25 / 60
+ Log.info(
+ f"[{self.dataset_name}] has {minutes:.1f} minutes motion -> Resampled to {len(self.idx2meta)} samples."
+ )
+
+ def _load_data(self, idx):
+ data = {}
+ vid, range1, range2 = self.idx2meta[idx]
+
+ # Random select a subset
+ mlength = range2 - range1
+ min_motion_len = self.min_motion_frames
+ max_motion_len = self.max_motion_frames
+
+ if (
+ mlength < min_motion_len
+ ): # this may happen, the minimal mlength is around 30
+ start = range1
+ length = mlength
+ else:
+ effect_max_motion_len = min(max_motion_len, mlength)
+ length = np.random.randint(
+ min_motion_len, effect_max_motion_len + 1
+ ) # [low, high)
+ start = np.random.randint(range1, range2 - length + 1)
+ end = start + length
+ data["length"] = length
+ data["meta"] = {
+ "data_name": self.dataset_name,
+ "idx": idx,
+ "vid": vid,
+ "start_end": (start, end),
+ }
+
+ # Select motion subset
+ data["smplx_params_incam"] = {
+ k: v[start:end]
+ for k, v in self.refit_smplx[vid]["smplx_params_incam"].items()
+ }
+ data["K_fullimg"] = self.train_labels[vid]["K_fullimg"]
+ data["T_w2c"] = self.train_labels[vid]["T_w2c"][start:end]
+
+ # Img (as feature):
+ f_img_dict = torch.load(self.f_img_folder / f"{vid}.pt")
+ data["bbx_xys"] = f_img_dict["bbx_xys"][start:end] # (F, 3)
+ data["f_imgseq"] = f_img_dict["features"][start:end].float() # (F, 3)
+ data["img_wh"] = f_img_dict["img_wh"] # (2)
+ data["kp2d"] = torch.zeros(
+ (end - start), 17, 3
+ ) # (L, 17, 3) # do not provide kp2d
+
+ return data
+
+ def _process_data(self, data, idx):
+ length = data["length"]
+
+ smpl_params_c = data["smplx_params_incam"]
+ smpl_params_w_zero = {k: torch.zeros_like(v) for k, v in smpl_params_c.items()}
+ K_fullimg = data["K_fullimg"][None].repeat(length, 1, 1)
+ T_w2c = data["T_w2c"]
+ normed_T_w2c = normalize_T_w2c(T_w2c)
+ noisy_normed_T_w2c = normed_T_w2c.clone()
+ noisy_t_w2c = noisy_normed_T_w2c[:, :3, 3]
+ rand_scale = min(max(0.1, torch.randn(1) + 3), 10)
+ noisy_t_w2c = noisy_t_w2c / rand_scale
+ noisy_normed_T_w2c[:, :3, 3] = noisy_t_w2c
+
+ cam_angvel = compute_cam_angvel(normed_T_w2c[:, :3, :3])
+ cam_tvel = compute_cam_tvel(normed_T_w2c[:, :3, 3])
+ noisy_cam_tvel = compute_cam_tvel(noisy_normed_T_w2c[:, :3, 3])
+ max_len = self.max_motion_frames
+ return_data = {
+ "meta": data["meta"],
+ "length": length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w_zero,
+ "R_c2gv": torch.zeros(length, 3, 3), # (F, 3, 3)
+ "gravity_vec": torch.zeros(3), # (3)
+ "bbx_xys": data["bbx_xys"], # (F, 3)
+ "K_fullimg": K_fullimg, # (F, 3, 3)
+ "f_imgseq": data["f_imgseq"], # (F, D)
+ "kp2d": data["kp2d"], # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "cam_tvel": cam_tvel, # (F, 3)
+ "noisy_cam_tvel": noisy_cam_tvel, # (F, 3)
+ "T_w2c": normed_T_w2c, # (F, 4, 4)
+ "mask": {
+ "valid": get_valid_mask(max_len, length),
+ "has_img_mask": get_valid_mask(max_len, length),
+ "has_2d_mask": get_valid_mask(max_len, length),
+ "has_cam_mask": get_valid_mask(max_len, length),
+ "has_audio_mask": get_valid_mask(max_len, 0),
+ "has_music_mask": get_valid_mask(max_len, 0),
+ "2d_only": False,
+ "vitpose": False,
+ "bbx_xys": True,
+ "f_imgseq": True,
+ "spv_incam_only": True,
+ "invalid_contact": True,
+ },
+ }
+
+ # Batchable
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(
+ return_data["smpl_params_c"], max_len
+ )
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(
+ return_data["smpl_params_w"], max_len
+ )
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["bbx_xys"] = repeat_to_max_len(return_data["bbx_xys"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(return_data["K_fullimg"], max_len)
+ return_data["f_imgseq"] = repeat_to_max_len(return_data["f_imgseq"], max_len)
+ return_data["kp2d"] = repeat_to_max_len(return_data["kp2d"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(
+ return_data["cam_angvel"], max_len
+ )
+ return_data["cam_tvel"] = repeat_to_max_len(return_data["cam_tvel"], max_len)
+ return_data["noisy_cam_tvel"] = repeat_to_max_len(
+ return_data["noisy_cam_tvel"], max_len
+ )
+ return_data["T_w2c"] = repeat_to_max_len(return_data["T_w2c"], max_len)
+
+ return return_data
diff --git a/genmo/datasets/threedpw/threedpw_occ_motion_test.py b/genmo/datasets/threedpw/threedpw_occ_motion_test.py
new file mode 100644
index 0000000000000000000000000000000000000000..44675f10c5daa3df174418177c352feb1c302001
--- /dev/null
+++ b/genmo/datasets/threedpw/threedpw_occ_motion_test.py
@@ -0,0 +1,248 @@
+from pathlib import Path
+
+import torch
+from torch.utils import data
+
+from genmo.utils.geo_transform import (
+ compute_cam_angvel,
+ compute_cam_tvel,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import get_valid_mask
+from genmo.utils.pylogger import Log
+from third_party.GVHMR.hmr4d.utils.geo.flip_utils import flip_kp2d_coco17
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import estimate_K, resize_K
+
+VID_HARD = []
+# VID_HARD = ["downtown_bar_00_1"]
+
+
+class ThreedpwOccSmplFullSeqDataset(data.Dataset):
+ def __init__(self, flip_test=False, skip_invalid=False):
+ super().__init__()
+ self.dataset_name = "3DPW_OCC"
+ self.skip_invalid = skip_invalid
+ Log.info(f"[{self.dataset_name}] Full sequence")
+
+ # Load evaluation protocol from WHAM labels
+ self.threedpw_dir = Path("inputs/3DPW/hmr4d_support")
+ # ['vname', 'K_fullimg', 'T_w2c', 'smpl_params', 'gender', 'mask_raw', 'mask_wham', 'img_wh']
+ self.labels = torch.load(self.threedpw_dir / "test_3dpw_gt_labels.pt")
+ self.vid2bbx = torch.load(self.threedpw_dir / "preproc_test_bbx.pt")
+ self.vid2kp2d = torch.load(self.threedpw_dir / "preproc_test_kp2d_v0.pt")
+
+ self.vimo_labels = torch.load(self.threedpw_dir / "test_3dpw_vimo_labels.pt")
+ self.droid_cam_traj = torch.load(self.threedpw_dir / "3dpw_test_slam_traj.pt")
+ # Setup dataset index
+ self.idx2meta = list(self.labels)
+ if len(VID_HARD) > 0: # Pick subsets for fast testing
+ self.idx2meta = VID_HARD
+ Log.info(f"[{self.dataset_name}] {len(self.idx2meta)} sequences.")
+
+ # If flip_test is enabled, we will return extra data for flipped test
+ self.flip_test = flip_test
+ if self.flip_test:
+ Log.info(f"[{self.dataset_name}] Flip test enabled")
+
+ def __len__(self):
+ return len(self.idx2meta)
+
+ def _load_data(self, idx):
+ data = {}
+ vid = self.idx2meta[idx]
+ meta = {"dataset_id": self.dataset_name, "vid": vid}
+ data.update({"meta": meta})
+
+ # Add useful data
+ label = self.labels[vid]
+ vimo_label = self.vimo_labels[vid]
+ droid_label = self.droid_cam_traj[vid]
+
+ mask = label["mask_wham"]
+ width_height = label["img_wh"]
+ vimo_smpl_params = {
+ "pred_cam": vimo_label["vimo_params"]["pred_cam"],
+ "pred_pose": vimo_label["vimo_params"]["pred_pose"],
+ "pred_shape": vimo_label["vimo_params"]["pred_shape"],
+ "pred_trans_c": vimo_label["vimo_params"]["pred_trans"],
+ }
+
+ length = len(mask)
+ data.update(
+ {
+ "length": length, # F
+ "smpl_params": label["smpl_params"], # world
+ "gender": label["gender"], # str
+ # "T_w2c": label["T_w2c"], # (F, 4, 4)
+ "mask": {
+ "valid": mask, # (F)
+ "has_img_mask": get_valid_mask(length, length),
+ "has_2d_mask": get_valid_mask(length, length),
+ "has_cam_mask": get_valid_mask(length, length),
+ "has_audio_mask": get_valid_mask(length, 0),
+ "has_music_mask": get_valid_mask(length, 0),
+ "2d_only": False,
+ },
+ }
+ )
+
+ gt_T_w2c = label["T_w2c"]
+ data.update({"vimo_smpl_params": vimo_smpl_params})
+
+ # camera
+ # load droid slam
+ R_c2w = torch.from_numpy(droid_label["pred_cam_R"]).float()
+ t_c2w = torch.from_numpy(droid_label["pred_cam_T"]).float()
+ scales = torch.from_numpy(droid_label["all_scales"]).float()
+ mean_scale = droid_label["scale"]
+ T_c2w = torch.eye(4)[None].repeat(length, 1, 1).to(R_c2w)
+ T_c2w[:, :3, :3] = R_c2w
+ T_c2w[:, :3, 3] = t_c2w
+ T_w2c = T_c2w.inverse()
+
+ K_fullimg = label["K_fullimg"] # (3, 3)
+ if False:
+ K_fullimg = estimate_K(*width_height)
+ data["K_fullimg"] = K_fullimg
+ data.update(
+ {
+ "T_w2c": T_w2c,
+ "scales": scales,
+ "mean_scale": mean_scale,
+ "gt_T_w2c": gt_T_w2c,
+ }
+ )
+
+ if "vimo_params_flip" in vimo_label:
+ flipped_trans_c = vimo_label["vimo_params_flip"]["pred_trans"]
+ orig_trans_c = data["vimo_smpl_params"]["pred_trans_c"]
+ tz = flipped_trans_c[..., 2]
+ tx = flipped_trans_c[..., 0]
+ focal = K_fullimg[0, 0]
+ cx = K_fullimg[0, 2]
+ width = width_height[0]
+
+ flipped_tx = tz * (width - 1 - 2 * cx) / focal - tx
+ avg_trans_c = torch.zeros_like(flipped_trans_c)
+ avg_trans_c[..., 0] = (flipped_tx + orig_trans_c[..., 0]) / 2
+ avg_trans_c[..., 0] = orig_trans_c[..., 0]
+ avg_trans_c[..., 1] = (flipped_trans_c[..., 1] + orig_trans_c[..., 1]) / 2
+ avg_trans_c[..., 2] = (tz + orig_trans_c[..., 2]) / 2
+ data["vimo_smpl_params"]["pred_trans_c"] = avg_trans_c
+
+ # Preprocessed: bbx, kp2d, image as feature
+ imgfeat_dir = self.threedpw_dir / "imgfeats/3dpw_occ_test"
+ f_img_dict = torch.load(imgfeat_dir / f"{vid}.pth")
+ # bbx_xys = self.vid2bbx[vid]["bbx_xys"] # (F, 3)
+ # kp2d = self.vid2kp2d[vid] # (F, 17, 3)
+ bbx_xys = f_img_dict["bbx_xys"].clone().float()
+ kp2d = f_img_dict["vitpose"].clone().float()
+
+ norm_T_w2c = normalize_T_w2c(data["T_w2c"])
+ cam_angvel = compute_cam_angvel(norm_T_w2c[:, :3, :3]) # (L, 6)
+ cam_tvel = compute_cam_tvel(norm_T_w2c[:, :3, 3]) # (L, 3)
+ data.update(
+ {
+ "bbx_xys": bbx_xys,
+ "kp2d": kp2d,
+ "cam_angvel": cam_angvel,
+ "cam_tvel": cam_tvel,
+ }
+ )
+ data["R_w2c"] = norm_T_w2c[:, :3, :3]
+
+ f_imgseq = f_img_dict["features"].float()
+ kp2d = f_img_dict["vitpose"].clone().float()
+ data["f_imgseq"] = f_imgseq # (F, 1024)
+
+ # to render a video
+ vname = label["vname"]
+ video_path = self.threedpw_dir / f"videos/{vname}.mp4"
+ frame_id = torch.where(mask)[0].long()
+ ds = 0.5
+ K_render = resize_K(K_fullimg, ds)
+ bbx_xys_render = bbx_xys * ds
+ kp2d_render = kp2d.clone()
+ kp2d_render[..., :2] *= ds
+ data["meta_render"] = {
+ "name": vid,
+ "video_path": str(video_path),
+ "ds": ds,
+ "frame_id": frame_id,
+ "K": K_render,
+ "bbx_xys": bbx_xys_render,
+ "kp2d": kp2d_render,
+ }
+
+ if self.flip_test:
+ imgfeat_dir = self.threedpw_dir / "imgfeats/3dpw_occ_test_flip"
+ f_img_dict = torch.load(imgfeat_dir / f"{vid}.pth")
+ flipped_bbx_xys = f_img_dict["bbx_xys"].float() # (L, 3)
+ flipped_features = f_img_dict["features"].float() # (L, 1024)
+ flipped_kp2d = flip_kp2d_coco17(kp2d, width_height[0]) # (L, 17, 3)
+
+ R_flip_x = torch.tensor([[1, 0, 0], [0, -1, 0], [0, 0, -1]]).float()
+ flipped_R_w2c = R_flip_x @ norm_T_w2c[:, :3, :3].clone()
+ flipped_t_w2c = (R_flip_x @ norm_T_w2c[:, :3, 3:].clone())[..., 0]
+ flipped_T_w2c = torch.eye(4)[None].repeat(length, 1, 1).to(flipped_R_w2c)
+ flipped_T_w2c[:, :3, :3] = flipped_R_w2c
+ flipped_T_w2c[:, :3, 3] = flipped_t_w2c
+
+ data_flip = {
+ "bbx_xys": flipped_bbx_xys,
+ "f_imgseq": flipped_features,
+ "kp2d": flipped_kp2d,
+ "cam_angvel": compute_cam_angvel(flipped_R_w2c),
+ "cam_tvel": compute_cam_tvel(flipped_t_w2c),
+ "R_w2c": flipped_R_w2c,
+ }
+ flipped_trans_c = vimo_label["vimo_params_flip"]["pred_trans"]
+ flipped_trans_c[..., 2] = avg_trans_c[..., 2]
+ vimo_smpl_params_flip = {
+ "pred_cam": vimo_label["vimo_params_flip"]["pred_cam"],
+ "pred_pose": vimo_label["vimo_params_flip"]["pred_pose"],
+ "pred_shape": vimo_label["vimo_params_flip"]["pred_shape"],
+ "pred_trans_c": flipped_trans_c,
+ }
+ data_flip["vimo_smpl_params"] = vimo_smpl_params_flip
+
+ flipped_K_fullimg = K_fullimg.clone()
+ data_flip.update(
+ {
+ "K_fullimg": flipped_K_fullimg,
+ "T_w2c": flipped_T_w2c,
+ "scales": scales,
+ "mean_scale": mean_scale,
+ }
+ )
+ data["flip_test"] = data_flip
+ return data
+
+ def _process_data(self, data):
+ length = data["length"]
+ data["K_fullimg"] = data["K_fullimg"][None].repeat(length, 1, 1)
+
+ if self.skip_invalid: # Drop all invalid frames
+ mask = data["mask"].clone()
+ data["length"] = sum(mask)
+ data["smpl_params"] = {
+ k: v[mask].clone() for k, v in data["smpl_params"].items()
+ }
+ data["T_w2c"] = data["T_w2c"][mask].clone()
+ data["mask"] = data["mask"][mask].clone()
+ data["K_fullimg"] = data["K_fullimg"][mask].clone()
+ data["bbx_xys"] = data["bbx_xys"][mask].clone()
+ data["kp2d"] = data["kp2d"][mask].clone()
+ data["cam_angvel"] = data["cam_angvel"][mask].clone()
+ data["cam_tvel"] = data["cam_tvel"][mask].clone()
+ data["f_imgseq"] = data["f_imgseq"][mask].clone()
+ data["flip_test"] = {
+ k: v[mask].clone() for k, v in data["flip_test"].items()
+ }
+
+ return data
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data)
+ return data
diff --git a/genmo/datasets/threedpw/threedpw_occ_motion_train.py b/genmo/datasets/threedpw/threedpw_occ_motion_train.py
new file mode 100644
index 0000000000000000000000000000000000000000..87dd35e6a82a91c5cb11e9cdf24ec4d5da858365
--- /dev/null
+++ b/genmo/datasets/threedpw/threedpw_occ_motion_train.py
@@ -0,0 +1,188 @@
+# from torch.utils import data
+from pathlib import Path
+
+import numpy as np
+import torch
+
+from genmo.datasets.imgfeat_motion.base_dataset import ImgfeatMotionDatasetBase
+from genmo.utils.geo_transform import (
+ compute_cam_angvel,
+ compute_cam_tvel,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import (
+ get_valid_mask,
+ repeat_to_max_len,
+ repeat_to_max_len_dict,
+)
+from genmo.utils.pylogger import Log
+
+
+class ThreedpwOccSmplDataset(ImgfeatMotionDatasetBase):
+ def __init__(self):
+ # Path
+ self.hmr4d_support_dir = Path("inputs/3DPW/hmr4d_support")
+ self.dataset_name = "3DPW_OCC"
+
+ # Setting
+ self.min_motion_frames = 60
+ self.max_motion_frames = 120
+ super().__init__()
+
+ def _load_dataset(self):
+ self.train_labels = torch.load(
+ self.hmr4d_support_dir / "train_3dpw_gt_labels.pt"
+ )
+ self.refit_smplx = torch.load(self.hmr4d_support_dir / "train_refit_smplx.pt")
+ if True: # Remove clips that have obvious error
+ update_list = {
+ "courtyard_basketball_00_1": [(0, 300), (340, 468)],
+ "courtyard_laceShoe_00_0": [(0, 620), (780, 931)],
+ "courtyard_rangeOfMotions_00_1": [(0, 370), (410, 601)],
+ "courtyard_shakeHands_00_1": [(0, 100), (120, 391)],
+ }
+ for k, v in update_list.items():
+ self.refit_smplx[k]["valid_range_list"] = v
+
+ self.f_img_folder = self.hmr4d_support_dir / "imgfeats/3dpw_occ_train"
+ Log.info(f"[{self.dataset_name}] Train")
+
+ def _get_idx2meta(self):
+ # We expect to see the entire sequence during one epoch,
+ # so each sequence will be sampled max(SeqLength // MotionFrames, 1) times
+ seq_lengths = []
+ self.idx2meta = []
+ for vid in self.refit_smplx:
+ valid_range_list = self.refit_smplx[vid]["valid_range_list"]
+ for start, end in valid_range_list:
+ seq_length = end - start
+ num_samples = max(seq_length // self.max_motion_frames, 1)
+ seq_lengths.append(seq_length)
+ self.idx2meta.extend([(vid, start, end)] * num_samples)
+ minutes = sum(seq_lengths) / 25 / 60
+ Log.info(
+ f"[{self.dataset_name}] has {minutes:.1f} minutes motion -> Resampled to {len(self.idx2meta)} samples."
+ )
+
+ def _load_data(self, idx):
+ data = {}
+ vid, range1, range2 = self.idx2meta[idx]
+
+ # Random select a subset
+ mlength = range2 - range1
+ min_motion_len = self.min_motion_frames
+ max_motion_len = self.max_motion_frames
+
+ if (
+ mlength < min_motion_len
+ ): # this may happen, the minimal mlength is around 30
+ start = range1
+ length = mlength
+ else:
+ effect_max_motion_len = min(max_motion_len, mlength)
+ length = np.random.randint(
+ min_motion_len, effect_max_motion_len + 1
+ ) # [low, high)
+ start = np.random.randint(range1, range2 - length + 1)
+ end = start + length
+ data["length"] = length
+ data["meta"] = {
+ "data_name": self.dataset_name,
+ "idx": idx,
+ "vid": vid,
+ "start_end": (start, end),
+ }
+
+ # Select motion subset
+ data["smplx_params_incam"] = {
+ k: v[start:end]
+ for k, v in self.refit_smplx[vid]["smplx_params_incam"].items()
+ }
+ data["K_fullimg"] = self.train_labels[vid]["K_fullimg"]
+ data["T_w2c"] = self.train_labels[vid]["T_w2c"][start:end]
+
+ # Img (as feature):
+ f_img_dict = torch.load(self.f_img_folder / f"{vid}.pth")
+
+ data["bbx_xys"] = f_img_dict["bbx_xys"][start:end] # (F, 3)
+ data["f_imgseq"] = f_img_dict["features"][start:end].float() # (F, 3)
+ data["img_wh"] = f_img_dict["img_wh"] # (2)
+ # data["kp2d"] = torch.zeros((end - start), 17, 3) # (L, 17, 3) # do not provide kp2d
+ data["kp2d"] = f_img_dict["vitpose"][start:end]
+ return data
+
+ def _process_data(self, data, idx):
+ length = data["length"]
+
+ smpl_params_c = data["smplx_params_incam"]
+ smpl_params_w_zero = {k: torch.zeros_like(v) for k, v in smpl_params_c.items()}
+ K_fullimg = data["K_fullimg"][None].repeat(length, 1, 1)
+ T_w2c = data["T_w2c"]
+ normed_T_w2c = normalize_T_w2c(T_w2c)
+ noisy_normed_T_w2c = normed_T_w2c.clone()
+ noisy_t_w2c = noisy_normed_T_w2c[:, :3, 3]
+ rand_scale = min(max(0.1, torch.randn(1) + 3), 10)
+ noisy_t_w2c = noisy_t_w2c / rand_scale
+ noisy_normed_T_w2c[:, :3, 3] = noisy_t_w2c
+
+ cam_angvel = compute_cam_angvel(normed_T_w2c[:, :3, :3])
+ cam_tvel = compute_cam_tvel(normed_T_w2c[:, :3, 3])
+ noisy_cam_tvel = compute_cam_tvel(noisy_normed_T_w2c[:, :3, 3])
+ max_len = self.max_motion_frames
+ return_data = {
+ "meta": data["meta"],
+ "length": length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w_zero,
+ "R_c2gv": torch.zeros(length, 3, 3), # (F, 3, 3)
+ "gravity_vec": torch.zeros(3), # (3)
+ "bbx_xys": data["bbx_xys"], # (F, 3)
+ "K_fullimg": K_fullimg, # (F, 3, 3)
+ "f_imgseq": data["f_imgseq"], # (F, D)
+ "kp2d": data["kp2d"], # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "cam_tvel": cam_tvel, # (F, 3)
+ "noisy_cam_tvel": noisy_cam_tvel, # (F, 3)
+ "T_w2c": normed_T_w2c, # (F, 4, 4)
+ "mask": {
+ "valid": get_valid_mask(max_len, length),
+ "has_img_mask": get_valid_mask(max_len, length),
+ "has_2d_mask": get_valid_mask(max_len, length),
+ "has_cam_mask": get_valid_mask(max_len, length),
+ "has_audio_mask": get_valid_mask(max_len, 0),
+ "has_music_mask": get_valid_mask(max_len, 0),
+ "2d_only": False,
+ "vitpose": True,
+ "bbx_xys": True,
+ "f_imgseq": True,
+ "spv_incam_only": True,
+ "invalid_contact": True,
+ },
+ "use_det_kp": torch.ones(length), # default: False
+ }
+
+ # Batchable
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(
+ return_data["smpl_params_c"], max_len
+ )
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(
+ return_data["smpl_params_w"], max_len
+ )
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["bbx_xys"] = repeat_to_max_len(return_data["bbx_xys"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(return_data["K_fullimg"], max_len)
+ return_data["f_imgseq"] = repeat_to_max_len(return_data["f_imgseq"], max_len)
+ return_data["kp2d"] = repeat_to_max_len(return_data["kp2d"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(
+ return_data["cam_angvel"], max_len
+ )
+ return_data["cam_tvel"] = repeat_to_max_len(return_data["cam_tvel"], max_len)
+ return_data["noisy_cam_tvel"] = repeat_to_max_len(
+ return_data["noisy_cam_tvel"], max_len
+ )
+ return_data["T_w2c"] = repeat_to_max_len(return_data["T_w2c"], max_len)
+ return_data["use_det_kp"] = repeat_to_max_len(
+ return_data["use_det_kp"], max_len
+ )
+
+ return return_data
diff --git a/genmo/datasets/threedpw/utils.py b/genmo/datasets/threedpw/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..ca71b1f662c43d41c22fcc1e6982ef6e6a7928d9
--- /dev/null
+++ b/genmo/datasets/threedpw/utils.py
@@ -0,0 +1,88 @@
+import json
+import pickle
+from collections import defaultdict
+from pathlib import Path
+
+import joblib
+import numpy as np
+import torch
+
+RESOURCE_FOLDER = Path(__file__).resolve().parent / "resource"
+
+
+def read_raw_pkl(pkl_path):
+ with open(pkl_path, "rb") as f:
+ data = pickle.load(f, encoding="bytes")
+
+ num_subjects = len(data[b"poses"])
+ F = data[b"poses"][0].shape[0]
+ smpl_params = []
+ for i in range(num_subjects):
+ smpl_params.append(
+ {
+ "body_pose": torch.from_numpy(
+ data[b"poses"][i][:, 3:72]
+ ).float(), # (F, 69)
+ "betas": torch.from_numpy(data[b"betas"][i][:10])
+ .repeat(F, 1)
+ .float(), # (F, 10)
+ "global_orient": torch.from_numpy(
+ data[b"poses"][i][:, :3]
+ ).float(), # (F, 3)
+ "transl": torch.from_numpy(data[b"trans"][i]).float(), # (F, 3)
+ }
+ )
+ genders = ["male" if g == "m" else "female" for g in data[b"genders"]]
+ campose_valid = [torch.from_numpy(v).bool() for v in data[b"campose_valid"]]
+
+ seq_name = data[b"sequence"]
+ K_fullimg = torch.from_numpy(data[b"cam_intrinsics"]).float()
+ T_w2c = torch.from_numpy(data[b"cam_poses"]).float()
+
+ return_data = {
+ "sequence": seq_name, # 'courtyard_bodyScannerMotions_00'
+ "K_fullimg": K_fullimg, # (3, 3), not 55FoV
+ "T_w2c": T_w2c, # (F, 4, 4)
+ "smpl_params": smpl_params, # list of dict
+ "genders": genders, # list of str
+ "campose_valid": campose_valid, # list of bool-array
+ # "jointPositions": data[b'jointPositions'], # SMPL, 24x3
+ # "poses2d": data[b"poses2d"], # COCO, 3x18(?)
+ }
+ return return_data
+
+
+def load_and_convert_wham_pth(pth):
+ """
+ Convert to {vid: DataDict} style, Add smpl_params_incam
+ """
+ # load
+ wham_labels_raw = joblib.load(pth)
+ # convert it to {vid: DataDict} style
+ wham_labels = {}
+ for i, vid in enumerate(wham_labels_raw["vid"]):
+ wham_labels[vid] = {k: wham_labels_raw[k][i] for k in wham_labels_raw}
+
+ # convert pose and betas as smpl_params_incam (without transl)
+ for vid in wham_labels:
+ pose = wham_labels[vid]["pose"]
+ global_orient = pose[:, :3] # (F, 3)
+ body_pose = pose[:, 3:] # (F, 69)
+ betas = wham_labels[vid]["betas"] # (F, 10), all frames are the same
+ wham_labels[vid]["smpl_params_incam"] = {
+ "body_pose": body_pose.float(), # (F, 69)
+ "betas": betas.float(), # (F, 10)
+ "global_orient": global_orient.float(), # (F, 3)
+ }
+
+ return wham_labels
+
+
+# Neural-Annot utils
+
+
+def na_cam_param_to_K_fullimg(cam_param):
+ K = torch.eye(3)
+ K[[0, 1], [0, 1]] = torch.tensor(cam_param["focal"])
+ K[[0, 1], [2, 2]] = torch.tensor(cam_param["princpt"])
+ return K
diff --git a/genmo/datasets/unity_dataset.py b/genmo/datasets/unity_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..257e74cab1f7530a1de74b976c48aa877fb4c559
--- /dev/null
+++ b/genmo/datasets/unity_dataset.py
@@ -0,0 +1,335 @@
+import torch
+from pathlib import Path
+from torch.utils.data import Dataset
+from genmo.utils.pylogger import Log
+from genmo.utils.net_utils import repeat_to_max_len, repeat_to_max_len_dict, get_valid_mask
+from genmo.utils.geo_transform import compute_cam_angvel, compute_cam_tvel, normalize_T_w2c
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import get_R_c2gv
+import numpy as np
+from pytorch3d.transforms import axis_angle_to_matrix, matrix_to_axis_angle, matrix_to_rotation_6d
+
+class UnityDataset(Dataset):
+ def __init__(
+ self,
+ root,
+ split,
+ motion_frames,
+ convert_world_to_az: bool = True,
+ raw_root=None,
+ vitpose_like: bool = False,
+ kp2d_clamp_to_image: bool = False,
+ kp2d_zero_oof: bool = True,
+ ):
+ self.root = Path(root)
+ self.raw_root = Path(raw_root) if raw_root else None
+ self.split = split
+ self.motion_frames = motion_frames
+ self.convert_world_to_az = bool(convert_world_to_az)
+ self.vitpose_like = bool(vitpose_like)
+ self.kp2d_clamp_to_image = bool(kp2d_clamp_to_image)
+ self.kp2d_zero_oof = bool(kp2d_zero_oof)
+ self._vid2mp4 = {}
+
+ # Path to the .pt files (Cached Features)
+ self.feat_dir = self.root / "genmo_features"
+
+ if not self.feat_dir.exists():
+ raise FileNotFoundError(f"Feature dir not found: {self.feat_dir}")
+
+ self.pt_files = sorted(list(self.feat_dir.glob("*.pt")))
+
+ if not self.pt_files:
+ raise FileNotFoundError(f"No .pt files found in {self.feat_dir}")
+
+ Log.info(f"[UnityDataset] Found {len(self.pt_files)} sequences.")
+
+ def _resolve_mp4_path(self, vid: str) -> str | None:
+ if not self.raw_root:
+ return None
+ vid = str(vid)
+ if vid in self._vid2mp4:
+ return self._vid2mp4[vid]
+
+ candidates = [
+ self.raw_root / f"{vid}.mp4",
+ self.raw_root / f"video_{vid}.mp4",
+ self.raw_root / "videos" / f"{vid}.mp4",
+ self.raw_root / "video" / f"{vid}.mp4",
+ self.raw_root / "mp4" / f"{vid}.mp4",
+ ]
+ for p in candidates:
+ try:
+ if p.exists():
+ self._vid2mp4[vid] = str(p)
+ return self._vid2mp4[vid]
+ except Exception:
+ continue
+
+ # Fallback: search (lazy, per-vid)
+ try:
+ hits = list(self.raw_root.rglob(f"{vid}.mp4"))
+ if hits:
+ self._vid2mp4[vid] = str(hits[0])
+ return self._vid2mp4[vid]
+ except Exception:
+ pass
+
+ self._vid2mp4[vid] = None
+ return None
+
+ def __len__(self):
+ return len(self.pt_files)
+
+ def __getitem__(self, idx):
+ pt_path = self.pt_files[idx]
+
+ # Load data (CPU)
+ data = torch.load(pt_path, map_location="cpu", weights_only=False)
+
+ # 1. Determine Length & Slice
+ full_len = data["f_imgseq"].shape[0]
+
+ if full_len > self.motion_frames:
+ start_idx = torch.randint(0, full_len - self.motion_frames, (1,)).item()
+ end_idx = start_idx + self.motion_frames
+ else:
+ start_idx = 0
+ end_idx = full_len
+
+ length = end_idx - start_idx
+ max_len = self.motion_frames
+
+ def repeat_list_to_max_len(items, max_len):
+ if len(items) == 0:
+ return [""] * max_len
+ if len(items) >= max_len:
+ return items[:max_len]
+ return items + [items[-1]] * (max_len - len(items))
+
+ # Helper to slice tensors
+ def sl(tensor_or_dict):
+ if isinstance(tensor_or_dict, dict):
+ return {k: sl(v) for k, v in tensor_or_dict.items()}
+ if isinstance(tensor_or_dict, (torch.Tensor, np.ndarray)):
+ return tensor_or_dict[start_idx:end_idx]
+ return tensor_or_dict
+
+ # --- 2. Extract Data ---
+ f_imgseq = sl(data["f_imgseq"])
+
+ # Check Dimension for Features (Must be 1024)
+ if f_imgseq.shape[-1] != 1024:
+ # If your features are not 1024, the Linear layer later will also crash.
+ # Assuming they are correct for now, or you might need a projection here.
+ pass
+
+ bbx_xys = sl(data["bbx_xys"]).float()
+ kp2d = sl(data["kp2d"]).float()
+ K_fullimg = sl(data["K_fullimg"])
+ T_w2c = sl(data["T_w2c"])
+
+ if (
+ (self.kp2d_clamp_to_image or self.kp2d_zero_oof)
+ and isinstance(kp2d, torch.Tensor)
+ and isinstance(K_fullimg, torch.Tensor)
+ and kp2d.ndim == 3
+ and K_fullimg.ndim == 3
+ ):
+ try:
+ # Infer image size from intrinsics (K assumes principal point at W/2, H/2).
+ W = (2.0 * K_fullimg[:, 0, 2]).clamp_min(1.0) # (L,)
+ H = (2.0 * K_fullimg[:, 1, 2]).clamp_min(1.0) # (L,)
+ x = kp2d[..., 0]
+ y = kp2d[..., 1]
+ c = kp2d[..., 2]
+ # If a keypoint is out-of-frame, treat it as non-detected (detector-like behavior).
+ if self.kp2d_zero_oof:
+ oof = (x < 0.0) | (x > (W[:, None] - 1.0)) | (y < 0.0) | (y > (H[:, None] - 1.0))
+ c = torch.where(oof, c.new_zeros(()).expand_as(c), c)
+
+ if self.kp2d_clamp_to_image:
+ # Clamp only joints that are considered "visible" by the conf threshold used in training.
+ m = c > 0.5
+ x = torch.where(m, x.clamp(min=0.0, max=W[:, None] - 1.0), x)
+ y = torch.where(m, y.clamp(min=0.0, max=H[:, None] - 1.0), y)
+
+ kp2d = torch.stack([x, y, c], dim=-1)
+ except Exception:
+ pass
+
+ if self.convert_world_to_az:
+ # Unity exports are in a Y-up world (x right, y up, z forward) after the CV conversion in
+ # `process_dataset.py`. GENMO datasets expect world ("az") to be Z-up (x right, y forward, z up).
+ #
+ # Convert basis:
+ # [x', y', z'] = [x, z, y]
+ # This makes z' the up axis and matches the gravity convention used across GENMO datasets.
+ P = torch.tensor(
+ [[1.0, 0.0, 0.0], [0.0, 0.0, 1.0], [0.0, 1.0, 0.0]],
+ dtype=T_w2c.dtype,
+ )
+
+ # World translation vectors (L,3): v' = P v
+ if "smpl_params_w" in data:
+ pass # keep mypy happy; actual dict is `smpl_params_w` below
+
+ # T_w2c rotation maps world->cam: R' = R @ P^{-1} = R @ P^T
+ T_w2c = T_w2c.clone()
+ T_w2c[:, :3, :3] = T_w2c[:, :3, :3] @ P.T
+
+ # Extrinsics calc
+ R_w2c = T_w2c[:, :3, :3]
+ # Gravity direction in the current world convention:
+ # - Unity exports (after `tools/demo/process_dataset.py`) are Y-up -> gravity is -Y.
+ # - GENMO "az" convention is Z-up -> gravity is -Z.
+ if self.convert_world_to_az:
+ gravity_vec = torch.tensor([0.0, 0.0, -1.0], dtype=R_w2c.dtype)
+ else:
+ gravity_vec = torch.tensor([0.0, -1.0, 0.0], dtype=R_w2c.dtype)
+ R_c2gv = get_R_c2gv(R_w2c, axis_gravity_in_w=gravity_vec)
+
+ smpl_params_c = sl(data["smpl_params_c"])
+ smpl_params_w = sl(data["smpl_params_w"])
+
+ if self.convert_world_to_az:
+ P = torch.tensor(
+ [[1.0, 0.0, 0.0], [0.0, 0.0, 1.0], [0.0, 1.0, 0.0]],
+ dtype=smpl_params_w["transl"].dtype,
+ )
+ smpl_params_w = dict(smpl_params_w)
+ smpl_params_w["transl"] = smpl_params_w["transl"] @ P.T
+
+ R_w_old = axis_angle_to_matrix(smpl_params_w["global_orient"])
+ R_w_new = P.to(R_w_old) @ R_w_old
+ smpl_params_w["global_orient"] = matrix_to_axis_angle(R_w_new)
+
+ # Prepare smpl_params for MetricMocap (needs 23 joints / 69 dims)
+ smpl_params_metric = {}
+ for k, v in smpl_params_w.items():
+ if k == "body_pose" and v.shape[-1] == 63:
+ padding = torch.zeros((v.shape[0], 6), dtype=v.dtype, device=v.device)
+ smpl_params_metric[k] = torch.cat([v, padding], dim=-1)
+ else:
+ smpl_params_metric[k] = v
+
+ cam_angvel = sl(data["cam_angvel"]).float() if "cam_angvel" in data else None
+ cam_tvel = sl(data["cam_tvel"]).float() if "cam_tvel" in data else None
+
+ # IMPORTANT: GENMO pretraining datasets (e.g., BEDLAM/3DPW/EMDB) compute camera velocities
+ # from a *normalized* camera trajectory (first-frame aligned), via `normalize_T_w2c`.
+ #
+ # For Unity, we may have both:
+ # - `T_w2c`: exported/GT camera (aligned to the Unity world used for GT SMPL).
+ # - `T_w2c_dpvo`: DPVO-estimated camera (inference-style), useful to match GENMO inference conditioning.
+ #
+ # Keep `T_w2c` as exported (for GT/metrics/vis), but compute vel features from:
+ # `T_w2c_dpvo` if present, else `T_w2c`,
+ # and always normalize the chosen trajectory before differentiating.
+ try:
+ T_for_vel = sl(data.get("T_w2c_dpvo", None))
+ if not isinstance(T_for_vel, torch.Tensor) or T_for_vel.ndim != 3:
+ T_for_vel = T_w2c
+ normed_T_w2c = normalize_T_w2c(T_for_vel)
+ cam_angvel = compute_cam_angvel(normed_T_w2c[:, :3, :3]).to(torch.float32)
+ cam_tvel = compute_cam_tvel(normed_T_w2c[:, :3, 3]).to(torch.float32)
+ except Exception:
+ # Fallback to exported velocities if present; otherwise compute a simple diff.
+ if cam_angvel is None or cam_tvel is None:
+ R_w2c = T_w2c[:, :3, :3]
+ t_w2c = T_w2c[:, :3, 3]
+ L = int(R_w2c.shape[0])
+
+ cam_angvel = torch.zeros((L, 6), dtype=R_w2c.dtype)
+ cam_angvel[0] = cam_angvel.new_tensor([1.0, 0.0, 0.0, 0.0, 1.0, 0.0])
+ if L > 1:
+ R_diff = R_w2c[1:] @ R_w2c[:-1].transpose(-1, -2) # (L-1, 3, 3)
+ cam_angvel[1:] = matrix_to_rotation_6d(R_diff).to(cam_angvel)
+
+ cam_tvel = torch.zeros((L, 3), dtype=t_w2c.dtype)
+ if L > 1:
+ cam_tvel[1:] = (t_w2c[1:] - t_w2c[:-1]).to(cam_tvel)
+
+ imgname = data.get("imgname", None)
+ frame_ids = None
+ if isinstance(imgname, list):
+ imgname = imgname[start_idx:end_idx]
+ img_paths = [str(self.root / p) for p in imgname]
+ img_paths = repeat_list_to_max_len(img_paths, max_len)
+ try:
+ # imgname like "images//img_00012.jpg"
+ frame_ids = []
+ for p in imgname:
+ s = str(p)
+ stem = s.split("/")[-1].split(".")[0]
+ frame_ids.append(int(stem.split("_")[-1]))
+ frame_ids = repeat_list_to_max_len(frame_ids, max_len)
+ except Exception:
+ frame_ids = None
+ else:
+ img_paths = None
+
+ gender = str(data.get("gender", "female"))
+
+ # --- 3. Construct Return Dict ---
+ ret = {
+ "meta": {
+ "id": pt_path.stem,
+ "dataset_id": "Unity",
+ "vid": pt_path.stem,
+ },
+ # Rendering-only metadata. Collate keeps `meta*` keys as lists without stacking.
+ "meta_render": {
+ "img_paths": img_paths,
+ "frame_ids": frame_ids,
+ "video_path": self._resolve_mp4_path(pt_path.stem),
+ },
+ "gender": gender,
+ "length": length,
+
+ # Input Features
+ "f_imgseq": repeat_to_max_len(f_imgseq, max_len),
+ "bbx_xys": repeat_to_max_len(bbx_xys, max_len),
+ "kp2d": repeat_to_max_len(kp2d, max_len),
+ "cam_angvel": repeat_to_max_len(cam_angvel, max_len),
+ "cam_tvel": repeat_to_max_len(cam_tvel, max_len),
+
+ # Ground Truth
+ "smpl_params": repeat_to_max_len_dict(smpl_params_metric, max_len),
+ "smpl_params_c": repeat_to_max_len_dict(smpl_params_c, max_len),
+ "smpl_params_w": repeat_to_max_len_dict(smpl_params_w, max_len),
+ "gt_T_w2c": repeat_to_max_len(T_w2c, max_len),
+ "T_w2c": repeat_to_max_len(T_w2c, max_len),
+ "R_c2gv": repeat_to_max_len(R_c2gv, max_len),
+ "K_fullimg": repeat_to_max_len(K_fullimg, max_len),
+
+ # --- Text Fix (Prevents KeyError/crash if text encoder is on) ---
+ "caption": "",
+ "has_text": False,
+
+ "mask": {
+ # 1. Tensor Masks (Per frame)
+ "valid": get_valid_mask(max_len, length),
+ "has_img_mask": get_valid_mask(max_len, length),
+ "has_2d_mask": get_valid_mask(max_len, length),
+ "has_cam_mask": get_valid_mask(max_len, length),
+
+ # 2. Scalar Flags
+ "bbx_xys": True,
+ "f_imgseq": True,
+ "2d_only": False,
+ "spv_incam_only": False,
+ # If you want to mimic 3DPW-Occ conditioning (detector keypoints), set `vitpose_like=True`.
+ "vitpose": bool(self.vitpose_like),
+ "invalid_contact": False,
+
+ # 3. Missing Modalities
+ "has_audio_mask": torch.zeros(max_len).bool(),
+ "has_music_mask": torch.zeros(max_len).bool(),
+ }
+ }
+
+ if self.vitpose_like:
+ # Match the key used by 3DPW-Occ so `genmo/genmo.py` can use detector keypoints.
+ ret["use_det_kp"] = torch.ones(max_len)
+
+ return ret
diff --git a/genmo/diffusion_utils/fp16_util.py b/genmo/diffusion_utils/fp16_util.py
new file mode 100644
index 0000000000000000000000000000000000000000..57d7bf438ca33ea02b08761b03201e5eaf20c3c6
--- /dev/null
+++ b/genmo/diffusion_utils/fp16_util.py
@@ -0,0 +1,236 @@
+"""
+Helpers to train with 16-bit precision.
+"""
+
+import numpy as np
+import torch as th
+import torch.nn as nn
+from torch._utils import _flatten_dense_tensors, _unflatten_dense_tensors
+
+from genmo.diffusion_utils import logger
+
+INITIAL_LOG_LOSS_SCALE = 20.0
+
+
+def convert_module_to_f16(l):
+ """
+ Convert primitive modules to float16.
+ """
+ if isinstance(l, (nn.Conv1d, nn.Conv2d, nn.Conv3d)):
+ l.weight.data = l.weight.data.half()
+ if l.bias is not None:
+ l.bias.data = l.bias.data.half()
+
+
+def convert_module_to_f32(l):
+ """
+ Convert primitive modules to float32, undoing convert_module_to_f16().
+ """
+ if isinstance(l, (nn.Conv1d, nn.Conv2d, nn.Conv3d)):
+ l.weight.data = l.weight.data.float()
+ if l.bias is not None:
+ l.bias.data = l.bias.data.float()
+
+
+def make_master_params(param_groups_and_shapes):
+ """
+ Copy model parameters into a (differently-shaped) list of full-precision
+ parameters.
+ """
+ master_params = []
+ for param_group, shape in param_groups_and_shapes:
+ master_param = nn.Parameter(
+ _flatten_dense_tensors(
+ [param.detach().float() for (_, param) in param_group]
+ ).view(shape)
+ )
+ master_param.requires_grad = True
+ master_params.append(master_param)
+ return master_params
+
+
+def model_grads_to_master_grads(param_groups_and_shapes, master_params):
+ """
+ Copy the gradients from the model parameters into the master parameters
+ from make_master_params().
+ """
+ for master_param, (param_group, shape) in zip(
+ master_params, param_groups_and_shapes
+ ):
+ master_param.grad = _flatten_dense_tensors(
+ [param_grad_or_zeros(param) for (_, param) in param_group]
+ ).view(shape)
+
+
+def master_params_to_model_params(param_groups_and_shapes, master_params):
+ """
+ Copy the master parameter data back into the model parameters.
+ """
+ # Without copying to a list, if a generator is passed, this will
+ # silently not copy any parameters.
+ for master_param, (param_group, _) in zip(master_params, param_groups_and_shapes):
+ for (_, param), unflat_master_param in zip(
+ param_group, unflatten_master_params(param_group, master_param.view(-1))
+ ):
+ param.detach().copy_(unflat_master_param)
+
+
+def unflatten_master_params(param_group, master_param):
+ return _unflatten_dense_tensors(master_param, [param for (_, param) in param_group])
+
+
+def get_param_groups_and_shapes(named_model_params):
+ named_model_params = list(named_model_params)
+ scalar_vector_named_params = (
+ [(n, p) for (n, p) in named_model_params if p.ndim <= 1],
+ (-1),
+ )
+ matrix_named_params = (
+ [(n, p) for (n, p) in named_model_params if p.ndim > 1],
+ (1, -1),
+ )
+ return [scalar_vector_named_params, matrix_named_params]
+
+
+def master_params_to_state_dict(
+ model, param_groups_and_shapes, master_params, use_fp16
+):
+ if use_fp16:
+ state_dict = model.state_dict()
+ for master_param, (param_group, _) in zip(
+ master_params, param_groups_and_shapes
+ ):
+ for (name, _), unflat_master_param in zip(
+ param_group, unflatten_master_params(param_group, master_param.view(-1))
+ ):
+ assert name in state_dict
+ state_dict[name] = unflat_master_param
+ else:
+ state_dict = model.state_dict()
+ for i, (name, _value) in enumerate(model.named_parameters()):
+ assert name in state_dict
+ state_dict[name] = master_params[i]
+ return state_dict
+
+
+def state_dict_to_master_params(model, state_dict, use_fp16):
+ if use_fp16:
+ named_model_params = [
+ (name, state_dict[name]) for name, _ in model.named_parameters()
+ ]
+ param_groups_and_shapes = get_param_groups_and_shapes(named_model_params)
+ master_params = make_master_params(param_groups_and_shapes)
+ else:
+ master_params = [state_dict[name] for name, _ in model.named_parameters()]
+ return master_params
+
+
+def zero_master_grads(master_params):
+ for param in master_params:
+ param.grad = None
+
+
+def zero_grad(model_params):
+ for param in model_params:
+ # Taken from https://pytorch.org/docs/stable/_modules/torch/optim/optimizer.html#Optimizer.add_param_group
+ if param.grad is not None:
+ param.grad.detach_()
+ param.grad.zero_()
+
+
+def param_grad_or_zeros(param):
+ if param.grad is not None:
+ return param.grad.data.detach()
+ else:
+ return th.zeros_like(param)
+
+
+class MixedPrecisionTrainer:
+ def __init__(
+ self,
+ *,
+ model,
+ use_fp16=False,
+ fp16_scale_growth=1e-3,
+ initial_lg_loss_scale=INITIAL_LOG_LOSS_SCALE,
+ ):
+ self.model = model
+ self.use_fp16 = use_fp16
+ self.fp16_scale_growth = fp16_scale_growth
+
+ self.model_params = list(self.model.parameters())
+ self.master_params = self.model_params
+ self.param_groups_and_shapes = None
+ self.lg_loss_scale = initial_lg_loss_scale
+
+ if self.use_fp16:
+ self.param_groups_and_shapes = get_param_groups_and_shapes(
+ self.model.named_parameters()
+ )
+ self.master_params = make_master_params(self.param_groups_and_shapes)
+ self.model.convert_to_fp16()
+
+ def zero_grad(self):
+ zero_grad(self.model_params)
+
+ def backward(self, loss: th.Tensor):
+ if self.use_fp16:
+ loss_scale = 2**self.lg_loss_scale
+ (loss * loss_scale).backward()
+ else:
+ loss.backward()
+
+ def optimize(self, opt: th.optim.Optimizer):
+ if self.use_fp16:
+ return self._optimize_fp16(opt)
+ else:
+ return self._optimize_normal(opt)
+
+ def _optimize_fp16(self, opt: th.optim.Optimizer):
+ logger.logkv_mean("lg_loss_scale", self.lg_loss_scale)
+ model_grads_to_master_grads(self.param_groups_and_shapes, self.master_params)
+ grad_norm, param_norm = self._compute_norms(grad_scale=2**self.lg_loss_scale)
+ if check_overflow(grad_norm):
+ self.lg_loss_scale -= 1
+ logger.log(f"Found NaN, decreased lg_loss_scale to {self.lg_loss_scale}")
+ zero_master_grads(self.master_params)
+ return False
+
+ logger.logkv_mean("grad_norm", grad_norm)
+ logger.logkv_mean("param_norm", param_norm)
+
+ self.master_params[0].grad.mul_(1.0 / (2**self.lg_loss_scale))
+ opt.step()
+ zero_master_grads(self.master_params)
+ master_params_to_model_params(self.param_groups_and_shapes, self.master_params)
+ self.lg_loss_scale += self.fp16_scale_growth
+ return True
+
+ def _optimize_normal(self, opt: th.optim.Optimizer):
+ grad_norm, param_norm = self._compute_norms()
+ logger.logkv_mean("grad_norm", grad_norm)
+ logger.logkv_mean("param_norm", param_norm)
+ opt.step()
+ return True
+
+ def _compute_norms(self, grad_scale=1.0):
+ grad_norm = 0.0
+ param_norm = 0.0
+ for p in self.master_params:
+ with th.no_grad():
+ param_norm += th.norm(p, p=2, dtype=th.float32).item() ** 2
+ if p.grad is not None:
+ grad_norm += th.norm(p.grad, p=2, dtype=th.float32).item() ** 2
+ return np.sqrt(grad_norm) / grad_scale, np.sqrt(param_norm)
+
+ def master_params_to_state_dict(self, master_params):
+ return master_params_to_state_dict(
+ self.model, self.param_groups_and_shapes, master_params, self.use_fp16
+ )
+
+ def state_dict_to_master_params(self, state_dict):
+ return state_dict_to_master_params(self.model, state_dict, self.use_fp16)
+
+
+def check_overflow(value):
+ return (value == float("inf")) or (value == -float("inf")) or (value != value)
diff --git a/genmo/diffusion_utils/gaussian_diffusion.py b/genmo/diffusion_utils/gaussian_diffusion.py
new file mode 100644
index 0000000000000000000000000000000000000000..cd583c88752a5626d8c64667303a82fc5dc9c8a4
--- /dev/null
+++ b/genmo/diffusion_utils/gaussian_diffusion.py
@@ -0,0 +1,2067 @@
+# This code is based on https://github.com/openai/guided-diffusion
+"""
+This code started out as a PyTorch port of Ho et al's diffusion models:
+https://github.com/hojonathanho/diffusion/blob/1e0dceb3b3495bbe19116a5e1b3596cd0706c543/diffusion_tf/diffusion_utils_2.py
+https://github.com/openai/guided-diffusion/blob/main/guided_diffusion/gaussian_diffusion.py
+Docstrings have been added, as well as DDIM sampling and a new collection of beta schedules.
+"""
+
+import enum
+import math
+import random
+from copy import deepcopy
+
+import numpy as np
+import torch
+import torch as th
+
+# from torchmin import minimize
+from torch.autograd import Variable
+
+from genmo.diffusion_utils.losses import discretized_gaussian_log_likelihood, normal_kl
+from genmo.diffusion_utils.nn import mean_flat, sum_flat
+from genmo.utils.rotation_conversions import (
+ matrix_to_rotation_6d,
+ rotation_6d_to_matrix,
+)
+
+
+def gmof(res, sigma):
+ """
+ Geman-McClure error function
+ - residual
+ - sigma scaling factor
+ """
+ x_squared = res**2
+ sigma_squared = sigma**2
+ return (sigma_squared * x_squared) / (sigma_squared + x_squared)
+
+
+def get_named_beta_schedule(schedule_name, num_diffusion_timesteps, scale_betas=1.0):
+ """
+ Get a pre-defined beta schedule for the given name.
+
+ The beta schedule library consists of beta schedules which remain similar
+ in the limit of num_diffusion_timesteps.
+ Beta schedules may be added, but should not be removed or changed once
+ they are committed to maintain backwards compatibility.
+ """
+ if schedule_name == "linear":
+ # Linear schedule from Ho et al, extended to work for any number of
+ # diffusion steps.
+ scale = scale_betas * 1000 / num_diffusion_timesteps
+ beta_start = scale * 0.0001
+ beta_end = scale * 0.02
+ return np.linspace(
+ beta_start, beta_end, num_diffusion_timesteps, dtype=np.float64
+ )
+ elif schedule_name == "cosine":
+ return betas_for_alpha_bar(
+ num_diffusion_timesteps,
+ lambda t: math.cos((t + 0.008) / 1.008 * math.pi / 2) ** 2,
+ )
+ else:
+ raise NotImplementedError(f"unknown beta schedule: {schedule_name}")
+
+
+def betas_for_alpha_bar(num_diffusion_timesteps, alpha_bar, max_beta=0.999):
+ """
+ Create a beta schedule that discretizes the given alpha_t_bar function,
+ which defines the cumulative product of (1-beta) over time from t = [0,1].
+
+ :param num_diffusion_timesteps: the number of betas to produce.
+ :param alpha_bar: a lambda that takes an argument t from 0 to 1 and
+ produces the cumulative product of (1-beta) up to that
+ part of the diffusion process.
+ :param max_beta: the maximum beta to use; use values lower than 1 to
+ prevent singularities.
+ """
+ betas = []
+ for i in range(num_diffusion_timesteps):
+ t1 = i / num_diffusion_timesteps
+ t2 = (i + 1) / num_diffusion_timesteps
+ betas.append(min(1 - alpha_bar(t2) / alpha_bar(t1), max_beta))
+ return np.array(betas)
+
+
+class ModelMeanType(enum.Enum):
+ """
+ Which type of output the model predicts.
+ """
+
+ PREVIOUS_X = enum.auto() # the model predicts x_{t-1}
+ START_X = enum.auto() # the model predicts x_0
+ EPSILON = enum.auto() # the model predicts epsilon
+
+
+class ModelVarType(enum.Enum):
+ """
+ What is used as the model's output variance.
+
+ The LEARNED_RANGE option has been added to allow the model to predict
+ values between FIXED_SMALL and FIXED_LARGE, making its job easier.
+ """
+
+ LEARNED = enum.auto()
+ FIXED_SMALL = enum.auto()
+ FIXED_LARGE = enum.auto()
+ LEARNED_RANGE = enum.auto()
+
+
+class LossType(enum.Enum):
+ MSE = enum.auto() # use raw MSE loss (and KL when learning variances)
+ RESCALED_MSE = (
+ enum.auto()
+ ) # use raw MSE loss (with RESCALED_KL when learning variances)
+ KL = enum.auto() # use the variational lower-bound
+ RESCALED_KL = enum.auto() # like KL, but rescale to estimate the full VLB
+
+ def is_vb(self):
+ return self == LossType.KL or self == LossType.RESCALED_KL
+
+
+class GaussianDiffusion:
+ """
+ Utilities for training and sampling diffusion models.
+
+ Ported directly from here, and then adapted over time to further experimentation.
+ https://github.com/hojonathanho/diffusion/blob/1e0dceb3b3495bbe19116a5e1b3596cd0706c543/diffusion_tf/diffusion_utils_2.py#L42
+
+ :param betas: a 1-D numpy array of betas for each diffusion timestep,
+ starting at T and going to 1.
+ :param model_mean_type: a ModelMeanType determining what the model outputs.
+ :param model_var_type: a ModelVarType determining how variance is output.
+ :param loss_type: a LossType determining the loss function to use.
+ :param rescale_timesteps: if True, pass floating point timesteps into the
+ model so that they are always scaled like in the
+ original paper (0 to 1000).
+ """
+
+ def __init__(
+ self,
+ *,
+ betas,
+ model_mean_type,
+ model_var_type,
+ loss_type,
+ rescale_timesteps=False,
+ ):
+ self.model_mean_type = model_mean_type
+ self.model_var_type = model_var_type
+ self.loss_type = loss_type
+ self.rescale_timesteps = rescale_timesteps
+
+ # Use float64 for accuracy.
+ betas = np.array(betas, dtype=np.float64)
+ self.betas = betas
+ assert len(betas.shape) == 1, "betas must be 1-D"
+ assert (betas > 0).all() and (betas <= 1).all()
+
+ self.num_timesteps = int(betas.shape[0])
+
+ alphas = 1.0 - betas
+ self.alphas_cumprod = np.cumprod(alphas, axis=0)
+ self.alphas_cumprod_prev = np.append(1.0, self.alphas_cumprod[:-1])
+ self.alphas_cumprod_next = np.append(self.alphas_cumprod[1:], 0.0)
+ assert self.alphas_cumprod_prev.shape == (self.num_timesteps,)
+
+ # calculations for diffusion q(x_t | x_{t-1}) and others
+ self.sqrt_alphas_cumprod = np.sqrt(self.alphas_cumprod)
+ self.sqrt_one_minus_alphas_cumprod = np.sqrt(1.0 - self.alphas_cumprod)
+ self.log_one_minus_alphas_cumprod = np.log(1.0 - self.alphas_cumprod)
+ self.sqrt_recip_alphas_cumprod = np.sqrt(1.0 / self.alphas_cumprod)
+ self.sqrt_recipm1_alphas_cumprod = np.sqrt(1.0 / self.alphas_cumprod - 1)
+
+ # calculations for posterior q(x_{t-1} | x_t, x_0)
+ self.posterior_variance = (
+ betas * (1.0 - self.alphas_cumprod_prev) / (1.0 - self.alphas_cumprod)
+ )
+ # log calculation clipped because the posterior variance is 0 at the
+ # beginning of the diffusion chain.
+ if len(self.posterior_variance) == 1:
+ self.posterior_log_variance_clipped = np.array([0.0])
+ else:
+ self.posterior_log_variance_clipped = np.log(
+ np.append(self.posterior_variance[1], self.posterior_variance[1:])
+ )
+ self.posterior_mean_coef1 = (
+ betas * np.sqrt(self.alphas_cumprod_prev) / (1.0 - self.alphas_cumprod)
+ )
+ self.posterior_mean_coef2 = (
+ (1.0 - self.alphas_cumprod_prev)
+ * np.sqrt(alphas)
+ / (1.0 - self.alphas_cumprod)
+ )
+
+ self.l2_loss = (
+ lambda a, b: (a - b) ** 2
+ ) # th.nn.MSELoss(reduction='none') # must be None for handling mask later on.
+
+ def masked_l2(self, a, b, mask):
+ # assuming a.shape == b.shape == bs, J, Jdim, seqlen
+ # assuming mask.shape == bs, 1, 1, seqlen
+ loss = self.l2_loss(a, b)
+ loss = sum_flat(
+ loss * mask.float()
+ ) # gives \sigma_euclidean over unmasked elements
+ n_entries = a.shape[1] * a.shape[2]
+ non_zero_elements = sum_flat(mask) * n_entries
+ # print('mask', mask.shape)
+ # print('non_zero_elements', non_zero_elements)
+ # print('loss', loss)
+ mse_loss_val = loss / non_zero_elements
+ # print('mse_loss_val', mse_loss_val)
+ return mse_loss_val
+
+ def q_mean_variance(self, x_start, t):
+ """
+ Get the distribution q(x_t | x_0).
+
+ :param x_start: the [N x C x ...] tensor of noiseless inputs.
+ :param t: the number of diffusion steps (minus 1). Here, 0 means one step.
+ :return: A tuple (mean, variance, log_variance), all of x_start's shape.
+ """
+ mean = (
+ _extract_into_tensor(self.sqrt_alphas_cumprod, t, x_start.shape) * x_start
+ )
+ variance = _extract_into_tensor(1.0 - self.alphas_cumprod, t, x_start.shape)
+ log_variance = _extract_into_tensor(
+ self.log_one_minus_alphas_cumprod, t, x_start.shape
+ )
+ return mean, variance, log_variance
+
+ def q_sample(self, x_start, t, noise=None):
+ """
+ Diffuse the dataset for a given number of diffusion steps.
+
+ In other words, sample from q(x_t | x_0).
+
+ :param x_start: the initial dataset batch.
+ :param t: the number of diffusion steps (minus 1). Here, 0 means one step.
+ :param noise: if specified, the split-out normal noise.
+ :return: A noisy version of x_start.
+ """
+ if noise is None:
+ noise = th.randn_like(x_start)
+ assert noise.shape == x_start.shape
+ return (
+ _extract_into_tensor(self.sqrt_alphas_cumprod, t, x_start.shape) * x_start
+ + _extract_into_tensor(self.sqrt_one_minus_alphas_cumprod, t, x_start.shape)
+ * noise
+ )
+
+ def q_posterior_mean_variance(self, x_start, x_t, t):
+ """
+ Compute the mean and variance of the diffusion posterior:
+
+ q(x_{t-1} | x_t, x_0)
+
+ """
+ assert x_start.shape == x_t.shape
+ posterior_mean = (
+ _extract_into_tensor(self.posterior_mean_coef1, t, x_t.shape) * x_start
+ + _extract_into_tensor(self.posterior_mean_coef2, t, x_t.shape) * x_t
+ )
+ posterior_variance = _extract_into_tensor(self.posterior_variance, t, x_t.shape)
+ posterior_log_variance_clipped = _extract_into_tensor(
+ self.posterior_log_variance_clipped, t, x_t.shape
+ )
+ assert (
+ posterior_mean.shape[0]
+ == posterior_variance.shape[0]
+ == posterior_log_variance_clipped.shape[0]
+ == x_start.shape[0]
+ )
+ return posterior_mean, posterior_variance, posterior_log_variance_clipped
+
+ def p_mean_variance_guided(
+ self,
+ model,
+ x,
+ t,
+ clip_denoised=True,
+ denoised_fn=None,
+ model_kwargs=None,
+ guide=None,
+ target_motion=None,
+ ):
+ """
+ Apply the model to get p(x_{t-1} | x_t), as well as a prediction of
+ the initial x, x_0.
+
+ :param model: the model, which takes a signal and a batch of timesteps
+ as input.
+ :param x: the [N x C x ...] tensor at time t.
+ :param t: a 1-D Tensor of timesteps.
+ :param clip_denoised: if True, clip the denoised signal into [-1, 1].
+ :param denoised_fn: if not None, a function which applies to the
+ x_start prediction before it is used to sample. Applies before
+ clip_denoised.
+ :param model_kwargs: if not None, a dict of extra keyword arguments to
+ pass to the model. This can be used for conditioning.
+ :return: a dict with the following keys:
+ - 'mean': the model mean output.
+ - 'variance': the model variance output.
+ - 'log_variance': the log of 'variance'.
+ - 'pred_xstart': the prediction for x_0.
+ """
+ if model_kwargs is None:
+ model_kwargs = {}
+
+ B, C = x.shape[:2]
+ assert t.shape == (B,)
+ # print(self._scale_timesteps(t).max())
+ with th.enable_grad():
+ x = x.detach().requires_grad_()
+ model_output = model(x, self._scale_timesteps(t), **model_kwargs)
+
+ if self.model_var_type in [ModelVarType.LEARNED, ModelVarType.LEARNED_RANGE]:
+ assert model_output.shape == (B, C * 2, *x.shape[2:])
+ model_output, model_var_values = th.split(model_output, C, dim=1)
+ if self.model_var_type == ModelVarType.LEARNED:
+ model_log_variance = model_var_values
+ model_variance = th.exp(model_log_variance)
+ else:
+ min_log = _extract_into_tensor(
+ self.posterior_log_variance_clipped, t, x.shape
+ )
+ max_log = _extract_into_tensor(np.log(self.betas), t, x.shape)
+ # The model_var_values is [-1, 1] for [min_var, max_var].
+ frac = (model_var_values + 1) / 2
+ model_log_variance = frac * max_log + (1 - frac) * min_log
+ model_variance = th.exp(model_log_variance)
+ else:
+ model_variance, model_log_variance = {
+ # for fixedlarge, we set the initial (log-)variance like so
+ # to get a better decoder log likelihood.
+ ModelVarType.FIXED_LARGE: (
+ np.append(self.posterior_variance[1], self.betas[1:]),
+ np.log(np.append(self.posterior_variance[1], self.betas[1:])),
+ ),
+ ModelVarType.FIXED_SMALL: (
+ self.posterior_variance,
+ self.posterior_log_variance_clipped,
+ ),
+ }[self.model_var_type]
+ model_variance = _extract_into_tensor(model_variance, t, x.shape)
+ model_log_variance = _extract_into_tensor(model_log_variance, t, x.shape)
+
+ if guide is not None:
+ model_output = guide.guide(
+ x, model_output, target_motion, model_variance, t
+ )
+
+ def process_xstart(x):
+ if denoised_fn is not None:
+ x = denoised_fn(x, t)
+ if clip_denoised:
+ return x.clamp(-1, 1)
+ return x
+
+ if self.model_mean_type == ModelMeanType.PREVIOUS_X:
+ pred_xstart = process_xstart(
+ self._predict_xstart_from_xprev(x_t=x, t=t, xprev=model_output)
+ )
+ model_mean = model_output
+ elif self.model_mean_type in [ModelMeanType.START_X, ModelMeanType.EPSILON]:
+ if self.model_mean_type == ModelMeanType.START_X:
+ pred_xstart = process_xstart(model_output)
+ else:
+ pred_xstart = process_xstart(
+ self._predict_xstart_from_eps(x_t=x, t=t, eps=model_output)
+ )
+ model_mean, _, _ = self.q_posterior_mean_variance(
+ x_start=pred_xstart, x_t=x, t=t
+ )
+ else:
+ raise NotImplementedError(self.model_mean_type)
+
+ assert (
+ model_mean.shape == model_log_variance.shape == pred_xstart.shape == x.shape
+ )
+ return {
+ "mean": model_mean,
+ "variance": model_variance,
+ "log_variance": model_log_variance,
+ "pred_xstart": pred_xstart,
+ }
+
+ def p_mean_variance(
+ self,
+ model,
+ x,
+ t,
+ clip_denoised=True,
+ denoised_fn=None,
+ model_kwargs=None,
+ model_output=None,
+ ):
+ """
+ Apply the model to get p(x_{t-1} | x_t), as well as a prediction of
+ the initial x, x_0.
+
+ :param model: the model, which takes a signal and a batch of timesteps
+ as input.
+ :param x: the [N x C x ...] tensor at time t.
+ :param t: a 1-D Tensor of timesteps.
+ :param clip_denoised: if True, clip the denoised signal into [-1, 1].
+ :param denoised_fn: if not None, a function which applies to the
+ x_start prediction before it is used to sample. Applies before
+ clip_denoised.
+ :param model_kwargs: if not None, a dict of extra keyword arguments to
+ pass to the model. This can be used for conditioning.
+ :return: a dict with the following keys:
+ - 'mean': the model mean output.
+ - 'variance': the model variance output.
+ - 'log_variance': the log of 'variance'.
+ - 'pred_xstart': the prediction for x_0.
+ """
+ if model_kwargs is None:
+ model_kwargs = {}
+
+ B, C = x.shape[:2]
+ assert t.shape == (B,), (t.shape, B, x.shape)
+
+ if model_output is None:
+ model_output = model(x, self._scale_timesteps(t), **model_kwargs)
+ aux_output = {}
+ if isinstance(model_output, dict):
+ for k, v in model_output.items():
+ if k != "pred_x_start":
+ aux_output[k] = v
+ model_output = model_output["pred_x_start"]
+ if self.model_var_type in [ModelVarType.LEARNED, ModelVarType.LEARNED_RANGE]:
+ assert model_output.shape == (B, C * 2, *x.shape[2:])
+ model_output, model_var_values = th.split(model_output, C, dim=1)
+ if self.model_var_type == ModelVarType.LEARNED:
+ model_log_variance = model_var_values
+ model_variance = th.exp(model_log_variance)
+ else:
+ min_log = _extract_into_tensor(
+ self.posterior_log_variance_clipped, t, x.shape
+ )
+ max_log = _extract_into_tensor(np.log(self.betas), t, x.shape)
+ # The model_var_values is [-1, 1] for [min_var, max_var].
+ frac = (model_var_values + 1) / 2
+ model_log_variance = frac * max_log + (1 - frac) * min_log
+ model_variance = th.exp(model_log_variance)
+ else:
+ model_variance, model_log_variance = {
+ # for fixedlarge, we set the initial (log-)variance like so
+ # to get a better decoder log likelihood.
+ ModelVarType.FIXED_LARGE: (
+ np.append(self.posterior_variance[1], self.betas[1:])
+ if len(self.posterior_variance) > 1
+ else self.betas,
+ np.log(np.append(self.posterior_variance[1], self.betas[1:]))
+ if len(self.posterior_variance) > 1
+ else np.log(self.betas),
+ ),
+ ModelVarType.FIXED_SMALL: (
+ self.posterior_variance,
+ self.posterior_log_variance_clipped,
+ ),
+ }[self.model_var_type]
+ model_variance = _extract_into_tensor(model_variance, t, x.shape)
+ model_log_variance = _extract_into_tensor(model_log_variance, t, x.shape)
+
+ # perturb model output according to the guidance objective
+ # alpha = 0.0001
+ # model_output = model_output - alpha * grad
+
+ def process_xstart(x):
+ if denoised_fn is not None:
+ x = denoised_fn(x, t)
+ if clip_denoised:
+ return x.clamp(-1, 1)
+ return x
+
+ if self.model_mean_type == ModelMeanType.PREVIOUS_X:
+ pred_xstart = process_xstart(
+ self._predict_xstart_from_xprev(x_t=x, t=t, xprev=model_output)
+ )
+ model_mean = model_output
+ elif self.model_mean_type in [ModelMeanType.START_X, ModelMeanType.EPSILON]:
+ if self.model_mean_type == ModelMeanType.START_X:
+ pred_xstart = process_xstart(model_output)
+ else:
+ pred_xstart = process_xstart(
+ self._predict_xstart_from_eps(x_t=x, t=t, eps=model_output)
+ )
+ model_mean, _, _ = self.q_posterior_mean_variance(
+ x_start=pred_xstart, x_t=x, t=t
+ )
+ else:
+ raise NotImplementedError(self.model_mean_type)
+
+ assert (
+ model_mean.shape == model_log_variance.shape == pred_xstart.shape == x.shape
+ )
+ return {
+ "mean": model_mean,
+ "variance": model_variance,
+ "log_variance": model_log_variance,
+ "pred_xstart": pred_xstart,
+ **aux_output,
+ }
+
+ def _predict_xstart_from_eps(self, x_t, t, eps):
+ assert x_t.shape == eps.shape
+ return (
+ _extract_into_tensor(self.sqrt_recip_alphas_cumprod, t, x_t.shape) * x_t
+ - _extract_into_tensor(self.sqrt_recipm1_alphas_cumprod, t, x_t.shape) * eps
+ )
+
+ def _predict_xstart_from_xprev(self, x_t, t, xprev):
+ assert x_t.shape == xprev.shape
+ return ( # (xprev - coef2*x_t) / coef1
+ _extract_into_tensor(1.0 / self.posterior_mean_coef1, t, x_t.shape) * xprev
+ - _extract_into_tensor(
+ self.posterior_mean_coef2 / self.posterior_mean_coef1, t, x_t.shape
+ )
+ * x_t
+ )
+
+ def _predict_eps_from_xstart(self, x_t, t, pred_xstart):
+ return (
+ _extract_into_tensor(self.sqrt_recip_alphas_cumprod, t, x_t.shape) * x_t
+ - pred_xstart
+ ) / _extract_into_tensor(self.sqrt_recipm1_alphas_cumprod, t, x_t.shape)
+
+ def _scale_timesteps(self, t):
+ if self.rescale_timesteps:
+ return t.float() * (1000.0 / self.num_timesteps)
+ return t
+
+ def condition_mean(self, cond_fn, p_mean_var, x, t, model_kwargs=None):
+ """
+ Compute the mean for the previous step, given a function cond_fn that
+ computes the gradient of a conditional log probability with respect to
+ x. In particular, cond_fn computes grad(log(p(y|x))), and we want to
+ condition on y.
+
+ This uses the conditioning strategy from Sohl-Dickstein et al. (2015).
+ """
+ gradient = cond_fn(x, self._scale_timesteps(t), **model_kwargs)
+ new_mean = (
+ p_mean_var["mean"].float() + p_mean_var["variance"] * gradient.float()
+ )
+ return new_mean
+
+ def condition_mean_with_grad(self, cond_fn, p_mean_var, x, t, model_kwargs=None):
+ """
+ Compute the mean for the previous step, given a function cond_fn that
+ computes the gradient of a conditional log probability with respect to
+ x. In particular, cond_fn computes grad(log(p(y|x))), and we want to
+ condition on y.
+
+ This uses the conditioning strategy from Sohl-Dickstein et al. (2015).
+ """
+ gradient = cond_fn(x, t, p_mean_var, **model_kwargs)
+ new_mean = (
+ p_mean_var["mean"].float() + p_mean_var["variance"] * gradient.float()
+ )
+ return new_mean
+
+ def condition_score(self, cond_fn, p_mean_var, x, t, model_kwargs=None):
+ """
+ Compute what the p_mean_variance output would have been, should the
+ model's score function be conditioned by cond_fn.
+
+ See condition_mean() for details on cond_fn.
+
+ Unlike condition_mean(), this instead uses the conditioning strategy
+ from Song et al (2020).
+ """
+ alpha_bar = _extract_into_tensor(self.alphas_cumprod, t, x.shape)
+
+ eps = self._predict_eps_from_xstart(x, t, p_mean_var["pred_xstart"])
+ eps = eps - (1 - alpha_bar).sqrt() * cond_fn(
+ x, self._scale_timesteps(t), **model_kwargs
+ )
+
+ out = p_mean_var.copy()
+ out["pred_xstart"] = self._predict_xstart_from_eps(x, t, eps)
+ out["mean"], _, _ = self.q_posterior_mean_variance(
+ x_start=out["pred_xstart"], x_t=x, t=t
+ )
+ return out
+
+ def condition_score_with_grad(self, cond_fn, p_mean_var, x, t, model_kwargs=None):
+ """
+ Compute what the p_mean_variance output would have been, should the
+ model's score function be conditioned by cond_fn.
+
+ See condition_mean() for details on cond_fn.
+
+ Unlike condition_mean(), this instead uses the conditioning strategy
+ from Song et al (2020).
+ """
+ alpha_bar = _extract_into_tensor(self.alphas_cumprod, t, x.shape)
+
+ eps = self._predict_eps_from_xstart(x, t, p_mean_var["pred_xstart"])
+ eps = eps - (1 - alpha_bar).sqrt() * cond_fn(x, t, p_mean_var, **model_kwargs)
+
+ out = p_mean_var.copy()
+ out["pred_xstart"] = self._predict_xstart_from_eps(x, t, eps)
+ out["mean"], _, _ = self.q_posterior_mean_variance(
+ x_start=out["pred_xstart"], x_t=x, t=t
+ )
+ return out
+
+ def p_sample(
+ self,
+ model,
+ x,
+ t,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ const_noise=False,
+ ):
+ """
+ Sample x_{t-1} from the model at the given timestep.
+
+ :param model: the model to sample from.
+ :param x: the current tensor at x_{t-1}.
+ :param t: the value of t, starting at 0 for the first diffusion step.
+ :param clip_denoised: if True, clip the x_start prediction to [-1, 1].
+ :param denoised_fn: if not None, a function which applies to the
+ x_start prediction before it is used to sample.
+ :param cond_fn: if not None, this is a gradient function that acts
+ similarly to the model.
+ :param model_kwargs: if not None, a dict of extra keyword arguments to
+ pass to the model. This can be used for conditioning.
+ :return: a dict containing the following keys:
+ - 'sample': a random sample from the model.
+ - 'pred_xstart': a prediction of x_0.
+ """
+ out = self.p_mean_variance(
+ model,
+ x,
+ t,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ model_kwargs=model_kwargs,
+ )
+ noise = th.randn_like(x)
+ # print('const_noise', const_noise)
+ if const_noise:
+ noise = noise[[0]].repeat(x.shape[0], 1, 1, 1)
+
+ nonzero_mask = (
+ (t != 0).float().view(-1, *([1] * (len(x.shape) - 1)))
+ ) # no noise when t == 0
+ if cond_fn is not None:
+ out["mean"] = self.condition_mean(
+ cond_fn, out, x, t, model_kwargs=model_kwargs
+ )
+ sample = out["mean"] + nonzero_mask * th.exp(0.5 * out["log_variance"]) * noise
+ return {"sample": sample, "pred_xstart": out["pred_xstart"]}
+
+ def p_sample_with_grad(
+ self,
+ model,
+ x,
+ t,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ ):
+ """
+ Sample x_{t-1} from the model at the given timestep.
+
+ :param model: the model to sample from.
+ :param x: the current tensor at x_{t-1}.
+ :param t: the value of t, starting at 0 for the first diffusion step.
+ :param clip_denoised: if True, clip the x_start prediction to [-1, 1].
+ :param denoised_fn: if not None, a function which applies to the
+ x_start prediction before it is used to sample.
+ :param cond_fn: if not None, this is a gradient function that acts
+ similarly to the model.
+ :param model_kwargs: if not None, a dict of extra keyword arguments to
+ pass to the model. This can be used for conditioning.
+ :return: a dict containing the following keys:
+ - 'sample': a random sample from the model.
+ - 'pred_xstart': a prediction of x_0.
+ """
+ with th.enable_grad():
+ x = x.detach().requires_grad_()
+ out = self.p_mean_variance(
+ model,
+ x,
+ t,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ model_kwargs=model_kwargs,
+ )
+ noise = th.randn_like(x)
+ nonzero_mask = (
+ (t != 0).float().view(-1, *([1] * (len(x.shape) - 1)))
+ ) # no noise when t == 0
+ if cond_fn is not None:
+ out["mean"] = self.condition_mean_with_grad(
+ cond_fn, out, x, t, model_kwargs=model_kwargs
+ )
+ sample = out["mean"] + nonzero_mask * th.exp(0.5 * out["log_variance"]) * noise
+ return {"sample": sample, "pred_xstart": out["pred_xstart"].detach()}
+
+ def p_sample_loop(
+ self,
+ model,
+ shape,
+ noise=None,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ device=None,
+ progress=False,
+ skip_timesteps=0,
+ init_image=None,
+ randomize_class=False,
+ cond_fn_with_grad=False,
+ dump_steps=None,
+ const_noise=False,
+ ):
+ """
+ Generate samples from the model.
+
+ :param model: the model module.
+ :param shape: the shape of the samples, (N, C, H, W).
+ :param noise: if specified, the noise from the encoder to sample.
+ Should be of the same shape as `shape`.
+ :param clip_denoised: if True, clip x_start predictions to [-1, 1].
+ :param denoised_fn: if not None, a function which applies to the
+ x_start prediction before it is used to sample.
+ :param cond_fn: if not None, this is a gradient function that acts
+ similarly to the model.
+ :param model_kwargs: if not None, a dict of extra keyword arguments to
+ pass to the model. This can be used for conditioning.
+ :param device: if specified, the device to create the samples on.
+ If not specified, use a model parameter's device.
+ :param progress: if True, show a tqdm progress bar.
+ :param const_noise: If True, will noise all samples with the same noise throughout sampling
+ :return: a non-differentiable batch of samples.
+ """
+ final = None
+ if dump_steps is not None:
+ dump = []
+
+ for i, sample in enumerate(
+ self.p_sample_loop_progressive(
+ model,
+ shape,
+ noise=noise,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ cond_fn=cond_fn,
+ model_kwargs=model_kwargs,
+ device=device,
+ progress=progress,
+ skip_timesteps=skip_timesteps,
+ init_image=init_image,
+ randomize_class=randomize_class,
+ cond_fn_with_grad=cond_fn_with_grad,
+ const_noise=const_noise,
+ )
+ ):
+ if dump_steps is not None and i in dump_steps:
+ dump.append(deepcopy(sample["sample"]))
+ final = sample
+ if dump_steps is not None:
+ return dump
+ return final["sample"]
+
+ def p_sample_loop_progressive(
+ self,
+ model,
+ shape,
+ noise=None,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ device=None,
+ progress=False,
+ skip_timesteps=0,
+ init_image=None,
+ randomize_class=False,
+ cond_fn_with_grad=False,
+ const_noise=False,
+ ):
+ """
+ Generate samples from the model and yield intermediate samples from
+ each timestep of diffusion.
+
+ Arguments are the same as p_sample_loop().
+ Returns a generator over dicts, where each dict is the return value of
+ p_sample().
+ """
+ if device is None:
+ device = next(model.parameters()).device
+ assert isinstance(shape, (tuple, list))
+ if noise is not None:
+ img = noise
+ else:
+ img = th.randn(*shape, device=device)
+
+ if skip_timesteps and init_image is None:
+ init_image = th.zeros_like(img)
+
+ indices = list(range(self.num_timesteps - skip_timesteps))[::-1]
+
+ if init_image is not None:
+ my_t = th.ones([shape[0]], device=device, dtype=th.long) * indices[0]
+ img = self.q_sample(init_image, my_t, img)
+
+ if progress:
+ # Lazy import so that we don't depend on tqdm.
+ from tqdm.auto import tqdm
+
+ indices = tqdm(indices)
+
+ for i in indices:
+ t = th.tensor([i] * shape[0], device=device)
+ if randomize_class and "y" in model_kwargs:
+ model_kwargs["y"] = th.randint(
+ low=0,
+ high=model.num_classes,
+ size=model_kwargs["y"].shape,
+ device=model_kwargs["y"].device,
+ )
+ with th.no_grad():
+ sample_fn = (
+ self.p_sample_with_grad if cond_fn_with_grad else self.p_sample
+ )
+ out = sample_fn(
+ model,
+ img,
+ t,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ cond_fn=cond_fn,
+ model_kwargs=model_kwargs,
+ const_noise=const_noise,
+ )
+ yield out
+ img = out["sample"]
+
+ def ddim_sample(
+ self,
+ model,
+ x,
+ t,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ eta=0.0,
+ target_motion=None,
+ guide=None,
+ guide_2d=None,
+ overwrite_2d=False,
+ overwrite_data=None,
+ model_output=None,
+ ):
+ """
+ Sample x_{t-1} from the model using DDIM.
+
+ Same usage as p_sample().
+ """
+ out_orig = self.p_mean_variance(
+ model,
+ x,
+ t,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ model_kwargs=model_kwargs,
+ model_output=model_output,
+ )
+ if cond_fn is not None:
+ out = self.condition_score(
+ cond_fn, out_orig, x, t, model_kwargs=model_kwargs
+ )
+ else:
+ out = out_orig
+
+ alpha_bar = _extract_into_tensor(self.alphas_cumprod, t, x.shape)
+ alpha_bar_prev = _extract_into_tensor(self.alphas_cumprod_prev, t, x.shape)
+
+ # Usually our model outputs epsilon, but we re-derive it
+ # in case we used x_start or x_prev prediction.
+ eps = self._predict_eps_from_xstart(x, t, out["pred_xstart"])
+
+ sigma = (
+ eta
+ * th.sqrt((1 - alpha_bar_prev) / (1 - alpha_bar))
+ * th.sqrt(1 - alpha_bar / alpha_bar_prev)
+ )
+ # Equation 12.
+ noise = th.randn_like(x)
+ mean_pred = (
+ out["pred_xstart"] * th.sqrt(alpha_bar_prev)
+ + th.sqrt(1 - alpha_bar_prev - sigma**2) * eps
+ )
+ nonzero_mask = (
+ (t != 0).float().view(-1, *([1] * (len(x.shape) - 1)))
+ ) # no noise when t == 0
+ sample = mean_pred + nonzero_mask * sigma * noise
+ return {"sample": sample, "pred_xstart": out_orig["pred_xstart"], **out_orig}
+
+ def ddim_get_xt(self, x, t, pred_xstart, eta):
+ eps = self._predict_eps_from_xstart(x, t, pred_xstart)
+
+ alpha_bar = _extract_into_tensor(self.alphas_cumprod, t, x.shape)
+ alpha_bar_prev = _extract_into_tensor(self.alphas_cumprod_prev, t, x.shape)
+ sigma = (
+ eta
+ * th.sqrt((1 - alpha_bar_prev) / (1 - alpha_bar))
+ * th.sqrt(1 - alpha_bar / alpha_bar_prev)
+ )
+ # Equation 12.
+ mean_pred = (
+ pred_xstart * th.sqrt(alpha_bar_prev)
+ + th.sqrt(1 - alpha_bar_prev - sigma**2) * eps
+ )
+ nonzero_mask = (
+ (t != 0).float().view(-1, *([1] * (len(x.shape) - 1)))
+ ) # no noise when t == 0
+ std = nonzero_mask * sigma
+ noise = th.randn_like(x)
+ x_t_1 = mean_pred + std * noise
+ return {"x_t-1": x_t_1, "x_t-1_mean": mean_pred, "std": std}
+
+ def ddim_sample_with_grad(
+ self,
+ model,
+ x,
+ t,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ eta=0.0,
+ guide=None,
+ ):
+ """
+ Sample x_{t-1} from the model using DDIM.
+
+ Same usage as p_sample().
+ """
+ with th.enable_grad():
+ x = x.detach().requires_grad_()
+ out_orig = self.p_mean_variance(
+ model,
+ x,
+ t,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ model_kwargs=model_kwargs,
+ )
+ if cond_fn is not None:
+ out = self.condition_score_with_grad(
+ cond_fn, out_orig, x, t, model_kwargs=model_kwargs
+ )
+ else:
+ out = out_orig
+
+ out["pred_xstart"] = out["pred_xstart"].detach()
+
+ # Usually our model outputs epsilon, but we re-derive it
+ # in case we used x_start or x_prev prediction.
+ eps = self._predict_eps_from_xstart(x, t, out["pred_xstart"])
+
+ alpha_bar = _extract_into_tensor(self.alphas_cumprod, t, x.shape)
+ alpha_bar_prev = _extract_into_tensor(self.alphas_cumprod_prev, t, x.shape)
+ sigma = (
+ eta
+ * th.sqrt((1 - alpha_bar_prev) / (1 - alpha_bar))
+ * th.sqrt(1 - alpha_bar / alpha_bar_prev)
+ )
+ # Equation 12.
+ noise = th.randn_like(x)
+ mean_pred = (
+ out["pred_xstart"] * th.sqrt(alpha_bar_prev)
+ + th.sqrt(1 - alpha_bar_prev - sigma**2) * eps
+ )
+ nonzero_mask = (
+ (t != 0).float().view(-1, *([1] * (len(x.shape) - 1)))
+ ) # no noise when t == 0
+ sample = mean_pred + nonzero_mask * sigma * noise
+ return {"sample": sample, "pred_xstart": out_orig["pred_xstart"].detach()}
+
+ def ddim_reverse_sample(
+ self,
+ model,
+ x,
+ t,
+ clip_denoised=True,
+ denoised_fn=None,
+ model_kwargs=None,
+ eta=0.0,
+ ):
+ """
+ Sample x_{t+1} from the model using DDIM reverse ODE.
+ """
+ assert eta == 0.0, "Reverse ODE only for deterministic path"
+ out = self.p_mean_variance(
+ model,
+ x,
+ t,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ model_kwargs=model_kwargs,
+ )
+ # Usually our model outputs epsilon, but we re-derive it
+ # in case we used x_start or x_prev prediction.
+ eps = (
+ _extract_into_tensor(self.sqrt_recip_alphas_cumprod, t, x.shape) * x
+ - out["pred_xstart"]
+ ) / _extract_into_tensor(self.sqrt_recipm1_alphas_cumprod, t, x.shape)
+ alpha_bar_next = _extract_into_tensor(self.alphas_cumprod_next, t, x.shape)
+
+ # Equation 12. reversed
+ mean_pred = (
+ out["pred_xstart"] * th.sqrt(alpha_bar_next)
+ + th.sqrt(1 - alpha_bar_next) * eps
+ )
+
+ return {"sample": mean_pred, "pred_xstart": out["pred_xstart"]}
+
+ def ddim_sample_loop(
+ self,
+ model,
+ shape,
+ noise=None,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ model_kwargs_modify_fn=None,
+ device=None,
+ progress=False,
+ eta=0.0,
+ skip_timesteps=0,
+ repeat_final_timesteps=None,
+ init_image=None,
+ randomize_class=False,
+ cond_fn_with_grad=False,
+ dump_steps=None,
+ const_noise=False,
+ target_motion=None,
+ guide=None,
+ update_sample_fn=None,
+ guide_2d=None,
+ overwrite_2d=False,
+ overwrite_data=None,
+ ):
+ """
+ Generate samples from the model using DDIM.
+
+ Same usage as p_sample_loop().
+ """
+ # print(eta)
+ if dump_steps is not None:
+ raise NotImplementedError()
+ if const_noise == True:
+ raise NotImplementedError()
+
+ final = None
+ for sample in self.ddim_sample_loop_progressive(
+ model,
+ shape,
+ noise=noise,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ cond_fn=cond_fn,
+ model_kwargs=model_kwargs,
+ model_kwargs_modify_fn=model_kwargs_modify_fn,
+ device=device,
+ progress=progress,
+ eta=eta,
+ skip_timesteps=skip_timesteps,
+ repeat_final_timesteps=repeat_final_timesteps,
+ init_image=init_image,
+ randomize_class=randomize_class,
+ cond_fn_with_grad=cond_fn_with_grad,
+ target_motion=target_motion,
+ guide=guide,
+ update_sample_fn=update_sample_fn,
+ guide_2d=guide_2d,
+ overwrite_2d=overwrite_2d,
+ overwrite_data=overwrite_data,
+ ):
+ final = sample
+ return final["sample"]
+
+ def ddim_sample_loop_with_aux(
+ self,
+ model,
+ shape,
+ noise=None,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ model_kwargs_modify_fn=None,
+ device=None,
+ progress=False,
+ eta=0.0,
+ skip_timesteps=0,
+ repeat_final_timesteps=None,
+ init_image=None,
+ randomize_class=False,
+ cond_fn_with_grad=False,
+ dump_steps=None,
+ const_noise=False,
+ target_motion=None,
+ guide=None,
+ update_sample_fn=None,
+ guide_2d=None,
+ overwrite_2d=False,
+ overwrite_data=None,
+ return_mid=False,
+ ):
+ """
+ Generate samples from the model using DDIM.
+
+ Same usage as p_sample_loop().
+ """
+ # print(eta)
+ if dump_steps is not None:
+ raise NotImplementedError()
+ if const_noise == True:
+ raise NotImplementedError()
+
+ final = None
+ intermediates = []
+ for sample in self.ddim_sample_loop_progressive(
+ model,
+ shape,
+ noise=noise,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ cond_fn=cond_fn,
+ model_kwargs=model_kwargs,
+ model_kwargs_modify_fn=model_kwargs_modify_fn,
+ device=device,
+ progress=progress,
+ eta=eta,
+ skip_timesteps=skip_timesteps,
+ repeat_final_timesteps=repeat_final_timesteps,
+ init_image=init_image,
+ randomize_class=randomize_class,
+ cond_fn_with_grad=cond_fn_with_grad,
+ target_motion=target_motion,
+ guide=guide,
+ update_sample_fn=update_sample_fn,
+ guide_2d=guide_2d,
+ overwrite_2d=overwrite_2d,
+ overwrite_data=overwrite_data,
+ ):
+ intermediates.append(sample)
+ final = sample
+ if return_mid:
+ final["intermediates"] = intermediates
+ return final
+
+ def ddim_sample_loop_progressive(
+ self,
+ model,
+ shape,
+ noise=None,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ model_kwargs_modify_fn=None,
+ device=None,
+ progress=False,
+ eta=0.0,
+ skip_timesteps=0,
+ repeat_final_timesteps=None,
+ init_image=None,
+ randomize_class=False,
+ cond_fn_with_grad=False,
+ target_motion=False,
+ guide=None,
+ update_sample_fn=None,
+ guide_2d=None,
+ overwrite_2d=False,
+ overwrite_data=None,
+ ):
+ """
+ Use DDIM to sample from the model and yield intermediate samples from
+ each timestep of DDIM.
+
+ Same usage as p_sample_loop_progressive().
+ """
+ if device is None:
+ device = next(model.parameters()).device
+ assert isinstance(shape, (tuple, list))
+ if noise is not None:
+ img = noise
+ else:
+ img = th.randn(*shape, device=device)
+ img_start = img.clone()
+
+ if skip_timesteps and init_image is None:
+ init_image = th.zeros_like(img)
+
+ indices = list(range(self.num_timesteps - skip_timesteps))[::-1]
+ if repeat_final_timesteps is not None:
+ if "%" in repeat_final_timesteps:
+ num_repeat_steps = int(
+ int(repeat_final_timesteps.replace("%", "")) / 100 * len(indices)
+ )
+ else:
+ num_repeat_steps = int(repeat_final_timesteps)
+ indices = indices + indices[-num_repeat_steps:]
+
+ if init_image is not None:
+ my_t = th.ones([shape[0]], device=device, dtype=th.long) * indices[0]
+ img = self.q_sample(init_image, my_t, img)
+
+ if progress:
+ # Lazy import so that we don't depend on tqdm.
+ from tqdm.auto import tqdm
+
+ indices = tqdm(indices)
+
+ for k, i in enumerate(indices):
+ t = th.tensor([i] * shape[0], device=device)
+ if randomize_class and "y" in model_kwargs:
+ model_kwargs["y"] = th.randint(
+ low=0,
+ high=model.num_classes,
+ size=model_kwargs["y"].shape,
+ device=model_kwargs["y"].device,
+ )
+ if model_kwargs_modify_fn is not None:
+ is_final_repeat_timestep = k >= len(indices) - num_repeat_steps
+ cur_model_kwargs = model_kwargs_modify_fn(
+ model_kwargs, img, i, is_final_repeat_timestep
+ )
+ else:
+ cur_model_kwargs = model_kwargs
+
+ with th.no_grad():
+ sample_fn = (
+ self.ddim_sample_with_grad
+ if cond_fn_with_grad
+ else self.ddim_sample
+ )
+ out = sample_fn(
+ model,
+ img,
+ t,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ cond_fn=cond_fn,
+ model_kwargs=cur_model_kwargs,
+ eta=eta,
+ target_motion=target_motion,
+ guide=guide,
+ guide_2d=guide_2d,
+ overwrite_2d=overwrite_2d,
+ overwrite_data=overwrite_data,
+ )
+ yield out
+ if update_sample_fn is not None:
+ before_repeat_timesteps = k == len(indices) - num_repeat_steps - 1
+ img = update_sample_fn(
+ img,
+ out,
+ i,
+ is_final_repeat_timestep,
+ before_repeat_timesteps,
+ img_start,
+ )
+ else:
+ img = out["sample"]
+
+ def ddim_sds_loop(
+ self,
+ model,
+ x0,
+ shape,
+ sds_weight_type="alphas",
+ noise=None,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ model_kwargs_modify_fn=None,
+ device=None,
+ progress=False,
+ eta=0.0,
+ skip_timesteps=0,
+ repeat_final_timesteps=None,
+ init_image=None,
+ randomize_class=False,
+ cond_fn_with_grad=False,
+ dump_steps=None,
+ opt_steps=500,
+ const_noise=False,
+ target_motion=None,
+ guide=None,
+ update_sample_fn=None,
+ guide_2d=None,
+ overwrite_2d=False,
+ overwrite_data=None,
+ ):
+ """
+ Generate samples from the model using DDIM.
+
+ Same usage as p_sample_loop().
+ """
+ # print(eta)
+ if dump_steps is not None:
+ raise NotImplementedError()
+ if const_noise == True:
+ raise NotImplementedError()
+
+ final = None
+ for sample in self.ddim_sds_loop_progressive(
+ model,
+ x0,
+ shape,
+ sds_weight_type=sds_weight_type,
+ noise=noise,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ cond_fn=cond_fn,
+ model_kwargs=model_kwargs,
+ model_kwargs_modify_fn=model_kwargs_modify_fn,
+ device=device,
+ progress=progress,
+ eta=eta,
+ opt_steps=opt_steps,
+ skip_timesteps=skip_timesteps,
+ repeat_final_timesteps=repeat_final_timesteps,
+ init_image=init_image,
+ randomize_class=randomize_class,
+ cond_fn_with_grad=cond_fn_with_grad,
+ target_motion=target_motion,
+ guide=guide,
+ update_sample_fn=update_sample_fn,
+ guide_2d=guide_2d,
+ overwrite_2d=overwrite_2d,
+ overwrite_data=overwrite_data,
+ ):
+ final = sample
+ return final
+
+ def ddim_sds_loop_progressive(
+ self,
+ model,
+ x0,
+ shape,
+ sds_weight_type="alphas",
+ noise=None,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ model_kwargs_modify_fn=None,
+ device=None,
+ progress=False,
+ eta=0.0,
+ opt_steps=500,
+ skip_timesteps=0,
+ repeat_final_timesteps=None,
+ init_image=None,
+ randomize_class=False,
+ cond_fn_with_grad=False,
+ target_motion=False,
+ guide=None,
+ update_sample_fn=None,
+ guide_2d=None,
+ overwrite_2d=False,
+ overwrite_data=None,
+ ):
+ """
+ Use DDIM to sample from the model and yield intermediate samples from
+ each timestep of DDIM.
+
+ Same usage as p_sample_loop_progressive().
+ """
+ if device is None:
+ device = next(model.parameters()).device
+ assert isinstance(shape, (tuple, list))
+ init_smpl_pose, init_smpl_transl = self.motion2global(
+ x0,
+ mean=self.motion_mean,
+ std=self.motion_std,
+ smpl=self.smpl,
+ return_jts=False,
+ )
+ init_smpl_pose_6d = matrix_to_rotation_6d(init_smpl_pose)
+ init_smpl_pose_6d = Variable(
+ init_smpl_pose_6d.clone().contiguous().detach(), requires_grad=True
+ )
+ init_smpl_transl = Variable(
+ init_smpl_transl.clone().contiguous().detach(), requires_grad=True
+ )
+ # x_start = Variable(x0.clone().contiguous().detach(), requires_grad=True)
+ # optim_type = 'LBFGS'
+ optim_type = "SGD"
+ bs, seqlen = init_smpl_pose_6d.shape[:2]
+ if optim_type == "LBFGS":
+ opt_steps = opt_steps // 10
+ # optimizer = torch.optim.LBFGS([x_start], max_iter=4, history_size=10, line_search_fn='strong_wolfe')
+ optimizer = torch.optim.LBFGS(
+ [init_smpl_pose_6d, init_smpl_transl],
+ max_iter=4,
+ history_size=10,
+ line_search_fn="strong_wolfe",
+ )
+ elif optim_type == "Adam":
+ # optimizer = torch.optim.Adam([x_start], lr=1e-2)
+ optimizer = torch.optim.Adam([init_smpl_pose_6d, init_smpl_transl], lr=1e-2)
+ elif optim_type == "SGD":
+ # optimizer = torch.optim.SGD([x_start], lr=1e-3)
+ optimizer = torch.optim.SGD([init_smpl_pose_6d, init_smpl_transl], lr=1e-3)
+ else:
+ raise ValueError(f"Unknown optimizer type: {optim_type}")
+
+ indices = list(range(self.num_timesteps - skip_timesteps))[::-1]
+
+ pbar = range(opt_steps)
+ if progress:
+ from tqdm.auto import tqdm
+
+ pbar = tqdm(pbar)
+
+ for _ in pbar:
+ ind = random.randint(0, len(indices) - 1)
+ t = th.tensor([indices[ind]] * shape[0], device=device)
+ cur_model_kwargs = model_kwargs
+ if sds_weight_type == "alphas":
+ w_t = _extract_into_tensor(self.alphas_cumprod, t, x0.shape)
+ elif sds_weight_type == "sqrt_alphas":
+ w_t = _extract_into_tensor(self.sqrt_alphas_cumprod, t, x0.shape)
+ elif sds_weight_type == "1minus_alphas":
+ w_t = _extract_into_tensor(1 - self.alphas_cumprod, t, x0.shape)
+ elif sds_weight_type == "sqrt_1minus_alphas":
+ w_t = _extract_into_tensor(
+ self.sqrt_one_minus_alphas_cumprod, t, x0.shape
+ )
+ elif sds_weight_type == "log_one_minus_alphas_cumprod":
+ w_t = _extract_into_tensor(
+ self.log_one_minus_alphas_cumprod, t, x0.shape
+ )
+ elif sds_weight_type == "constant":
+ w_t = th.ones_like(x0)
+ else:
+ raise ValueError(f"Unknown sds weight type: {sds_weight_type}")
+
+ w_t = w_t[:, :1, :1, :1]
+
+ init_smpl_pose = rotation_6d_to_matrix(init_smpl_pose_6d)
+ with th.no_grad():
+ orient_mat = init_smpl_pose[:, :, 0]
+ pose_feat = matrix_to_rotation_6d(init_smpl_pose[:, :, 1:]).reshape(
+ bs, seqlen, 23 * 6
+ )
+ x_start = self.smpl2motion(
+ orient_mat, init_smpl_transl, pose_feat, None, None
+ )
+ x_start = (x_start - self.motion_mean) / self.motion_std
+ x_start = x_start.transpose(1, 2)[:, :, None, :]
+ xt = self.q_sample(x_start, t)
+ sample_fn = self.ddim_sample
+ out = sample_fn(
+ model,
+ xt,
+ t,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ cond_fn=cond_fn,
+ model_kwargs=cur_model_kwargs,
+ eta=eta,
+ target_motion=target_motion,
+ guide=guide,
+ guide_2d=guide_2d,
+ overwrite_2d=overwrite_2d,
+ overwrite_data=overwrite_data,
+ )
+ sampled_x0 = out["pred_xstart"]
+
+ def closure():
+ optimizer.zero_grad()
+ cam2world = cur_model_kwargs["y"]["cam2world"]
+ local_kpt2d = cur_model_kwargs["y"]["local_kpt2d"] # [B, T, 17, 2]
+ intrinsics = cur_model_kwargs["y"]["cam_intrinsics"] # [B, 1, 3, 3]
+ cam_orient = rotation_6d_to_matrix(cam2world[:, :, :6]) # [B, T, 3, 3]
+ cam_pos = cam2world[:, :, 6:] # [B, T, 3]
+ local_orient = cur_model_kwargs["y"]["local_orient"]
+ local_transl = cur_model_kwargs["y"]["local_transl"]
+ betas = cur_model_kwargs["y"]["betas"]
+ kpt2d_score = cur_model_kwargs["y"]["kpt2d_score"]
+ R0 = local_orient[:, :1] # [B, 1, 3, 3]
+ t0 = local_transl[:, :1] # [B, 1, 3]
+ cx, cy = intrinsics[:, :, 0, 2], intrinsics[:, :, 1, 2]
+ focal = intrinsics[:, :, 0, 0]
+ scale = focal
+ bs = local_kpt2d.size(0)
+
+ # smpl_pose, smpl_transl = self.motion2global(
+ # x_start,
+ # mean=self.motion_mean,
+ # std=self.motion_std,
+ # smpl=self.smpl,
+ # return_jts=False
+ # )
+ smpl_pose = init_smpl_pose
+ smpl_transl = init_smpl_transl
+ global_orient = smpl_pose[:, :, 0]
+ global_orient = R0 @ global_orient
+ smpl_transl = (R0 @ smpl_transl[..., None])[..., 0] + t0
+ local_pose = smpl_pose[:, :, 1:]
+ bs, seqlen = smpl_pose.shape[:2]
+ smpl_pose = torch.cat([global_orient[:, :, None], local_pose], dim=2)
+
+ # take the first and last frame
+ # select_idx = [0, seqlen // 2, -1]
+ select_idx = list(range(seqlen))
+
+ select_global_orient = global_orient[:, select_idx]
+ select_smpl_transl = smpl_transl[:, select_idx]
+ select_local_pose = local_pose[:, select_idx]
+ select_betas = betas[:, select_idx]
+ select_kpt2d_score = kpt2d_score[:, select_idx]
+ select_local_kpt2d = local_kpt2d[:, select_idx]
+ new_len = select_global_orient.shape[1]
+ sout = self.smpl(
+ global_orient=select_global_orient.reshape(bs * new_len, 1, 3, 3),
+ body_pose=select_local_pose.reshape(bs * new_len, 23, 3, 3),
+ betas=select_betas.reshape(bs * new_len, 10),
+ root_trans=select_smpl_transl.reshape(bs * new_len, 3),
+ orig_joints=True,
+ pose2rot=False,
+ )
+ joints17 = torch.einsum(
+ "bik,ji->bjk", [sout.vertices, self.smpl.J_regressor_wham]
+ )[:, :17]
+ joints17 = joints17.reshape(bs, new_len, 17, 3)
+
+ select_cam_orient = cam_orient[:, select_idx]
+ select_cam_pos = cam_pos[:, select_idx]
+ # project 3D joints to 2D
+ joints = (
+ select_cam_orient.transpose(2, 3)
+ @ (joints17 - select_cam_pos[:, :, None]).transpose(2, 3)
+ ).transpose(2, 3)
+ joints = joints / joints[..., 2:3]
+ joints = (intrinsics @ joints.transpose(2, 3)).transpose(2, 3)
+ joints = joints[..., :2]
+ diff = joints - select_local_kpt2d
+ robust_sqr_dist = gmof(diff, sigma=scale[..., None, None])
+ robust_sqr_dist = (
+ select_kpt2d_score**2
+ ) * robust_sqr_dist # / scale[..., None, None]
+ # coeff = torch.where(robust_sqr_dist > 10, 10 / robust_sqr_dist.detach(), torch.ones_like(robust_sqr_dist))
+ # alpha_bar_batch = _extract_into_tensor(self.alphas_cumprod, t, robust_sqr_dist.shape[:1])
+ # alpha_bar_batch = alpha_bar_batch[:, None, None, None].clamp(min=0.01, max=0.5)
+ # robust_sqr_dist = robust_sqr_dist * coeff
+ # loss_reproj = (robust_sqr_dist * alpha_bar_batch ** 2).mean()
+
+ # just take the first and last frame
+ loss_reproj = robust_sqr_dist.reshape(bs, -1).mean(dim=-1).sum()
+
+ # loss_reproj = (diff.abs() * kpt2d_score / scale[..., None, None]).reshape(x_start.shape[0], -1).mean(dim=-1).sum()
+ loss_sds = ((sampled_x0 - x_start) ** 2) * w_t
+ # loss_sds[:, select_idx] = loss_sds[:, select_idx] * 0.0
+ loss_sds = loss_sds.reshape(bs, -1).mean(dim=-1).sum()
+
+ sampled_x0_pose, sampled_x0_transl = self.motion2global(
+ sampled_x0,
+ mean=self.motion_mean,
+ std=self.motion_std,
+ smpl=self.smpl,
+ return_jts=False,
+ )
+ loss_sds_pose = (
+ (sampled_x0_pose - smpl_pose) ** 2 * w_t.reshape(bs, 1, 1, 1, 1)
+ )[:, :, 1:]
+ loss_sds_transl = (sampled_x0_transl - smpl_transl) ** 2 * w_t.reshape(
+ bs, 1, 1
+ )
+ # loss_sds_pose[:, select_idx] = loss_sds_pose[:, select_idx] * 0.0
+ # loss_sds_transl[:, select_idx] = loss_sds_transl[:, select_idx] * 0.0
+
+ loss_sds_pose = loss_sds_pose.reshape(bs, -1).mean(dim=-1).sum()
+ loss_sds_transl = loss_sds_transl.reshape(bs, -1).mean(dim=-1).sum()
+ # loss = loss_reproj * 0.1 + loss_sds * 0.001 + loss_sds_pose * 1 + loss_sds_transl * 1
+ loss = (
+ loss_sds * 0.1 + loss_sds_pose * 0.0001 + loss_sds_transl * 0.0001
+ )
+ # loss = loss_reproj * 0.1
+ # loss = loss_reproj * 0.01 + loss_sds * 1 # + loss_sds_pose * 1 + loss_sds_transl * 1
+
+ loss.backward()
+ # gradient clipping
+ optimizer.step()
+ return loss_sds
+
+ optimizer.step(closure)
+ yield x_start.detach()
+
+ def plms_sample(
+ self,
+ model,
+ x,
+ t,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ cond_fn_with_grad=False,
+ order=2,
+ old_out=None,
+ ):
+ """
+ Sample x_{t-1} from the model using Pseudo Linear Multistep.
+
+ Same usage as p_sample().
+ """
+ if not int(order) or not 1 <= order <= 4:
+ raise ValueError("order is invalid (should be int from 1-4).")
+
+ def get_model_output(x, t):
+ with th.set_grad_enabled(cond_fn_with_grad and cond_fn is not None):
+ x = x.detach().requires_grad_() if cond_fn_with_grad else x
+ out_orig = self.p_mean_variance(
+ model,
+ x,
+ t,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ model_kwargs=model_kwargs,
+ )
+ if cond_fn is not None:
+ if cond_fn_with_grad:
+ out = self.condition_score_with_grad(
+ cond_fn, out_orig, x, t, model_kwargs=model_kwargs
+ )
+ x = x.detach()
+ else:
+ out = self.condition_score(
+ cond_fn, out_orig, x, t, model_kwargs=model_kwargs
+ )
+ else:
+ out = out_orig
+
+ # Usually our model outputs epsilon, but we re-derive it
+ # in case we used x_start or x_prev prediction.
+ eps = self._predict_eps_from_xstart(x, t, out["pred_xstart"])
+ return eps, out, out_orig
+
+ alpha_bar = _extract_into_tensor(self.alphas_cumprod, t, x.shape)
+ alpha_bar_prev = _extract_into_tensor(self.alphas_cumprod_prev, t, x.shape)
+ eps, out, out_orig = get_model_output(x, t)
+
+ if order > 1 and old_out is None:
+ # Pseudo Improved Euler
+ old_eps = [eps]
+ mean_pred = (
+ out["pred_xstart"] * th.sqrt(alpha_bar_prev)
+ + th.sqrt(1 - alpha_bar_prev) * eps
+ )
+ eps_2, _, _ = get_model_output(mean_pred, t - 1)
+ eps_prime = (eps + eps_2) / 2
+ pred_prime = self._predict_xstart_from_eps(x, t, eps_prime)
+ mean_pred = (
+ pred_prime * th.sqrt(alpha_bar_prev)
+ + th.sqrt(1 - alpha_bar_prev) * eps_prime
+ )
+ else:
+ # Pseudo Linear Multistep (Adams-Bashforth)
+ old_eps = old_out["old_eps"]
+ old_eps.append(eps)
+ cur_order = min(order, len(old_eps))
+ if cur_order == 1:
+ eps_prime = old_eps[-1]
+ elif cur_order == 2:
+ eps_prime = (3 * old_eps[-1] - old_eps[-2]) / 2
+ elif cur_order == 3:
+ eps_prime = (23 * old_eps[-1] - 16 * old_eps[-2] + 5 * old_eps[-3]) / 12
+ elif cur_order == 4:
+ eps_prime = (
+ 55 * old_eps[-1]
+ - 59 * old_eps[-2]
+ + 37 * old_eps[-3]
+ - 9 * old_eps[-4]
+ ) / 24
+ else:
+ raise RuntimeError("cur_order is invalid.")
+ pred_prime = self._predict_xstart_from_eps(x, t, eps_prime)
+ mean_pred = (
+ pred_prime * th.sqrt(alpha_bar_prev)
+ + th.sqrt(1 - alpha_bar_prev) * eps_prime
+ )
+
+ if len(old_eps) >= order:
+ old_eps.pop(0)
+
+ nonzero_mask = (t != 0).float().view(-1, *([1] * (len(x.shape) - 1)))
+ sample = mean_pred * nonzero_mask + out["pred_xstart"] * (1 - nonzero_mask)
+
+ return {
+ "sample": sample,
+ "pred_xstart": out_orig["pred_xstart"],
+ "old_eps": old_eps,
+ }
+
+ def plms_sample_loop(
+ self,
+ model,
+ shape,
+ noise=None,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ device=None,
+ progress=False,
+ skip_timesteps=0,
+ init_image=None,
+ randomize_class=False,
+ cond_fn_with_grad=False,
+ order=2,
+ ):
+ """
+ Generate samples from the model using Pseudo Linear Multistep.
+
+ Same usage as p_sample_loop().
+ """
+ final = None
+ for sample in self.plms_sample_loop_progressive(
+ model,
+ shape,
+ noise=noise,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ cond_fn=cond_fn,
+ model_kwargs=model_kwargs,
+ device=device,
+ progress=progress,
+ skip_timesteps=skip_timesteps,
+ init_image=init_image,
+ randomize_class=randomize_class,
+ cond_fn_with_grad=cond_fn_with_grad,
+ order=order,
+ ):
+ final = sample
+ return final["sample"]
+
+ def plms_sample_loop_progressive(
+ self,
+ model,
+ shape,
+ noise=None,
+ clip_denoised=True,
+ denoised_fn=None,
+ cond_fn=None,
+ model_kwargs=None,
+ device=None,
+ progress=False,
+ skip_timesteps=0,
+ init_image=None,
+ randomize_class=False,
+ cond_fn_with_grad=False,
+ order=2,
+ ):
+ """
+ Use PLMS to sample from the model and yield intermediate samples from each
+ timestep of PLMS.
+
+ Same usage as p_sample_loop_progressive().
+ """
+ if device is None:
+ device = next(model.parameters()).device
+ assert isinstance(shape, (tuple, list))
+ if noise is not None:
+ img = noise
+ else:
+ img = th.randn(*shape, device=device)
+
+ if skip_timesteps and init_image is None:
+ init_image = th.zeros_like(img)
+
+ indices = list(range(self.num_timesteps - skip_timesteps))[::-1]
+
+ if init_image is not None:
+ my_t = th.ones([shape[0]], device=device, dtype=th.long) * indices[0]
+ img = self.q_sample(init_image, my_t, img)
+
+ if progress:
+ # Lazy import so that we don't depend on tqdm.
+ from tqdm.auto import tqdm
+
+ indices = tqdm(indices)
+
+ old_out = None
+
+ for i in indices:
+ t = th.tensor([i] * shape[0], device=device)
+ if randomize_class and "y" in model_kwargs:
+ model_kwargs["y"] = th.randint(
+ low=0,
+ high=model.num_classes,
+ size=model_kwargs["y"].shape,
+ device=model_kwargs["y"].device,
+ )
+ with th.no_grad():
+ out = self.plms_sample(
+ model,
+ img,
+ t,
+ clip_denoised=clip_denoised,
+ denoised_fn=denoised_fn,
+ cond_fn=cond_fn,
+ model_kwargs=model_kwargs,
+ cond_fn_with_grad=cond_fn_with_grad,
+ order=order,
+ old_out=old_out,
+ )
+ yield out
+ old_out = out
+ img = out["sample"]
+
+ def _vb_terms_bpd(
+ self, model, x_start, x_t, t, clip_denoised=True, model_kwargs=None
+ ):
+ """
+ Get a term for the variational lower-bound.
+
+ The resulting units are bits (rather than nats, as one might expect).
+ This allows for comparison to other papers.
+
+ :return: a dict with the following keys:
+ - 'output': a shape [N] tensor of NLLs or KLs.
+ - 'pred_xstart': the x_0 predictions.
+ """
+ true_mean, _, true_log_variance_clipped = self.q_posterior_mean_variance(
+ x_start=x_start, x_t=x_t, t=t
+ )
+ out = self.p_mean_variance(
+ model, x_t, t, clip_denoised=clip_denoised, model_kwargs=model_kwargs
+ )
+ kl = normal_kl(
+ true_mean, true_log_variance_clipped, out["mean"], out["log_variance"]
+ )
+ kl = mean_flat(kl) / np.log(2.0)
+
+ decoder_nll = -discretized_gaussian_log_likelihood(
+ x_start, means=out["mean"], log_scales=0.5 * out["log_variance"]
+ )
+ assert decoder_nll.shape == x_start.shape
+ decoder_nll = mean_flat(decoder_nll) / np.log(2.0)
+
+ # At the first timestep return the decoder NLL,
+ # otherwise return KL(q(x_{t-1}|x_t,x_0) || p(x_{t-1}|x_t))
+ output = th.where((t == 0), decoder_nll, kl)
+ return {"output": output, "pred_xstart": out["pred_xstart"]}
+
+ def training_losses(
+ self, model, x_start, t, model_kwargs=None, noise=None, dataset=None
+ ):
+ """
+ Compute training losses for a single timestep.
+
+ :param model: the model to evaluate loss on.
+ :param x_start: the [N x C x ...] tensor of inputs.
+ :param t: a batch of timestep indices.
+ :param model_kwargs: if not None, a dict of extra keyword arguments to
+ pass to the model. This can be used for conditioning.
+ :param noise: if specified, the specific Gaussian noise to try to remove.
+ :return: a dict with the key "loss" containing a tensor of shape [N].
+ Some mean or variance settings may also have other keys.
+ """
+ if model_kwargs is None:
+ model_kwargs = {}
+ if noise is None:
+ noise = th.randn_like(x_start)
+ x_t = self.q_sample(x_start, t, noise=noise)
+
+ terms = {}
+
+ if self.loss_type == LossType.KL or self.loss_type == LossType.RESCALED_KL:
+ terms["loss"] = self._vb_terms_bpd(
+ model=model,
+ x_start=x_start,
+ x_t=x_t,
+ t=t,
+ clip_denoised=False,
+ model_kwargs=model_kwargs,
+ )["output"]
+ if self.loss_type == LossType.RESCALED_KL:
+ terms["loss"] *= self.num_timesteps
+ elif self.loss_type == LossType.MSE or self.loss_type == LossType.RESCALED_MSE:
+ model_output = model(x_t, self._scale_timesteps(t), **model_kwargs)
+
+ if self.model_var_type in [
+ ModelVarType.LEARNED,
+ ModelVarType.LEARNED_RANGE,
+ ]:
+ B, C = x_t.shape[:2]
+ assert model_output.shape == (B, C * 2, *x_t.shape[2:])
+ model_output, model_var_values = th.split(model_output, C, dim=1)
+ # Learn the variance using the variational bound, but don't let
+ # it affect our mean prediction.
+ frozen_out = th.cat([model_output.detach(), model_var_values], dim=1)
+ terms["vb"] = self._vb_terms_bpd(
+ model=lambda *args, r=frozen_out: r,
+ x_start=x_start,
+ x_t=x_t,
+ t=t,
+ clip_denoised=False,
+ )["output"]
+ if self.loss_type == LossType.RESCALED_MSE:
+ # Divide by 1000 for equivalence with initial implementation.
+ # Without a factor of 1/1000, the VB term hurts the MSE term.
+ terms["vb"] *= self.num_timesteps / 1000.0
+
+ target = {
+ ModelMeanType.PREVIOUS_X: self.q_posterior_mean_variance(
+ x_start=x_start, x_t=x_t, t=t
+ )[0],
+ ModelMeanType.START_X: x_start,
+ ModelMeanType.EPSILON: noise,
+ }[self.model_mean_type]
+ assert (
+ model_output.shape == target.shape == x_start.shape
+ ) # [bs, njoints, nfeats, nframes]
+
+ mask = model_kwargs["y"]["mask"]
+ terms["rot_mse"] = self.masked_l2(
+ target, model_output, mask
+ ) # mean_flat(rot_mse)
+
+ terms["loss"] = terms["rot_mse"] + terms.get("vb", 0.0)
+
+ else:
+ raise NotImplementedError(self.loss_type)
+
+ return terms
+
+ def get_vb_term(self, x_t, x_start, t, model_output):
+ vb = None
+ if self.model_var_type in [
+ ModelVarType.LEARNED,
+ ModelVarType.LEARNED_RANGE,
+ ]:
+ B, C = x_t.shape[:2]
+ assert model_output.shape == (B, C * 2, *x_t.shape[2:])
+ model_output, model_var_values = th.split(model_output, C, dim=1)
+ # Learn the variance using the variational bound, but don't let
+ # it affect our mean prediction.
+ frozen_out = th.cat([model_output.detach(), model_var_values], dim=1)
+ vb = self._vb_terms_bpd(
+ model=lambda *args, r=frozen_out: r,
+ x_start=x_start,
+ x_t=x_t,
+ t=t,
+ clip_denoised=False,
+ )["output"]
+ if self.loss_type == LossType.RESCALED_MSE:
+ # Divide by 1000 for equivalence with initial implementation.
+ # Without a factor of 1/1000, the VB term hurts the MSE term.
+ vb *= self.num_timesteps / 1000.0
+ return vb
+
+ def _prior_bpd(self, x_start):
+ """
+ Get the prior KL term for the variational lower-bound, measured in
+ bits-per-dim.
+
+ This term can't be optimized, as it only depends on the encoder.
+
+ :param x_start: the [N x C x ...] tensor of inputs.
+ :return: a batch of [N] KL values (in bits), one per batch element.
+ """
+ batch_size = x_start.shape[0]
+ t = th.tensor([self.num_timesteps - 1] * batch_size, device=x_start.device)
+ qt_mean, _, qt_log_variance = self.q_mean_variance(x_start, t)
+ kl_prior = normal_kl(
+ mean1=qt_mean, logvar1=qt_log_variance, mean2=0.0, logvar2=0.0
+ )
+ return mean_flat(kl_prior) / np.log(2.0)
+
+ def calc_bpd_loop(self, model, x_start, clip_denoised=True, model_kwargs=None):
+ """
+ Compute the entire variational lower-bound, measured in bits-per-dim,
+ as well as other related quantities.
+
+ :param model: the model to evaluate loss on.
+ :param x_start: the [N x C x ...] tensor of inputs.
+ :param clip_denoised: if True, clip denoised samples.
+ :param model_kwargs: if not None, a dict of extra keyword arguments to
+ pass to the model. This can be used for conditioning.
+
+ :return: a dict containing the following keys:
+ - total_bpd: the total variational lower-bound, per batch element.
+ - prior_bpd: the prior term in the lower-bound.
+ - vb: an [N x T] tensor of terms in the lower-bound.
+ - xstart_mse: an [N x T] tensor of x_0 MSEs for each timestep.
+ - mse: an [N x T] tensor of epsilon MSEs for each timestep.
+ """
+ device = x_start.device
+ batch_size = x_start.shape[0]
+
+ vb = []
+ xstart_mse = []
+ mse = []
+ for t in list(range(self.num_timesteps))[::-1]:
+ t_batch = th.tensor([t] * batch_size, device=device)
+ noise = th.randn_like(x_start)
+ x_t = self.q_sample(x_start=x_start, t=t_batch, noise=noise)
+ # Calculate VLB term at the current timestep
+ with th.no_grad():
+ out = self._vb_terms_bpd(
+ model,
+ x_start=x_start,
+ x_t=x_t,
+ t=t_batch,
+ clip_denoised=clip_denoised,
+ model_kwargs=model_kwargs,
+ )
+ vb.append(out["output"])
+ xstart_mse.append(mean_flat((out["pred_xstart"] - x_start) ** 2))
+ eps = self._predict_eps_from_xstart(x_t, t_batch, out["pred_xstart"])
+ mse.append(mean_flat((eps - noise) ** 2))
+
+ vb = th.stack(vb, dim=1)
+ xstart_mse = th.stack(xstart_mse, dim=1)
+ mse = th.stack(mse, dim=1)
+
+ prior_bpd = self._prior_bpd(x_start)
+ total_bpd = vb.sum(dim=1) + prior_bpd
+ return {
+ "total_bpd": total_bpd,
+ "prior_bpd": prior_bpd,
+ "vb": vb,
+ "xstart_mse": xstart_mse,
+ "mse": mse,
+ }
+
+
+def _extract_into_tensor(arr, timesteps, broadcast_shape):
+ """
+ Extract values from a 1-D numpy array for a batch of indices.
+
+ :param arr: the 1-D numpy array.
+ :param timesteps: a tensor of indices into the array to extract.
+ :param broadcast_shape: a larger shape of K dimensions with the batch
+ dimension equal to the length of timesteps.
+ :return: a tensor of shape [batch_size, 1, ...] where the shape has K dims.
+ """
+ res = th.from_numpy(arr).to(device=timesteps.device)[timesteps].float()
+ while len(res.shape) < len(broadcast_shape):
+ res = res[..., None]
+ return res.expand(broadcast_shape)
diff --git a/genmo/diffusion_utils/logger.py b/genmo/diffusion_utils/logger.py
new file mode 100644
index 0000000000000000000000000000000000000000..84f470a3d2564b10c6fe437c75459f012839d304
--- /dev/null
+++ b/genmo/diffusion_utils/logger.py
@@ -0,0 +1,494 @@
+"""
+Logger copied from OpenAI baselines to avoid extra RL-based dependencies:
+https://github.com/openai/baselines/blob/ea25b9e8b234e6ee1bca43083f8f3cf974143998/baselines/logger.py
+"""
+
+import datetime
+import json
+import os
+import os.path as osp
+import shutil
+import sys
+import tempfile
+import time
+import warnings
+from collections import defaultdict
+from contextlib import contextmanager
+
+DEBUG = 10
+INFO = 20
+WARN = 30
+ERROR = 40
+
+DISABLED = 50
+
+
+class KVWriter(object):
+ def writekvs(self, kvs):
+ raise NotImplementedError
+
+
+class SeqWriter(object):
+ def writeseq(self, seq):
+ raise NotImplementedError
+
+
+class HumanOutputFormat(KVWriter, SeqWriter):
+ def __init__(self, filename_or_file):
+ if isinstance(filename_or_file, str):
+ self.file = open(filename_or_file, "wt")
+ self.own_file = True
+ else:
+ assert hasattr(filename_or_file, "read"), (
+ "expected file or str, got %s" % filename_or_file
+ )
+ self.file = filename_or_file
+ self.own_file = False
+
+ def writekvs(self, kvs):
+ # Create strings for printing
+ key2str = {}
+ for key, val in sorted(kvs.items()):
+ if hasattr(val, "__float__"):
+ valstr = "%-8.3g" % val
+ else:
+ valstr = str(val)
+ key2str[self._truncate(key)] = self._truncate(valstr)
+
+ # Find max widths
+ if len(key2str) == 0:
+ print("WARNING: tried to write empty key-value dict")
+ return
+ else:
+ keywidth = max(map(len, key2str.keys()))
+ valwidth = max(map(len, key2str.values()))
+
+ # Write out the data
+ dashes = "-" * (keywidth + valwidth + 7)
+ lines = [dashes]
+ for key, val in sorted(key2str.items(), key=lambda kv: kv[0].lower()):
+ lines.append(
+ "| %s%s | %s%s |"
+ % (key, " " * (keywidth - len(key)), val, " " * (valwidth - len(val)))
+ )
+ lines.append(dashes)
+ self.file.write("\n".join(lines) + "\n")
+
+ # Flush the output to the file
+ self.file.flush()
+
+ def _truncate(self, s):
+ maxlen = 30
+ return s[: maxlen - 3] + "..." if len(s) > maxlen else s
+
+ def writeseq(self, seq):
+ seq = list(seq)
+ for i, elem in enumerate(seq):
+ self.file.write(elem)
+ if i < len(seq) - 1: # add space unless this is the last one
+ self.file.write(" ")
+ self.file.write("\n")
+ self.file.flush()
+
+ def close(self):
+ if self.own_file:
+ self.file.close()
+
+
+class JSONOutputFormat(KVWriter):
+ def __init__(self, filename):
+ self.file = open(filename, "wt")
+
+ def writekvs(self, kvs):
+ for k, v in sorted(kvs.items()):
+ if hasattr(v, "dtype"):
+ kvs[k] = float(v)
+ self.file.write(json.dumps(kvs) + "\n")
+ self.file.flush()
+
+ def close(self):
+ self.file.close()
+
+
+class CSVOutputFormat(KVWriter):
+ def __init__(self, filename):
+ self.file = open(filename, "w+t")
+ self.keys = []
+ self.sep = ","
+
+ def writekvs(self, kvs):
+ # Add our current row to the history
+ extra_keys = list(kvs.keys() - self.keys)
+ extra_keys.sort()
+ if extra_keys:
+ self.keys.extend(extra_keys)
+ self.file.seek(0)
+ lines = self.file.readlines()
+ self.file.seek(0)
+ for i, k in enumerate(self.keys):
+ if i > 0:
+ self.file.write(",")
+ self.file.write(k)
+ self.file.write("\n")
+ for line in lines[1:]:
+ self.file.write(line[:-1])
+ self.file.write(self.sep * len(extra_keys))
+ self.file.write("\n")
+ for i, k in enumerate(self.keys):
+ if i > 0:
+ self.file.write(",")
+ v = kvs.get(k)
+ if v is not None:
+ self.file.write(str(v))
+ self.file.write("\n")
+ self.file.flush()
+
+ def close(self):
+ self.file.close()
+
+
+class TensorBoardOutputFormat(KVWriter):
+ """
+ Dumps key/value pairs into TensorBoard's numeric format.
+ """
+
+ def __init__(self, dir):
+ os.makedirs(dir, exist_ok=True)
+ self.dir = dir
+ self.step = 1
+ prefix = "events"
+ path = osp.join(osp.abspath(dir), prefix)
+ import tensorflow as tf
+ from tensorflow.core.util import event_pb2
+ from tensorflow.python import pywrap_tensorflow
+ from tensorflow.python.util import compat
+
+ self.tf = tf
+ self.event_pb2 = event_pb2
+ self.pywrap_tensorflow = pywrap_tensorflow
+ self.writer = pywrap_tensorflow.EventsWriter(compat.as_bytes(path))
+
+ def writekvs(self, kvs):
+ def summary_val(k, v):
+ kwargs = {"tag": k, "simple_value": float(v)}
+ return self.tf.Summary.Value(**kwargs)
+
+ summary = self.tf.Summary(value=[summary_val(k, v) for k, v in kvs.items()])
+ event = self.event_pb2.Event(wall_time=time.time(), summary=summary)
+ event.step = (
+ self.step
+ ) # is there any reason why you'd want to specify the step?
+ self.writer.WriteEvent(event)
+ self.writer.Flush()
+ self.step += 1
+
+ def close(self):
+ if self.writer:
+ self.writer.Close()
+ self.writer = None
+
+
+def make_output_format(format, ev_dir, log_suffix=""):
+ os.makedirs(ev_dir, exist_ok=True)
+ if format == "stdout":
+ return HumanOutputFormat(sys.stdout)
+ elif format == "log":
+ return HumanOutputFormat(osp.join(ev_dir, "log%s.txt" % log_suffix))
+ elif format == "json":
+ return JSONOutputFormat(osp.join(ev_dir, "progress%s.json" % log_suffix))
+ elif format == "csv":
+ return CSVOutputFormat(osp.join(ev_dir, "progress%s.csv" % log_suffix))
+ elif format == "tensorboard":
+ return TensorBoardOutputFormat(osp.join(ev_dir, "tb%s" % log_suffix))
+ else:
+ raise ValueError("Unknown format specified: %s" % (format,))
+
+
+# ================================================================
+# API
+# ================================================================
+
+
+def logkv(key, val):
+ """
+ Log a value of some diagnostic
+ Call this once for each diagnostic quantity, each iteration
+ If called many times, last value will be used.
+ """
+ get_current().logkv(key, val)
+
+
+def logkv_mean(key, val):
+ """
+ The same as logkv(), but if called many times, values averaged.
+ """
+ get_current().logkv_mean(key, val)
+
+
+def logkvs(d):
+ """
+ Log a dictionary of key-value pairs
+ """
+ for k, v in d.items():
+ logkv(k, v)
+
+
+def dumpkvs():
+ """
+ Write all of the diagnostics from the current iteration
+ """
+ return get_current().dumpkvs()
+
+
+def getkvs():
+ return get_current().name2val
+
+
+def log(*args, level=INFO):
+ """
+ Write the sequence of args, with no separators, to the console and output files (if you've configured an output file).
+ """
+ get_current().log(*args, level=level)
+
+
+def debug(*args):
+ log(*args, level=DEBUG)
+
+
+def info(*args):
+ log(*args, level=INFO)
+
+
+def warn(*args):
+ log(*args, level=WARN)
+
+
+def error(*args):
+ log(*args, level=ERROR)
+
+
+def set_level(level):
+ """
+ Set logging threshold on current logger.
+ """
+ get_current().set_level(level)
+
+
+def set_comm(comm):
+ get_current().set_comm(comm)
+
+
+def get_dir():
+ """
+ Get directory that log files are being written to.
+ will be None if there is no output directory (i.e., if you didn't call start)
+ """
+ return get_current().get_dir()
+
+
+record_tabular = logkv
+dump_tabular = dumpkvs
+
+
+@contextmanager
+def profile_kv(scopename):
+ logkey = "wait_" + scopename
+ tstart = time.time()
+ try:
+ yield
+ finally:
+ get_current().name2val[logkey] += time.time() - tstart
+
+
+def profile(n):
+ """
+ Usage:
+ @profile("my_func")
+ def my_func(): code
+ """
+
+ def decorator_with_name(func):
+ def func_wrapper(*args, **kwargs):
+ with profile_kv(n):
+ return func(*args, **kwargs)
+
+ return func_wrapper
+
+ return decorator_with_name
+
+
+# ================================================================
+# Backend
+# ================================================================
+
+
+def get_current():
+ if Logger.CURRENT is None:
+ _configure_default_logger()
+
+ return Logger.CURRENT
+
+
+class Logger(object):
+ DEFAULT = None # A logger with no output files. (See right below class definition)
+ # So that you can still log to the terminal without setting up any output files
+ CURRENT = None # Current logger being used by the free functions above
+
+ def __init__(self, dir, output_formats, comm=None):
+ self.name2val = defaultdict(float) # values this iteration
+ self.name2cnt = defaultdict(int)
+ self.level = INFO
+ self.dir = dir
+ self.output_formats = output_formats
+ self.comm = comm
+
+ # Logging API, forwarded
+ # ----------------------------------------
+ def logkv(self, key, val):
+ self.name2val[key] = val
+
+ def logkv_mean(self, key, val):
+ oldval, cnt = self.name2val[key], self.name2cnt[key]
+ self.name2val[key] = oldval * cnt / (cnt + 1) + val / (cnt + 1)
+ self.name2cnt[key] = cnt + 1
+
+ def dumpkvs(self):
+ if self.comm is None:
+ d = self.name2val
+ else:
+ d = mpi_weighted_mean(
+ self.comm,
+ {
+ name: (val, self.name2cnt.get(name, 1))
+ for (name, val) in self.name2val.items()
+ },
+ )
+ if self.comm.rank != 0:
+ d["dummy"] = 1 # so we don't get a warning about empty dict
+ out = d.copy() # Return the dict for unit testing purposes
+ for fmt in self.output_formats:
+ if isinstance(fmt, KVWriter):
+ fmt.writekvs(d)
+ self.name2val.clear()
+ self.name2cnt.clear()
+ return out
+
+ def log(self, *args, level=INFO):
+ if self.level <= level:
+ self._do_log(args)
+
+ # Configuration
+ # ----------------------------------------
+ def set_level(self, level):
+ self.level = level
+
+ def set_comm(self, comm):
+ self.comm = comm
+
+ def get_dir(self):
+ return self.dir
+
+ def close(self):
+ for fmt in self.output_formats:
+ fmt.close()
+
+ # Misc
+ # ----------------------------------------
+ def _do_log(self, args):
+ for fmt in self.output_formats:
+ if isinstance(fmt, SeqWriter):
+ fmt.writeseq(map(str, args))
+
+
+def get_rank_without_mpi_import():
+ # check environment variables here instead of importing mpi4py
+ # to avoid calling MPI_Init() when this module is imported
+ for varname in ["PMI_RANK", "OMPI_COMM_WORLD_RANK"]:
+ if varname in os.environ:
+ return int(os.environ[varname])
+ return 0
+
+
+def mpi_weighted_mean(comm, local_name2valcount):
+ """
+ Copied from: https://github.com/openai/baselines/blob/ea25b9e8b234e6ee1bca43083f8f3cf974143998/baselines/common/mpi_util.py#L110
+ Perform a weighted average over dicts that are each on a different node
+ Input: local_name2valcount: dict mapping key -> (value, count)
+ Returns: key -> mean
+ """
+ all_name2valcount = comm.gather(local_name2valcount)
+ if comm.rank == 0:
+ name2sum = defaultdict(float)
+ name2count = defaultdict(float)
+ for n2vc in all_name2valcount:
+ for name, (val, count) in n2vc.items():
+ try:
+ val = float(val)
+ except ValueError:
+ if comm.rank == 0:
+ warnings.warn(
+ "WARNING: tried to compute mean on non-float {}={}".format(
+ name, val
+ )
+ )
+ else:
+ name2sum[name] += val * count
+ name2count[name] += count
+ return {name: name2sum[name] / name2count[name] for name in name2sum}
+ else:
+ return {}
+
+
+def configure(dir=None, format_strs=None, comm=None, log_suffix=""):
+ """
+ If comm is provided, average all numerical stats across that comm
+ """
+ if dir is None:
+ dir = os.getenv("OPENAI_LOGDIR")
+ if dir is None:
+ dir = osp.join(
+ tempfile.gettempdir(),
+ datetime.datetime.now().strftime("openai-%Y-%m-%d-%H-%M-%S-%f"),
+ )
+ assert isinstance(dir, str)
+ dir = os.path.expanduser(dir)
+ os.makedirs(os.path.expanduser(dir), exist_ok=True)
+
+ rank = get_rank_without_mpi_import()
+ if rank > 0:
+ log_suffix = log_suffix + "-rank%03i" % rank
+
+ if format_strs is None:
+ if rank == 0:
+ format_strs = os.getenv("OPENAI_LOG_FORMAT", "stdout,log,csv").split(",")
+ else:
+ format_strs = os.getenv("OPENAI_LOG_FORMAT_MPI", "log").split(",")
+ format_strs = filter(None, format_strs)
+ output_formats = [make_output_format(f, dir, log_suffix) for f in format_strs]
+
+ Logger.CURRENT = Logger(dir=dir, output_formats=output_formats, comm=comm)
+ if output_formats:
+ log("Logging to %s" % dir)
+
+
+def _configure_default_logger():
+ configure()
+ Logger.DEFAULT = Logger.CURRENT
+
+
+def reset():
+ if Logger.CURRENT is not Logger.DEFAULT:
+ Logger.CURRENT.close()
+ Logger.CURRENT = Logger.DEFAULT
+ log("Reset logger")
+
+
+@contextmanager
+def scoped_configure(dir=None, format_strs=None, comm=None):
+ prevlogger = Logger.CURRENT
+ configure(dir=dir, format_strs=format_strs, comm=comm)
+ try:
+ yield
+ finally:
+ Logger.CURRENT.close()
+ Logger.CURRENT = prevlogger
diff --git a/genmo/diffusion_utils/losses.py b/genmo/diffusion_utils/losses.py
new file mode 100644
index 0000000000000000000000000000000000000000..e3fded1953584eaaf183f3d2399be545a5003e0a
--- /dev/null
+++ b/genmo/diffusion_utils/losses.py
@@ -0,0 +1,77 @@
+# This code is based on https://github.com/openai/guided-diffusion
+"""
+Helpers for various likelihood-based losses. These are ported from the original
+Ho et al. diffusion models codebase:
+https://github.com/hojonathanho/diffusion/blob/1e0dceb3b3495bbe19116a5e1b3596cd0706c543/diffusion_tf/utils.py
+"""
+
+import numpy as np
+import torch as th
+
+
+def normal_kl(mean1, logvar1, mean2, logvar2):
+ """
+ Compute the KL divergence between two gaussians.
+
+ Shapes are automatically broadcasted, so batches can be compared to
+ scalars, among other use cases.
+ """
+ tensor = None
+ for obj in (mean1, logvar1, mean2, logvar2):
+ if isinstance(obj, th.Tensor):
+ tensor = obj
+ break
+ assert tensor is not None, "at least one argument must be a Tensor"
+
+ # Force variances to be Tensors. Broadcasting helps convert scalars to
+ # Tensors, but it does not work for th.exp().
+ logvar1, logvar2 = [
+ x if isinstance(x, th.Tensor) else th.tensor(x).to(tensor)
+ for x in (logvar1, logvar2)
+ ]
+
+ return 0.5 * (
+ -1.0
+ + logvar2
+ - logvar1
+ + th.exp(logvar1 - logvar2)
+ + ((mean1 - mean2) ** 2) * th.exp(-logvar2)
+ )
+
+
+def approx_standard_normal_cdf(x):
+ """
+ A fast approximation of the cumulative distribution function of the
+ standard normal.
+ """
+ return 0.5 * (1.0 + th.tanh(np.sqrt(2.0 / np.pi) * (x + 0.044715 * th.pow(x, 3))))
+
+
+def discretized_gaussian_log_likelihood(x, *, means, log_scales):
+ """
+ Compute the log-likelihood of a Gaussian distribution discretizing to a
+ given image.
+
+ :param x: the target images. It is assumed that this was uint8 values,
+ rescaled to the range [-1, 1].
+ :param means: the Gaussian mean Tensor.
+ :param log_scales: the Gaussian log stddev Tensor.
+ :return: a tensor like x of log probabilities (in nats).
+ """
+ assert x.shape == means.shape == log_scales.shape
+ centered_x = x - means
+ inv_stdv = th.exp(-log_scales)
+ plus_in = inv_stdv * (centered_x + 1.0 / 255.0)
+ cdf_plus = approx_standard_normal_cdf(plus_in)
+ min_in = inv_stdv * (centered_x - 1.0 / 255.0)
+ cdf_min = approx_standard_normal_cdf(min_in)
+ log_cdf_plus = th.log(cdf_plus.clamp(min=1e-12))
+ log_one_minus_cdf_min = th.log((1.0 - cdf_min).clamp(min=1e-12))
+ cdf_delta = cdf_plus - cdf_min
+ log_probs = th.where(
+ x < -0.999,
+ log_cdf_plus,
+ th.where(x > 0.999, log_one_minus_cdf_min, th.log(cdf_delta.clamp(min=1e-12))),
+ )
+ assert log_probs.shape == x.shape
+ return log_probs
diff --git a/genmo/diffusion_utils/model_util.py b/genmo/diffusion_utils/model_util.py
new file mode 100644
index 0000000000000000000000000000000000000000..4a9819ba8950b63394e5195734021a4ed9f25b91
--- /dev/null
+++ b/genmo/diffusion_utils/model_util.py
@@ -0,0 +1,41 @@
+from genmo.diffusion_utils import gaussian_diffusion as gd
+from genmo.diffusion_utils.respace import SpacedDiffusion, space_timesteps
+
+
+def create_gaussian_diffusion(cfg, training):
+ # default params
+ predict_xstart = True # we always predict x_start (a.k.a. x0), that's our deal!
+ steps = 1000
+ scale_beta = 1.0 # no scaling
+ timestep_respacing = (
+ cfg.train_timestep_respacing if training else cfg.test_timestep_respacing
+ ) # '' # can be used for ddim sampling, we don't use it.
+ if type(timestep_respacing) is not str:
+ timestep_respacing = str(timestep_respacing)
+ learn_sigma = False
+ rescale_timesteps = False
+
+ betas = gd.get_named_beta_schedule(cfg.noise_schedule, steps, scale_beta)
+ loss_type = gd.LossType.MSE
+
+ if not timestep_respacing:
+ timestep_respacing = [steps]
+
+ return SpacedDiffusion(
+ use_timesteps=space_timesteps(steps, timestep_respacing),
+ betas=betas,
+ model_mean_type=(
+ gd.ModelMeanType.EPSILON if not predict_xstart else gd.ModelMeanType.START_X
+ ),
+ model_var_type=(
+ (
+ gd.ModelVarType.FIXED_LARGE
+ if not cfg.sigma_small
+ else gd.ModelVarType.FIXED_SMALL
+ )
+ if not learn_sigma
+ else gd.ModelVarType.LEARNED_RANGE
+ ),
+ loss_type=loss_type,
+ rescale_timesteps=rescale_timesteps,
+ )
diff --git a/genmo/diffusion_utils/nn.py b/genmo/diffusion_utils/nn.py
new file mode 100644
index 0000000000000000000000000000000000000000..12089d7453a4c7b8da747a350b2c8a1ee1e2e0d8
--- /dev/null
+++ b/genmo/diffusion_utils/nn.py
@@ -0,0 +1,200 @@
+# This code is based on https://github.com/openai/guided-diffusion
+"""
+Various utilities for neural networks.
+"""
+
+import math
+
+import torch as th
+import torch.nn as nn
+
+
+# PyTorch 1.7 has SiLU, but we support PyTorch 1.5.
+class SiLU(nn.Module):
+ def forward(self, x):
+ return x * th.sigmoid(x)
+
+
+class GroupNorm32(nn.GroupNorm):
+ def forward(self, x):
+ return super().forward(x.float()).type(x.dtype)
+
+
+def conv_nd(dims, *args, **kwargs):
+ """
+ Create a 1D, 2D, or 3D convolution module.
+ """
+ if dims == 1:
+ return nn.Conv1d(*args, **kwargs)
+ elif dims == 2:
+ return nn.Conv2d(*args, **kwargs)
+ elif dims == 3:
+ return nn.Conv3d(*args, **kwargs)
+ raise ValueError(f"unsupported dimensions: {dims}")
+
+
+def linear(*args, **kwargs):
+ """
+ Create a linear module.
+ """
+ return nn.Linear(*args, **kwargs)
+
+
+def avg_pool_nd(dims, *args, **kwargs):
+ """
+ Create a 1D, 2D, or 3D average pooling module.
+ """
+ if dims == 1:
+ return nn.AvgPool1d(*args, **kwargs)
+ elif dims == 2:
+ return nn.AvgPool2d(*args, **kwargs)
+ elif dims == 3:
+ return nn.AvgPool3d(*args, **kwargs)
+ raise ValueError(f"unsupported dimensions: {dims}")
+
+
+def update_ema(target_params, source_params, rate=0.99):
+ """
+ Update target parameters to be closer to those of source parameters using
+ an exponential moving average.
+
+ :param target_params: the target parameter sequence.
+ :param source_params: the source parameter sequence.
+ :param rate: the EMA rate (closer to 1 means slower).
+ """
+ for targ, src in zip(target_params, source_params):
+ targ.detach().mul_(rate).add_(src, alpha=1 - rate)
+
+
+def zero_module(module):
+ """
+ Zero out the parameters of a module and return it.
+ """
+ for p in module.parameters():
+ p.detach().zero_()
+ return module
+
+
+def scale_module(module, scale):
+ """
+ Scale the parameters of a module and return it.
+ """
+ for p in module.parameters():
+ p.detach().mul_(scale)
+ return module
+
+
+def mean_flat(tensor):
+ """
+ Take the mean over all non-batch dimensions.
+ """
+ return tensor.mean(dim=list(range(1, len(tensor.shape))))
+
+
+def sum_flat(tensor):
+ """
+ Take the sum over all non-batch dimensions.
+ """
+ return tensor.sum(dim=list(range(1, len(tensor.shape))))
+
+
+def normalization(channels):
+ """
+ Make a standard normalization layer.
+
+ :param channels: number of input channels.
+ :return: an nn.Module for normalization.
+ """
+ return GroupNorm32(32, channels)
+
+
+def timestep_embedding(timesteps, dim, max_period=10000):
+ """
+ Create sinusoidal timestep embeddings.
+
+ :param timesteps: a 1-D Tensor of N indices, one per batch element.
+ These may be fractional.
+ :param dim: the dimension of the output.
+ :param max_period: controls the minimum frequency of the embeddings.
+ :return: an [N x dim] Tensor of positional embeddings.
+ """
+ half = dim // 2
+ freqs = th.exp(
+ -math.log(max_period) * th.arange(start=0, end=half, dtype=th.float32) / half
+ ).to(device=timesteps.device)
+ args = timesteps[:, None].float() * freqs[None]
+ embedding = th.cat([th.cos(args), th.sin(args)], dim=-1)
+ if dim % 2:
+ embedding = th.cat([embedding, th.zeros_like(embedding[:, :1])], dim=-1)
+ return embedding
+
+
+def checkpoint(func, inputs, params, flag):
+ """
+ Evaluate a function without caching intermediate activations, allowing for
+ reduced memory at the expense of extra compute in the backward pass.
+ :param func: the function to evaluate.
+ :param inputs: the argument sequence to pass to `func`.
+ :param params: a sequence of parameters `func` depends on but does not
+ explicitly take as arguments.
+ :param flag: if False, disable gradient checkpointing.
+ """
+ if flag:
+ args = tuple(inputs) + tuple(params)
+ return CheckpointFunction.apply(func, len(inputs), *args)
+ else:
+ return func(*inputs)
+
+
+class CheckpointFunction(th.autograd.Function):
+ @staticmethod
+ @th.cuda.amp.custom_fwd
+ def forward(ctx, run_function, length, *args):
+ ctx.run_function = run_function
+ ctx.input_length = length
+ ctx.save_for_backward(*args)
+ with th.no_grad():
+ output_tensors = ctx.run_function(*args[:length])
+ return output_tensors
+
+ @staticmethod
+ @th.cuda.amp.custom_bwd
+ def backward(ctx, *output_grads):
+ args = list(ctx.saved_tensors)
+
+ # Filter for inputs that require grad. If none, exit early.
+ input_indices = [i for (i, x) in enumerate(args) if x.requires_grad]
+ if not input_indices:
+ return (None, None) + tuple(None for _ in args)
+
+ with th.enable_grad():
+ for i in input_indices:
+ if i < ctx.input_length:
+ # Not sure why the OAI code does this little
+ # dance. It might not be necessary.
+ args[i] = args[i].detach().requires_grad_()
+ args[i] = args[i].view_as(args[i])
+ output_tensors = ctx.run_function(*args[: ctx.input_length])
+
+ if isinstance(output_tensors, th.Tensor):
+ output_tensors = [output_tensors]
+
+ # Filter for outputs that require grad. If none, exit early.
+ out_and_grads = [
+ (o, g) for (o, g) in zip(output_tensors, output_grads) if o.requires_grad
+ ]
+ if not out_and_grads:
+ return (None, None) + tuple(None for _ in args)
+
+ # Compute gradients on the filtered tensors.
+ computed_grads = th.autograd.grad(
+ [o for (o, g) in out_and_grads],
+ [args[i] for i in input_indices],
+ [g for (o, g) in out_and_grads],
+ )
+
+ # Reassemble the complete gradient tuple.
+ input_grads = [None for _ in args]
+ for i, g in zip(input_indices, computed_grads):
+ input_grads[i] = g
+ return (None, None) + tuple(input_grads)
diff --git a/genmo/diffusion_utils/resample.py b/genmo/diffusion_utils/resample.py
new file mode 100644
index 0000000000000000000000000000000000000000..3f4854386cc8b6f38db6975f0681fb645c180a9f
--- /dev/null
+++ b/genmo/diffusion_utils/resample.py
@@ -0,0 +1,154 @@
+from abc import ABC, abstractmethod
+
+import numpy as np
+import torch as th
+import torch.distributed as dist
+
+
+def create_named_schedule_sampler(name, diffusion):
+ """
+ Create a ScheduleSampler from a library of pre-defined samplers.
+
+ :param name: the name of the sampler.
+ :param diffusion: the diffusion object to sample for.
+ """
+ if name == "uniform":
+ return UniformSampler(diffusion)
+ elif name == "loss-second-moment":
+ return LossSecondMomentResampler(diffusion)
+ else:
+ raise NotImplementedError(f"unknown schedule sampler: {name}")
+
+
+class ScheduleSampler(ABC):
+ """
+ A distribution over timesteps in the diffusion process, intended to reduce
+ variance of the objective.
+
+ By default, samplers perform unbiased importance sampling, in which the
+ objective's mean is unchanged.
+ However, subclasses may override sample() to change how the resampled
+ terms are reweighted, allowing for actual changes in the objective.
+ """
+
+ @abstractmethod
+ def weights(self):
+ """
+ Get a numpy array of weights, one per diffusion step.
+
+ The weights needn't be normalized, but must be positive.
+ """
+
+ def sample(self, batch_size, device):
+ """
+ Importance-sample timesteps for a batch.
+
+ :param batch_size: the number of timesteps.
+ :param device: the torch device to save to.
+ :return: a tuple (timesteps, weights):
+ - timesteps: a tensor of timestep indices.
+ - weights: a tensor of weights to scale the resulting losses.
+ """
+ w = self.weights()
+ p = w / np.sum(w)
+ indices_np = np.random.choice(len(p), size=(batch_size,), p=p)
+ indices = th.from_numpy(indices_np).long().to(device)
+ weights_np = 1 / (len(p) * p[indices_np])
+ weights = th.from_numpy(weights_np).float().to(device)
+ return indices, weights
+
+
+class UniformSampler(ScheduleSampler):
+ def __init__(self, diffusion):
+ self.diffusion = diffusion
+ self._weights = np.ones([diffusion.num_timesteps])
+
+ def weights(self):
+ return self._weights
+
+
+class LossAwareSampler(ScheduleSampler):
+ def update_with_local_losses(self, local_ts, local_losses):
+ """
+ Update the reweighting using losses from a model.
+
+ Call this method from each rank with a batch of timesteps and the
+ corresponding losses for each of those timesteps.
+ This method will perform synchronization to make sure all of the ranks
+ maintain the exact same reweighting.
+
+ :param local_ts: an integer Tensor of timesteps.
+ :param local_losses: a 1D Tensor of losses.
+ """
+ batch_sizes = [
+ th.tensor([0], dtype=th.int32, device=local_ts.device)
+ for _ in range(dist.get_world_size())
+ ]
+ dist.all_gather(
+ batch_sizes,
+ th.tensor([len(local_ts)], dtype=th.int32, device=local_ts.device),
+ )
+
+ # Pad all_gather batches to be the maximum batch size.
+ batch_sizes = [x.item() for x in batch_sizes]
+ max_bs = max(batch_sizes)
+
+ timestep_batches = [th.zeros(max_bs).to(local_ts) for bs in batch_sizes]
+ loss_batches = [th.zeros(max_bs).to(local_losses) for bs in batch_sizes]
+ dist.all_gather(timestep_batches, local_ts)
+ dist.all_gather(loss_batches, local_losses)
+ timesteps = [
+ x.item() for y, bs in zip(timestep_batches, batch_sizes) for x in y[:bs]
+ ]
+ losses = [x.item() for y, bs in zip(loss_batches, batch_sizes) for x in y[:bs]]
+ self.update_with_all_losses(timesteps, losses)
+
+ @abstractmethod
+ def update_with_all_losses(self, ts, losses):
+ """
+ Update the reweighting using losses from a model.
+
+ Sub-classes should override this method to update the reweighting
+ using losses from the model.
+
+ This method directly updates the reweighting without synchronizing
+ between workers. It is called by update_with_local_losses from all
+ ranks with identical arguments. Thus, it should have deterministic
+ behavior to maintain state across workers.
+
+ :param ts: a list of int timesteps.
+ :param losses: a list of float losses, one per timestep.
+ """
+
+
+class LossSecondMomentResampler(LossAwareSampler):
+ def __init__(self, diffusion, history_per_term=10, uniform_prob=0.001):
+ self.diffusion = diffusion
+ self.history_per_term = history_per_term
+ self.uniform_prob = uniform_prob
+ self._loss_history = np.zeros(
+ [diffusion.num_timesteps, history_per_term], dtype=np.float64
+ )
+ self._loss_counts = np.zeros([diffusion.num_timesteps], dtype=np.int)
+
+ def weights(self):
+ if not self._warmed_up():
+ return np.ones([self.diffusion.num_timesteps], dtype=np.float64)
+ weights = np.sqrt(np.mean(self._loss_history**2, axis=-1))
+ weights /= np.sum(weights)
+ weights *= 1 - self.uniform_prob
+ weights += self.uniform_prob / len(weights)
+ return weights
+
+ def update_with_all_losses(self, ts, losses):
+ for t, loss in zip(ts, losses):
+ if self._loss_counts[t] == self.history_per_term:
+ # Shift out the oldest loss term.
+ self._loss_history[t, :-1] = self._loss_history[t, 1:]
+ self._loss_history[t, -1] = loss
+ else:
+ self._loss_history[t, self._loss_counts[t]] = loss
+ self._loss_counts[t] += 1
+
+ def _warmed_up(self):
+ return (self._loss_counts == self.history_per_term).all()
diff --git a/genmo/diffusion_utils/respace.py b/genmo/diffusion_utils/respace.py
new file mode 100644
index 0000000000000000000000000000000000000000..2f4a36748630818c6b964856dda877f13f670957
--- /dev/null
+++ b/genmo/diffusion_utils/respace.py
@@ -0,0 +1,128 @@
+# This code is based on https://github.com/openai/guided-diffusion
+import numpy as np
+import torch as th
+
+from .gaussian_diffusion import GaussianDiffusion
+
+
+def space_timesteps(num_timesteps, section_counts):
+ """
+ Create a list of timesteps to use from an original diffusion process,
+ given the number of timesteps we want to take from equally-sized portions
+ of the original process.
+
+ For example, if there's 300 timesteps and the section counts are [10,15,20]
+ then the first 100 timesteps are strided to be 10 timesteps, the second 100
+ are strided to be 15 timesteps, and the final 100 are strided to be 20.
+
+ If the stride is a string starting with "ddim", then the fixed striding
+ from the DDIM paper is used, and only one section is allowed.
+
+ :param num_timesteps: the number of diffusion steps in the original
+ process to divide up.
+ :param section_counts: either a list of numbers, or a string containing
+ comma-separated numbers, indicating the step count
+ per section. As a special case, use "ddimN" where N
+ is a number of steps to use the striding from the
+ DDIM paper.
+ :return: a set of diffusion steps from the original process to use.
+ """
+ if isinstance(section_counts, str):
+ if section_counts.startswith("ddim"):
+ desired_count = int(section_counts[len("ddim") :])
+ for i in range(1, num_timesteps):
+ if len(range(0, num_timesteps, i)) == desired_count:
+ return set(range(0, num_timesteps, i))
+ raise ValueError(
+ f"cannot create exactly {num_timesteps} steps with an integer stride"
+ )
+ section_counts = [int(x) for x in section_counts.split(",")]
+ size_per = num_timesteps // len(section_counts)
+ extra = num_timesteps % len(section_counts)
+ start_idx = 0
+ all_steps = []
+ for i, section_count in enumerate(section_counts):
+ size = size_per + (1 if i < extra else 0)
+ if size < section_count:
+ raise ValueError(
+ f"cannot divide section of {size} steps into {section_count}"
+ )
+ if section_count <= 1:
+ frac_stride = 1
+ else:
+ frac_stride = (size - 1) / (section_count - 1)
+ cur_idx = 0.0
+ taken_steps = []
+ for _ in range(section_count):
+ taken_steps.append(start_idx + round(cur_idx))
+ cur_idx += frac_stride
+ all_steps += taken_steps
+ start_idx += size
+ return set(all_steps)
+
+
+class SpacedDiffusion(GaussianDiffusion):
+ """
+ A diffusion process which can skip steps in a base diffusion process.
+
+ :param use_timesteps: a collection (sequence or set) of timesteps from the
+ original diffusion process to retain.
+ :param kwargs: the kwargs to create the base diffusion process.
+ """
+
+ def __init__(self, use_timesteps, **kwargs):
+ self.use_timesteps = set(use_timesteps)
+ self.timestep_map = []
+ self.original_num_steps = len(kwargs["betas"])
+
+ base_diffusion = GaussianDiffusion(**kwargs) # pylint: disable=missing-kwoa
+ last_alpha_cumprod = 1.0
+ new_betas = []
+ for i, alpha_cumprod in enumerate(base_diffusion.alphas_cumprod):
+ if i in self.use_timesteps:
+ new_betas.append(1 - alpha_cumprod / last_alpha_cumprod)
+ last_alpha_cumprod = alpha_cumprod
+ self.timestep_map.append(i)
+ kwargs["betas"] = np.array(new_betas)
+ super().__init__(**kwargs)
+
+ def p_mean_variance(self, model, *args, **kwargs): # pylint: disable=signature-differs
+ return super().p_mean_variance(self._wrap_model(model), *args, **kwargs)
+
+ def p_mean_variance_guided(self, model, *args, **kwargs): # pylint: disable=signature-differs
+ return super().p_mean_variance_guided(self._wrap_model(model), *args, **kwargs)
+
+ def training_losses(self, model, *args, **kwargs): # pylint: disable=signature-differs
+ return super().training_losses(self._wrap_model(model), *args, **kwargs)
+
+ def condition_mean(self, cond_fn, *args, **kwargs):
+ return super().condition_mean(self._wrap_model(cond_fn), *args, **kwargs)
+
+ def condition_score(self, cond_fn, *args, **kwargs):
+ return super().condition_score(self._wrap_model(cond_fn), *args, **kwargs)
+
+ def _wrap_model(self, model):
+ if isinstance(model, _WrappedModel):
+ return model
+ return _WrappedModel(
+ model, self.timestep_map, self.rescale_timesteps, self.original_num_steps
+ )
+
+ def _scale_timesteps(self, t):
+ # Scaling is done by the wrapped model.
+ return t
+
+
+class _WrappedModel:
+ def __init__(self, model, timestep_map, rescale_timesteps, original_num_steps):
+ self.model = model
+ self.timestep_map = timestep_map
+ self.rescale_timesteps = rescale_timesteps
+ self.original_num_steps = original_num_steps
+
+ def __call__(self, x, ts, **kwargs):
+ map_tensor = th.tensor(self.timestep_map, device=ts.device, dtype=ts.dtype)
+ new_ts = map_tensor[ts]
+ if self.rescale_timesteps:
+ new_ts = new_ts.float() * (1000.0 / self.original_num_steps)
+ return self.model(x, new_ts, **kwargs)
diff --git a/genmo/genmo.py b/genmo/genmo.py
new file mode 100644
index 0000000000000000000000000000000000000000..b8090c8040b67f9bfcc857837405fc0ca5413aa7
--- /dev/null
+++ b/genmo/genmo.py
@@ -0,0 +1,1307 @@
+import os
+import time
+
+from hydra.utils import instantiate
+import numpy as np
+try:
+ import pytorch_lightning as pl
+except Exception: # pragma: no cover
+ import types
+ import torch
+
+ class _LightningModule(torch.nn.Module):
+ pass
+
+ pl = types.SimpleNamespace(LightningModule=_LightningModule)
+from timm.models.vision_transformer import Mlp
+import torch
+import torch.nn as nn
+try:
+ from transformers import T5EncoderModel, T5Tokenizer
+except Exception: # pragma: no cover
+ T5EncoderModel = None
+ T5Tokenizer = None
+
+from genmo.network.base_arch.transformer.layer import BasicBlock, zero_module
+from genmo.utils.geo_transform import get_bbx_xys, normalize_kp2d
+from genmo.utils.net_utils import length_to_mask
+from genmo.utils.pylogger import Log
+from genmo.utils.tools import Timer
+from third_party.GVHMR.hmr4d.model.gvhmr.utils import stats_compose
+from third_party.GVHMR.hmr4d.model.gvhmr.utils.postprocess import (
+ pp_static_joint,
+ process_ik,
+)
+from third_party.GVHMR.hmr4d.utils.geo.augment_noisy_pose import (
+ get_invisible_legs_mask,
+ get_visible_mask,
+ get_wham_aug_kp3d,
+ randomly_modify_hands_legs,
+)
+from third_party.GVHMR.hmr4d.utils.geo.flip_utils import avg_smplx_aa, flip_smplx_params
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import (
+ compute_bbox_info_bedlam,
+ perspective_projection,
+ safely_render_x3d_K,
+)
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+reproj_z_thr = 0.3
+
+
+class GENMO(pl.LightningModule):
+ def __init__(
+ self,
+ pipeline,
+ optimizer=None,
+ scheduler=None,
+ model_cfg=None,
+ ignored_weights_prefix=["smplx", "pipeline.endecoder"],
+ ):
+ super().__init__()
+ self.pipeline = instantiate(pipeline, _recursive_=False)
+ self.endecoder = self.pipeline.endecoder
+ self.optimizer = instantiate(optimizer, _partial_=True)
+ self.model_cfg = model_cfg
+ self.scheduler = scheduler
+ self.enable_test_time_opt = model_cfg.get("enable_test_time_opt", False)
+ self.train_modes = model_cfg.get("train_modes", [])
+ if isinstance(self.train_modes, str):
+ self.train_modes = [self.train_modes]
+
+ # Options
+ self.ignored_weights_prefix = ignored_weights_prefix
+
+ # The test step is the same as validation
+ self.test_step = self.predict_step = self.validation_step
+ self.timing = os.environ.get("DEBUG_TIMING", "FALSE") == "TRUE"
+
+ # SMPLX
+ self.smplx = make_smplx("supermotion_v437coco17")
+
+ if "text_encoder" in model_cfg:
+ self.max_text_len = model_cfg.text_encoder.max_text_len
+
+ self.use_text_encoder = True
+ if model_cfg.text_encoder.get("load_llm", False):
+ llm_version = model_cfg.text_encoder.llm_version
+ self.max_text_len = model_cfg.text_encoder.max_text_len
+ text_encoder, self.tokenizer = self.load_and_freeze_llm(llm_version)
+ self.text_encoder = [text_encoder.cuda()]
+ else:
+ self.text_encoder = self.tokenizer = None
+ else:
+ self.use_text_encoder = False
+
+ self.f_condition_dim = {
+ "obs": (17, 3),
+ "f_cliffcam": (3,),
+ "f_cam_angvel": (6,),
+ "f_cam_t_vel": (3,),
+ "f_imgseq": (1024,),
+ # "encoded_music": 438,
+ "encoded_music": (self.pipeline.args.encoded_music_dim,),
+ "encoded_audio": (128,),
+ "observed_motion_3d": (151,),
+ }
+
+ self.not_add_features = [
+ "obs",
+ "f_cliffcam",
+ "f_cam_angvel",
+ "f_cam_t_vel",
+ "f_imgseq",
+ "observed_motion_3d",
+ "multi_text_embed",
+ "encoded_music",
+ "encoded_audio",
+ ]
+
+ dropout = self.pipeline.args_denoiser3d.get("dropout", 0.1)
+
+ latent_dim = self.pipeline.args_denoiser3d.get("latent_dim", 512)
+ self.latent_dim = latent_dim
+ if "obs" in self.pipeline.args.in_attr:
+ self.learned_pos_linear = nn.Linear(2, 32)
+ self.learned_pos_params = nn.Parameter(
+ torch.randn(17, 32), requires_grad=True
+ )
+ self.embed_noisyobs = Mlp(
+ 17 * 32,
+ hidden_features=latent_dim * 2,
+ out_features=latent_dim,
+ drop=dropout,
+ )
+
+ if "f_cliffcam" in self.pipeline.args.in_attr:
+ self.cliffcam_embedder = nn.Sequential(
+ nn.Linear(self.f_condition_dim["f_cliffcam"][0], latent_dim),
+ nn.SiLU(),
+ nn.Dropout(dropout),
+ zero_module(nn.Linear(latent_dim, latent_dim)),
+ )
+
+ if "f_imgseq" in self.pipeline.args.in_attr:
+ self.imgseq_embedder = nn.Sequential(
+ nn.LayerNorm(self.f_condition_dim["f_imgseq"][0]),
+ zero_module(nn.Linear(self.f_condition_dim["f_imgseq"][0], latent_dim)),
+ )
+
+ if "f_cam_angvel" in self.pipeline.args.in_attr:
+ self.cam_angvel_embedder = nn.Sequential(
+ nn.Linear(self.f_condition_dim["f_cam_angvel"][0], latent_dim),
+ nn.SiLU(),
+ nn.Dropout(dropout),
+ zero_module(nn.Linear(latent_dim, latent_dim)),
+ )
+
+ if "f_cam_t_vel" in self.pipeline.args.in_attr:
+ self.cam_t_vel_embedder = nn.Sequential(
+ nn.Linear(self.f_condition_dim["f_cam_t_vel"][0], latent_dim),
+ nn.SiLU(),
+ nn.Dropout(dropout),
+ zero_module(nn.Linear(latent_dim, latent_dim)),
+ )
+
+ if "encoded_music" in self.pipeline.args.in_attr:
+ self.music_embedder = Mlp(
+ self.f_condition_dim["encoded_music"][0],
+ hidden_features=latent_dim * 2,
+ out_features=latent_dim,
+ drop=dropout,
+ )
+ self.music_mask_prob = model_cfg.music_mask_prob
+
+ if "encoded_audio" in self.pipeline.args.in_attr:
+ self.audio_encoder = torch.nn.Sequential(
+ BasicBlock(1, 32, 15, 5),
+ BasicBlock(32, 32, 15, 6),
+ BasicBlock(32, 32, 15, 1),
+ BasicBlock(32, 64, 15, 5),
+ BasicBlock(64, 64, 15, 1),
+ BasicBlock(64, 128, 15, 4),
+ )
+ self.audio_embedder = nn.Sequential(
+ nn.LayerNorm(self.f_condition_dim["encoded_audio"][0]),
+ zero_module(
+ nn.Linear(self.f_condition_dim["encoded_audio"][0], latent_dim)
+ ),
+ )
+
+ self.audio_mask_prob = model_cfg.audio_mask_prob
+
+ if "multi_text_embed" in self.pipeline.args.in_attr:
+ multi_text_module_cfg = model_cfg.get("multi_text_module_cfg", {})
+ text_embed_dim = multi_text_module_cfg.get("text_embed_dim", 1024)
+ self.multi_text_embedder = nn.Linear(text_embed_dim, latent_dim)
+ encoder_layer = nn.TransformerEncoderLayer(
+ d_model=latent_dim, # Input dimension
+ nhead=multi_text_module_cfg.get(
+ "nhead", 8
+ ), # Number of attention heads
+ dim_feedforward=multi_text_module_cfg.get("dim_feedforward", 2048),
+ dropout=dropout,
+ batch_first=True,
+ )
+ self.multi_text_transformer = nn.TransformerEncoder(
+ encoder_layer, num_layers=multi_text_module_cfg.get("num_layers", 3)
+ )
+
+ self.condition_source = {
+ "image": ["f_imgseq"],
+ "2d": ["obs", "f_cliffcam"],
+ "camera": ["f_cam_angvel", "f_cam_tvel"],
+ "audio": ["encoded_audio"],
+ "music": ["encoded_music"],
+ }
+
+ if self.model_cfg.normalize_cam_angvel:
+ cam_angvel_stats = stats_compose.cam_angvel["manual"]
+ self.register_buffer(
+ "cam_angvel_mean",
+ torch.tensor(cam_angvel_stats["mean"]),
+ persistent=False,
+ )
+ self.register_buffer(
+ "cam_angvel_std",
+ torch.tensor(cam_angvel_stats["std"]),
+ persistent=False,
+ )
+
+ # Load normalizer stats
+ self.normalizer_stats = {}
+ if "norm_attr_stats" in self.model_cfg:
+ for key, stats_path in self.model_cfg.norm_attr_stats.items():
+ self.normalizer_stats[key] = torch.load(
+ stats_path, map_location="cpu", weights_only=False
+ )
+
+ self.no_exist_keys = ["obs", "observed_motion_3d", "multi_text_embed"]
+ # self.no_exist_keys = ["observed_motion_3d", "multi_text_embed"]
+ if self.model_cfg.use_cond_exists_as_input:
+ if self.model_cfg.cond_merge_strategy == "add":
+ self.cond_exists_embedder = nn.ModuleDict()
+ for k in self.pipeline.args.in_attr:
+ if k not in self.no_exist_keys:
+ self.cond_exists_embedder[k] = nn.Sequential(
+ nn.Linear(latent_dim + 1, latent_dim),
+ nn.SiLU(),
+ zero_module(nn.Linear(latent_dim, latent_dim)),
+ )
+ elif self.model_cfg.cond_merge_strategy == "concat":
+ raise NotImplementedError("Concat is not implemented")
+
+ def normalize_attr(self, x, key):
+ """Normalize input tensor using stored statistics"""
+ mean = self.normalizer_stats[key]["mean"].to(x)
+ std = self.normalizer_stats[key]["std"].to(x)
+ return (x - mean) / std
+
+ def load_and_freeze_llm(self, llm_version):
+ if T5Tokenizer is None or T5EncoderModel is None:
+ raise RuntimeError(
+ "Text encoder requested but `transformers` is not available. "
+ "Install it (e.g. `pip install transformers sentencepiece`) or set "
+ "`model.model_cfg.text_encoder.load_llm=false` / `model.model_cfg.text_encoder=null`."
+ )
+ tokenizer = T5Tokenizer.from_pretrained(llm_version)
+ model = T5EncoderModel.from_pretrained(llm_version)
+ # Freeze llm weights
+ model.eval()
+ for p in model.parameters():
+ p.requires_grad = False
+ return model, tokenizer
+
+ def encode_text(self, raw_text, has_text=None):
+ # raw_text - list (batch_size length) of strings with input text prompts
+ device = next(self.parameters()).device
+ with torch.no_grad():
+ with torch.cuda.amp.autocast(enabled=False):
+ max_text_len = self.max_text_len
+
+ encoded = self.tokenizer.batch_encode_plus(
+ raw_text,
+ return_tensors="pt",
+ padding="max_length",
+ max_length=max_text_len,
+ truncation=True,
+ )
+ # We expect all the processing is done in GPU.
+ input_ids = encoded.input_ids.to(device)
+ attn_mask = encoded.attention_mask.to(device)
+
+ with torch.no_grad():
+ output = self.text_encoder[0](
+ input_ids=input_ids, attention_mask=attn_mask
+ )
+ encoded_text = output.last_hidden_state.detach()
+
+ encoded_text = encoded_text[:, :max_text_len]
+ attn_mask = attn_mask[:, :max_text_len]
+ encoded_text *= attn_mask.unsqueeze(-1)
+ # for bnum in range(encoded_text.shape[0]):
+ # nvalid_elem = attn_mask[bnum].sum().item()
+ # encoded_text[bnum][nvalid_elem:] = 0
+ if has_text is not None:
+ no_text = ~has_text
+ encoded_text[no_text] = 0
+ return encoded_text
+
+ def generate_mask(self, mask_cfg, orig_mask, length):
+ _cfg = mask_cfg
+ mask = torch.ones_like(orig_mask)
+ drop_prob = _cfg.get("drop_prob", 0.0)
+ if drop_prob <= 0:
+ return mask
+ max_num_drops = _cfg.get("max_num_drops", 1)
+ min_drop_nframes = _cfg.get("min_drop_nframes", 1)
+ max_drop_nframes = _cfg.get("max_drop_nframes", 30)
+ joint_drop_prob = _cfg.get("joint_drop_prob", 0.0)
+ for i in range(orig_mask.shape[0]):
+ mlen = length[i].item()
+ if np.random.rand() < drop_prob:
+ num_drops = np.random.randint(1, max_num_drops + 1)
+ for _ in range(num_drops):
+ drop_len = np.random.randint(
+ min_drop_nframes, min(max_drop_nframes, mlen) + 1
+ )
+ drop_start = np.random.randint(0, max(mlen - drop_len, 1))
+ if joint_drop_prob > 0:
+ drop_joints = np.random.rand(17) < joint_drop_prob
+ mask[i, drop_start : drop_start + drop_len, drop_joints] = False
+ else:
+ mask[i, drop_start : drop_start + drop_len] = False
+ # print(f"Drop {i} {drop_start} {drop_len}")
+ if joint_drop_prob > 0:
+ COCO17_TREE = [
+ [5, 6],
+ 0,
+ 0,
+ 1,
+ 2,
+ -1,
+ -1,
+ 5,
+ 6,
+ 7,
+ 8,
+ -1,
+ -1,
+ 11,
+ 12,
+ 13,
+ 14,
+ 15,
+ 15,
+ 15,
+ 16,
+ 16,
+ 16,
+ ]
+ for child in range(17):
+ parent = COCO17_TREE[child]
+ if parent == -1:
+ continue
+ if isinstance(parent, list):
+ mask[..., child] *= mask[..., parent[0]] * mask[..., parent[1]]
+ else:
+ mask[..., child] *= mask[..., parent]
+ return mask
+
+ def training_step(self, batch, batch_idx):
+ def append_mode_to_loss(outputs, mode, suffix=""):
+ if suffix != "":
+ suffix = f"_{suffix}"
+ for k in list(outputs.keys()):
+ if "_loss" in k or k in {"loss"}:
+ outputs[f"Loss_{mode}{suffix}/{k}"] = outputs.pop(k)
+ return outputs
+
+ outputs = {"loss": 0}
+
+ with Timer("train_step", enabled=self.timing):
+ for mode in self.train_modes:
+ self.prepare_batch(batch, mode) # set "obs" (2d keypoints)
+ outputs_mode = self.train_step(batch, batch_idx, mode=mode)
+ outputs["loss"] += outputs_mode["loss"]
+ append_mode_to_loss(outputs_mode, mode)
+ outputs.update(outputs_mode)
+ if mode == "regression" and "diffusion" in self.train_modes:
+ batch["regression_outputs"] = outputs_mode.copy()
+ # batch[f"{mode}_condition"] = outputs_mode[f"{mode}_condition"]
+
+ # Log
+ log_kwargs = {
+ "on_epoch": True,
+ "prog_bar": True,
+ "logger": True,
+ "sync_dist": True,
+ "batch_size": outputs["batch_size"],
+ }
+ self.log("train/loss", outputs["loss"], **log_kwargs)
+ for k, v in outputs.items():
+ if "_loss" in k:
+ self.log(f"{k}", v, **log_kwargs)
+
+ return outputs
+
+
+
+ def prepare_batch(self, batch, mode):
+ target_x = self.endecoder.encode(batch) # (B, L, C)
+ if mode == "diffusion":
+ target_x[batch["mask"]["2d_only"]] = batch["regression_outputs"][
+ "model_output"
+ ]["pred_x_start"][batch["mask"]["2d_only"]]
+ else:
+ target_x[batch["mask"]["2d_only"]] = 0
+ valid_mask = batch["mask"]["valid"]
+ target_x_mask = torch.ones_like(target_x).bool()
+ target_x_mask[batch["mask"]["2d_only"]] = False
+ target_x_mask[batch["mask"]["spv_incam_only"], :, 142:] = False
+ target_x_mask = target_x_mask & valid_mask[:, :, None]
+ # Optional: do not supervise vertical component of local_transl_vel (helps prevent "flying" roots).
+ # local_transl_vel is [148:151] in the gvhmr encoding; y is index 149.
+ if bool(self.model_cfg.get("mask_transl_vel_y", False)):
+ target_x_mask[..., 149] = False
+
+ batch["target_x"] = target_x
+ batch["target_x_mask"] = target_x_mask
+
+ batch["device"] = batch["target_x"].device
+ batch["B"], batch["L"] = B, L = batch["target_x"].shape[:2]
+
+ if "text_embed" in batch:
+ batch["encoded_text"] = batch["text_embed"].cuda()
+ elif self.use_text_encoder:
+ batch["encoded_text"] = self.encode_text(
+ batch["caption"], batch["has_text"]
+ )
+
+ # Create augmented noisy-obs : gt_j3d(coco17)
+ with torch.cuda.amp.autocast(enabled=False):
+ with torch.no_grad():
+ gt_verts437, gt_j3d = self.smplx(**batch["smpl_params_c"])
+ root_ = gt_j3d[:, :, [11, 12], :].mean(-2, keepdim=True)
+ batch["gt_j3d"] = gt_j3d
+ batch["gt_cr_coco17"] = gt_j3d - root_
+ batch["gt_c_verts437"] = gt_verts437
+ batch["gt_cr_verts437"] = gt_verts437 - root_
+
+ # compute bbx_xys from GT Vertices
+ i_x2d = safely_render_x3d_K(gt_verts437, batch["K_fullimg"], thr=0.3)
+ det_kp2d = batch["kp2d"]
+ assert det_kp2d.ndim == 4 and det_kp2d.shape[-1] == 3, (
+ f"det_kp2d.shape: {det_kp2d.shape}"
+ )
+ det_kp2d_conf = det_kp2d[..., 2]
+ batch["det_kp2d_conf"] = det_kp2d_conf
+ det_kp2d = det_kp2d[..., :2]
+ det_kp2d_mask = det_kp2d_conf > 0.5
+ bbx_xys = get_bbx_xys(i_x2d, do_augment=True)
+ bbx_xys_detected = get_bbx_xys(det_kp2d, det_kp2d_mask, do_augment=True)
+ bbx_xys[batch["mask"]["2d_only"]] = bbx_xys_detected[batch["mask"]["2d_only"]]
+
+ # TODO: Need to compare
+ if False: # trust image bbx_xys seems better
+ batch["bbx_xys"] = bbx_xys
+ else:
+ mask_bbx_xys = batch["mask"]["bbx_xys"]
+ batch["bbx_xys"][~mask_bbx_xys] = bbx_xys[~mask_bbx_xys].to(
+ batch["bbx_xys"]
+ )
+
+ with torch.cuda.amp.autocast(enabled=False):
+ gt_kp2d = perspective_projection(gt_j3d, batch["K_fullimg"]) # (B, L, J, 2)
+ gt_kp2d[batch["mask"]["2d_only"]] = det_kp2d[batch["mask"]["2d_only"]]
+ gt_kp2d_normed = normalize_kp2d(gt_kp2d, batch["bbx_xys"]) # (B, L, J, 3)
+ valid_mask_j17 = (
+ (gt_j3d[..., 2] > reproj_z_thr)
+ * (gt_kp2d_normed[..., 0] > 0.0)
+ * (gt_kp2d_normed[..., 0] < 1.0)
+ * (gt_kp2d_normed[..., 1] > 0.0)
+ * (gt_kp2d_normed[..., 1] < 1.0)
+ )[..., None]
+
+ batch["gt_kp2d_normed"] = gt_kp2d_normed
+ batch["valid_mask_j17"] = valid_mask_j17
+
+ # IMPORTANT: stochastic 2D corruption is training-time augmentation.
+ # For validation/inference (`self.training == False`), keep conditioning deterministic.
+ if self.training:
+ noisy_j3d = gt_j3d + get_wham_aug_kp3d(gt_j3d.shape[:2])
+ else:
+ noisy_j3d = gt_j3d
+ obs_i_j2d = perspective_projection(
+ noisy_j3d, batch["K_fullimg"]
+ ) # (B, L, J, 2)
+ noisy_det_j2d = det_kp2d.clone()
+ if self.training:
+ aug = get_wham_aug_kp3d(noisy_det_j2d.shape[:2])[..., :2]
+ f = torch.tensor([1024.0, 1024.0]).to(aug) / 4.0
+ aug *= f * self.model_cfg.kp2d_noise_scale
+ noisy_det_j2d = noisy_det_j2d + aug
+
+ # Use some detected vitpose (presave data)
+ prob = 0.5
+ mask_real_vitpose = (torch.rand(batch["B"]).to(obs_i_j2d) < prob) * batch[
+ "mask"
+ ]["vitpose"]
+ mask_real_vitpose = mask_real_vitpose | batch["mask"]["2d_only"]
+
+ assert batch["mask"]["2d_only"].sum() == 0, batch["mask"]["2d_only"].sum()
+
+ obs_i_j2d[mask_real_vitpose] = noisy_det_j2d[mask_real_vitpose]
+
+ if self.training:
+ obs_i_j2d = randomly_modify_hands_legs(obs_i_j2d)
+ j2d_visible_mask = get_visible_mask(gt_j3d.shape[:2]).cuda() # (B, L, J)
+ else:
+ j2d_visible_mask = torch.ones(gt_j3d.shape[:3], device=gt_j3d.device, dtype=torch.bool)
+ j2d_visible_mask = j2d_visible_mask & batch["mask"]["has_2d_mask"][:, :, None]
+
+ j2d_visible_mask[mask_real_vitpose] *= det_kp2d_mask[mask_real_vitpose]
+ close_mask = (noisy_j3d[..., 2] < 0.3) & (~mask_real_vitpose)[:, None, None]
+ j2d_visible_mask[close_mask] = (
+ False # Set close-to-image-plane points as invisible
+ )
+
+ if self.training:
+ legs_invisible_mask = get_invisible_legs_mask(
+ gt_j3d.shape[:2]
+ ).cuda() # (B, L, J)
+ j2d_visible_mask[legs_invisible_mask] = False
+ if "mask_cfg" in self.model_cfg:
+ if self.training:
+ mask = self.generate_mask(
+ self.model_cfg.mask_cfg, j2d_visible_mask, batch["length"]
+ )
+ j2d_visible_mask = j2d_visible_mask & mask
+ if "body_mask_cfg" in self.model_cfg:
+ if self.training:
+ mask = self.generate_mask(
+ self.model_cfg.body_mask_cfg, j2d_visible_mask, batch["length"]
+ )
+ j2d_visible_mask = j2d_visible_mask & mask
+
+ occluded_img_mask = j2d_visible_mask.sum(dim=-1) <= 3
+ f_cliffcam = compute_bbox_info_bedlam(
+ batch["bbx_xys"], batch["K_fullimg"]
+ ) # (B, L, 3)
+ f_cliffcam[occluded_img_mask] = 0
+ batch["f_cliffcam"] = f_cliffcam
+ condition_mask = dict()
+ condition_mask["has_img_mask"] = batch["mask"]["has_img_mask"] & (
+ ~occluded_img_mask
+ )
+ condition_mask["has_2d_mask"] = batch["mask"]["has_2d_mask"] & (
+ ~occluded_img_mask
+ )
+ condition_mask["has_cam_mask"] = batch["mask"]["has_cam_mask"].clone()
+ condition_mask["has_audio_mask"] = batch["mask"]["has_audio_mask"].clone()
+ condition_mask["has_music_mask"] = batch["mask"]["has_music_mask"].clone()
+ batch["condition_mask"] = condition_mask
+
+ obs_kp2d = torch.cat(
+ [obs_i_j2d, j2d_visible_mask[:, :, :, None].float()], dim=-1
+ ) # (B, L, J, 3)
+ obs = normalize_kp2d(obs_kp2d, batch["bbx_xys"]) # (B, L, J, 3)
+ obs[~j2d_visible_mask] = 0 # if not visible, set to (0,0,0)
+ j2d_visible_mask[~batch["mask"]["valid"]] = False
+ batch["obs"] = obs
+ condition_mask["j2d_visible_mask"] = j2d_visible_mask
+ batch["obs"][~batch["mask"]["valid"]] = 0
+
+ batch["static_gt"] = self.endecoder.get_static_gt(
+ batch, self.pipeline.args.static_conf.vel_thr
+ ) # (B, L, 6)
+ batch["static_gt_mask"] = ~batch["mask"]["invalid_contact"]
+
+ f_cam_angvel = batch["cam_angvel"]
+ if self.model_cfg.normalize_cam_angvel:
+ f_cam_angvel = (f_cam_angvel - self.cam_angvel_mean) / self.cam_angvel_std
+ batch["f_cam_angvel"] = f_cam_angvel
+
+ for k in self.normalizer_stats:
+ if k in batch:
+ batch[k] = self.normalize_attr(batch[k], k)
+
+ def create_condition_mask(
+ self, batch, cond_mask_cfg, mode, train, first_k_frames=None
+ ):
+ B, L = batch["B"], batch["L"]
+ device = batch["device"]
+
+ has_text = batch["has_text"]
+ condition_mask = batch["condition_mask"]
+ has_img_mask = condition_mask["has_img_mask"].clone()
+ has_2d_mask = condition_mask["has_2d_mask"].clone()
+ has_cam_mask = condition_mask["has_cam_mask"].clone()
+ has_audio_mask = condition_mask["has_audio_mask"].clone()
+ has_music_mask = condition_mask["has_music_mask"].clone()
+ j2d_visible_mask = condition_mask["j2d_visible_mask"].clone()
+
+ if train:
+ regression_no_img_mask = cond_mask_cfg.get("regression_no_img_mask", False)
+ mask_text_prob = cond_mask_cfg.get("mask_text_prob", {}).get(mode, 0.0)
+ mask_img_prob = cond_mask_cfg.get("mask_img_prob", 0.0)
+ mask_cam_prob = cond_mask_cfg.get("mask_cam_prob", 0.0)
+ mask_f_imgseq_prob = cond_mask_cfg.get("mask_f_imgseq_prob", 0.0)
+
+ if mask_text_prob > 0:
+ mask_text = (torch.rand(batch["B"]) < mask_text_prob).to(device)
+ batch["text_mask"] = mask_text
+ else:
+ batch["text_mask"] = None
+ if batch.get("text_mask", None) is not None:
+ batch["has_text"][batch["text_mask"]] = False
+
+ if regression_no_img_mask and mode == "regression":
+ mask_img_prob = 0
+ mask_f_imgseq_prob = 0
+ has_2d_mask[~batch["mask"]["2d_only"]] = True
+
+ if mask_img_prob > 0:
+ mask_img = (has_text[:, None] | has_audio_mask | has_music_mask) & (
+ torch.rand(batch["B"]) < mask_img_prob
+ ).to(device)[:, None]
+ has_img_mask = has_img_mask & ~mask_img
+ has_2d_mask = has_2d_mask & ~mask_img
+ j2d_visible_mask = j2d_visible_mask & ~mask_img[..., None]
+
+ if mask_cam_prob > 0:
+ mask_cam = (has_text[:, None] | has_music_mask | has_audio_mask) & (
+ torch.rand(batch["B"]) < mask_cam_prob
+ ).to(device)[:, None]
+ has_cam_mask = has_cam_mask & ~mask_cam
+
+ has_music_mask = (
+ has_music_mask
+ & (torch.rand((B,), device=device) > self.music_mask_prob)[:, None]
+ )
+ has_audio_mask = (
+ has_audio_mask
+ & (torch.rand((B,), device=device) > self.audio_mask_prob)[:, None]
+ )
+
+ j2d_visible_mask = j2d_visible_mask & has_2d_mask[:, :, None]
+ has_2d_mask = j2d_visible_mask.sum(dim=-1) > 3
+
+ f_condition_exists = dict()
+ # f_condition = dict()
+ for k in self.condition_source["image"]:
+ f_condition_exists[k] = has_img_mask.clone()
+ for k in self.condition_source["2d"]:
+ if k == "obs":
+ f_condition_exists[k] = j2d_visible_mask.clone()
+ else:
+ f_condition_exists[k] = has_2d_mask.clone()
+ for k in self.condition_source["camera"]:
+ f_condition_exists[k] = has_cam_mask.clone()
+ for k in self.condition_source["audio"]:
+ f_condition_exists[k] = has_audio_mask.clone()
+ for k in self.condition_source["music"]:
+ f_condition_exists[k] = has_music_mask.clone()
+
+ if train and mask_f_imgseq_prob > 0:
+ mask_f_imgseq = (torch.rand(batch["B"]) < mask_f_imgseq_prob).to(device)
+ f_condition_exists["f_imgseq"] = f_condition_exists["f_imgseq"] & (
+ ~mask_f_imgseq
+ )
+
+ # randomly set null condition
+ skip_keys = self.pipeline.args.get("skip_keys_for_null_condition", [])
+ uncond_prob = self.pipeline.args.get("uncond_prob", 0.1)
+ if train and not self.pipeline.args.get("disable_random_null_condition", False):
+ for k in self.pipeline.args.in_attr:
+ if k in skip_keys:
+ continue
+ mask = torch.rand(f_condition_exists[k].shape[:2]) < uncond_prob
+ f_condition_exists[k][mask] = False
+
+ f_cond_dict = {}
+ f_uncond_dict = {}
+ f_uncond_exists = {k: f_condition_exists[k].clone() for k in f_condition_exists}
+
+ length = batch["length"]
+ end_fr = first_k_frames if first_k_frames is not None else None
+ if first_k_frames is not None:
+ length = length.clamp(max=first_k_frames)
+ for k in self.pipeline.args.in_attr:
+ if k == "obs":
+ obs = batch["obs"][:, :end_fr]
+ B, L, J, C = obs.shape
+ assert J == 17 and C == 3
+ obs = obs.clone()
+ obs = obs * j2d_visible_mask[:, :, :, None]
+ visible_mask = obs[..., [2]] > 0.5 # (B, L, J, 1)
+ obs[~visible_mask[..., 0]] = 0 # set low-conf to all zeros
+ f_obs = self.learned_pos_linear(obs[..., :2]) # (B, L, J, 32)
+ f_obs = (
+ f_obs * visible_mask
+ + self.learned_pos_params.repeat(B, L, 1, 1) * ~visible_mask
+ ) # (B, L, J, 32)
+ f_obs = self.embed_noisyobs(
+ f_obs.view(B, L, -1)
+ ) # (B, L, J*32) -> (B, L, C)
+ f_cond_dict["obs"] = f_obs
+ f_uncond_dict["obs"] = f_obs
+ elif k == "f_cliffcam":
+ f_cliffcam = batch["f_cliffcam"][:, :end_fr] # (B, L, 3)
+ f_cliffcam = self.cliffcam_embedder(f_cliffcam)
+ mask = f_condition_exists[k][:, :, None]
+ f_cond_dict["f_cliffcam"] = f_cliffcam * mask.float()
+ f_uncond_dict["f_cliffcam"] = f_cliffcam * mask.float()
+ elif k == "f_cam_angvel":
+ f_cam_angvel = batch["f_cam_angvel"][:, :end_fr] # (B, L, 6)
+ f_cam_angvel = self.cam_angvel_embedder(f_cam_angvel)
+ mask = f_condition_exists[k][:, :, None]
+ f_cond_dict["f_cam_angvel"] = f_cam_angvel * mask.float()
+ f_uncond_dict["f_cam_angvel"] = f_cam_angvel * mask.float()
+ elif k == "f_cam_t_vel":
+ f_cam_t_vel = batch["f_cam_t_vel"][:, :end_fr] # (B, L, 3)
+ f_cam_t_vel = self.cam_t_vel_embedder(f_cam_t_vel)
+ mask = f_condition_exists[k][:, :, None]
+ f_cond_dict["f_cam_t_vel"] = f_cam_t_vel * mask.float()
+ f_uncond_dict["f_cam_t_vel"] = f_cam_t_vel * mask.float()
+ elif k == "f_imgseq":
+ f_imgseq = batch["f_imgseq"][:, :end_fr] # (B, L, C)
+ f_imgseq = self.imgseq_embedder(f_imgseq)
+ mask = f_condition_exists[k][:, :, None]
+ f_cond_dict["f_imgseq"] = f_imgseq * mask.float()
+ f_uncond_dict["f_imgseq"] = f_imgseq * mask.float()
+ elif k == "encoded_music":
+ if "music_embed" in batch:
+ f_encoded_music = batch["music_embed"][:, :end_fr] # (B, L, C)
+ f_encoded_music = self.music_embedder(f_encoded_music)
+ mask = f_condition_exists[k][:, :, None]
+ f_cond_dict["encoded_music"] = f_encoded_music * mask.float()
+ else:
+ f_cond_dict["encoded_music"] = torch.zeros(
+ B, L, self.latent_dim
+ ).to(batch["device"])
+ f_uncond_dict["encoded_music"] = torch.zeros(B, L, self.latent_dim).to(
+ batch["device"]
+ )
+ f_uncond_exists["encoded_music"] = torch.zeros_like(
+ f_condition_exists["encoded_music"]
+ )
+ elif k == "encoded_audio":
+ if "audio_array" in batch:
+ encoded_audio = (
+ self.audio_encoder(batch["audio_array"].cuda().unsqueeze(1))
+ .transpose(1, 2)
+ .contiguous()
+ )[:, :end_fr]
+ mask = f_condition_exists[k][:, :, None]
+ encoded_audio = self.audio_embedder(encoded_audio)
+ f_cond_dict["encoded_audio"] = encoded_audio * mask.float()
+ else:
+ f_cond_dict["encoded_audio"] = torch.zeros(
+ B, L, self.latent_dim
+ ).to(batch["device"])
+ f_uncond_dict["encoded_audio"] = torch.zeros(B, L, self.latent_dim).to(
+ batch["device"]
+ )
+ f_uncond_exists["encoded_audio"] = torch.zeros_like(
+ f_condition_exists["encoded_audio"]
+ )
+ elif k == "observed_motion_3d":
+ motion_mask_3d = batch.get(
+ "motion_mask_3d",
+ torch.zeros_like(batch["observed_motion_3d"]),
+ )[:, :end_fr]
+ f_observed_motion_3d = torch.cat(
+ [batch["observed_motion_3d"][:, :end_fr], motion_mask_3d],
+ dim=-1,
+ )
+ f_observed_motion_3d = self.observed_motion_3d_embedder(
+ f_observed_motion_3d
+ )
+ f_cond_dict["observed_motion_3d"] = f_observed_motion_3d
+ f_uncond_dict["observed_motion_3d"] = torch.zeros_like(
+ f_observed_motion_3d
+ )
+ else:
+ assert False, f"Unknown condition key: {k}"
+
+ if k not in self.not_add_features:
+ f_cond_dict[k] = self.add_feature_embedders[k](batch[k][:, :end_fr])
+
+ if self.model_cfg.use_cond_exists_as_input:
+ if k not in self.no_exist_keys:
+ if k == "obs":
+ exist_mask = f_condition_exists[k][:, :end_fr]
+ exist_mask = exist_mask.sum(dim=-1, keepdim=True) > 0
+ uncond_exist_mask = f_uncond_exists[k][:, :end_fr]
+ uncond_exist_mask = (
+ uncond_exist_mask.sum(dim=-1, keepdim=True) > 0
+ )
+ else:
+ exist_mask = f_condition_exists[k][:, :end_fr, None]
+ uncond_exist_mask = f_uncond_exists[k][:, :end_fr, None]
+ f_cond_dict[k] = torch.cat(
+ [
+ f_cond_dict[k],
+ exist_mask.float(),
+ ],
+ dim=-1,
+ )
+ f_cond_dict[k] = self.cond_exists_embedder[k](f_cond_dict[k])
+ f_uncond_dict[k] = torch.cat(
+ [f_uncond_dict[k], uncond_exist_mask.float()],
+ dim=-1,
+ )
+ f_uncond_dict[k] = self.cond_exists_embedder[k](f_uncond_dict[k])
+
+ f_cond = sum(f_cond_dict.values())
+ f_uncond = sum(f_uncond_dict.values())
+ batch["f_cond"] = f_cond
+ batch["f_uncond"] = f_uncond
+
+ if batch.get("text_mask", None) is not None:
+ batch["encoded_text"] = batch["encoded_text"] * (
+ 1 - batch["text_mask"][:, None, None].float()
+ )
+ vis_mask = length_to_mask(length, f_cond.shape[1])[:, :end_fr] # (B, L)
+ motion = batch["target_x"] * vis_mask[..., None]
+ batch["motion"] = motion[:, :end_fr]
+
+ return batch
+
+ def train_step(self, batch, batch_idx, mode):
+ batch = batch.copy()
+ for k, v in batch.items():
+ if isinstance(v, torch.Tensor):
+ batch[k] = v.detach().clone()
+
+ cond_mask_cfg = self.model_cfg.get("condition_mask", {})
+ batch = self.create_condition_mask(batch, cond_mask_cfg, mode, train=True)
+
+ # Forward and get loss
+ outputs = self.pipeline.forward(
+ batch,
+ train=True,
+ global_step=self.trainer.global_step,
+ mode=mode,
+ normalizer_stats=self.normalizer_stats,
+ )
+ outputs["batch_size"] = batch["B"]
+ return outputs
+
+ def validation_step(self, batch, batch_idx, dataloader_idx=0):
+ test_mode = batch["meta"][0].get("mode", "default")
+ return self.validation(batch, test_mode, batch_idx, dataloader_idx)
+
+ def validation(self, batch, test_mode, batch_idx, dataloader_idx=0):
+ # Options & Check
+ try:
+ stage = self.trainer.state.stage
+ global_step = self.trainer.global_step
+ except Exception:
+ stage = "test"
+ global_step = 0
+ do_postproc = stage == "test" # Only apply postproc in test
+ do_flip_test = "flip_test" in batch
+ do_postproc_not_flip_test = (
+ do_postproc and not do_flip_test
+ ) # later pp when flip_test
+
+ # ROPE inference
+ obs = normalize_kp2d(batch["kp2d"], batch["bbx_xys"])
+ B, L = obs.shape[:2]
+
+ if "mask" in batch:
+ mask = batch["mask"]
+ if isinstance(mask, dict):
+ mask = mask["valid"]
+ obs[0, ~mask[0]] = 0
+
+ test_mode = batch["meta"][0].get("mode", "default")
+ batch_ = {
+ "length": batch["length"],
+ "obs": obs,
+ "bbx_xys": batch["bbx_xys"],
+ "K_fullimg": batch["K_fullimg"],
+ "cam_angvel": batch["cam_angvel"].clone(),
+ "f_cam_angvel": batch["cam_angvel"].clone(),
+ "f_imgseq": batch["f_imgseq"],
+ "caption": batch.get("caption", [""] * B),
+ "has_text": batch.get("has_text", torch.zeros(B).to(obs.device).bool()),
+ # "eval_gen_only": eval_gen_only,
+ "mode": test_mode,
+ "meta": batch["meta"],
+ "B": batch["B"],
+ "L": obs.shape[1],
+ "device": obs.device,
+ "target_x": torch.zeros(B, L, self.endecoder.get_motion_dim()).to(
+ obs.device
+ ),
+ "mask": batch["mask"],
+ }
+ if "music_embed" in batch:
+ batch_["music_embed"] = batch["music_embed"]
+ if "audio_array" in batch:
+ batch_["audio_array"] = batch["audio_array"]
+ det_kp2d = batch["kp2d"]
+ det_kp2d_conf = det_kp2d[..., 2]
+ j2d_visible_mask = det_kp2d_conf > 0.5
+ f_cliffcam = compute_bbox_info_bedlam(
+ batch_["bbx_xys"], batch_["K_fullimg"]
+ ) # (B, L, 3)
+ batch_["f_cliffcam"] = f_cliffcam
+
+ condition_mask = dict()
+ condition_mask["has_img_mask"] = batch["mask"]["has_img_mask"]
+ condition_mask["has_2d_mask"] = batch["mask"]["has_2d_mask"]
+ condition_mask["has_cam_mask"] = batch["mask"]["has_cam_mask"].clone()
+ condition_mask["has_audio_mask"] = batch["mask"]["has_audio_mask"].clone()
+ condition_mask["has_music_mask"] = batch["mask"]["has_music_mask"].clone()
+ condition_mask["j2d_visible_mask"] = j2d_visible_mask
+ batch_["condition_mask"] = condition_mask
+
+ if self.model_cfg.normalize_cam_angvel:
+ batch_["f_cam_angvel"] = (
+ batch_["f_cam_angvel"] - self.cam_angvel_mean
+ ) / self.cam_angvel_std
+
+ if "text_embed" in batch:
+ batch_["encoded_text"] = batch["text_embed"].cuda()
+ elif self.use_text_encoder:
+ batch_["encoded_text"] = self.encode_text(
+ batch["caption"], batch["has_text"]
+ )
+
+ if test_mode == "infilling":
+ batch["target_x"] = self.endecoder.encode(batch) # (B, L, C)
+ rng = np.random.RandomState(
+ batch["meta"][0].get("eval_seed", 7) + batch_idx
+ )
+ assert "motion_3d_mask_cfg" in self.model_cfg
+ all_mask_types = [
+ x
+ for x in self.model_cfg.motion_3d_mask_cfg.mask_types
+ if x != "no_mask"
+ ]
+ use_mask_type = all_mask_types[batch_idx % len(all_mask_types)]
+ mask_res = self.generate_motion_3d_mask(
+ self.model_cfg.motion_3d_mask_cfg,
+ batch["target_x"],
+ batch["length"],
+ rng=rng,
+ use_mask_type=use_mask_type,
+ )
+ batch_.update(mask_res)
+
+ if "inpainting_3d" in self.model_cfg:
+ batch_["observed_motion_3d"] = self.endecoder.encode(batch)
+ motion_mask_3d = torch.zeros_like(batch_["observed_motion_3d"]).cuda()
+ L = batch["length"][0]
+ keyframes = [i for i in range(L)]
+ if self.model_cfg["inpainting_3d"]["mode"] == "body_pose_dense":
+ motion_mask_3d[:, :, : 126 + 10] = 1
+ elif self.model_cfg["inpainting_3d"]["mode"] == "body_pose_root_rot_dense":
+ motion_mask_3d[:, :, : 126 + 10 + 12] = 1
+ elif (
+ self.model_cfg["inpainting_3d"]["mode"]
+ == "body_pose_root_rot_keyframe2"
+ ):
+ # keyframes = [0, L-1] # start and fix end
+ keyframes = [
+ 0,
+ np.random.choice(keyframes[L // 2 :], 1)[0],
+ ] # start and random end
+ motion_mask_3d[:, keyframes, : 126 + 10 + 12] = 1
+ elif (
+ self.model_cfg["inpainting_3d"]["mode"]
+ == "body_pose_root_rot_keyframe5"
+ ):
+ keyframes = [int((L - 1) * i / 4) for i in range(5)]
+ motion_mask_3d[:, keyframes, : 126 + 10 + 12] = 1
+ elif self.model_cfg["inpainting_3d"]["mode"] == "root_rot_vel_dense":
+ motion_mask_3d[:, :, 126:] = 1
+ else:
+ raise ValueError(
+ f"Unknown inpainting mode [{self.model_cfg['inpainting_3d']['mode']}]"
+ )
+ batch_["motion_mask_3d"] = motion_mask_3d
+ batch["keyframes"] = keyframes
+
+ for k in self.normalizer_stats:
+ if k in batch_:
+ batch_[k] = self.normalize_attr(batch_[k], k)
+
+ batch_ = self.create_condition_mask(
+ batch_, cond_mask_cfg=None, mode=None, train=False
+ )
+
+ outputs = self.pipeline.forward(
+ batch_,
+ train=False,
+ postproc=do_postproc_not_flip_test,
+ global_step=global_step,
+ test_mode=test_mode,
+ )
+ if "pred_smpl_params_global" in outputs:
+ outputs["pred_smpl_params_global"] = {
+ k: v[0] for k, v in outputs["pred_smpl_params_global"].items()
+ }
+ if "pred_smpl_params_incam" in outputs:
+ outputs["pred_smpl_params_incam"] = {
+ k: v[0] for k, v in outputs["pred_smpl_params_incam"].items()
+ }
+
+ if test_mode == "infilling":
+ outputs.update(mask_res)
+
+ if do_flip_test:
+ flip_test = batch["flip_test"]
+ obs = normalize_kp2d(flip_test["kp2d"], flip_test["bbx_xys"])
+ if "mask" in batch:
+ mask = batch["mask"]
+ if isinstance(mask, dict):
+ mask = mask["valid"]
+ obs[0, ~mask[0]] = 0
+
+ batch_ = {
+ "length": batch["length"],
+ "obs": obs,
+ "bbx_xys": flip_test["bbx_xys"],
+ "K_fullimg": batch["K_fullimg"],
+ "cam_angvel": flip_test["cam_angvel"].clone(),
+ "f_cam_angvel": flip_test["cam_angvel"].clone(),
+ "f_imgseq": flip_test["f_imgseq"],
+ "caption": flip_test.get("caption", [""] * B),
+ "has_text": flip_test.get(
+ "has_text", torch.zeros(B).to(obs.device).bool()
+ ),
+ "meta": batch["meta"],
+ "B": batch["B"],
+ "L": obs.shape[1],
+ "device": obs.device,
+ "target_x": torch.zeros(B, L, self.endecoder.get_motion_dim()).to(
+ obs.device
+ ),
+ "mask": batch["mask"],
+ }
+
+ det_kp2d = flip_test["kp2d"]
+ det_kp2d_conf = det_kp2d[..., 2]
+ j2d_visible_mask = det_kp2d_conf > 0.5
+
+ f_cliffcam = compute_bbox_info_bedlam(
+ batch_["bbx_xys"], batch_["K_fullimg"]
+ ) # (B, L, 3)
+ batch_["f_cliffcam"] = f_cliffcam
+
+ condition_mask = dict()
+ condition_mask["has_img_mask"] = batch["mask"]["has_img_mask"]
+ condition_mask["has_2d_mask"] = batch["mask"]["has_2d_mask"]
+ condition_mask["has_cam_mask"] = batch["mask"]["has_cam_mask"].clone()
+ condition_mask["has_audio_mask"] = batch["mask"]["has_audio_mask"].clone()
+ condition_mask["has_music_mask"] = batch["mask"]["has_music_mask"].clone()
+ condition_mask["j2d_visible_mask"] = j2d_visible_mask
+ batch_["condition_mask"] = condition_mask
+
+ if self.model_cfg.normalize_cam_angvel:
+ batch_["f_cam_angvel"] = (
+ batch_["f_cam_angvel"] - self.cam_angvel_mean
+ ) / self.cam_angvel_std
+ for k in self.normalizer_stats:
+ if k in batch_:
+ batch_[k] = self.normalize_attr(batch_[k], k)
+
+ if "text_embed" in batch:
+ batch_["encoded_text"] = batch["text_embed"].cuda()
+ elif self.use_text_encoder:
+ batch_["encoded_text"] = self.encode_text(
+ batch["caption"], batch["has_text"]
+ )
+ batch_ = self.create_condition_mask(
+ batch_, cond_mask_cfg=None, mode=None, train=False
+ )
+
+ flipped_outputs = self.pipeline.forward(
+ batch_, train=False, global_step=global_step, test_mode=test_mode
+ )
+ # First update incam results
+ flipped_outputs["pred_smpl_params_incam"] = {
+ k: v[0] for k, v in flipped_outputs["pred_smpl_params_incam"].items()
+ }
+ smpl_params1 = outputs["pred_smpl_params_incam"]
+ smpl_params2 = flip_smplx_params(flipped_outputs["pred_smpl_params_incam"])
+
+ smpl_params_avg = smpl_params1.copy()
+ smpl_params_avg["betas"] = (
+ smpl_params1["betas"] + smpl_params2["betas"]
+ ) / 2
+ smpl_params_avg["body_pose"] = avg_smplx_aa(
+ smpl_params1["body_pose"], smpl_params2["body_pose"]
+ )
+ smpl_params_avg["global_orient"] = avg_smplx_aa(
+ smpl_params1["global_orient"], smpl_params2["global_orient"]
+ )
+ outputs["pred_smpl_params_incam"] = smpl_params_avg
+
+ # Then update global results
+ outputs["pred_smpl_params_global"]["betas"] = smpl_params_avg["betas"]
+ outputs["pred_smpl_params_global"]["body_pose"] = smpl_params_avg[
+ "body_pose"
+ ]
+
+ # Finally, apply postprocess
+ if do_postproc:
+ # temporarily recover the original batch-dim
+ outputs["pred_smpl_params_global"] = {
+ k: v[None] for k, v in outputs["pred_smpl_params_global"].items()
+ }
+ outputs["pred_smpl_params_global"]["transl"] = pp_static_joint(
+ outputs, self.pipeline.endecoder
+ )
+ body_pose = process_ik(outputs, self.pipeline.endecoder)
+ outputs["pred_smpl_params_global"] = {
+ k: v[0] for k, v in outputs["pred_smpl_params_global"].items()
+ }
+
+ outputs["pred_smpl_params_global"]["body_pose"] = body_pose[0]
+ # outputs["pred_smpl_params_incam"]["body_pose"] = body_pose[0]
+
+ return outputs
+
+ @torch.no_grad()
+ def predict(self, data, static_cam=False, postproc=True):
+ now = time.time()
+ # ROPE inference
+ test_mode = data["meta"][0].get("mode", "default")
+ batch = {
+ "length": data["length"][None].cuda(),
+ "obs": normalize_kp2d(data["kp2d"], data["bbx_xys"])[None].cuda(),
+ "bbx_xys": data["bbx_xys"][None].cuda(),
+ "K_fullimg": data["K_fullimg"][None].cuda(),
+ "cam_angvel": data["cam_angvel"][None].cuda(),
+ "f_cam_angvel": data["cam_angvel"][None].cuda(),
+ "cam_tvel": data["cam_tvel"][None].cuda(),
+ "R_w2c": data["R_w2c"][None].cuda(),
+ "f_imgseq": data["f_imgseq"][None].cuda(),
+ # "text_embed": data["text_embed"][None].cuda(),
+ "has_text": data["has_text"].cuda(),
+ "B": 1,
+ "L": data["f_imgseq"].shape[0],
+ "mode": test_mode,
+ "target_x": torch.zeros(
+ 1, data["f_imgseq"].shape[0], self.endecoder.get_motion_dim()
+ ).cuda(),
+ }
+ if "music_embed" in data:
+ batch["music_embed"] = data["music_embed"][None].cuda()
+ if "audio_array" in data:
+ batch["audio_array"] = data["audio_array"][None].cuda()
+
+ if "fast_rollout" in data:
+ batch["fast_rollout"] = data["fast_rollout"]
+ batch["device"] = batch["f_imgseq"].device
+
+ if "meta" in data:
+ batch["meta"] = data["meta"]
+ else:
+ batch["meta"] = None
+
+ if "text_embed" in data:
+ batch["encoded_text"] = data["text_embed"][None].cuda()
+ elif "encoded_text" in data:
+ batch["encoded_text"] = data["encoded_text"][None].cuda()
+
+ batch["caption"] = [data.get("caption", "")]
+ if "encoded_text" not in batch:
+ has_text = batch.get("has_text", torch.zeros(1, device=batch["device"]).bool())
+ has_text = has_text.to(device=batch["device"]).bool()
+ has_llm = self.text_encoder is not None and self.tokenizer is not None
+ if bool(has_text.item()) and has_llm:
+ batch["encoded_text"] = self.encode_text(batch["caption"], has_text)
+ else:
+ # Keep the denoiser's text cross-attention path valid without loading an LLM.
+ # NOTE: `NetworkEncoderRoPE` expects `encoded_text_dim`, typically 1024.
+ try:
+ encoded_text_dim = int(self.pipeline.denoiser3d.denoiser.encoded_text_dim)
+ except Exception:
+ encoded_text_dim = 1024
+ text_len = int(getattr(self, "max_text_len", 1))
+ batch["encoded_text"] = torch.zeros(
+ (1, max(text_len, 1), encoded_text_dim),
+ device=batch["device"],
+ dtype=batch["f_imgseq"].dtype,
+ )
+ batch["has_text"] = torch.zeros_like(has_text)
+
+ batch["f_cliffcam"] = compute_bbox_info_bedlam(
+ batch["bbx_xys"], batch["K_fullimg"]
+ ).cuda()
+
+ condition_mask = dict()
+ condition_mask["has_img_mask"] = data["mask"]["has_img_mask"][None].cuda()
+ condition_mask["has_2d_mask"] = data["mask"]["has_2d_mask"][None].cuda()
+ condition_mask["has_cam_mask"] = (
+ data["mask"]["has_cam_mask"][None].cuda().clone()
+ )
+ condition_mask["has_audio_mask"] = (
+ data["mask"]["has_audio_mask"][None].cuda().clone()
+ )
+ condition_mask["has_music_mask"] = (
+ data["mask"]["has_music_mask"][None].cuda().clone()
+ )
+ kp2d_conf = data["kp2d"][..., 2][None].cuda()
+ condition_mask["j2d_visible_mask"] = kp2d_conf > 0.7
+ batch["condition_mask"] = condition_mask
+
+ if self.model_cfg.normalize_cam_angvel:
+ batch["f_cam_angvel"] = (
+ batch["f_cam_angvel"] - self.cam_angvel_mean
+ ) / self.cam_angvel_std
+ for k in self.normalizer_stats:
+ if k in batch:
+ batch[k] = self.normalize_attr(batch[k], k)
+
+ if "multi_text_data" in batch["meta"][0]:
+ if "text_embed" not in batch["meta"][0]["multi_text_data"]:
+ multi_text_data = batch["meta"][0]["multi_text_data"]
+ num_text = len(multi_text_data["caption"])
+ text_embed = self.encode_text(
+ multi_text_data["caption"], torch.tensor([True] * num_text)
+ )
+ batch["meta"][0]["multi_text_data"]["text_embed"] = text_embed
+
+ batch = self.create_condition_mask(
+ batch, cond_mask_cfg=None, mode=None, train=False
+ )
+
+ if self.pipeline.args.infer_version == 3:
+ postproc = False
+ else:
+ postproc = postproc
+ print(f"Preproc taken: {time.time() - now}")
+ now = time.time()
+ outputs = self.pipeline.forward(
+ batch,
+ train=False,
+ postproc=postproc,
+ static_cam=static_cam,
+ test_mode=test_mode,
+ )
+
+ pred = {
+ "smpl_params_global": {
+ k: v[0] for k, v in outputs["pred_smpl_params_global"].items()
+ },
+ "smpl_params_incam": {
+ k: v[0] for k, v in outputs["pred_smpl_params_incam"].items()
+ },
+ "K_fullimg": data["K_fullimg"],
+ "net_outputs": outputs, # intermediate outputs
+ }
+ print(f"Demo taken: {time.time() - now}")
+ return pred
+
+ def configure_optimizers(self):
+ params = []
+ for k, v in self.named_parameters():
+ if v.requires_grad:
+ params.append(v)
+ optimizer = self.optimizer(params=params)
+
+ if self.scheduler is None or self.scheduler["scheduler"] is None:
+ return optimizer
+
+ scheduler = dict(self.scheduler)
+ scheduler["scheduler"] = instantiate(
+ scheduler["scheduler"], optimizer=optimizer
+ )
+ return [optimizer], [scheduler]
+
+ # ============== Utils ================= #
+ def on_save_checkpoint(self, checkpoint) -> None:
+ for ig_keys in self.ignored_weights_prefix:
+ for k in list(checkpoint["state_dict"].keys()):
+ if k.startswith(ig_keys):
+ # Log.info(f"Remove key `{ig_keys}' from checkpoint.")
+ checkpoint["state_dict"].pop(k)
+
+ def load_pretrained_model(self, ckpt_path):
+ """Load pretrained checkpoint, and assign each weight to the corresponding part."""
+ Log.info(f"[PL-Trainer] Loading ckpt: {ckpt_path}")
+
+ ckpt = torch.load(ckpt_path, "cpu")
+ state_dict = ckpt["state_dict"]
+ missing, unexpected = self.load_state_dict(state_dict, strict=False)
+ real_missing = []
+ for k in missing:
+ ignored_when_saving = any(
+ k.startswith(ig_keys) for ig_keys in self.ignored_weights_prefix
+ )
+ if not ignored_when_saving:
+ real_missing.append(k)
+
+ if len(real_missing) > 0:
+ Log.warn(f"Missing keys: {real_missing}")
+ if len(unexpected) > 0:
+ Log.warn(f"Unexpected keys: {unexpected}")
+ return ckpt
diff --git a/genmo/network/base_arch/embeddings/pe.py b/genmo/network/base_arch/embeddings/pe.py
new file mode 100644
index 0000000000000000000000000000000000000000..d1ae876862271f117802b9f281bcc518e42de44f
--- /dev/null
+++ b/genmo/network/base_arch/embeddings/pe.py
@@ -0,0 +1,131 @@
+import numpy as np
+import torch
+import torch.nn as nn
+from torch.nn import functional as F
+
+
+class PositionalEncoding(nn.Module):
+ def __init__(self, d_model, dropout=0.1, max_len=5000):
+ super(PositionalEncoding, self).__init__()
+ self.dropout = nn.Dropout(p=dropout)
+
+ pe = torch.zeros(max_len, d_model)
+ position = torch.arange(0, max_len, dtype=torch.float).unsqueeze(1)
+ div_term = torch.exp(
+ torch.arange(0, d_model, 2).float() * (-np.log(10000.0) / d_model)
+ )
+ pe[:, 0::2] = torch.sin(position * div_term)
+ pe[:, 1::2] = torch.cos(position * div_term)
+ pe = pe.unsqueeze(0).transpose(0, 1)
+
+ self.pe = nn.Parameter(pe, requires_grad=False)
+
+ def forward(self, x, batch_first=False):
+ # not used in the final model
+ if batch_first:
+ pe = self.pe.transpose(0, 1)
+ x = x + pe[:, : x.shape[1], :]
+ else:
+ x = x + self.pe[: x.shape[0], :]
+ return self.dropout(x)
+
+
+class TimestepEmbedder(nn.Module):
+ def __init__(self, latent_dim, sequence_pos_encoder):
+ super().__init__()
+ self.latent_dim = latent_dim
+ self.sequence_pos_encoder = sequence_pos_encoder
+
+ time_embed_dim = self.latent_dim
+ self.time_embed = nn.Sequential(
+ nn.Linear(self.latent_dim, time_embed_dim),
+ nn.SiLU(),
+ nn.Linear(time_embed_dim, time_embed_dim),
+ )
+
+ def forward(self, timesteps):
+ return self.time_embed(self.sequence_pos_encoder.pe[timesteps]).permute(1, 0, 2)
+
+
+class InputProcess(nn.Module):
+ def __init__(self, data_rep, input_feats, latent_dim):
+ super().__init__()
+ self.data_rep = data_rep
+ self.input_feats = input_feats
+ self.latent_dim = latent_dim
+ self.poseEmbedding = nn.Linear(self.input_feats, self.latent_dim)
+ if self.data_rep == "rot_vel":
+ self.velEmbedding = nn.Linear(self.input_feats, self.latent_dim)
+
+ def forward(self, x):
+ bs, njoints, nfeats, nframes = x.shape
+ x = x.permute((3, 0, 1, 2)).reshape(nframes, bs, njoints * nfeats)
+
+ if self.data_rep in ["rot", "xyz", "hml_vec", "root"]:
+ x = self.poseEmbedding(x) # [seqlen, bs, d]
+ return x
+ elif self.data_rep == "rot_vel":
+ first_pose = x[[0]] # [1, bs, 150]
+ first_pose = self.poseEmbedding(first_pose) # [1, bs, d]
+ vel = x[1:] # [seqlen-1, bs, 150]
+ vel = self.velEmbedding(vel) # [seqlen-1, bs, d]
+ return torch.cat((first_pose, vel), axis=0) # [seqlen, bs, d]
+ else:
+ raise ValueError
+
+
+class OutputProcess(nn.Module):
+ def __init__(self, data_rep, input_feats, latent_dim, njoints, nfeats):
+ super().__init__()
+ self.data_rep = data_rep
+ self.input_feats = input_feats
+ self.latent_dim = latent_dim
+ self.njoints = njoints
+ self.nfeats = nfeats
+ self.poseFinal = nn.Linear(self.latent_dim, self.input_feats)
+ if self.data_rep == "rot_vel":
+ self.velFinal = nn.Linear(self.latent_dim, self.input_feats)
+
+ def forward(self, output):
+ nframes, bs, d = output.shape
+ if self.data_rep in ["rot", "xyz", "hml_vec", "root"]:
+ output = self.poseFinal(output) # [seqlen, bs, 150]
+ elif self.data_rep == "rot_vel":
+ first_pose = output[[0]] # [1, bs, d]
+ first_pose = self.poseFinal(first_pose) # [1, bs, 150]
+ vel = output[1:] # [seqlen-1, bs, d]
+ vel = self.velFinal(vel) # [seqlen-1, bs, 150]
+ output = torch.cat((first_pose, vel), axis=0) # [seqlen, bs, 150]
+ else:
+ raise ValueError
+ output = output.reshape(nframes, bs, self.njoints, self.nfeats)
+ output = output.permute(1, 2, 3, 0) # [bs, njoints, nfeats, nframes]
+ return output
+
+
+class EmbedActionScalar(nn.Module):
+ def __init__(self, in_features, out_features, activation):
+ super().__init__()
+ self.in_features = in_features
+ self.out_features = out_features
+ mid_features = int(out_features / 2)
+ self.lin1 = nn.Linear(in_features, mid_features)
+ self.activation = eval(f"F.{activation}")
+ self.lin2 = nn.Linear(mid_features, out_features)
+
+ def forward(self, input):
+ output = self.lin1(input)
+ output = self.activation(output)
+ output = self.lin2(output)
+ return output
+
+
+class EmbedActionTensor(nn.Module):
+ def __init__(self, num_actions, latent_dim):
+ super().__init__()
+ self.action_embedding = nn.Parameter(torch.randn(num_actions, latent_dim))
+
+ def forward(self, input):
+ idx = input[:, 0].to(torch.long) # an index array must be long
+ output = self.action_embedding[idx]
+ return output
diff --git a/genmo/network/base_arch/embeddings/rotary_embedding.py b/genmo/network/base_arch/embeddings/rotary_embedding.py
new file mode 100644
index 0000000000000000000000000000000000000000..d14114aedd6bfb1dbe9360257c1a3c2f982e011b
--- /dev/null
+++ b/genmo/network/base_arch/embeddings/rotary_embedding.py
@@ -0,0 +1,78 @@
+import torch
+import torch.nn as nn
+from einops import rearrange, repeat
+from torch.cuda.amp import autocast
+
+
+def rotate_half(x):
+ x = rearrange(x, "... (d r) -> ... d r", r=2)
+ x1, x2 = x.unbind(dim=-1)
+ x = torch.stack((-x2, x1), dim=-1)
+ return rearrange(x, "... d r -> ... (d r)")
+
+
+@autocast(enabled=False)
+def apply_rotary_emb(freqs, t, start_index=0, scale=1.0, seq_dim=-2):
+ if t.ndim == 3:
+ seq_len = t.shape[seq_dim]
+ freqs = freqs[-seq_len:].to(t)
+
+ rot_dim = freqs.shape[-1]
+ end_index = start_index + rot_dim
+
+ assert rot_dim <= t.shape[-1], (
+ f"feature dimension {t.shape[-1]} is not of sufficient size to rotate in all the positions {rot_dim}"
+ )
+
+ t_left, t, t_right = (
+ t[..., :start_index],
+ t[..., start_index:end_index],
+ t[..., end_index:],
+ )
+ t = (t * freqs.cos() * scale) + (rotate_half(t) * freqs.sin() * scale)
+ return torch.cat((t_left, t, t_right), dim=-1)
+
+
+def get_encoding(d_model, max_seq_len=4096):
+ """Return: (L, D)"""
+ t = torch.arange(max_seq_len).float()
+ freqs = 1.0 / (10000 ** (torch.arange(0, d_model, 2).float() / d_model))
+ freqs = torch.einsum("i, j -> i j", t, freqs)
+ freqs = repeat(freqs, "i j -> i (j r)", r=2)
+ return freqs
+
+
+class ROPE(nn.Module):
+ """Minimal impl of a lang-style positional encoding."""
+
+ def __init__(self, d_model, max_seq_len=4096):
+ super().__init__()
+ self.d_model = d_model
+ self.max_seq_len = max_seq_len
+
+ # Pre-cache a freqs tensor
+ encoding = get_encoding(d_model, max_seq_len)
+ self.register_buffer("encoding", encoding, False)
+
+ def rotate_queries_or_keys(self, x):
+ """
+ Args:
+ x : (B, H, L, D)
+ Returns:
+ rotated_x: (B, H, L, D)
+ """
+
+ seq_len, d_model = x.shape[-2:]
+ assert d_model == self.d_model
+
+ # encoding: (L, D)s
+ if seq_len > self.max_seq_len:
+ encoding = get_encoding(d_model, seq_len).to(x)
+ else:
+ encoding = self.encoding[:seq_len]
+
+ # encoding: (L, D)
+ # x: (B, H, L, D)
+ rotated_x = apply_rotary_emb(encoding, x, seq_dim=-2)
+
+ return rotated_x
diff --git a/genmo/network/base_arch/transformer/encoder_rope.py b/genmo/network/base_arch/transformer/encoder_rope.py
new file mode 100644
index 0000000000000000000000000000000000000000..b3c44dfb8b207b1be482d6dbf4ad1fb00f3e6257
--- /dev/null
+++ b/genmo/network/base_arch/transformer/encoder_rope.py
@@ -0,0 +1,263 @@
+import math
+
+import numpy as np
+import torch
+import torch.nn as nn
+from einops import einsum
+from timm.models.vision_transformer import Mlp
+
+from genmo.network.base_arch.embeddings.rotary_embedding import ROPE
+
+
+class PositionalEncoding(nn.Module):
+ def __init__(self, d_model, dropout=0.1, max_len=5000):
+ super(PositionalEncoding, self).__init__()
+ self.dropout = nn.Dropout(p=dropout)
+
+ pe = torch.zeros(max_len, d_model)
+ position = torch.arange(0, max_len, dtype=torch.float).unsqueeze(1)
+ div_term = torch.exp(
+ torch.arange(0, d_model, 2).float() * (-np.log(10000.0) / d_model)
+ )
+ pe[:, 0::2] = torch.sin(position * div_term)
+ pe[:, 1::2] = torch.cos(position * div_term)
+ pe = pe.unsqueeze(0).transpose(0, 1)
+
+ self.pe = nn.Parameter(pe, requires_grad=False)
+
+ def forward(self, x, motion_text_pos_enc=None):
+ pe = self.pe.transpose(0, 1)
+ if "clamp_" in motion_text_pos_enc:
+ clamp_len = int(motion_text_pos_enc.split("_")[-1])
+ pe = pe[:, :clamp_len, :]
+ pe = torch.cat([pe, pe[:, [-1]].repeat(1, x.size(1) - clamp_len, 1)], dim=1)
+ else:
+ pe = pe[:, : x.shape[1], :]
+ pe = self.dropout(pe)
+ x = x + pe
+ return x
+
+
+class RoPEAttention(nn.Module):
+ def __init__(self, embed_dim, num_heads, dropout=0.1):
+ super().__init__()
+ self.embed_dim = embed_dim
+ self.num_heads = num_heads
+ self.head_dim = embed_dim // num_heads
+
+ self.rope = ROPE(self.head_dim, max_seq_len=4096)
+
+ self.query = nn.Linear(embed_dim, embed_dim)
+ self.key = nn.Linear(embed_dim, embed_dim)
+ self.value = nn.Linear(embed_dim, embed_dim)
+ self.dropout = nn.Dropout(dropout)
+ self.proj = nn.Linear(embed_dim, embed_dim)
+
+ def forward(self, x, context=None, attn_mask=None, key_padding_mask=None):
+ # x: (B, L, C)
+ # context: (B, L_ctx, C) or None
+ # attn_mask: (L, L) or (L, L_ctx)
+ # key_padding_mask: (B, L) or (B, L_ctx)
+ B, L, _ = x.shape
+ if context is None:
+ context = x
+ L_ctx = context.shape[1]
+
+ xq = self.query(x)
+ xk = self.key(context)
+ xv = self.value(context)
+
+ xq = xq.reshape(B, L, self.num_heads, -1).transpose(1, 2)
+ xk = xk.reshape(B, L_ctx, self.num_heads, -1).transpose(1, 2)
+ xv = xv.reshape(B, L_ctx, self.num_heads, -1).transpose(1, 2)
+
+ xq = self.rope.rotate_queries_or_keys(xq) # B, N, L, C
+ xk = self.rope.rotate_queries_or_keys(xk) # B, N, L_ctx, C
+
+ attn_score = einsum(xq, xk, "b n i c, b n j c -> b n i j") / math.sqrt(
+ self.head_dim
+ )
+ if attn_mask is not None:
+ if len(attn_mask.shape) == 2:
+ attn_mask = attn_mask.reshape(1, 1, L, L_ctx).expand(
+ B, self.num_heads, -1, -1
+ )
+ else:
+ attn_mask = attn_mask.reshape(B, 1, L, L_ctx).expand(
+ B, self.num_heads, -1, -1
+ )
+ attn_score = attn_score.masked_fill(attn_mask, float("-inf"))
+ if key_padding_mask is not None:
+ key_padding_mask = key_padding_mask.reshape(B, 1, 1, L_ctx).expand(
+ -1, self.num_heads, L, -1
+ )
+ attn_score = attn_score.masked_fill(key_padding_mask, float("-inf"))
+
+ attn_score = torch.softmax(attn_score, dim=-1)
+ attn_score = self.dropout(attn_score)
+ output = einsum(attn_score, xv, "b n i j, b n j c -> b n i c") # B, N, L, C
+ output = output.transpose(1, 2).reshape(B, L, -1) # B, L, C
+ output = self.proj(output) # B, L, C
+ return output
+
+
+class EncoderRoPEBlock(nn.Module):
+ def __init__(
+ self, hidden_size, num_heads, mlp_ratio=4.0, dropout=0.1, **block_kwargs
+ ):
+ super().__init__()
+ self.norm1 = nn.LayerNorm(hidden_size, elementwise_affine=True, eps=1e-6)
+ self.attn = RoPEAttention(hidden_size, num_heads, dropout)
+ self.norm2 = nn.LayerNorm(hidden_size, elementwise_affine=True, eps=1e-6)
+ mlp_hidden_dim = int(hidden_size * mlp_ratio)
+ approx_gelu = lambda: nn.GELU(approximate="tanh")
+ self.mlp = Mlp(
+ in_features=hidden_size,
+ hidden_features=mlp_hidden_dim,
+ act_layer=approx_gelu,
+ drop=dropout,
+ )
+
+ self.gate_msa = nn.Parameter(torch.zeros(1, 1, hidden_size))
+ self.gate_mlp = nn.Parameter(torch.zeros(1, 1, hidden_size))
+
+ # Zero-out adaLN modulation layers
+ nn.init.constant_(self.gate_msa, 0)
+ nn.init.constant_(self.gate_mlp, 0)
+
+ def forward(self, x, attn_mask=None, tgt_key_padding_mask=None):
+ x = x + self.gate_msa * self._sa_block(
+ self.norm1(x), attn_mask=attn_mask, key_padding_mask=tgt_key_padding_mask
+ )
+ x = x + self.gate_mlp * self.mlp(self.norm2(x))
+ return x
+
+ def _sa_block(self, x, attn_mask=None, key_padding_mask=None):
+ # x: (B, L, C)
+ x = self.attn(x, attn_mask=attn_mask, key_padding_mask=key_padding_mask)
+ return x
+
+
+class DecoderRoPEBlock(nn.Module):
+ def __init__(
+ self,
+ hidden_size,
+ num_heads,
+ mlp_ratio=4.0,
+ dropout=0.1,
+ use_self_attn=True,
+ cross_attn_type="rope",
+ pos_enc_dropout=0.0,
+ **block_kwargs,
+ ):
+ super().__init__()
+ self.use_self_attn = use_self_attn
+ if self.use_self_attn:
+ self.norm1 = nn.LayerNorm(hidden_size, elementwise_affine=True, eps=1e-6)
+ self.self_attn = RoPEAttention(hidden_size, num_heads, dropout)
+ self.gate_msa = nn.Parameter(torch.zeros(1, 1, hidden_size))
+ nn.init.constant_(self.gate_msa, 0)
+ self.norm2 = nn.LayerNorm(hidden_size, elementwise_affine=True, eps=1e-6)
+ self.cross_attn_type = cross_attn_type
+ if cross_attn_type == "rope":
+ self.cross_attn = RoPEAttention(hidden_size, num_heads, dropout)
+ elif cross_attn_type == "mha":
+ self.cross_attn = nn.MultiheadAttention(
+ hidden_size, num_heads, dropout=dropout, batch_first=True
+ )
+ self.norm3 = nn.LayerNorm(hidden_size, elementwise_affine=True, eps=1e-6)
+ mlp_hidden_dim = int(hidden_size * mlp_ratio)
+ approx_gelu = lambda: nn.GELU(approximate="tanh")
+ self.mlp = Mlp(
+ in_features=hidden_size,
+ hidden_features=mlp_hidden_dim,
+ act_layer=approx_gelu,
+ drop=dropout,
+ )
+
+ self.gate_cross_attn = nn.Parameter(torch.zeros(1, 1, hidden_size))
+ self.gate_mlp = nn.Parameter(torch.zeros(1, 1, hidden_size))
+
+ self.motion_pos_encoder = PositionalEncoding(hidden_size, pos_enc_dropout)
+
+ # Zero-out adaLN modulation layers
+ nn.init.constant_(self.gate_cross_attn, 0)
+ nn.init.constant_(self.gate_mlp, 0)
+
+ def forward(
+ self,
+ x,
+ context,
+ attn_mask=None,
+ tgt_key_padding_mask=None,
+ memory_key_padding_mask=None,
+ multi_text_data=None,
+ motion_text_pos_enc=None,
+ ):
+ if self.use_self_attn:
+ x = x + self.gate_msa * self._sa_block(
+ self.norm1(x),
+ attn_mask=attn_mask,
+ key_padding_mask=tgt_key_padding_mask,
+ )
+ x = x + self.gate_cross_attn * self._ca_block(
+ self.norm2(x),
+ context=context,
+ key_padding_mask=memory_key_padding_mask,
+ multi_text_data=multi_text_data,
+ motion_text_pos_enc=motion_text_pos_enc,
+ )
+ x = x + self.gate_mlp * self.mlp(self.norm3(x))
+ return x
+
+ def _sa_block(self, x, attn_mask=None, key_padding_mask=None):
+ # x: (B, L, C)
+ x = self.self_attn(x, attn_mask=attn_mask, key_padding_mask=key_padding_mask)
+ return x
+
+ def _ca_block(
+ self,
+ x,
+ context,
+ key_padding_mask=None,
+ multi_text_data=None,
+ motion_text_pos_enc=None,
+ ):
+ # x: (B, L, C)
+ if self.cross_attn_type == "rope":
+ if motion_text_pos_enc is not None:
+ x = self.motion_pos_encoder(x, motion_text_pos_enc)
+ x = self.cross_attn(x, context=context, key_padding_mask=key_padding_mask)
+ elif self.cross_attn_type == "mha":
+ if multi_text_data is not None:
+ # TODO: implement positional encoding for MHA
+ out = []
+ window_start = (
+ (multi_text_data["window_start"] * x.size(1)).round().long()
+ )
+ window_end = (multi_text_data["window_end"] * x.size(1)).round().long()
+ for i in range(len(multi_text_data["text_embed_feats"])):
+ text_embed_i = multi_text_data["text_embed_feats"][i].unsqueeze(0)
+ attn_mask = (
+ torch.ones(x.size(1), text_embed_i.size(1)).to(x.device).bool()
+ )
+ attn_mask[window_start[i] : window_end[i], :] = 0
+ out_i = self.cross_attn(
+ x,
+ text_embed_i,
+ text_embed_i,
+ attn_mask=attn_mask,
+ key_padding_mask=key_padding_mask,
+ )[0]
+ out_i[out_i.isnan()] = 0
+ out.append(out_i)
+ x = torch.sum(torch.stack(out), dim=0)
+ else:
+ if motion_text_pos_enc is not None:
+ x = self.motion_pos_encoder(x, motion_text_pos_enc)
+ x = self.cross_attn(
+ x, context, context, key_padding_mask=key_padding_mask
+ )[0]
+ else:
+ raise ValueError(f"Invalid cross_attn_type: {self.cross_attn_type}")
+ return x
diff --git a/genmo/network/base_arch/transformer/layer.py b/genmo/network/base_arch/transformer/layer.py
new file mode 100644
index 0000000000000000000000000000000000000000..6db33fce344c058d289d6e6ae61eb1b065f772d1
--- /dev/null
+++ b/genmo/network/base_arch/transformer/layer.py
@@ -0,0 +1,72 @@
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+
+
+def zero_module(module):
+ """
+ Zero out the parameters of a module and return it.
+ """
+ for p in module.parameters():
+ p.detach().zero_()
+ return module
+
+
+class BasicBlock(nn.Module):
+ """Basic 1D residual block."""
+
+ def __init__(
+ self,
+ inplanes,
+ planes,
+ ker_size,
+ stride=1,
+ dropout=0.1,
+ norm_layer=nn.BatchNorm1d,
+ act_layer=nn.LeakyReLU,
+ ):
+ super().__init__()
+ self.conv1 = nn.Conv1d(
+ inplanes,
+ planes,
+ kernel_size=ker_size,
+ stride=stride,
+ padding=ker_size // 2,
+ dilation=1,
+ bias=True,
+ )
+ self.bn1 = norm_layer(planes)
+ self.act1 = act_layer(inplace=True)
+ self.conv2 = nn.Conv1d(
+ planes, planes, kernel_size=ker_size, padding=ker_size // 2, bias=True
+ )
+ self.bn2 = norm_layer(planes)
+ self.act2 = act_layer(inplace=True)
+ self.downsample = None
+ if stride != 1 or inplanes != planes:
+ self.downsample = nn.Sequential(
+ nn.Conv1d(
+ inplanes,
+ planes,
+ stride=stride,
+ kernel_size=ker_size,
+ padding=ker_size // 2,
+ bias=True,
+ ),
+ norm_layer(planes),
+ )
+ self.dropout = nn.Dropout(dropout)
+
+ def forward(self, x):
+ shortcut = x
+ x = self.conv1(x)
+ x = self.bn1(x)
+ x = self.act1(x)
+ x = self.dropout(x)
+ x = self.conv2(x)
+ x = self.bn2(x)
+ if self.downsample is not None:
+ shortcut = self.downsample(shortcut)
+ x += shortcut
+ x = self.act2(x)
+ return x
diff --git a/genmo/network/endecoder.py b/genmo/network/endecoder.py
new file mode 100644
index 0000000000000000000000000000000000000000..8a1836ce187221582b5f45d48c2588dfa7529c5a
--- /dev/null
+++ b/genmo/network/endecoder.py
@@ -0,0 +1,465 @@
+# This code is based on git@github.com:zju3dv/GVHMR.git
+
+import numpy as np
+import torch
+import torch.nn as nn
+
+import genmo.utils.matrix as matrix
+from genmo.utils.rotation_conversions import (
+ axis_angle_to_matrix,
+ matrix_to_axis_angle,
+ matrix_to_rotation_6d,
+ rotation_6d_to_matrix,
+)
+from genmo.utils.torch_transform import (
+ angle_axis_to_quaternion,
+ get_y_heading_q,
+ quat_apply,
+ quat_conjugate,
+ quat_mul,
+ quaternion_to_angle_axis,
+)
+from third_party.GVHMR.hmr4d.utils.geo.augment_noisy_pose import gaussian_augment
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import (
+ get_local_transl_vel,
+ get_static_joint_mask,
+ rollout_local_transl_vel,
+)
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+from . import stats_compose
+
+
+class EnDecoder(nn.Module):
+ def __init__(
+ self,
+ stats_name="DEFAULT_01",
+ encode_type="gvhmr",
+ feature_arr=None,
+ stats_arr=None,
+ noise_pose_k=10,
+ clip_std=False,
+ ):
+ super().__init__()
+
+ if encode_type in ["gvhmr", "humanml3d"]:
+ feature_arr = [encode_type]
+ stats_arr = [stats_name]
+
+ # Define feature dimensions as a class attribute
+ self.FEATURE_DIMS = {
+ "gvhmr": 151,
+ "humanml3d": 143,
+ }
+
+ # Store stats for each feature type
+ self.stats_dict = {}
+
+ for feature, stats_name in zip(feature_arr, stats_arr):
+ stats = getattr(stats_compose, stats_name)
+ mean = torch.tensor(stats["mean"]).float()
+ std = torch.tensor(stats["std"]).float()
+
+ feature_dim = self.FEATURE_DIMS[feature]
+ if stats_name != "DEFAULT_01":
+ assert mean.shape[-1] == feature_dim
+ assert std.shape[-1] == feature_dim
+
+ if clip_std:
+ std = torch.clamp(std, 0.1, 1)
+
+ self.stats_dict[feature] = {"mean": mean, "std": std}
+
+ # Store feature configuration
+ self.feature_arr = feature_arr
+ self.stats_arr = stats_arr
+ self.clip_std = clip_std
+
+ # option
+ self.noise_pose_k = noise_pose_k
+ self.encode_type = encode_type
+ self.obs_indices_dict = None
+
+ # smpl
+ self.smplx_model = make_smplx("supermotion_v437coco17")
+ parents = self.smplx_model.parents[:22]
+ self.register_buffer("parents_tensor", parents, False)
+ self.parents = parents.tolist()
+
+ def normalize(self, x, feature_type):
+ """Normalize input using stats for specific feature type"""
+ stats = self.stats_dict[feature_type]
+ return (x - stats["mean"].to(x)) / stats["std"].to(x)
+
+ def denormalize(self, x_norm, feature_type):
+ """Denormalize input using stats for specific feature type"""
+ stats = self.stats_dict[feature_type]
+ return x_norm * stats["std"].to(x_norm) + stats["mean"].to(x_norm)
+
+ def get_noisyobs(self, data, return_type="r6d"):
+ """
+ Noisy observation contains local pose with noise
+ Args:
+ data (dict):
+ body_pose: (B, L, J*3) or (B, L, J, 3)
+ Returns:
+ noisy_bosy_pose: (B, L, J, 6) or (B, L, J, 3) or (B, L, 3, 3) depends on return_type
+ """
+ body_pose = data["body_pose"] # (B, L, 63)
+ B, L, _ = body_pose.shape
+ body_pose = body_pose.reshape(B, L, -1, 3)
+
+ # (B, L, J, C)
+ return_mapping = {"R": 0, "r6d": 1, "aa": 2}
+ return_id = return_mapping[return_type]
+ noisy_bosy_pose = gaussian_augment(body_pose, self.noise_pose_k, to_R=True)[
+ return_id
+ ]
+ return noisy_bosy_pose
+
+ def normalize_body_pose_r6d(self, body_pose_r6d):
+ """body_pose_r6d: (B, L, {J*6}/{J, 6}) -> (B, L, J*6)"""
+ B, L = body_pose_r6d.shape[:2]
+ body_pose_r6d = body_pose_r6d.reshape(B, L, -1)
+ if (
+ self.stats_dict[self.encode_type]["mean"].shape[-1] == 1
+ ): # no mean, std provided
+ return body_pose_r6d
+ body_pose_r6d = (
+ body_pose_r6d - self.stats_dict["gvhmr"]["mean"]
+ ) / self.stats_dict["gvhmr"]["std"] # (B, L, C)
+ return body_pose_r6d
+
+ def fk_v2(
+ self, body_pose, betas, global_orient=None, transl=None, get_intermediate=False
+ ):
+ """
+ Args:
+ body_pose: (B, L, 63)
+ betas: (B, L, 10)
+ global_orient: (B, L, 3)
+ Returns:
+ joints: (B, L, 22, 3)
+ """
+ B, L = body_pose.shape[:2]
+ if global_orient is None:
+ global_orient = torch.zeros((B, L, 3), device=body_pose.device)
+ aa = torch.cat([global_orient, body_pose], dim=-1).reshape(B, L, -1, 3)
+ rotmat = axis_angle_to_matrix(aa) # (B, L, 22, 3, 3)
+
+ skeleton = self.smplx_model.get_skeleton(betas)[..., :22, :] # (B, L, 22, 3)
+ local_skeleton = skeleton - skeleton[:, :, self.parents_tensor]
+ local_skeleton = torch.cat(
+ [skeleton[:, :, :1], local_skeleton[:, :, 1:]], dim=2
+ )
+
+ if transl is not None:
+ local_skeleton[..., 0, :] += transl # B, L, 22, 3
+
+ mat = matrix.get_TRS(rotmat, local_skeleton) # B, L, 22, 4, 4
+ fk_mat = matrix.forward_kinematics(mat, self.parents) # B, L, 22, 4, 4
+ joints = matrix.get_position(fk_mat) # B, L, 22, 3
+ if not get_intermediate:
+ return joints
+ else:
+ return joints, mat, fk_mat
+
+ def get_local_pos(self, betas):
+ skeleton = self.smplx_model.get_skeleton(betas)[..., :22, :] # (B, L, 22, 3)
+ local_skeleton = skeleton - skeleton[:, :, self.parents_tensor]
+ local_skeleton = torch.cat(
+ [skeleton[:, :, :1], local_skeleton[:, :, 1:]], dim=2
+ )
+ return local_skeleton
+
+ def get_static_gt(self, inputs, vel_thr):
+ joint_ids = [
+ 7,
+ 10,
+ 8,
+ 11,
+ 20,
+ 21,
+ ] # [L_Ankle, L_foot, R_Ankle, R_foot, L_wrist, R_wrist]
+ gt_w_j3d = self.fk_v2(**inputs["smpl_params_w"]) # (B, L, J=22, 3)
+ static_gt = get_static_joint_mask(
+ gt_w_j3d, vel_thr=vel_thr, repeat_last=True
+ ) # (B, L, J)
+ static_gt = static_gt[:, :, joint_ids].float() # (B, L, J')
+ return static_gt
+
+ def encode(self, inputs):
+ """Composite encoder that combines multiple feature types"""
+ encoded_features = []
+
+ for feature in self.feature_arr:
+ if feature == "gvhmr":
+ encoded = self.encode_gvhmr(inputs)
+ elif feature == "humanml3d":
+ encoded = self.encode_humanml3d(inputs)
+ encoded_features.append(encoded)
+
+ # Concatenate all encoded features
+ return torch.cat(encoded_features, dim=-1)
+
+ def encode_humanml3d(self, inputs):
+ """
+ definition: {
+ body_pose_r6d, # (B, L, (J-1)*6) -> 0:126
+ betas, # (B, L, 10) -> 126:136
+ root_data, # (B, L, 10) -> 136:143
+ }
+ """
+ self.obs_indices_dict = {
+ "body_pose": torch.arange(126),
+ "betas": torch.arange(126, 136),
+ "root_data": torch.arange(136, 143),
+ }
+ B, L = inputs["smpl_params_w"]["body_pose"].shape[:2]
+ # cam
+ smpl_params_w = inputs["smpl_params_w"]
+ body_pose = smpl_params_w["body_pose"].reshape(B, L, 21, 3)
+ body_pose_r6d = matrix_to_rotation_6d(axis_angle_to_matrix(body_pose)).flatten(
+ -2
+ )
+ betas = smpl_params_w["betas"]
+ global_orient = smpl_params_w["global_orient"]
+ trans = smpl_params_w["transl"].clone()
+
+ root_quat = angle_axis_to_quaternion(global_orient)
+ heading_quat = get_y_heading_q(root_quat)
+ heading_quat_inv = quat_conjugate(heading_quat)
+ root_quat_wo_heading = quat_mul(heading_quat_inv, root_quat)
+ # root_quat_wo_heading = quaternion_to_cont6d(root_quat_wo_heading)
+ root_quat_wo_heading = quaternion_to_angle_axis(root_quat_wo_heading)
+
+ init_heading_quat_inv = heading_quat_inv[:, [0]].repeat(1, L, 1)
+
+ """XZ at origin"""
+ root_y = trans[..., [1]]
+ root_pos_init = trans[:, [0]]
+ root_pose_init_xz = root_pos_init * torch.tensor([1, 0, 1]).to(root_pos_init)
+ trans = trans - root_pose_init_xz
+
+ """All initially face Z+"""
+ trans = quat_apply(init_heading_quat_inv, trans)
+ heading_quat = quat_mul(
+ heading_quat, init_heading_quat_inv
+ ) # normalize heading coordiante, so the heading is 0 for the first frame
+ heading_quat_inv = quat_conjugate(heading_quat)
+
+ """Root Linear Velocity"""
+ # (seq_len - 1, 3)
+ velocity = trans[:, 1:] - trans[:, :-1]
+ velocity = torch.cat([velocity, velocity[:, [-1]]], axis=1)
+ # print(r_rot.shape, velocity.shape)
+ velocity = quat_apply(heading_quat_inv, velocity)
+ l_velocity = velocity[..., [0, 2]]
+ """Root Angular Velocity"""
+ # (seq_len - 1, 4)
+ r_angles = torch.arctan2(heading_quat[..., 2:3], heading_quat[..., :1]) * 2
+ r_velocity = r_angles[:, 1:] - r_angles[:, :-1]
+ r_velocity[r_velocity > np.pi] -= 2 * np.pi
+ r_velocity[r_velocity < -np.pi] += 2 * np.pi
+ r_velocity = torch.cat([r_velocity, r_velocity[:, [-1]]], axis=1)
+
+ root_data = torch.cat(
+ [r_velocity, l_velocity, root_y, root_quat_wo_heading], axis=-1
+ )
+ # 126 + 10 + 7 = 143d
+ x = torch.cat([body_pose_r6d, betas, root_data], dim=-1)
+ return self.normalize(x, "humanml3d")
+
+ def encode_gvhmr(self, inputs):
+ """
+ definition: {
+ body_pose_r6d, # (B, L, (J-1)*6) -> 0:126
+ betas, # (B, L, 10) -> 126:136
+ global_orient_r6d, # (B, L, 6) -> 136:142 incam
+ global_orient_gv_r6d: # (B, L, 6) -> 142:148 gv
+ local_transl_vel, # (B, L, 3) -> 148:151, smpl-coord
+ }
+ """
+ self.obs_indices_dict = {
+ "body_pose": torch.arange(126),
+ "betas": torch.arange(126, 136),
+ "global_orient": torch.arange(136, 142),
+ "global_orient_gv": torch.arange(142, 148),
+ "local_transl_vel": torch.arange(148, 151),
+ }
+
+ B, L = inputs["smpl_params_c"]["body_pose"].shape[:2]
+ # cam
+ smpl_params_c = inputs["smpl_params_c"]
+ body_pose = smpl_params_c["body_pose"].reshape(B, L, 21, 3)
+ body_pose_r6d = matrix_to_rotation_6d(axis_angle_to_matrix(body_pose)).flatten(
+ -2
+ )
+ betas = smpl_params_c["betas"]
+ global_orient_R = axis_angle_to_matrix(smpl_params_c["global_orient"])
+ global_orient_r6d = matrix_to_rotation_6d(global_orient_R)
+
+ # global
+ R_c2gv = inputs["R_c2gv"] # (B, L, 3, 3)
+ global_orient_gv_r6d = matrix_to_rotation_6d(R_c2gv @ global_orient_R)
+
+ # local_transl_vel
+ smpl_params_w = inputs["smpl_params_w"]
+ local_transl_vel = get_local_transl_vel(
+ smpl_params_w["transl"], smpl_params_w["global_orient"]
+ )
+ if False: # debug
+ transl_recover = rollout_local_transl_vel(
+ local_transl_vel,
+ smpl_params_w["global_orient"],
+ smpl_params_w["transl"][:, [0]],
+ )
+ print((transl_recover - smpl_params_w["transl"]).abs().max())
+
+ # returns
+ x = torch.cat(
+ [
+ body_pose_r6d,
+ betas,
+ global_orient_r6d,
+ global_orient_gv_r6d,
+ local_transl_vel,
+ ],
+ dim=-1,
+ )
+ return self.normalize(x, "gvhmr")
+
+ def encode_translw(self, inputs):
+ """
+ definition: {
+ body_pose_r6d, # (B, L, (J-1)*6) -> 0:126
+ betas, # (B, L, 10) -> 126:136
+ global_orient_r6d, # (B, L, 6) -> 136:142 incam
+ global_orient_gv_r6d: # (B, L, 6) -> 142:148 gv
+ local_transl_vel, # (B, L, 3) -> 148:151, smpl-coord
+ }
+ """
+ # local_transl_vel
+ smpl_params_w = inputs["smpl_params_w"]
+ local_transl_vel = get_local_transl_vel(
+ smpl_params_w["transl"], smpl_params_w["global_orient"]
+ )
+
+ # returns
+ x = local_transl_vel
+ mean = self.stats_dict["gvhmr"]["mean"][-3:].to(x)
+ std = self.stats_dict["gvhmr"]["std"][-3:].to(x)
+ x_norm = (x - mean) / std
+ return x_norm
+
+ def decode_translw(self, x_norm):
+ std = self.stats_dict["gvhmr"]["std"][-3:].to(x_norm)
+ mean = self.stats_dict["gvhmr"]["mean"][-3:].to(x_norm)
+ return x_norm * std + mean
+
+ def decode(self, x_norm):
+ """Composite decoder that handles multiple feature types"""
+ current_idx = 0
+ decoded_outputs = {}
+
+ for feature in self.feature_arr:
+ feature_size = self.FEATURE_DIMS[feature]
+ feature_norm = x_norm[..., current_idx : current_idx + feature_size]
+
+ if feature == "gvhmr":
+ decoded = self.decode_gvhmr(feature_norm)
+ elif feature == "humanml3d":
+ decoded = self.decode_humanml3d(feature_norm)
+
+ decoded_outputs.update(decoded)
+ current_idx += feature_size
+
+ return decoded_outputs
+
+ def decode_humanml3d(self, x_norm):
+ """x_norm: (B, L, C)"""
+ B, L, C = x_norm.shape
+ x = self.denormalize(x_norm, "humanml3d")
+
+ body_pose_r6d = x[:, :, :126]
+ betas = x[:, :, 126:136]
+ root_data = x[:, :, 136:143]
+
+ body_pose = matrix_to_axis_angle(
+ rotation_6d_to_matrix(body_pose_r6d.reshape(B, L, -1, 6))
+ )
+ body_pose = body_pose.flatten(-2)
+ offset = self.smplx_model.get_skeleton(betas)[:, :, 0]
+
+ rot_vel = root_data[..., 0]
+ r_rot_ang = torch.zeros_like(rot_vel).to(root_data.device)
+ """Get Y-axis rotation from rotation velocity"""
+ r_rot_ang[..., 1:] = rot_vel[..., :-1]
+ r_rot_ang = torch.cumsum(r_rot_ang, dim=-1)
+ r_rot_quat = torch.zeros(root_data.shape[:-1] + (4,)).to(root_data)
+ r_rot_quat[..., 0] = torch.cos(r_rot_ang / 2)
+ r_rot_quat[..., 2] = torch.sin(r_rot_ang / 2)
+
+ r_pos = torch.zeros(root_data.shape[:-1] + (3,)).to(root_data)
+ r_pos[..., 1:, [0, 2]] = root_data[..., :-1, 1:3]
+ """Add Y-axis rotation to root position"""
+ r_pos = quat_apply(r_rot_quat, r_pos)
+ r_pos = torch.cumsum(r_pos, dim=-2)
+ r_pos[..., 1] = root_data[..., 3]
+ # return r_rot_quat, r_pos, r_rot_ang
+ # root_rot_wo_heading = rotation_6d_to_matrix(root_data[..., 4:])
+ # root_rot_wo_heading = cont6d_to_matrix(root_data[..., 4:])
+ # root_rot = quaternion_to_matrix(r_rot_quat) @ root_rot_wo_heading
+ # global_orient_w = matrix_to_axis_angle(root_rot)
+ global_orient_w = quaternion_to_angle_axis(
+ quat_mul(r_rot_quat, angle_axis_to_quaternion(root_data[..., 4:]))
+ )
+
+ output = {
+ "body_pose": body_pose,
+ "betas": betas,
+ "global_orient_w": global_orient_w,
+ "transl_w": r_pos,
+ "offset": offset,
+ }
+
+ return output
+
+ def decode_gvhmr(self, x_norm):
+ """x_norm: (B, L, C)"""
+ B, L, C = x_norm.shape
+ x = self.denormalize(x_norm, "gvhmr")
+
+ body_pose_r6d = x[:, :, :126]
+ betas = x[:, :, 126:136]
+ global_orient_r6d = x[:, :, 136:142]
+ global_orient_gv_r6d = x[:, :, 142:148]
+ local_transl_vel = x[:, :, 148:151]
+
+ body_pose = matrix_to_axis_angle(
+ rotation_6d_to_matrix(body_pose_r6d.reshape(B, L, -1, 6))
+ )
+ body_pose = body_pose.flatten(-2)
+ global_orient_c = matrix_to_axis_angle(rotation_6d_to_matrix(global_orient_r6d))
+ global_orient_gv = matrix_to_axis_angle(
+ rotation_6d_to_matrix(global_orient_gv_r6d)
+ )
+
+ offset = self.smplx_model.get_skeleton(betas)[:, :, 0]
+ output = {
+ "body_pose": body_pose,
+ "betas": betas,
+ "global_orient": global_orient_c,
+ "global_orient_gv": global_orient_gv,
+ "local_transl_vel": local_transl_vel,
+ "offset": offset,
+ }
+
+ return output
+
+ def get_motion_dim(self):
+ """Calculate total dimension based on enabled features"""
+ return sum(self.FEATURE_DIMS[feature] for feature in self.feature_arr)
+
+ def get_obs_indices(self, obs):
+ return self.obs_indices_dict[obs]
diff --git a/genmo/network/genmo_cfg_sampler.py b/genmo/network/genmo_cfg_sampler.py
new file mode 100644
index 0000000000000000000000000000000000000000..55c25c11aaaf2b5309080af473748d8bfd4de0c6
--- /dev/null
+++ b/genmo/network/genmo_cfg_sampler.py
@@ -0,0 +1,32 @@
+from copy import deepcopy
+
+import torch
+
+
+# A wrapper model for Classifier-free guidance **SAMPLING** only
+# https://arxiv.org/abs/2207.12598
+class ClassifierFreeSampleModel:
+ def __init__(self, model):
+ self.model = model # model is the actual model to run
+
+ def __call__(self, x, timesteps, y=None, **kwargs):
+ y_uncond = deepcopy(y)
+ y_uncond["encoded_text"] = torch.zeros_like(y["encoded_text"])
+ y_uncond["f_cond"] = y["f_uncond"]
+ if "multi_text_data" in y:
+ y_uncond["multi_text_data"]["text_embed"] = torch.zeros_like(
+ y["multi_text_data"]["text_embed"]
+ )
+
+ out = self.model(x, timesteps, y, **kwargs)
+ out_uncond = self.model(x, timesteps, y_uncond, **kwargs)
+ outputs = dict()
+ for k in out:
+ outputs[k] = out_uncond[k] + y["scale"] * (out[k] - out_uncond[k])
+ return outputs
+
+ def parameters(self):
+ return self.model.parameters()
+
+ def named_parameters(self):
+ return self.model.named_parameters()
diff --git a/genmo/network/genmo_denoiser.py b/genmo/network/genmo_denoiser.py
new file mode 100644
index 0000000000000000000000000000000000000000..dc89c669b4630c505d984caa1e6403311992d35b
--- /dev/null
+++ b/genmo/network/genmo_denoiser.py
@@ -0,0 +1,326 @@
+import torch
+import torch.nn as nn
+from einops import repeat
+from timm.models.vision_transformer import Mlp
+
+from genmo.network.base_arch.embeddings.pe import PositionalEncoding
+from genmo.network.base_arch.transformer.encoder_rope import (
+ DecoderRoPEBlock,
+ EncoderRoPEBlock,
+)
+from genmo.network.base_arch.transformer.layer import zero_module
+from genmo.utils.net_utils import length_to_mask
+
+
+class TimestepEmbedder(nn.Module):
+ def __init__(self, latent_dim, sequence_pos_encoder):
+ super().__init__()
+ self.latent_dim = latent_dim
+ self.sequence_pos_encoder = sequence_pos_encoder
+
+ time_embed_dim = self.latent_dim
+ self.time_embed = nn.Sequential(
+ nn.Linear(self.latent_dim, time_embed_dim),
+ nn.SiLU(),
+ nn.Linear(time_embed_dim, time_embed_dim),
+ )
+
+ def forward(self, timesteps):
+ return self.time_embed(self.sequence_pos_encoder.pe[timesteps])
+
+
+class NetworkEncoderRoPE(nn.Module):
+ def __init__(
+ self,
+ # x
+ output_dim=151,
+ xt_dim=157,
+ max_len=120,
+ # condition
+ cliffcam_dim=3,
+ cam_angvel_dim=6,
+ imgseq_dim=1024,
+ # intermediate
+ latent_dim=512,
+ num_layers=12,
+ num_heads=8,
+ mlp_ratio=4.0,
+ # output
+ pred_cam_dim=3,
+ static_conf_dim=6,
+ # training
+ dropout=0.1,
+ # other
+ avgbeta=True,
+ njoints=None,
+ encoded_text_dim=1024,
+ use_text_pos_enc=True,
+ text_encoder_cfg={},
+ motion_text_pos_enc=None,
+ text_mask_prob=0.0,
+ input_remove_global=False,
+ input_remove_condition=False,
+ allow_autoregressive=True,
+ **kwargs,
+ ):
+ super().__init__()
+
+ # input
+ self.output_dim = output_dim
+ self.max_len = max_len
+
+ # condition
+ self.cliffcam_dim = cliffcam_dim
+ self.cam_angvel_dim = cam_angvel_dim
+ self.imgseq_dim = imgseq_dim
+
+ # intermediate
+ self.latent_dim = latent_dim
+ self.num_layers = num_layers
+ self.num_heads = num_heads
+ self.dropout = dropout
+ self.njoints = njoints
+ self.nfeats = 1
+ self.encoded_text_dim = encoded_text_dim
+ self.text_mask_prob = text_mask_prob
+ self.use_text_pos_enc = use_text_pos_enc
+ self.input_remove_global = input_remove_global
+ self.input_remove_condition = input_remove_condition
+ self.allow_autoregressive = allow_autoregressive
+
+ # ===== build model ===== #
+ # Input (Kp2d)
+ # Main token: map d_obs 2 to 32
+ self.learned_pos_linear = nn.Linear(2, 32)
+ self.learned_pos_params = nn.Parameter(torch.randn(17, 32), requires_grad=True)
+ self.embed_noisyobs = Mlp(
+ 17 * 32,
+ hidden_features=self.latent_dim * 2,
+ out_features=self.latent_dim,
+ drop=dropout,
+ )
+
+ self._build_condition_embedder()
+
+ # Transformer
+ self.blocks = nn.ModuleList(
+ [
+ EncoderRoPEBlock(
+ self.latent_dim,
+ self.num_heads,
+ mlp_ratio=mlp_ratio,
+ dropout=dropout,
+ )
+ for _ in range(self.num_layers)
+ ]
+ )
+ self.sequence_pos_encoder = PositionalEncoding(self.latent_dim, dropout=0)
+ self.embed_timestep = TimestepEmbedder(
+ self.latent_dim, self.sequence_pos_encoder
+ )
+ self.embed_text = nn.Linear(self.encoded_text_dim, self.latent_dim)
+ self.text_encoder_cfg = text_encoder_cfg
+ text_encode_mode = text_encoder_cfg.get("mode", "first")
+ if text_encode_mode == "first":
+ self.text_encode_layer_idx = [0]
+ elif text_encode_mode == "all":
+ self.text_encode_layer_idx = list(range(num_layers))
+ elif text_encode_mode.startswith("every_"):
+ self.text_encode_layer_idx = list(
+ range(0, num_layers, int(text_encode_mode.split("_")[1]))
+ )
+ elif text_encode_mode == "none":
+ self.text_encode_layer_idx = []
+ else:
+ raise ValueError(f"Invalid text_encode_mode {text_encode_mode}")
+ use_self_attn = text_encoder_cfg.get("use_self_attn", False)
+ net_type = text_encoder_cfg.get("net_type", "rope_decoder")
+ cross_attn_type = text_encoder_cfg.get("cross_attn_type", "rope")
+ pos_enc_dropout = text_encoder_cfg.get("pos_enc_dropout", 0.0)
+ self.text_encoder_layers = nn.ModuleDict()
+ for idx in self.text_encode_layer_idx:
+ if net_type == "rope_decoder":
+ text_block = DecoderRoPEBlock(
+ self.latent_dim,
+ self.num_heads,
+ use_self_attn=use_self_attn,
+ mlp_ratio=mlp_ratio,
+ dropout=dropout,
+ cross_attn_type=cross_attn_type,
+ pos_enc_dropout=pos_enc_dropout,
+ )
+ else:
+ raise ValueError(f"Invalid net_type {net_type}")
+ self.text_encoder_layers[f"{idx}"] = text_block
+
+ self.motion_text_pos_enc = motion_text_pos_enc
+
+ # Output heads
+ self.final_layer = Mlp(self.latent_dim, out_features=self.output_dim)
+ self.pred_cam_head = (
+ pred_cam_dim > 0
+ ) # keep extra_output for easy-loading old ckpt
+ if self.pred_cam_head:
+ self.pred_cam_head = Mlp(self.latent_dim, out_features=pred_cam_dim)
+ self.register_buffer(
+ "pred_cam_mean", torch.tensor([1.0606, -0.0027, 0.2702]), False
+ )
+ self.register_buffer(
+ "pred_cam_std", torch.tensor([0.1784, 0.0956, 0.0764]), False
+ )
+
+ self.static_conf_head = static_conf_dim > 0
+ if self.static_conf_head:
+ self.static_conf_head = Mlp(self.latent_dim, out_features=static_conf_dim)
+
+ self.add_cond_linear = nn.Linear(xt_dim + self.latent_dim, self.latent_dim)
+
+ self.avgbeta = avgbeta
+
+ def _build_condition_embedder(self):
+ latent_dim = self.latent_dim
+ dropout = self.dropout
+ self.cliffcam_embedder = nn.Sequential(
+ nn.Linear(self.cliffcam_dim, latent_dim),
+ nn.SiLU(),
+ nn.Dropout(dropout),
+ zero_module(nn.Linear(latent_dim, latent_dim)),
+ )
+ if self.cam_angvel_dim > 0:
+ self.cam_angvel_embedder = nn.Sequential(
+ nn.Linear(self.cam_angvel_dim, latent_dim),
+ nn.SiLU(),
+ nn.Dropout(dropout),
+ zero_module(nn.Linear(latent_dim, latent_dim)),
+ )
+ if self.imgseq_dim > 0:
+ self.imgseq_embedder = nn.Sequential(
+ nn.LayerNorm(self.imgseq_dim),
+ zero_module(nn.Linear(self.imgseq_dim, latent_dim)),
+ )
+
+ def forward(
+ self,
+ xt,
+ timesteps,
+ y=None,
+ inputs=None,
+ observed_motion_3d=None,
+ motion_mask_3d=None,
+ rm_text_flag=None,
+ **kwargs,
+ ):
+ """
+ Args:
+ x: None we do not use it
+ timesteps: (B,)
+ length: (B), valid length of x, if None then use x.shape[2]
+ f_imgseq: (B, L, C)
+ f_cliffcam: (B, L, 3), CLIFF-Cam parameters (bbx-detection in the full-image)
+ f_noisyobs: (B, L, C), nosiy pose observation
+ f_cam_angvel: (B, L, 6), Camera angular velocity
+ """
+ x = y["f_cond"]
+ length = y["length"]
+ multi_text_data = y.get("multi_text_data", None)
+ L = xt.size(1)
+ B = xt.size(0)
+
+ if self.input_remove_condition:
+ x = torch.zeros_like(x)
+
+ if self.input_remove_global:
+ xt[..., -15:] = 0
+
+ if motion_mask_3d is not None:
+ xt = xt * (1 - motion_mask_3d) + observed_motion_3d * motion_mask_3d
+
+ emb = self.embed_timestep(timesteps) # [1, bs, d]
+ x = x + emb
+
+ x = self.add_cond_linear(torch.cat([x, xt], dim=-1))
+
+ if "encoded_text" in y and len(self.text_encode_layer_idx) > 0:
+ enc_text = y["encoded_text"].clone()
+ if self.training and self.text_mask_prob > 0:
+ mask = torch.rand((B,), device=x.device) < self.text_mask_prob
+ enc_text = enc_text * (1 - mask[:, None, None].float())
+ if rm_text_flag is not None:
+ enc_text = enc_text * (1 - rm_text_flag[:, None, None].float())
+ emb_text = self.embed_text(enc_text)
+ if self.use_text_pos_enc:
+ emb_text = self.sequence_pos_encoder(emb_text, batch_first=True)
+
+ if multi_text_data is not None:
+ multi_text_data["text_embed_feats"] = self.embed_text(
+ multi_text_data["text_embed"]
+ )
+ if self.use_text_pos_enc:
+ multi_text_data["text_embed_feats"] = self.sequence_pos_encoder(
+ multi_text_data["text_embed_feats"], batch_first=True
+ )
+
+ # Setup length and make padding mask
+ assert B == length.size(0)
+ pmask = ~length_to_mask(length, L) # (B, L)
+
+ if L > self.max_len:
+ attnmask = torch.ones((B, L, L), device=x.device, dtype=torch.bool)
+ attnmask_noar = torch.ones((L, L), device=x.device, dtype=torch.bool)
+ attnmask_ar = torch.ones((L, L), device=x.device, dtype=torch.bool)
+ for i in range(L):
+ min_ind = max(0, i - self.max_len // 2)
+ max_ind = min(L, i + self.max_len // 2)
+ eff_max_len = min(self.max_len, L)
+ max_ind_exp = max(eff_max_len, max_ind)
+ min_ind_exp = min(L - eff_max_len, min_ind)
+ attnmask_ar[i, min_ind:max_ind] = False
+ attnmask_noar[i, min_ind_exp:max_ind_exp] = False
+ attnmask[:] = attnmask_noar.unsqueeze(0)
+ else:
+ attnmask = None
+
+ # Transformer
+ for i, block in enumerate(self.blocks):
+ if i in self.text_encode_layer_idx:
+ text_block = self.text_encoder_layers[f"{i}"]
+ x = text_block(
+ x,
+ emb_text,
+ attn_mask=attnmask,
+ tgt_key_padding_mask=pmask,
+ multi_text_data=multi_text_data,
+ motion_text_pos_enc=self.motion_text_pos_enc,
+ )
+ x = block(x, attn_mask=attnmask, tgt_key_padding_mask=pmask)
+
+ # Output
+ sample = self.final_layer(x) # (B, L, C)
+ if self.avgbeta: # TODO: fix based on beta dims
+ betas = (sample[..., 126:136] * (~pmask[..., None])).sum(1) / length[
+ :, None
+ ] # (B, C)
+ betas = repeat(betas, "b c -> b l c", l=L)
+ sample = torch.cat([sample[..., :126], betas, sample[..., 136:]], dim=-1)
+
+ # Output (extra)
+ pred_cam = None
+ if self.pred_cam_head:
+ pred_cam = self.pred_cam_head(x)
+ pred_cam = pred_cam * self.pred_cam_std + self.pred_cam_mean
+ torch.clamp_min_(
+ pred_cam[..., 0], 0.25
+ ) # min_clamp s to 0.25 (prevent negative prediction)
+
+ static_conf_logits = None
+ if self.static_conf_head:
+ static_conf_logits = self.static_conf_head(x) # (B, L, C')
+
+ output = {
+ "pred_context": x,
+ "pred_x": sample,
+ "pred_x_start": sample,
+ "pred_cam": pred_cam,
+ "static_conf_logits": static_conf_logits,
+ }
+ return output
diff --git a/genmo/network/genmo_diffusion.py b/genmo/network/genmo_diffusion.py
new file mode 100644
index 0000000000000000000000000000000000000000..574442b3bbaaa25eeb615d62f1d836303e5d1ed1
--- /dev/null
+++ b/genmo/network/genmo_diffusion.py
@@ -0,0 +1,239 @@
+from copy import deepcopy
+
+import torch
+import torch.nn as nn
+from hydra.utils import instantiate
+
+from genmo.diffusion_utils.model_util import create_gaussian_diffusion
+from genmo.diffusion_utils.resample import create_named_schedule_sampler
+from genmo.utils.net_utils import length_to_mask
+
+from .genmo_cfg_sampler import ClassifierFreeSampleModel
+
+
+class GENMODiffusion(nn.Module):
+ def __init__(
+ self,
+ model_cfg,
+ max_len=120,
+ # condition
+ cliffcam_dim=3,
+ cam_angvel_dim=6,
+ cam_t_vel_dim=3,
+ imgseq_dim=1024,
+ observed_motion_3d_dim=151,
+ encoded_music_dim=438,
+ encoded_audio_dim=128,
+ latent_dim=512,
+ dropout=0.1,
+ args=None,
+ cond_merge_strategy="add",
+ cond_exists_dim=512,
+ music_mask_prob=0.1,
+ img_process_modules=None,
+ img_process_modules_enable_grad={},
+ multi_text_module_cfg={},
+ **kwargs,
+ ):
+ super().__init__()
+ self.model_cfg = model_cfg
+ self.args = args
+ self.max_len = max_len
+
+ self.regression_input_type = self.args.get("regression_input_type", "zero")
+
+ self.denoiser = instantiate(self.model_cfg.denoiser)
+ self.init_diffusion()
+ self.text_encoder, self.tokenizer = None, None
+
+ def init_diffusion(self):
+ self.train_diffusion = create_gaussian_diffusion(
+ self.model_cfg.diffusion, training=True
+ )
+ self.test_diffusion = create_gaussian_diffusion(
+ self.model_cfg.diffusion, training=False
+ )
+ gen_only_diffusion = deepcopy(self.model_cfg.diffusion)
+ gen_only_diffusion.test_timestep_respacing = self.model_cfg.diffusion.get(
+ "gen_only_test_timestep_respacing", "50"
+ )
+ print(
+ f"Gen only test timestep respacing: {gen_only_diffusion.test_timestep_respacing}"
+ )
+ self.test_gen_only_diffusion = create_gaussian_diffusion(
+ gen_only_diffusion, training=False
+ )
+ self.schedule_sampler = create_named_schedule_sampler(
+ self.model_cfg.diffusion.schedule_sampler_type, self.train_diffusion
+ )
+ return
+
+ def forward_train(self, inputs, mode):
+ assert self.training, "forward_train should only be called during training"
+ diffusion = self.train_diffusion if self.training else self.test_diffusion
+ length = inputs["length"]
+ # target_x = inputs["target_x"]
+ motion = inputs["motion"]
+ f_cond = inputs["f_cond"]
+ B, L, _ = motion.shape
+
+ vis_mask = length_to_mask(length, L) # (B, L)
+ valid_mask = inputs["mask"]["valid"]
+ assert (vis_mask == valid_mask).all()
+
+ denoiser_kwargs = {
+ "y": {
+ "text": inputs.get("caption", [""] * B),
+ "f_cond": f_cond,
+ "mask": vis_mask,
+ "length": length,
+ },
+ "inputs": inputs,
+ }
+ if "encoded_text" in inputs:
+ denoiser_kwargs["y"]["encoded_text"] = inputs["encoded_text"]
+ if "observed_motion_3d" in inputs:
+ denoiser_kwargs["observed_motion_3d"] = inputs["observed_motion_3d"]
+ denoiser_kwargs["motion_mask_3d"] = inputs["motion_mask_3d"]
+ denoiser_kwargs["rm_text_flag"] = inputs["rm_text_flag"]
+
+ if mode == "regression":
+ t = (
+ (torch.ones(B) * (diffusion.original_num_steps - 1))
+ .long()
+ .to(motion.device)
+ )
+ t_weights = torch.ones(B).to(motion.device)
+ x_start = motion
+ if self.regression_input_type == "zero":
+ x_t = torch.zeros_like(motion)
+ elif self.regression_input_type == "normal":
+ x_t = torch.randn_like(motion)
+ else:
+ raise ValueError(
+ f"Unsupported regression_input_type: {self.regression_input_type}"
+ )
+ elif mode == "diffusion":
+ t, t_weights = self.schedule_sampler.sample(motion.shape[0], motion.device)
+ if "regression_outputs" in inputs:
+ pred_x_start_regression = inputs["regression_outputs"]["model_output"][
+ "pred_x_start"
+ ].detach()
+ else:
+ raise ValueError("No regression outputs found")
+ # pred_x_start_regression = torch.zeros_like(motion)
+ x_start_reg = pred_x_start_regression.clone()
+ x_start = motion.clone()
+ x_start[inputs["mask"]["2d_only"]] = x_start_reg[inputs["mask"]["2d_only"]]
+ # regression_mask = (
+ # torch.rand(B).to(motion.device) < self.args.use_regression_outputs_prob
+ # ).float()
+ # if "gen_only" in inputs and self.args.get("use_gt_for_gen_only", True):
+ # regression_mask[inputs["gen_only"]] = 0
+ # x_start = x_start_reg * regression_mask[:, None, None] + x_start_gt * (
+ # 1 - regression_mask[:, None, None]
+ # )
+ noise = torch.randn_like(x_start)
+ x_t = self.train_diffusion.q_sample(x_start.clone(), t, noise=noise)
+
+ denoise_out = self.denoiser(
+ x_t, diffusion._scale_timesteps(t), return_aux=False, **denoiser_kwargs
+ )
+
+ output = {
+ "target_x_start": x_start,
+ "t_weights": t_weights,
+ }
+ output.update(denoise_out)
+ for x in self.args.out_attr:
+ assert x in output, f"Output {x} not found in denoise_out"
+
+ return output
+
+ def forward_test(self, inputs, progress=False):
+ assert not self.training, "forward_test should only be called during inference"
+ diffusion = self.test_gen_only_diffusion
+
+ denoiser = self.denoiser
+ length = inputs["length"]
+ B, L = inputs["B"], inputs["L"]
+
+ motion = inputs["motion"]
+ f_cond, f_uncond = inputs["f_cond"], inputs["f_uncond"]
+
+ vis_mask = length_to_mask(length, L) # (B, L)
+
+ denoiser_kwargs = {
+ "y": {
+ "text": inputs.get("caption", [""] * B),
+ "f_cond": f_cond,
+ "f_uncond": f_uncond,
+ "mask": vis_mask,
+ "length": length,
+ },
+ "inputs": inputs,
+ }
+ if "encoded_text" in inputs:
+ denoiser_kwargs["y"]["encoded_text"] = inputs["encoded_text"]
+ if "meta" in inputs and "multi_text_data" in inputs["meta"][0]:
+ denoiser_kwargs["y"]["multi_text_data"] = inputs["meta"][0][
+ "multi_text_data"
+ ]
+ if "observed_motion_3d" in inputs:
+ denoiser_kwargs["observed_motion_3d"] = inputs["observed_motion_3d"]
+ denoiser_kwargs["motion_mask_3d"] = inputs["motion_mask_3d"]
+ denoiser_kwargs["rm_text_flag"] = inputs.get("rm_text_flag", None)
+
+ if self.args.get("use_cfg_sampler_for_gen", False):
+ denoiser = ClassifierFreeSampleModel(denoiser)
+ denoiser_kwargs["y"]["scale"] = self.model_cfg.diffusion.guidance_param
+ diff_sampler = self.model_cfg.diffusion.get("sampler", "ddim")
+ if diff_sampler == "ddim":
+ sample_fn = diffusion.ddim_sample_loop_with_aux
+ kwargs = {"eta": self.model_cfg.diffusion.ddim_eta}
+ else:
+ raise NotImplementedError(f"Sampler {diff_sampler} not implemented")
+
+ if self.args.get("force_zero_noise", False):
+ noise = torch.zeros_like(motion)
+ elif self.args.get("force_rand_noise", False):
+ noise = torch.randn_like(motion)
+ else:
+ noise = torch.randn_like(motion)
+
+ if self.args.get("return_mid", False):
+ kwargs["return_mid"] = True
+
+ denoise_out = sample_fn(
+ denoiser,
+ motion.shape,
+ clip_denoised=False,
+ model_kwargs=denoiser_kwargs,
+ skip_timesteps=0, # 0 is the default value - i.e. don't skip any step
+ init_image=None,
+ progress=progress,
+ dump_steps=None,
+ noise=noise,
+ const_noise=False,
+ **kwargs,
+ )
+ output = denoise_out.copy()
+
+ for x in self.args.out_attr:
+ assert x in output, f"Output {x} not found in denoise_out"
+ return output
+
+ def forward(
+ self,
+ inputs,
+ train=False,
+ postproc=False,
+ static_cam=False,
+ mode=None,
+ test_mode=None,
+ normalizer_stats=None,
+ ):
+ if train:
+ return self.forward_train(inputs, mode=mode)
+ else:
+ return self.forward_test(inputs)
diff --git a/genmo/network/stats_compose.py b/genmo/network/stats_compose.py
new file mode 100644
index 0000000000000000000000000000000000000000..2549931d8203d79fd6dae1154022010b730ab3b5
--- /dev/null
+++ b/genmo/network/stats_compose.py
@@ -0,0 +1,868 @@
+# fmt:off
+body_pose_r6d = {
+ "bedlam": {
+ "count": 5417929,
+ "mean": [ 0.9772, -0.0925, 0.0028, 0.1058, 0.9111, 0.1373, 0.9796, 0.0711,
+ -0.0193, -0.0816, 0.8910, 0.1953, 0.9935, 0.0072, 0.0270, -0.0046,
+ 0.9200, -0.2511, 0.9752, 0.0477, -0.0990, -0.0613, 0.8242, -0.2730,
+ 0.9836, -0.0400, 0.0067, 0.0148, 0.7836, -0.3471, 0.9931, -0.0300,
+ -0.0469, 0.0244, 0.9825, -0.0513, 0.9777, 0.0206, 0.1444, -0.0470,
+ 0.9603, 0.1521, 0.9804, -0.0362, -0.0902, 0.0500, 0.9546, 0.1337,
+ 0.9969, -0.0105, 0.0076, 0.0090, 0.9914, 0.0150, 0.9953, -0.0607,
+ 0.0089, 0.0602, 0.9942, 0.0146, 0.9934, -0.0682, -0.0171, 0.0680,
+ 0.9932, -0.0017, 0.9790, 0.0294, 0.0065, -0.0338, 0.9706, -0.0456,
+ 0.9056, 0.2457, -0.1029, -0.2279, 0.9262, 0.0145, 0.9233, -0.1301,
+ 0.1550, 0.1140, 0.9476, 0.0534, 0.9769, -0.0572, -0.0095, 0.0569,
+ 0.9690, 0.0472, 0.6782, 0.5746, -0.2378, -0.5546, 0.7212, 0.0917,
+ 0.6489, -0.5955, 0.2424, 0.5821, 0.6797, 0.0563, 0.5562, -0.1252,
+ -0.5860, 0.0937, 0.9176, -0.1287, 0.4453, 0.1421, 0.6119, -0.1427,
+ 0.8996, -0.1136, 0.9186, -0.0881, -0.1463, 0.1087, 0.8692, 0.0845,
+ 0.9175, 0.0257, 0.0663, -0.0385, 0.8603, 0.1020],
+ "std": [0.0429, 0.1392, 0.1236, 0.1323, 0.1645, 0.3086, 0.0375, 0.1406, 0.1172,
+ 0.1275, 0.1934, 0.3280, 0.0119, 0.0835, 0.0716, 0.0741, 0.1528, 0.2484,
+ 0.0349, 0.0947, 0.1633, 0.0924, 0.3469, 0.3370, 0.0273, 0.1009, 0.1411,
+ 0.0680, 0.3876, 0.3323, 0.0103, 0.0735, 0.0712, 0.0690, 0.0246, 0.1617,
+ 0.0216, 0.1097, 0.1016, 0.0924, 0.0509, 0.2035, 0.0245, 0.1188, 0.1212,
+ 0.1056, 0.0634, 0.2308, 0.0054, 0.0579, 0.0517, 0.0575, 0.0124, 0.1158,
+ 0.0076, 0.0654, 0.0367, 0.0644, 0.0118, 0.0592, 0.0116, 0.0829, 0.0361,
+ 0.0832, 0.0124, 0.0422, 0.0343, 0.1060, 0.1680, 0.1075, 0.0473, 0.2023,
+ 0.0701, 0.2344, 0.2213, 0.2632, 0.0589, 0.1318, 0.0767, 0.2456, 0.2009,
+ 0.2666, 0.0542, 0.1106, 0.0347, 0.1080, 0.1718, 0.1117, 0.0459, 0.2025,
+ 0.1882, 0.2769, 0.2032, 0.3072, 0.1447, 0.2204, 0.2018, 0.2820, 0.2126,
+ 0.3213, 0.1760, 0.2486, 0.4749, 0.1677, 0.2791, 0.2239, 0.0963, 0.2705,
+ 0.5540, 0.1846, 0.2572, 0.2411, 0.1287, 0.2878, 0.1151, 0.2993, 0.1557,
+ 0.2812, 0.1880, 0.3334, 0.1286, 0.3355, 0.1553, 0.3216, 0.1880, 0.3306]
+ },
+ "amass": {
+ "count": 7114038,
+ "mean": [ 9.6969e-01, -5.9719e-02, -3.7700e-02, 5.8256e-02, 9.0800e-01,
+ 1.0972e-01, 9.7636e-01, 4.3401e-02, 4.3110e-03, -4.3032e-02,
+ 9.0261e-01, 1.4478e-01, 9.9288e-01, 3.5673e-03, 1.6264e-02,
+ -2.2260e-03, 9.3470e-01, -2.3495e-01, 9.7147e-01, 5.2553e-02,
+ -9.3666e-02, -5.4550e-02, 8.3321e-01, -2.4246e-01, 9.7971e-01,
+ -3.8429e-02, 5.3575e-03, 1.5537e-02, 8.1449e-01, -3.0926e-01,
+ 9.9532e-01, -9.4398e-03, -3.8328e-02, 8.5141e-03, 9.8880e-01,
+ 1.9976e-04, 9.5602e-01, -3.9528e-02, 2.0017e-01, 1.0363e-02,
+ 9.5965e-01, 1.3770e-01, 9.6223e-01, -4.6278e-02, -1.5177e-01,
+ 6.6705e-02, 9.5545e-01, 1.2519e-01, 9.9767e-01, -1.2616e-02,
+ -2.5442e-04, 1.1661e-02, 9.9376e-01, -3.6222e-02, 9.9511e-01,
+ -1.0583e-02, 1.2130e-02, 7.6461e-03, 9.9137e-01, 2.0029e-02,
+ 9.9295e-01, 7.2917e-03, 4.9454e-03, -8.0286e-03, 9.9137e-01,
+ 2.3707e-03, 9.7698e-01, 1.9943e-02, 1.3808e-03, -2.2006e-02,
+ 9.7375e-01, -6.7936e-02, 9.2804e-01, 2.5005e-01, -5.7167e-02,
+ -2.4047e-01, 9.4246e-01, 2.5863e-02, 9.2957e-01, -2.1329e-01,
+ 1.1112e-01, 2.0741e-01, 9.4876e-01, 2.9901e-02, 9.7683e-01,
+ -4.1210e-02, 2.3248e-03, 4.0967e-02, 9.7365e-01, 5.7309e-03,
+ 6.4513e-01, 6.1999e-01, -2.5469e-01, -6.2342e-01, 6.8177e-01,
+ 3.5524e-02, 6.6192e-01, -5.9341e-01, 2.7136e-01, 5.9269e-01,
+ 6.8966e-01, 3.1309e-02, 6.8946e-01, -1.1676e-01, -4.9859e-01,
+ 4.0969e-02, 9.3656e-01, -1.4875e-01, 6.2787e-01, 1.3793e-01,
+ 5.4289e-01, -9.1946e-02, 9.2868e-01, -1.1927e-01, 9.3012e-01,
+ -8.3810e-02, -1.1951e-01, 9.7211e-02, 8.9118e-01, 5.9887e-02,
+ 9.3033e-01, 7.1047e-02, 7.5264e-02, -8.0679e-02, 8.8562e-01,
+ 4.8960e-02],
+ "std": [0.0612, 0.1390, 0.1779, 0.1415, 0.1826, 0.3268, 0.0440, 0.1382, 0.1542,
+ 0.1348, 0.1930, 0.3272, 0.0132, 0.0801, 0.0855, 0.0729, 0.1255, 0.2238,
+ 0.0554, 0.1088, 0.1727, 0.0939, 0.3294, 0.3559, 0.0532, 0.1082, 0.1554,
+ 0.0768, 0.3446, 0.3407, 0.0120, 0.0650, 0.0584, 0.0632, 0.0198, 0.1335,
+ 0.0631, 0.1250, 0.1574, 0.1047, 0.0730, 0.2091, 0.0759, 0.1241, 0.1667,
+ 0.1112, 0.0831, 0.2185, 0.0060, 0.0441, 0.0502, 0.0441, 0.0102, 0.0946,
+ 0.0237, 0.0722, 0.0610, 0.0738, 0.0479, 0.0949, 0.0369, 0.0943, 0.0610,
+ 0.0966, 0.0498, 0.0729, 0.0425, 0.1001, 0.1824, 0.0972, 0.0408, 0.1887,
+ 0.0594, 0.1842, 0.1884, 0.2020, 0.0457, 0.1018, 0.0640, 0.1990, 0.1854,
+ 0.2133, 0.0467, 0.0910, 0.0392, 0.1049, 0.1776, 0.1037, 0.0413, 0.1945,
+ 0.1733, 0.2612, 0.1905, 0.2963, 0.1512, 0.1861, 0.1710, 0.2663, 0.1896,
+ 0.3135, 0.1568, 0.2219, 0.3976, 0.1594, 0.2810, 0.1855, 0.0845, 0.2398,
+ 0.4398, 0.1629, 0.2685, 0.1990, 0.0998, 0.2556, 0.1137, 0.2837, 0.1419,
+ 0.2761, 0.1678, 0.2973, 0.1172, 0.3010, 0.1394, 0.2910, 0.1724, 0.3039]
+ }
+}
+
+betas = {
+ "bedlam": {
+ "count": 37855, # so many subjects?
+ "mean": [ 0.0378, -0.3562, 0.1185, 0.2245, 0.0204, 0.0929, 0.0537, 0.1006,
+ -0.1180, 0.0936],
+ "std":[0.8070, 1.3480, 0.8964, 0.7390, 0.6433, 0.6089, 0.5374, 0.6984, 0.7263,
+ 0.5395],
+ },
+ "amass": {
+ "count": 18086,
+ "mean": [ 0.2310, 0.1750, 0.2931, -0.1859, -1.1163, -1.1028, -0.2573, 0.3555,
+ 0.3732, 0.2852],
+ "std": [0.8831, 0.7965, 1.0899, 1.1788, 1.2128, 1.1081, 0.9780, 1.1434, 0.8498,
+ 1.1462],
+ }
+}
+
+global_orient_c_r6d = {
+ "bedlam": {
+ "count": 5417929,
+ "mean": [-4.9862e-03, -8.7136e-04, -1.4187e-03, 1.4825e-02, -9.4419e-01,
+ -5.1653e-02],
+ "std": [0.7048, 0.1713, 0.6884, 0.1548, 0.1546, 0.2403],
+ },
+}
+
+global_orient_gv_r6d = {
+ "bedlam": {
+ "count": 5134187,
+ "mean": [ 3.6018e-04, -2.2327e-04, 2.2316e-03, -4.4879e-02, -9.7435e-01,
+ 1.0021e-01],
+ "std": [0.6070, 0.5355, 0.5873, 0.6285, 0.2336, 0.7675],
+ },
+}
+
+local_transl_vel = {
+ "none":{
+ "mean": [0., 0., 0.],
+ "std": [1., 1., 1.]
+ },
+ "1e-2":{
+ "mean": [0., 0., 0.],
+ "std": [1e-2, 1e-2, 1e-2]
+ },
+ "bedlam": {
+ "count": 5417929,
+ "mean": [7.3057e-05, -2.2142e-04, 3.2444e-03],
+ "std": [0.0065, 0.0091, 0.0114],
+ },
+ "amass": {
+ "count": 7113068,
+ "mean": [-0.0002, -0.0006, 0.0069],
+ "std": [0.0064, 0.0070, 0.0138],
+ },
+ "alignhead":{
+ "count": 7113068,
+ "mean":[-2.0822e-04, -1.7966e-06, 6.9816e-03],
+ "std":[0.0065, 0.0066, 0.0139],
+ },
+ "alignhead_absy":{
+ "count": 7113068,
+ "mean":[-0.0002, -0.0316, 0.0070],
+ "std":[0.0065, 0.1351, 0.0139],
+ },
+ "alignhead_absgy":{
+ "count": 7113068,
+ "mean":[[-2.0822e-04, 1.2627e+00, 6.9816e-03]],
+ "std":[0.0065, 0.1516, 0.0139],
+ }
+
+}
+
+pred_cam = {
+ "bedlam": {
+ "count": 5096332,
+ "mean": [1.0606, -0.0027, 0.2702],
+ "std": [0.1784, 0.0956, 0.0764],
+ }
+}
+
+vitfeat = {
+ "bedlam": {
+ "count": 5546332,
+ "mean": [-1.3772, 0.2490, 0.0602, -0.1834, 0.2458, 0.5372, 0.3343, -0.3476, -0.1017, -0.0362, -0.0678, 0.2150, -0.2534, 0.1029, 0.8199, -0.4676, 0.6259, -0.3350, 0.0549, -0.4469, 0.2751, -0.1763, 0.1114, -0.2115, -0.0264, 0.5294, 0.8212, -0.4562, 0.4147, -0.0256, -0.1019, 0.2798, 0.9284, 0.4652, 0.6365, 0.6785, -0.0765, 0.0337, -0.2566, -0.0335, -0.1799, 0.7426, 0.2810, -0.7121, -0.0893, 0.1608, -0.2483, 1.5094, -1.4395, -0.3682, -0.4157, -0.0032, -0.0376, -0.0043, 0.2092, 0.3038, -0.2077, -0.4868, -0.1534, 0.2668, 1.2773, 0.2838, -0.4863, -1.2300, 0.0581, -0.3041, 0.1518, 0.7955, -0.4293, 1.4666, 0.3077, 0.3918, 0.1418, 0.1590, 0.8671, -0.3527, 0.5629, 0.1414, 0.0964, -0.1094, -0.0211, -0.0937, 0.1606, -0.7900, 0.0397, 0.0570, 0.7083, -0.5732, 0.1430, -0.2571, 0.5275, 0.6603, 0.3265, 0.4574, -0.3361, -0.1267, 0.3841, 0.1758, -0.6207, -0.3673, 0.8914, 0.4297, -0.8118, 0.2229, -0.2876, 0.2460, 0.4856, -0.1446, -0.2416, 0.1229, 0.2865, 0.7023, -0.2883, 0.3940, -1.5496, 0.4456, 0.6445, 0.2058, -0.4265, 0.3724, 0.1557, -1.4208, -0.1246, 0.1237, -0.3965, 0.0105, -0.0780, 0.6448, -0.1132, 0.8500, -0.2828, 0.4447, 0.6257, -0.2664, -0.8384, -1.8091, -0.2769, 0.1866, 0.6051, -0.2548, 0.9823, -0.2985, -0.2773, -0.4383, 0.1886, 0.2411, 0.2546, 0.2195, -0.0041, 0.1038, -0.6804, 1.2364, 0.5393, 0.0351, 0.4537, -0.8044, -0.1993, -2.1097, -0.8458, 0.1497, 1.6042, 0.6458, -0.5455, 0.0778, 0.0504, -0.5242, -0.3215, -0.0199, 1.1461, -0.3355, -0.3421, -0.3951, 0.0184, -0.0261, 0.2048, 0.0080, 0.6553, -1.3221, 0.5140, 0.5958, -0.2523, 0.9434, -0.0727, 0.1978, 1.1105, -0.4992, 0.3990, 0.2074, 0.3843, -0.0444, 0.0624, -0.8442, -0.0724, -0.5328, 1.1723, 0.8043, 0.6674, 1.5283, 4.2502, 0.0935, 0.3733, 0.1569, 0.0154, 0.0674, 0.0862, -0.2744, -0.4537, 0.1588, -1.9156, 0.0149, -1.0498, -0.0790, 0.0851, -0.5007, 0.3323, -0.1065, 0.0782, 0.0725, -0.5921, -0.1876, 0.0094, -0.3631, 0.0951, 0.1318, 0.0936, 0.5668, -0.0875, -0.4576, -0.4306, 0.5458, 1.0761, 1.1740, -0.0337, 1.3718, -0.2913, -0.3433, 0.5338, -0.4577, -0.4966, 0.2704, 0.3236, 0.4053, 0.0360, 1.1616, -0.2012, 0.7373, 0.0779, -0.0280, -0.4426, 0.0450, 0.2923, 0.0161, -0.4788, 0.1924, -0.3012, 0.0298, -0.7776, -0.2215, 0.4494, -0.1677, 0.2214, 0.0762, -0.3088, 0.4230, 0.0673, -1.0233, 0.0748, -0.4358, -0.2497, -0.0066, 0.1679, -0.1077, -0.4290, 2.5254, -0.8819, -0.8073, 0.2535, 2.0680, -0.4715, 0.3614, -2.9281, 3.1536, 0.3118, -0.0239, 0.7064, -0.6935, -1.1070, -0.1715, -0.0920, -0.2133, -1.0173, 0.0084, -0.1721, 0.2605, -0.6607, -0.0788, -0.3479, -0.2187, 1.0605, 0.2857, 0.7464, 0.9612, -1.1332, 1.5708, -1.0264, 0.6070, 0.4103, -0.1950, -0.0629, -0.0958, -0.2199, -0.2198, -0.4019, 0.2478, -0.3576, 0.0191, -5.8435, 0.0145, -0.2312, 0.9872, 1.1159, 0.3775, 0.1960, -0.5968, -0.2611, -0.0634, -0.1003, 0.7411, -0.8298, -0.1743, 1.8418, 0.3692, -0.4321, 0.0613, -1.9046, 0.5812, 0.2805, 0.1703, -0.2212, -0.0740, -0.2737, -0.3084, 2.9787, -0.1392, 0.3347, 0.0866, -0.8654, -0.4564, -0.7839, 0.1033, -0.0204, 0.1558, -0.1469, 0.2850, -0.1139, 0.8253, 0.7352, -0.6132, 0.0566, 0.3087, -0.1189, 0.1640, 0.2511, 0.5230, -0.0972, -0.5621, -2.5404, 0.3529, -0.2543, -0.6757, 0.2045, -0.0511, -0.2204, 0.1023, 0.0143, 0.4191, -0.3946, -1.0912, 0.8555, 1.0751, -0.0184, -0.3162, 0.1910, 0.6522, -0.5801, 0.2091, -0.8254, -0.3425, 0.3368, -0.0384, -0.4570, 2.5288, -0.3513, -0.1630, 0.1096, -0.5936, 1.5303, -0.4135, -0.2418, -0.0564, -2.6344, -0.1054, 0.8866, -0.2946, -0.4564, -0.6220, 0.2672, -0.9012, 0.3535, 0.2344, -0.0718, 0.0782, 0.0133, 0.2032, -1.2768, 0.1271, -0.5114, -0.0584, -0.8219, -0.1069, 1.5577, -0.1432, -0.6794, 0.9101, 0.6390, 0.3547, -0.6126, -0.1885, 0.2462, -1.1864, 0.0653, -0.7940, 0.5204, 0.5372, 0.5353, -0.4268, -0.2003, -0.2496, -0.0405, 0.3615, -0.1635, 0.1908, -0.0467, 0.7167, 0.1465, 0.4621, 0.1190, -1.6899, 0.6512, 1.3150, -0.1273, 0.0507, 0.2058, -0.1855, 0.1316, 0.1280, 0.5049, 0.0262, -0.0329, 2.0327, -0.6410, 0.4536, 0.0609, 0.1883, -0.5454, -0.5247, 0.1856, 0.7238, 1.4886, -0.1068, 1.7239, -0.8228, -0.2155, 0.5159, 0.2941, -0.0782, -0.0159, 0.1844, -0.1808, -0.1132, 0.4861, 4.0106, 0.0130, 0.2455, -0.1101, 0.0792, 0.4720, -0.1022, 2.0154, -0.4013, 0.5604, 1.3600, -0.5614, 0.3793, -0.1245, 0.2444, 0.1657, 1.7616, 0.6198, 0.1761, -0.6036, -0.1931, 0.4449, 0.2574, -0.2360, 1.1118, 0.0804, 1.1533, 0.2549, 0.3386, 0.2463, 0.0930, -0.6093, -0.1464, 0.2889, 0.2294, -0.5943, 0.1323, 0.5119, 0.1093, -1.0178, 0.4735, 0.3068, 0.3213, -0.0585, -0.3682, -0.6105, -0.7776, 0.1999, 0.9439, -0.4209, 0.1488, 1.3119, -0.4679, -0.3882, 0.2677, -0.1673, -0.5921, -1.2811, -1.0972, 0.3873, 0.0798, -0.0538, 0.0659, -0.1439, -1.3106, -0.5175, 0.4538, -1.0376, -0.9015, 0.7454, -0.0714, -0.4641, 0.2083, 0.0596, -2.9637, 0.3057, 0.2121, -0.2399, 0.6963, 0.1400, 1.7446, 0.9707, -0.3118, -0.3371, 0.0130, 1.0006, -0.2740, 0.1100, -0.9666, 0.7636, 1.2002, -0.0018, -0.3380, 0.1262, 0.5829, -0.0374, 0.0689, 0.2022, -2.0056, -0.2051, -0.4549, 0.0519, 0.4217, -0.7413, 0.0601, 0.4385, 2.8503, -2.7656, 1.2281, -0.1280, 0.6028, 0.4995, 0.0638, -0.3376, 0.2527, -0.1572, -0.4385, -0.6372, 0.2569, 0.4115, 0.4507, 0.6063, -0.1051, 1.2529, 0.2453, -0.7905, -0.3797, -0.2674, 0.2662, 1.5347, -0.3908, 0.8839, -0.6054, -0.4827, -0.3495, 1.2107, -0.4419, -0.6177, 0.1054, 1.0132, -0.3246, -0.1776, 1.1740, -0.0252, 0.0368, -0.7937, -0.9988, -0.0228, 0.0742, -2.4925, 0.5785, 2.3900, 1.2726, -0.3682, -0.8625, -0.3299, 0.3934, 1.4045, -0.6200, -0.0024, 0.2348, -0.1827, -0.5913, -0.6982, 0.2648, 0.2601, 0.9986, 0.1636, 0.8982, -0.4269, 1.7454, -1.9136, -0.9865, -0.0451, 0.2851, -0.5938, -0.3066, 0.0910, -0.3150, -0.4002, 0.4789, 0.0337, -0.6997, -0.2555, -0.6602, -3.0103, 0.2491, -1.0346, 0.3651, 0.2319, 1.0224, -0.2613, 1.6970, 0.7515, 2.1477, 0.1310, 0.2060, 0.1372, 1.0049, -0.8758, -0.3804, -2.1513, 0.8010, -0.2271, -0.2108, 0.3728, -1.7321, -1.0250, -0.2584, -0.2513, 0.2418, -0.7641, 0.2084, -1.3560, 0.5803, 0.1556, -0.3612, 1.3099, -0.2673, 0.4371, -0.8022, 0.1776, -0.5019, 0.1880, -0.2093, 0.0750, -0.7228, -1.3950, 0.1944, -1.5994, -0.2832, 0.0507, 0.1917, 1.2954, 0.0471, 0.3115, -2.2382, -0.3891, -0.0704, 0.3897, 0.0347, 0.9186, -0.8407, 0.9456, 0.5629, 0.3474, -0.4869, 0.4696, -0.4438, 0.0860, -0.8313, -0.0383, 0.2055, 0.4822, -0.1455, -0.1719, -0.2346, -0.4606, 0.8018, 0.3767, -0.0613, 1.9429, -0.6558, -0.0772, -0.1592, -0.1413, 0.4759, -0.0686, 0.9243, -0.2413, -0.1084, -0.2248, -0.0776, 1.4193, -0.0605, 0.1305, -0.2055, 0.0917, 0.6884, -0.0152, 0.1215, 0.2920, -0.0781, -0.0256, 0.3789, -0.1933, 0.1759, 2.3899, 1.0915, -0.7082, -0.4519, -0.2648, -1.2404, -0.2485, 1.0713, 0.1662, -0.1268, 0.3338, -0.0319, 0.1692, -0.5161, 0.9351, 0.1996, -0.2743, 0.0492, -0.0171, 0.1546, 0.2533, -0.0102, 0.6147, 0.0035, -0.2468, -0.2116, -1.7912, 0.2735, 0.4147, 0.4458, 0.6123, 0.0860, 0.2098, -0.3691, -0.2297, -0.6086, -1.0407, -0.7736, -0.3087, -0.0900, -0.1007, -0.3801, -0.3408, -0.4853, -0.3101, -0.8812, 0.0187, -0.9697, -0.2393, 0.1129, -0.5682, 0.4349, 0.1017, 0.2173, -0.0644, -0.9307, 0.9754, 0.2189, 0.2966, -0.4089, -0.2471, -0.7549, 0.3300, 0.7856, 0.1262, 0.2097, -0.5872, 0.9896, 0.5100, 1.0608, -0.7974, 0.1549, -0.1020, 0.4286, 0.0603, -0.6836, -0.4662, -1.2350, -0.0858, -0.5552, 0.0383, 0.2145, -0.4324, -0.5896, 0.9709, -0.0827, -0.2574, 0.2436, -0.1460, 0.5862, 0.4329, -1.2421, 0.0497, -0.0034, 0.2385, -0.1346, 2.0652, 0.8790, -0.2033, -2.6427, 0.3654, -0.1929, -0.0753, -0.9107, 0.9437, 0.3717, -0.7058, -0.2487, -1.0937, -0.7612, 0.9516, -0.7426, -0.0736, 1.2167, 0.6336, 0.2707, -0.7666, -0.1272, -0.8960, 0.3748, 0.7344, 0.7257, 0.3686, -0.5036, -0.2829, 0.0548, 0.3034, -0.2335, -0.3215, 0.0566, -0.2733, -0.3644, 0.0467, -0.0924, -0.5145, -1.7089, 0.4896, 0.0074, 0.2840, 0.1140, -0.0409, -0.3251, 1.0805, 3.0856, -0.3409, 1.2684, -0.0245, -0.0636, -0.0090, 0.1293, -0.3410, -0.0482, 0.1482, 0.2027, 0.5623, 0.0566, 0.6453, -0.0126, 0.0720, -0.0277, 0.0531, 0.1860, -0.1044, -0.6973, 0.3026, 0.4733, -0.1590, 0.4727, 0.8486, 0.4478, 0.1814, 1.0862, 0.0478, 0.2437, -0.5269, -0.0796, -0.4291, 0.4937, -0.0407, -0.6961, -0.0412, 0.6865, 0.0457, 0.1085, -0.4717, -0.1339, 0.8600, 0.6718, -0.3542, -0.5655, 1.3711, 0.0034, 0.3077, 0.0903, 0.3618, 0.3287, -0.1007, 0.0332, -0.3841, -0.3981, 0.1079, -0.4399, 0.1836, 0.0939, -0.1425, -0.2531, -1.2103, 0.0234, -1.3023, -0.0570, -0.0587, 1.1733, 0.0079, 1.0809, 0.4697, -0.1427, 3.3793, -0.1503, 0.4354, 0.0274, 0.3112, -0.3816, 0.0187, -0.1282, -0.4136, 0.3684, 0.6930, 1.3605, 0.4949, 0.4162, -2.2398, 0.4104, 0.6839, 0.4519, 0.0546, -0.0816, 0.0357, 0.1977, -0.8450, 0.1481, 0.1588, -0.1392, -0.3304, -0.3499, -0.8669, 0.1510, 0.1127, 0.9853, -0.3019, -0.3493, -0.0783, -0.8491, 0.0696, 0.7295, -1.0612, 0.1232],
+ "std": [0.9277, 0.7470, 0.6154, 0.8520, 0.8682, 0.7121, 0.7048, 0.6865, 0.7543, 0.6952, 0.6186, 0.4204, 0.4614, 0.4731, 0.4421, 0.4068, 0.6927, 0.6540, 0.4717, 0.4993, 0.5945, 0.5480, 0.4898, 0.6438, 0.5551, 0.5686, 0.7287, 0.6033, 0.5590, 0.3768, 0.5304, 0.6748, 0.5559, 0.5265, 0.6214, 0.6490, 0.4639, 0.6465, 0.5575, 0.6202, 0.5369, 1.2466, 0.7340, 0.5462, 0.6508, 0.5766, 0.5405, 0.5581, 0.5687, 0.7549, 0.5743, 0.4748, 0.6308, 0.6292, 0.6391, 0.6284, 0.4202, 0.5970, 0.5587, 0.5364, 0.4655, 0.5201, 0.7140, 0.6220, 0.4978, 0.4479, 0.5452, 0.7489, 0.5866, 0.4592, 0.7493, 0.6548, 0.5497, 0.4658, 0.8663, 0.4574, 0.5351, 0.5595, 0.4579, 0.5141, 0.4824, 0.5504, 0.5468, 0.5726, 0.5155, 0.6679, 0.8433, 0.5278, 0.5666, 0.7699, 0.5682, 0.9431, 0.5344, 0.6562, 0.4749, 0.5241, 0.6869, 0.4117, 0.5839, 0.5115, 0.8811, 0.5335, 0.6476, 0.4883, 0.6034, 0.5778, 0.4764, 0.8787, 0.8589, 0.5168, 0.4548, 0.8146, 0.5860, 0.6087, 0.6758, 0.7049, 0.8292, 0.6547, 0.6043, 0.7242, 0.6158, 0.6435, 0.5219, 0.6148, 0.7738, 0.4871, 0.7944, 0.7605, 0.6120, 0.5482, 0.6107, 0.6106, 0.4295, 0.4549, 0.4167, 0.6142, 0.6368, 0.5432, 0.5412, 0.6568, 0.9641, 0.6413, 0.6634, 0.4222, 0.6917, 0.5664, 0.5554, 0.4098, 0.6949, 0.5890, 0.4995, 0.5475, 0.6446, 0.5599, 0.6439, 0.6220, 0.5761, 0.5862, 0.5126, 0.6037, 0.5377, 0.5817, 0.6216, 0.5986, 0.4834, 0.6929, 0.5819, 0.6781, 0.6088, 0.5425, 0.7211, 0.6253, 0.5408, 0.6826, 0.5454, 0.7614, 0.9767, 0.8721, 0.7527, 0.4022, 0.5061, 0.5921, 0.5945, 0.6048, 0.7206, 0.5533, 0.5506, 0.6816, 0.6116, 0.6424, 0.7484, 0.6350, 0.5953, 0.4941, 0.7675, 0.8244, 0.6885, 0.5751, 0.9304, 0.5252, 0.5741, 0.4537, 0.5610, 0.9873, 0.5155, 0.7180, 0.4421, 0.5171, 0.5343, 0.5225, 0.7952, 0.6149, 0.6401, 0.5667, 0.6946, 0.8172, 0.5188, 0.5082, 0.6298, 0.6904, 0.4820, 0.5600, 0.5584, 0.5600, 0.4776, 0.5008, 0.7215, 0.6071, 0.5571, 0.6174, 0.4049, 0.7368, 0.5996, 0.7888, 0.7609, 0.5913, 0.8778, 0.4462, 0.7460, 0.7240, 0.5705, 0.6267, 0.5684, 0.5707, 0.6560, 0.5310, 0.5278, 0.6833, 0.6420, 0.6696, 0.8815, 0.4767, 0.7171, 0.4826, 0.6736, 0.5483, 0.4913, 0.5840, 0.5242, 0.4310, 0.5846, 0.4389, 0.5164, 0.6203, 0.5625, 0.8495, 0.5091, 0.6904, 0.5490, 0.5467, 0.4746, 0.8446, 0.6030, 0.6563, 1.0108, 0.5633, 0.6324, 0.6339, 0.6269, 1.2128, 0.6877, 0.5998, 0.4763, 0.4979, 0.7968, 0.6549, 1.0234, 0.5385, 0.6164, 0.5485, 0.8526, 0.5776, 0.5292, 0.5716, 0.5458, 0.5332, 0.5264, 0.6239, 0.6668, 0.7481, 0.3929, 0.5932, 0.5741, 0.4433, 0.7519, 0.4940, 0.7438, 0.5315, 0.3895, 0.5528, 0.6656, 0.6665, 0.9897, 0.8098, 0.6000, 0.5226, 1.2953, 0.5624, 0.6416, 0.5880, 0.5828, 0.4779, 0.6721, 0.6273, 0.7918, 0.5498, 0.5262, 0.6396, 0.6185, 0.6117, 0.8871, 0.5688, 0.5335, 0.6402, 0.5994, 0.9472, 0.5072, 0.7688, 0.6257, 0.6548, 0.6070, 0.7646, 0.5362, 0.5151, 0.6852, 0.4533, 0.6976, 0.6170, 0.5700, 0.5819, 0.4350, 0.5755, 0.4902, 0.9396, 0.5110, 0.5461, 0.6380, 1.0192, 0.5009, 0.8211, 0.6223, 0.5970, 0.5465, 0.8314, 0.4997, 0.5066, 0.5824, 0.6241, 0.4910, 0.4849, 0.5292, 0.5357, 0.4856, 0.6120, 0.4212, 0.6712, 0.4599, 0.4625, 0.7568, 0.8765, 0.8095, 0.7385, 0.5748, 0.7405, 0.6474, 0.6466, 0.6481, 0.5660, 0.6876, 0.9852, 0.5923, 0.6319, 0.6818, 0.4716, 0.6599, 0.5343, 0.5384, 0.9786, 0.4421, 0.5543, 1.0386, 0.5640, 0.5990, 0.5060, 0.6141, 0.3880, 0.6767, 0.5753, 0.4797, 0.4623, 0.5802, 0.6813, 0.5792, 0.4790, 0.6855, 0.5186, 0.4890, 0.5740, 0.6117, 0.5177, 0.5032, 0.6367, 0.4555, 0.6749, 0.6680, 0.6878, 0.7425, 0.8106, 0.5460, 1.0575, 0.5022, 0.7639, 0.5132, 0.5433, 0.7702, 0.4572, 0.4274, 0.6779, 0.5277, 0.5634, 0.4814, 0.5491, 0.5790, 0.5750, 0.5573, 0.4652, 0.5240, 0.6244, 0.6247, 0.7397, 0.7107, 0.5964, 0.4891, 0.7089, 0.6531, 0.6979, 0.4630, 0.5348, 0.4308, 0.8983, 0.5416, 0.4521, 0.6261, 0.4931, 0.7247, 0.5689, 0.5254, 0.4913, 0.6307, 0.5586, 0.5804, 0.5692, 0.5211, 0.6549, 0.6069, 0.5216, 0.4617, 0.7538, 0.4234, 0.4868, 0.7661, 1.1726, 0.8879, 0.4984, 0.6142, 0.4203, 0.5944, 0.6758, 0.5682, 0.6554, 0.7316, 0.5552, 0.7454, 0.3907, 0.7559, 0.4752, 0.5638, 0.7824, 0.7995, 0.5728, 0.8546, 0.5663, 0.5545, 0.4785, 1.0497, 0.7177, 0.5461, 0.5134, 0.5432, 0.5964, 0.5879, 0.7046, 0.7501, 0.5707, 0.9907, 0.9337, 0.5682, 0.4887, 0.5970, 0.6229, 0.6501, 0.7529, 0.7062, 0.6775, 0.7286, 0.6250, 0.4521, 0.5357, 0.5479, 0.7957, 0.4596, 0.6440, 0.8665, 0.6024, 0.7485, 0.6478, 0.6483, 0.5785, 0.5500, 0.4802, 0.4465, 0.6829, 0.6890, 0.6180, 0.8767, 0.7419, 0.6193, 0.3918, 0.5888, 0.5440, 0.5146, 0.4297, 0.4410, 0.4894, 0.4422, 0.9614, 0.6290, 0.6717, 0.5415, 0.5442, 0.5862, 0.4967, 0.7102, 1.1356, 0.4818, 0.4557, 0.6403, 0.4971, 0.7491, 0.8534, 0.8754, 0.5308, 0.5591, 0.6415, 0.7715, 0.8137, 0.4898, 0.5460, 0.5476, 0.9199, 0.6195, 0.5949, 0.7990, 0.4444, 0.6199, 0.5166, 0.4646, 0.9060, 0.6261, 0.5149, 0.6533, 0.7420, 0.4830, 0.5314, 0.5503, 0.5777, 0.6284, 0.7288, 0.5743, 0.6041, 0.5674, 0.4661, 0.6211, 0.6172, 0.4094, 0.5787, 0.8089, 0.6061, 0.5882, 0.5498, 0.7239, 0.6387, 0.7910, 0.5267, 0.5569, 0.6382, 0.5492, 0.5444, 0.6476, 0.8666, 0.9807, 0.5594, 0.6814, 0.5467, 0.8900, 0.5321, 0.5516, 1.0188, 0.7193, 0.5044, 0.5717, 0.9741, 0.7856, 0.6849, 0.5604, 1.0236, 0.8399, 0.5065, 0.6475, 0.4055, 0.7975, 0.4454, 0.5726, 0.4489, 0.6851, 0.6504, 0.4737, 0.5995, 0.6226, 0.5917, 0.5394, 0.5240, 0.7863, 0.6008, 0.5330, 0.4760, 0.6163, 0.4679, 0.5712, 0.7180, 0.4908, 1.0175, 0.5942, 0.5170, 0.7534, 0.5569, 0.8764, 0.7314, 0.5474, 0.9083, 0.6677, 0.6286, 0.6759, 0.5397, 0.5748, 0.6215, 0.4800, 0.5206, 0.5591, 0.5884, 0.6291, 0.6633, 0.7693, 0.5104, 0.6564, 0.5489, 0.6270, 0.5935, 0.6236, 0.6108, 0.4794, 0.5974, 0.7061, 0.6686, 0.6512, 0.4998, 0.5933, 0.4956, 0.6610, 0.7542, 0.5869, 0.8418, 0.9938, 0.9021, 0.6323, 0.5777, 0.4343, 0.6098, 0.5338, 0.5906, 0.7783, 0.7423, 0.6426, 0.6236, 0.9643, 0.5780, 1.0100, 1.1266, 0.7556, 0.5229, 0.8272, 0.6900, 0.5175, 0.4124, 0.5741, 0.4516, 0.6266, 0.5630, 0.5275, 0.5692, 0.5075, 0.7549, 0.6359, 0.5804, 0.6680, 0.7558, 0.6250, 0.4314, 0.6496, 0.5479, 0.7524, 0.7088, 0.6644, 0.7214, 0.6450, 0.4467, 0.7789, 0.5168, 0.6297, 0.6242, 0.4410, 0.8372, 0.5758, 0.4997, 0.8915, 0.6473, 0.5974, 0.5293, 0.7941, 0.4605, 0.9110, 0.5919, 0.5139, 0.5003, 0.4500, 0.6182, 0.5807, 0.4562, 0.5618, 0.6794, 0.7201, 0.6143, 0.8797, 0.8171, 0.6225, 0.7453, 0.7611, 0.4696, 1.0906, 0.8825, 0.7207, 0.5523, 0.7120, 0.5194, 0.5321, 1.0233, 0.5618, 0.5410, 0.4300, 0.7191, 0.5373, 0.4795, 0.4450, 0.6546, 0.7965, 0.7454, 0.6264, 0.5576, 0.7710, 0.5527, 0.6586, 0.5177, 0.4858, 0.5005, 0.5372, 0.5766, 0.4508, 0.5238, 0.8275, 0.4104, 0.5535, 0.8077, 0.4460, 0.7125, 0.7166, 0.6107, 0.4561, 0.6620, 0.4635, 0.6397, 0.4391, 0.6880, 0.6801, 0.5627, 0.8076, 0.7918, 1.0309, 0.5832, 0.6152, 0.7971, 0.4539, 0.5846, 0.7248, 0.4455, 0.6318, 0.6118, 0.4552, 0.6757, 0.5354, 0.6566, 0.6728, 0.4383, 0.6899, 1.0565, 0.6028, 0.6937, 0.5518, 0.8039, 0.4296, 0.6068, 0.5736, 0.4923, 0.7643, 0.7391, 0.4975, 0.5006, 0.5674, 0.5170, 0.4835, 0.4286, 0.5667, 0.6109, 0.6465, 0.6281, 0.7791, 0.5174, 0.5058, 0.6196, 0.6593, 0.5999, 0.5012, 0.5414, 0.7151, 0.6546, 0.6790, 0.5412, 0.4801, 0.6561, 1.0082, 0.5567, 0.6362, 0.4540, 0.8812, 0.6893, 0.6420, 0.6078, 0.5117, 0.7079, 0.8240, 0.7587, 0.6344, 0.6848, 0.4633, 0.5352, 0.6077, 0.5436, 0.7223, 0.5001, 0.9734, 0.5155, 0.5549, 0.4711, 0.9038, 0.5415, 1.0173, 0.5001, 0.5290, 0.5228, 0.5619, 0.9670, 0.7854, 0.5350, 0.5183, 0.9770, 0.5547, 0.9710, 0.5050, 0.4584, 0.6438, 0.4854, 0.5949, 0.6611, 0.4676, 0.4815, 0.8837, 0.6425, 0.6257, 0.6896, 0.4465, 0.7492, 0.6293, 0.7096, 0.5578, 0.5117, 0.4909, 0.5773, 0.4800, 0.5488, 0.6336, 0.6863, 0.5035, 0.6682, 0.7245, 0.5524, 0.4594, 0.5816, 0.5698, 0.6140, 0.5816, 0.5242, 0.4088, 0.4358, 0.6426, 0.4777, 0.6115, 0.4383, 0.5957, 0.8423, 0.5353, 0.5407, 0.8497, 0.6962, 0.7542, 0.5981, 0.5121, 0.6232, 0.5306, 0.5416, 0.5217, 0.5437, 0.5349, 0.5111, 0.8627, 0.6092, 0.5850, 0.5851, 0.7203, 0.3688, 0.5063, 0.5650, 0.5444, 0.5657, 0.7461, 0.4447, 0.7153, 0.4738, 0.5730, 0.4605, 0.4905, 0.6253, 0.8114, 0.8273, 0.5052, 0.6180, 0.6496, 0.4037, 0.5635, 0.5212, 0.7652, 0.4872, 0.5764, 0.7834, 0.6888, 0.5313, 0.5379, 0.5710, 0.7474, 0.6535, 0.9660, 0.5257, 0.7157, 0.7150, 0.5430, 0.5331, 0.6820, 0.6872, 0.4904, 0.6592, 0.6256, 0.6107, 0.4939, 0.5986, 0.5172, 0.4583],
+ },
+ "emdb": {
+ "count": 62707,
+ "mean": [-1.1869, 0.1485, 0.1933, -0.6247, 0.0793, 0.5762, 0.1835, -0.2564, 0.1285, 0.3221, 0.0577, 0.1154, -0.0818, -0.2512, 0.9673, -0.5680, 0.5968, -0.2124, -0.0112, -0.5576, 0.5339, -0.1490, 0.3102, -0.4012, -0.0570, 0.6416, 0.9359, -0.2932, 0.8544, 0.1719, -0.4534, 0.1316, 0.8625, 0.3806, 0.4884, 1.0853, -0.3872, -0.2403, -0.4274, 0.1319, -0.3334, 0.6352, 0.5748, -0.8850, -0.4331, 0.3662, -0.3324, 1.3993, -1.5142, -0.3082, -0.5491, -0.1847, 0.0145, -0.0726, 0.0015, -0.0358, -0.2815, -0.4356, -0.3842, 0.1150, 1.1513, 0.6343, -0.7336, -1.1613, 0.1020, -0.1291, 0.1560, 0.4854, -0.4191, 1.6794, 0.4274, 0.4792, 0.3570, 0.0811, 1.0886, 0.0670, 0.5227, 0.1891, 0.1121, 0.1495, -0.2090, -0.2156, -0.2512, -0.9291, 0.1287, -0.0481, 0.6701, -0.4579, 0.2352, -0.1056, 0.5551, 0.4357, 0.8168, 0.6344, -0.6445, -0.1965, 0.5587, 0.3860, -0.2466, -0.1542, 0.6825, 0.5875, -0.5208, 0.1500, -0.3980, 0.2157, 0.8368, -0.1356, -0.3387, 0.1747, 0.1467, 0.2282, -0.1412, 0.6216, -1.8406, 0.0150, 0.2891, 0.0280, 0.0461, 0.8558, 0.2929, -1.3753, -0.5792, 0.2089, -0.3524, -0.1849, -0.0157, 0.4454, -0.5306, 0.8238, -0.3160, 0.3760, 0.8978, -0.1943, -0.9474, -1.7321, -0.0149, 0.2338, 0.6087, -0.4851, 0.5210, -0.4042, -0.5368, -0.6220, 0.1245, 0.3112, 0.6360, -0.1522, 0.0540, -0.2380, -0.8354, 1.7591, 0.5687, 0.1732, 0.7923, -0.5383, -0.3271, -2.0050, -0.5563, 0.2979, 1.6609, 0.7108, -1.0155, 0.3591, 0.0136, -0.4743, -0.5401, -0.0176, 1.3333, -0.2973, -0.1114, -0.1616, 0.1160, 0.1152, 0.0057, 0.2067, 0.3876, -1.5311, 0.0636, 0.4566, -0.2653, 1.0534, -0.4638, 0.2166, 0.8686, -0.1447, 0.5605, -0.3841, 0.7015, 0.0418, 0.0811, -0.6406, -0.2929, -0.6821, 1.3678, 0.7574, 0.8315, 2.0377, 4.9034, -0.0097, 0.0165, 0.3248, 0.2994, 0.0210, 0.2276, -0.6580, -0.6899, 0.1981, -2.3205, 0.0059, -0.9412, -0.3191, 0.0389, -0.4170, 0.3391, -0.1346, 0.1567, 0.1838, -0.4176, -0.2758, 0.1495, -0.2977, 0.0929, 0.7186, 0.1230, 0.8780, -0.1240, -0.7370, -0.7551, 0.3830, 1.0824, 1.4500, -0.1040, 1.4225, 0.0929, 0.4612, 0.5167, -0.7093, -0.4729, 0.2321, 0.4156, -0.0696, -0.0626, 1.3341, -0.2398, 0.8453, 0.4048, 0.1690, 0.0074, -0.0474, 0.4134, 0.2043, -0.5962, 0.1643, -0.3821, 0.3012, -0.5690, 0.0133, 0.1876, -0.0727, 0.2896, 0.3253, 0.0313, 0.5141, -0.0055, -1.2889, -0.0983, -0.3212, -0.4173, -0.0804, 0.2591, -0.4160, -0.4815, 2.2822, -1.0033, -0.9814, 0.5290, 1.7943, -0.4217, -0.0373, -3.3970, 3.3067, 0.1174, -0.1369, 0.3847, -0.6960, -0.8867, -0.3825, -0.0134, -0.4367, -1.0273, -0.0623, 0.1520, 0.3816, -0.6543, -0.0118, -0.3019, -0.1190, 1.0490, 0.6255, 0.8503, 0.9500, -1.1942, 1.6886, -1.3958, 0.9389, 0.2318, -0.0460, 0.1140, -0.2352, -0.5648, 0.0363, -0.5636, 0.0661, -0.8680, -0.1223, -6.5336, 0.2139, -0.2734, 1.1739, 0.6003, 0.2183, 0.2154, -0.5902, -0.2916, -0.2748, 0.0787, 0.9065, -0.9764, -0.2278, 1.6248, 0.7941, -0.5014, 0.2422, -2.1474, 0.7818, 0.4370, 0.1361, -0.3936, -0.7724, 0.0941, -0.5762, 3.2182, -0.1101, 0.2677, -0.0101, -1.1798, -0.0122, -0.8163, 0.1115, -0.1697, -0.1466, -0.3549, 0.5360, -0.5183, 0.7519, 0.7093, -0.5946, 0.2787, 0.4822, -0.2680, 0.0934, 0.1483, 0.6706, -0.1150, -0.1945, -2.6643, 0.2194, -0.5014, -0.5869, 0.1022, 0.1988, -0.2558, 0.3732, -0.0644, 0.6440, -0.7403, -1.0228, 0.8158, 0.9543, -0.1226, -0.0929, 0.2716, 0.7962, -0.5293, 0.1538, -1.2074, -0.5093, 0.2037, 0.2156, -0.4407, 2.6976, -0.3653, 0.0458, -0.0899, -0.7584, 1.8329, -0.5082, -0.4776, -0.0265, -2.9437, -0.1675, 1.2358, 0.1571, -0.5022, -0.6370, 0.4087, -0.9664, 0.3533, 0.0928, -0.5308, 0.4462, 0.2476, 0.0976, -1.8347, 0.0468, -0.9309, -0.3712, -0.8578, -0.0568, 1.7377, -0.1299, -0.7187, 0.9764, 0.6858, 0.4272, -0.9588, 0.1038, 0.2520, -1.3775, 0.1491, -0.8507, 0.7052, 0.6483, 0.2818, -0.3305, -0.5913, -0.0907, -0.2438, -0.1932, -0.0564, -0.0777, -0.0748, 0.6530, 0.2393, 0.4476, 0.3941, -1.7061, 0.8876, 1.1888, 0.1423, 0.1737, 0.1330, 0.1115, 0.1525, -0.3715, 0.4657, -0.4010, -0.3089, 2.0455, -0.9555, 0.5093, 0.1502, -0.0865, -0.7851, -0.5175, 0.1613, 0.8113, 1.1943, 0.0612, 1.7087, -1.1616, -0.3204, 0.4428, 0.6120, -0.2282, 0.0174, -0.3141, -0.0045, 0.2204, 0.3966, 4.1174, -0.1531, 0.4325, -0.0245, -0.0310, 0.6541, 0.2904, 1.9309, -0.5405, 0.8576, 1.0352, -0.3592, -0.1056, -0.0047, 0.7218, 0.2350, 1.8817, 0.7558, -0.1575, -0.0544, 0.0234, 0.5841, 0.0996, -0.0503, 1.4150, 0.2260, 0.9152, 0.0688, 0.5286, 0.5885, 0.4606, -0.9186, 0.0441, 0.5233, 0.5305, -0.9086, 0.3728, 0.6752, 0.5453, -1.1360, 0.0613, -0.2365, 0.8856, -0.0512, -0.2589, -0.7055, -0.8111, 0.1787, 1.0393, -0.2469, -0.0922, 1.1790, -0.3284, 0.0402, 0.0746, -0.1033, -0.7248, -1.3859, -1.0511, 0.2797, 0.2777, -0.0877, 0.0271, 0.0740, -1.5863, -0.7014, 0.3677, -1.6786, -1.0769, 0.5594, 0.2428, -0.2664, 0.3454, -0.0490, -3.3762, 0.2004, 0.1913, -0.6461, 0.7643, -0.1239, 1.6487, 0.4942, -0.3305, -0.5069, -0.2183, 1.1533, -0.4380, 0.0219, -0.6319, 0.6743, 1.0648, 0.0587, -0.0989, -0.0995, 0.3757, 0.1813, 0.2854, 0.4345, -2.2154, 0.3601, -0.6406, -0.1099, 0.3583, -0.3726, 0.2892, 0.5897, 3.4282, -2.8781, 0.8985, 0.1550, 0.1102, 0.8008, -0.0811, -0.4199, 0.3145, -0.3236, -0.2425, -0.4502, 0.2431, 0.8504, 0.4597, 0.6396, 0.0902, 1.3885, 0.1297, -1.1721, -0.3227, -0.4472, 0.2575, 1.6201, -0.5444, 0.8665, -0.9622, 0.0035, -0.5908, 1.6270, 0.0351, -0.3419, 0.0039, 1.1001, -0.3767, -0.2270, 1.3332, 0.3555, 0.0667, -0.5392, -1.3500, -0.0842, 0.2591, -2.8862, 0.3166, 2.3757, 1.1254, -0.5208, -0.7074, -0.8110, 0.3715, 1.3720, -0.7236, -0.0665, 0.2772, -0.2840, -0.3515, -0.4777, 0.3030, 0.5417, 0.7752, -0.0182, 1.1569, -0.1614, 1.6521, -2.2844, -0.9332, -0.1472, 0.6151, -0.5020, -0.0719, 0.3361, -0.2722, -0.1500, 0.5092, -0.0348, -0.6530, -0.4159, -0.6603, -3.6738, 0.1421, -1.1267, 0.4267, 0.0699, 1.6415, 0.1451, 1.3309, 0.7792, 2.1801, -0.0886, 0.4233, 0.2828, 1.3708, -1.2021, -0.2627, -2.1505, 0.7701, -0.0167, -0.0247, 0.4665, -1.5951, -0.9997, -0.1568, -0.1108, 0.1543, -1.0055, 0.0001, -1.0355, 0.8421, -0.0485, -0.3064, 1.2358, -0.0448, 0.4038, -0.7671, 0.3624, -0.6197, 0.7966, -0.2266, 0.1130, -0.5302, -1.5468, 0.0700, -1.1711, -0.3307, 0.0086, -0.0416, 1.2763, -0.0574, 0.0121, -2.6334, -0.3180, -0.1954, 0.3944, 0.0076, 1.2025, -0.5634, 0.9271, 0.4198, 0.3251, -0.0041, 0.5236, -0.5314, 0.0639, -0.8840, -0.2680, 0.4958, 0.7804, 0.2942, -0.1935, -0.1405, -0.5670, 0.9489, 0.5726, -0.2529, 1.8878, -0.7204, -0.0050, -0.2448, 0.1725, 0.4253, 0.0058, 1.0247, -0.2908, -0.3978, -0.0963, 0.2107, 1.3576, 0.3074, 0.5527, -0.0927, 0.1521, 0.6300, -0.1377, -0.0497, 0.0425, -0.2248, -0.1534, 0.5778, 0.0033, 0.1789, 2.4935, 1.3225, -0.8038, -0.8864, 0.1176, -1.0532, -0.2375, 1.4582, -0.1168, 0.0548, 0.4221, -0.3585, 0.4043, -0.4371, 1.3289, -0.3674, -0.4286, -0.1730, 0.0535, 0.1441, 0.2703, 0.3826, 0.5123, -0.0401, -0.1230, -0.3143, -1.7583, 0.2582, 0.3484, 0.5722, 0.8621, 0.4420, 0.4442, -0.2445, 0.0532, -0.8102, -1.4058, -0.6382, -0.5799, -0.2456, -0.0906, -0.3191, -0.3395, -0.4364, -0.5810, -0.7970, 0.0831, -1.1570, -0.2573, -0.0644, -0.7106, 0.1313, 0.1944, -0.2329, 0.1409, -1.2096, 1.0822, 0.5523, 0.2151, -0.1106, -0.1034, -0.4873, 0.6932, 1.0196, -0.0521, 0.0569, -0.8759, 1.0084, 0.6800, 1.0768, -1.2878, -0.1161, 0.0447, 0.1888, -0.2371, -1.0470, -0.4027, -1.4363, 0.1606, -0.8026, -0.0244, -0.2893, -0.4938, -0.6921, 1.0140, -0.4158, -0.5957, 0.3313, -0.2462, 0.7703, 0.3403, -1.5113, -0.1231, -0.3776, 0.3326, 0.1634, 2.1520, 0.7302, -0.0300, -2.8234, 0.4553, -0.4652, -0.3331, -1.0286, 1.2882, -0.2797, -0.4759, 0.1470, -1.0253, -0.8175, 0.6936, -0.3728, -0.4594, 1.0876, 0.6229, -0.0461, -0.4342, -0.1686, -1.3960, 0.5283, 0.4002, 0.8179, 0.4787, -0.7147, -0.5052, -0.2552, 0.2817, -0.4022, -0.5289, 0.0815, -0.4814, -0.5451, -0.1384, -0.4303, -0.4506, -1.9036, 0.6884, 0.1361, 0.2678, -0.0052, 0.0119, -0.1882, 1.0507, 3.1094, -0.5746, 1.3087, -0.1831, -0.1917, 0.0633, 0.5083, -0.1448, -0.0134, 0.5002, 0.2579, 0.7755, 0.1579, 0.4157, -0.2610, -0.4953, 0.1709, 0.4063, 0.2068, 0.2666, -0.7872, 0.5325, 0.4910, -0.1599, 0.4387, 0.9262, 0.9245, 0.5763, 0.9292, -0.4531, -0.5367, -0.4911, 0.2302, -0.4182, 0.7188, 0.0342, -0.2079, 0.1310, 0.5718, -0.0331, 0.1861, -0.1287, -0.0427, 0.8478, 0.7278, -0.5664, -0.5335, 1.3976, 0.1697, 0.6063, -0.0220, 0.4921, -0.1349, -0.0531, -0.2408, -0.3858, -0.2741, 0.2285, -0.5532, 0.2704, -0.2687, -0.2161, -0.1179, -1.5228, -0.3683, -1.3004, 0.2431, -0.3305, 1.6118, -0.0328, 1.1503, 0.5712, -0.0423, 3.4830, -0.2760, 0.6307, -0.0419, 0.1553, -0.5602, 0.2106, -0.2213, -0.4543, 0.3034, 0.9189, 1.5738, 0.5071, 0.2238, -2.2069, 0.4104, 0.6224, 0.2836, -0.1620, -0.3043, -0.4012, 0.2410, -0.6261, -0.2435, 0.0211, -0.2227, -0.2392, -0.3634, -0.9207, 0.2260, 0.0929, 0.8206, -0.3214, -0.2296, 0.1274, -0.8615, 0.2329, 1.1085, -1.0565, 0.2258],
+ "std": [0.9963, 0.6391, 0.4956, 0.6280, 0.7591, 0.5610, 0.8236, 0.7139, 0.7494, 0.5686, 0.5042, 0.3464, 0.4228, 0.4171, 0.3526, 0.3710, 0.6288, 0.4674, 0.4413, 0.4741, 0.6553, 0.4882, 0.3697, 0.5507, 0.4961, 0.3683, 0.5604, 0.5302, 0.6027, 0.3023, 0.4882, 0.5746, 0.5314, 0.5031, 0.6145, 0.5994, 0.4285, 0.6399, 0.5362, 0.5403, 0.4677, 1.2902, 0.6126, 0.4145, 0.5068, 0.4667, 0.4825, 0.4275, 0.4381, 0.6758, 0.4866, 0.4136, 0.5262, 0.5698, 0.6550, 0.6492, 0.3450, 0.5948, 0.4219, 0.4973, 0.4483, 0.4336, 0.7440, 0.4595, 0.4366, 0.3634, 0.4430, 0.6587, 0.5073, 0.3533, 0.7036, 0.7039, 0.5312, 0.4701, 0.7512, 0.4102, 0.4227, 0.4488, 0.4158, 0.4676, 0.4521, 0.4560, 0.3917, 0.4757, 0.4348, 0.6013, 0.6715, 0.5179, 0.4834, 0.7451, 0.4845, 0.8893, 0.4188, 0.5963, 0.4306, 0.4551, 0.6417, 0.2886, 0.5378, 0.4316, 0.7568, 0.4818, 0.5494, 0.4736, 0.5841, 0.5043, 0.4265, 0.6994, 0.7652, 0.4344, 0.3931, 0.7198, 0.4169, 0.5794, 0.6720, 0.5694, 0.8603, 0.5307, 0.5893, 0.5763, 0.5292, 0.5228, 0.4156, 0.4901, 0.8334, 0.4574, 0.7241, 0.5346, 0.4063, 0.4147, 0.4979, 0.6599, 0.4173, 0.3715, 0.3828, 0.4492, 0.5576, 0.4060, 0.4353, 0.5315, 0.9834, 0.5548, 0.5679, 0.3506, 0.5419, 0.4256, 0.4187, 0.3570, 0.6316, 0.5870, 0.4832, 0.4862, 0.6072, 0.6781, 0.6152, 0.6708, 0.5008, 0.4435, 0.4229, 0.4973, 0.4301, 0.5363, 0.5478, 0.5388, 0.3952, 0.5961, 0.4721, 0.6389, 0.4450, 0.4841, 0.5594, 0.5234, 0.5224, 0.6326, 0.4469, 0.7397, 0.9551, 0.8426, 0.7576, 0.3893, 0.4382, 0.5222, 0.5234, 0.6035, 0.5764, 0.4043, 0.4741, 0.5471, 0.4229, 0.5962, 0.7127, 0.6205, 0.5671, 0.3766, 0.7455, 0.7315, 0.5891, 0.5372, 0.5957, 0.5342, 0.4010, 0.4453, 0.4609, 0.8789, 0.4353, 0.6297, 0.4126, 0.4149, 0.4597, 0.4859, 0.6733, 0.6096, 0.5719, 0.4494, 0.6353, 0.7537, 0.4643, 0.4577, 0.6485, 0.6069, 0.3603, 0.5821, 0.4807, 0.5192, 0.5329, 0.4153, 0.7329, 0.5444, 0.5742, 0.4593, 0.4003, 0.6770, 0.5428, 0.6781, 0.7920, 0.5037, 0.7615, 0.4537, 0.5931, 0.7333, 0.4880, 0.5469, 0.4698, 0.4917, 0.6256, 0.4947, 0.3974, 0.7559, 0.5916, 0.6547, 0.7502, 0.4682, 0.4517, 0.4888, 0.6472, 0.4755, 0.3927, 0.5845, 0.4135, 0.4091, 0.5860, 0.4544, 0.4051, 0.5547, 0.5322, 0.7200, 0.4595, 0.5484, 0.4758, 0.5259, 0.4137, 0.7149, 0.5638, 0.6221, 0.9309, 0.5637, 0.5657, 0.5711, 0.5651, 1.0484, 0.4435, 0.4587, 0.3716, 0.4108, 0.8114, 0.5531, 1.0675, 0.5825, 0.3841, 0.4500, 0.7335, 0.4767, 0.4162, 0.5679, 0.4880, 0.4614, 0.5118, 0.5198, 0.5619, 0.6869, 0.3536, 0.5128, 0.4722, 0.3722, 0.7705, 0.4556, 0.5365, 0.4999, 0.3254, 0.5268, 0.7580, 0.5932, 0.9908, 0.6171, 0.4912, 0.4439, 0.9135, 0.4658, 0.6566, 0.5500, 0.5423, 0.4725, 0.5415, 0.5550, 0.7519, 0.4220, 0.6024, 0.4821, 0.5268, 0.4583, 0.7421, 0.5200, 0.4541, 0.5197, 0.4562, 0.8381, 0.4423, 0.7400, 0.6578, 0.6459, 0.5316, 0.6877, 0.5362, 0.4215, 0.6455, 0.4363, 0.6716, 0.5795, 0.5587, 0.5234, 0.4456, 0.4991, 0.4244, 0.8959, 0.4744, 0.4440, 0.4437, 0.8485, 0.4237, 0.6907, 0.5582, 0.4315, 0.5458, 0.7341, 0.4731, 0.5065, 0.6181, 0.5643, 0.4407, 0.4353, 0.4732, 0.3769, 0.4162, 0.5028, 0.3689, 0.6656, 0.4598, 0.3735, 0.6801, 0.7902, 0.7101, 0.6292, 0.5732, 0.7452, 0.6803, 0.5065, 0.5261, 0.4644, 0.5021, 0.6714, 0.5226, 0.4455, 0.7599, 0.4380, 0.5468, 0.4595, 0.5308, 0.8445, 0.4413, 0.5196, 0.9241, 0.5414, 0.5018, 0.3832, 0.4950, 0.3185, 0.5330, 0.4844, 0.4481, 0.4517, 0.5104, 0.6092, 0.5712, 0.4164, 0.6590, 0.4888, 0.3930, 0.5419, 0.5486, 0.5165, 0.4390, 0.5542, 0.3883, 0.4074, 0.6213, 0.6185, 0.7711, 0.6565, 0.4925, 1.0624, 0.4690, 0.7498, 0.5333, 0.5290, 0.6258, 0.4473, 0.3862, 0.6571, 0.4873, 0.5240, 0.4127, 0.4445, 0.5094, 0.4754, 0.5769, 0.4786, 0.4510, 0.5130, 0.4897, 0.7568, 0.7398, 0.5718, 0.4229, 0.4929, 0.7470, 0.5901, 0.3772, 0.4914, 0.4074, 0.9471, 0.4967, 0.4323, 0.5259, 0.3591, 0.7202, 0.6012, 0.4573, 0.4296, 0.5578, 0.5218, 0.4640, 0.4522, 0.4029, 0.8071, 0.6086, 0.4832, 0.4202, 0.6781, 0.3862, 0.3920, 0.7543, 1.0257, 0.8849, 0.4181, 0.4722, 0.4069, 0.4854, 0.5405, 0.4676, 0.5547, 0.6282, 0.4275, 0.8011, 0.3308, 0.7135, 0.4315, 0.4915, 0.6616, 0.7376, 0.5742, 0.7461, 0.5443, 0.4749, 0.4906, 1.0020, 0.6306, 0.4435, 0.4559, 0.4360, 0.4047, 0.5802, 0.6109, 0.7836, 0.5163, 0.9777, 0.9272, 0.4618, 0.3534, 0.5218, 0.4479, 0.6498, 0.7145, 0.6224, 0.5671, 0.5042, 0.3885, 0.4079, 0.4481, 0.5406, 0.6944, 0.3744, 0.5942, 0.6770, 0.5934, 0.7417, 0.5662, 0.4753, 0.5063, 0.5003, 0.4510, 0.4358, 0.6455, 0.7740, 0.4780, 0.8687, 0.5533, 0.5700, 0.3518, 0.4868, 0.4154, 0.4798, 0.3266, 0.3536, 0.3789, 0.3805, 0.7909, 0.5760, 0.5784, 0.4993, 0.5787, 0.5324, 0.4496, 0.8483, 1.0794, 0.4820, 0.4135, 0.6231, 0.4668, 0.6684, 0.7052, 0.7616, 0.4881, 0.4150, 0.5793, 0.8068, 0.7793, 0.4721, 0.5230, 0.4810, 0.9577, 0.5537, 0.5583, 0.6645, 0.4334, 0.6398, 0.5011, 0.4081, 0.6255, 0.5372, 0.4846, 0.6125, 0.6509, 0.4413, 0.4762, 0.4917, 0.5940, 0.4950, 0.6753, 0.6653, 0.5210, 0.5599, 0.4678, 0.4868, 0.5985, 0.4160, 0.4874, 0.8380, 0.5382, 0.5701, 0.5448, 0.6131, 0.5674, 0.7120, 0.4070, 0.4434, 0.5725, 0.4919, 0.4805, 0.5997, 0.7108, 0.9824, 0.4765, 0.7575, 0.4452, 0.8892, 0.4639, 0.4962, 1.0346, 0.7584, 0.4312, 0.4835, 0.8968, 0.4799, 0.6864, 0.5641, 1.0694, 0.6750, 0.4288, 0.5159, 0.3649, 0.7699, 0.4386, 0.4449, 0.3923, 0.6499, 0.5612, 0.4541, 0.6261, 0.5444, 0.4369, 0.4124, 0.4174, 0.6129, 0.5005, 0.4779, 0.3929, 0.4865, 0.4338, 0.4114, 0.6266, 0.3669, 1.0147, 0.4856, 0.4867, 0.6250, 0.5368, 0.6699, 0.6411, 0.5296, 0.7614, 0.5643, 0.5843, 0.6846, 0.3923, 0.3928, 0.4964, 0.4490, 0.4755, 0.4104, 0.5468, 0.6040, 0.5808, 0.6283, 0.4316, 0.6127, 0.4635, 0.5303, 0.4261, 0.4668, 0.6121, 0.4063, 0.5571, 0.6130, 0.5874, 0.4987, 0.4113, 0.5401, 0.4028, 0.6598, 0.7740, 0.5384, 0.7890, 0.9379, 0.8801, 0.6222, 0.5356, 0.3990, 0.4802, 0.4107, 0.5475, 0.6936, 0.6865, 0.4776, 0.5211, 0.8844, 0.6517, 1.0729, 0.9252, 0.6953, 0.4177, 0.7587, 0.6628, 0.3629, 0.3685, 0.3758, 0.4439, 0.5236, 0.4905, 0.5290, 0.4184, 0.3940, 0.6498, 0.5411, 0.5662, 0.5519, 0.6107, 0.6385, 0.4127, 0.6277, 0.5255, 0.5926, 0.5653, 0.6570, 0.6034, 0.5312, 0.4128, 0.7292, 0.3620, 0.5067, 0.5314, 0.3908, 0.7561, 0.4494, 0.4501, 0.7682, 0.4939, 0.4198, 0.5256, 0.6339, 0.5123, 0.9018, 0.5054, 0.4879, 0.4567, 0.4145, 0.6046, 0.3835, 0.4289, 0.5254, 0.6191, 0.6610, 0.5933, 0.7890, 0.7817, 0.6299, 0.5977, 0.7094, 0.3737, 1.0318, 0.7045, 0.7785, 0.5376, 0.5861, 0.4233, 0.5538, 1.0604, 0.5690, 0.5249, 0.3747, 0.6036, 0.4707, 0.3617, 0.3665, 0.6184, 0.4878, 0.6193, 0.5311, 0.6187, 0.6748, 0.4493, 0.6137, 0.4601, 0.3855, 0.4183, 0.4986, 0.4832, 0.4192, 0.4416, 0.7202, 0.3724, 0.4899, 0.6939, 0.4272, 0.7122, 0.6950, 0.5565, 0.4417, 0.6186, 0.4753, 0.5919, 0.3763, 0.5643, 0.5347, 0.5454, 0.9336, 0.6594, 0.9747, 0.4970, 0.4725, 0.7820, 0.4113, 0.4942, 0.6699, 0.4159, 0.6766, 0.6564, 0.3947, 0.5381, 0.3874, 0.6686, 0.5628, 0.3904, 0.6647, 0.9821, 0.4343, 0.5455, 0.4879, 0.8165, 0.4153, 0.5544, 0.5179, 0.3821, 0.6678, 0.7883, 0.3372, 0.4702, 0.5044, 0.4584, 0.4769, 0.3787, 0.4377, 0.5435, 0.5899, 0.5378, 0.5986, 0.4887, 0.5390, 0.5464, 0.6330, 0.5010, 0.4244, 0.5249, 0.6770, 0.6314, 0.6404, 0.4605, 0.3649, 0.6489, 1.0657, 0.5497, 0.5357, 0.3651, 0.8484, 0.8126, 0.4873, 0.6711, 0.4401, 0.6181, 0.8585, 0.6000, 0.5654, 0.5416, 0.3504, 0.4671, 0.5499, 0.4409, 0.7650, 0.4980, 0.9734, 0.3568, 0.6037, 0.4361, 0.7880, 0.4726, 0.9902, 0.5020, 0.5178, 0.5065, 0.4543, 0.9039, 0.8296, 0.4451, 0.4436, 0.8518, 0.5201, 0.8668, 0.5122, 0.3412, 0.5849, 0.4815, 0.5795, 0.5664, 0.4384, 0.4593, 0.7974, 0.6570, 0.6522, 0.5490, 0.4195, 0.6821, 0.6133, 0.5692, 0.4780, 0.4574, 0.5090, 0.4488, 0.4269, 0.4153, 0.5143, 0.6560, 0.4480, 0.5482, 0.6997, 0.4377, 0.4166, 0.6103, 0.4671, 0.4449, 0.5672, 0.3296, 0.3898, 0.3778, 0.6572, 0.5555, 0.4047, 0.3720, 0.5728, 0.6867, 0.5435, 0.5001, 0.6808, 0.6373, 0.6849, 0.4826, 0.4767, 0.3736, 0.5070, 0.4442, 0.4302, 0.4339, 0.4614, 0.4735, 0.7977, 0.5657, 0.4047, 0.5261, 0.6204, 0.3413, 0.3996, 0.4236, 0.3303, 0.4193, 0.6074, 0.3941, 0.4802, 0.4114, 0.3880, 0.3460, 0.3767, 0.6491, 0.6893, 0.8560, 0.4244, 0.4307, 0.5702, 0.3635, 0.5170, 0.3975, 0.6187, 0.5012, 0.4976, 0.7149, 0.7001, 0.4834, 0.3844, 0.5179, 0.6909, 0.5862, 1.0062, 0.5099, 0.6410, 0.7432, 0.4219, 0.4655, 0.6067, 0.6674, 0.4618, 0.7115, 0.5300, 0.5284, 0.4208, 0.4955, 0.4561, 0.3723],
+ }
+}
+
+cam_angvel = {
+ "emdb_none_test": {
+ "count": 42622,
+ "mean": [1., 0., 0., 0., 1., 0.],
+ "std": [5.5702e-05, 3.2200e-03, 5.6530e-03, 3.2191e-03, 2.4738e-05, 3.3406e-03],
+ },
+ "manual": {
+ "mean": [1., 0., 0., 0., 1., 0.],
+ "std": [0.001, 0.1, 0.1, 0.1, 0.001, 0.1], # manually
+ }
+}
+
+cam_tvel = {
+ 'manual': {
+ 'mean': [-1.7233e-05, 6.1458e-06, 7.7848e-04],
+ 'std': [0.0139, 0.0111, 0.0146]
+ }
+}
+# fmt:on
+
+# ====== Compose ====== #
+
+
+def compose(targets, sources):
+ if len(sources) == 1:
+ sources = sources * len(targets)
+ mean = []
+ std = []
+ for t, s in zip(targets, sources):
+ mean.extend(t[s]["mean"])
+ std.extend(t[s]["std"])
+ return {"mean": mean, "std": std}
+
+
+DEFAULT_01 = {"mean": [0.0], "std": [1.0]}
+
+MM_V1 = compose(
+ [body_pose_r6d, betas, global_orient_c_r6d, global_orient_gv_r6d, local_transl_vel],
+ ["bedlam"] * 5,
+)
+MM_V1_AMASS_LOCAL_BEDLAM_CAM = compose(
+ [body_pose_r6d, betas, global_orient_c_r6d, global_orient_gv_r6d, local_transl_vel],
+ ["amass", "amass", "bedlam", "bedlam", "amass"],
+)
+
+MM_V2 = compose(
+ [body_pose_r6d, betas, global_orient_c_r6d, global_orient_gv_r6d, local_transl_vel],
+ ["bedlam", "bedlam", "bedlam", "bedlam", "none"],
+)
+
+MM_V2_1 = compose(
+ [body_pose_r6d, betas, global_orient_c_r6d, global_orient_gv_r6d, local_transl_vel],
+ ["bedlam", "bedlam", "bedlam", "bedlam", "1e-2"],
+)
+
+# ====== Unity Dataset Statistics ====== #
+# Computed from Unity processed data (11997 frames)
+
+body_pose_r6d["unity"] = {
+ "count": 11997,
+ "mean": [0.9924, -0.0491, 0.0246, 0.0435, 0.9702, 0.1913, 0.9851, 0.1239, -0.0341, -0.1161, 0.9608, 0.1993, 0.9954, 0.0867, -7.9529e-03, -0.0873, 0.9926, -0.0668, 0.9999, -5.4245e-03, 7.6656e-04, 5.3467e-03, 0.9432, -0.2797, 0.9997, 0.0115, -2.7577e-03, -0.0112, 0.9211, -0.2825, 0.9915, -0.1267, -2.3587e-03, 0.1266, 0.9911, 0.0400, 0.9931, 0.0408, 5.0597e-03, -0.0424, 0.9783, 0.1149, 0.9894, -0.0670, -0.0346, 0.0696, 0.9813, 0.0514, 1.0000, 0.0, 0.0, 0.0, 1.0000, 0.0, 0.9999, -1.3002e-03, -6.7745e-04, 1.4068e-03, 0.9982, 8.8789e-03, 0.9999, 1.9599e-03, 1.0788e-03, -2.1213e-03, 0.9978, 0.0112, 0.9888, 0.0920, 0.0708, -0.1028, 0.9804, 0.1672, 0.9925, 0.0972, 0.0164, -0.0949, 0.9880, -0.0987, 0.9870, -0.1341, -0.0418, 0.1288, 0.9825, -0.1130, 0.9872, -0.0796, 0.0647, 0.0944, 0.9582, -0.2484, 0.3496, 0.8362, -0.1972, -0.7904, 0.4130, 0.2523, 0.4682, -0.7962, 0.2015, 0.7959, 0.4985, 0.1363, 0.8203, -0.0135, -0.2049, 0.0620, 0.9244, -0.0711, 0.7697, 0.0526, 0.2749, -0.0567, 0.9009, -0.1330, 0.9366, -0.0967, -0.1415, 0.0674, 0.9297, -0.1863, 0.9387, 0.0489, 0.0141, -0.0374, 0.9005, -0.2217],
+ "std": [8.3272e-03, 0.0892, 0.0633, 0.0896, 0.0294, 0.1068, 0.0163, 0.0940, 0.0636, 0.0956, 0.0362, 0.1152, 2.0401e-03, 0.0291, 0.0261, 0.0294, 4.3569e-03, 0.0415, 1.8386e-04, 0.0127, 2.9381e-03, 0.0125, 0.0786, 0.1604, 1.1208e-03, 0.0184, 0.0117, 0.0178, 0.1785, 0.1988, 1.3900e-03, 8.0863e-03, 0.0296, 8.1536e-03, 1.1073e-03, 0.0112, 9.6823e-03, 0.0614, 0.0908, 0.0616, 0.0491, 0.1477, 0.0104, 0.0523, 0.1120, 0.0527, 0.0554, 0.1543, 1.0e-06, 1.0e-06, 1.0e-06, 1.0e-06, 1.0e-06, 1.0e-06, 6.9881e-04, 8.9390e-03, 6.1391e-03, 9.0626e-03, 0.0207, 0.0552, 9.8994e-04, 0.0112, 7.4486e-03, 0.0118, 0.0227, 0.0600, 0.0117, 0.0116, 0.0924, 8.1376e-03, 2.2220e-03, 0.0166, 3.8557e-03, 0.0571, 0.0442, 0.0545, 5.3657e-03, 0.0453, 6.2895e-03, 0.0605, 0.0482, 0.0568, 6.1017e-03, 0.0454, 0.0192, 0.0763, 0.0929, 0.0627, 0.0280, 0.0800, 0.2001, 0.2204, 0.2262, 0.2535, 0.2111, 0.1795, 0.1490, 0.1964, 0.2131, 0.2117, 0.1377, 0.1887, 0.3686, 0.1906, 0.3356, 0.2142, 0.1650, 0.2519, 0.4057, 0.2444, 0.3239, 0.1914, 0.1934, 0.3058, 0.0861, 0.2387, 0.1705, 0.2379, 0.0873, 0.1796, 0.1005, 0.2807, 0.1655, 0.2843, 0.1099, 0.2136],
+}
+
+betas["unity"] = {
+ "count": 11997,
+ "mean": [0.1385, -1.9147, -0.5287, 0.6716, 0.0760, 0.8861, -0.3723, 0.7269, -0.4031, 1.0523],
+ "std": [0.01, 0.01, 0.01, 0.01, 0.01, 0.01, 0.01, 0.01, 0.01, 0.01], # Use small std since all same character
+}
+
+global_orient_c_r6d["unity"] = {
+ "count": 11997,
+ "mean": [0.6875, -0.0190, -0.0269, -0.0195, -0.9795, 0.0464],
+ "std": [0.3081, 0.0465, 0.6552, 0.1071, 0.0323, 0.1595],
+}
+
+global_orient_gv_r6d["unity"] = {
+ "count": 11997,
+ "mean": [0.6875, -0.0189, -0.0269, -0.0215, -0.9971, 0.0325],
+ "std": [0.3081, 0.0466, 0.6552, 0.0476, 4.8359e-03, 0.0439],
+}
+
+local_transl_vel["unity"] = {
+ "count": 11997,
+ "mean": [7.4442e-05, -2.7914e-05, 9.1121e-05],
+ "std": [5.7417e-03, 5.2172e-03, 2.4900e-03],
+}
+
+# Unity-specific composition
+MM_UNITY = compose(
+ [body_pose_r6d, betas, global_orient_c_r6d, global_orient_gv_r6d, local_transl_vel],
+ ["unity", "unity", "unity", "unity", "unity"],
+)
+
+HUMANML3D_V1 = {
+ "mean": [
+ 0.9821,
+ -0.0786,
+ 0.0149,
+ 0.0839,
+ 0.9101,
+ 0.1669,
+ 0.9821,
+ 0.0789,
+ -0.0153,
+ -0.084,
+ 0.9106,
+ 0.1661,
+ 0.9938,
+ 0.0001,
+ 0.0002,
+ -0.0,
+ 0.9312,
+ -0.2309,
+ 0.979,
+ 0.0637,
+ -0.0519,
+ -0.0555,
+ 0.8071,
+ -0.3257,
+ 0.979,
+ -0.0635,
+ 0.0519,
+ 0.0554,
+ 0.8073,
+ -0.3252,
+ 0.995,
+ 0.0,
+ -0.0001,
+ -0.0,
+ 0.9882,
+ -0.0333,
+ 0.9783,
+ 0.013,
+ 0.1243,
+ -0.0376,
+ 0.9587,
+ 0.1688,
+ 0.9783,
+ -0.013,
+ -0.1246,
+ 0.0378,
+ 0.9587,
+ 0.169,
+ 0.9978,
+ -0.0,
+ 0.0,
+ 0.0,
+ 0.9959,
+ 0.0069,
+ 0.9956,
+ -0.0145,
+ 0.0064,
+ 0.0129,
+ 0.9949,
+ 0.0126,
+ 0.9956,
+ 0.0145,
+ -0.0065,
+ -0.0129,
+ 0.9949,
+ 0.0126,
+ 0.9791,
+ -0.0001,
+ -0.0003,
+ 0.0001,
+ 0.9747,
+ -0.0,
+ 0.8793,
+ 0.3254,
+ -0.1433,
+ -0.3195,
+ 0.9095,
+ 0.0186,
+ 0.8793,
+ -0.3254,
+ 0.1434,
+ 0.3195,
+ 0.9095,
+ 0.0188,
+ 0.98,
+ -0.0001,
+ -0.0002,
+ 0.0002,
+ 0.9679,
+ -0.0583,
+ 0.6338,
+ 0.6403,
+ -0.2582,
+ -0.6392,
+ 0.6684,
+ 0.0207,
+ 0.6335,
+ -0.6405,
+ 0.2584,
+ 0.6396,
+ 0.6683,
+ 0.0207,
+ 0.5877,
+ -0.2228,
+ -0.5617,
+ 0.1476,
+ 0.9108,
+ -0.1817,
+ 0.5875,
+ 0.2228,
+ 0.5614,
+ -0.1477,
+ 0.9108,
+ -0.1821,
+ 0.9151,
+ -0.0893,
+ -0.1287,
+ 0.1066,
+ 0.875,
+ 0.0094,
+ 0.9151,
+ 0.0891,
+ 0.1288,
+ -0.1064,
+ 0.8751,
+ 0.0089,
+ 0.1178,
+ 0.0762,
+ 0.4235,
+ 0.0272,
+ 0.1815,
+ -0.1596,
+ 0.5019,
+ -0.2575,
+ -0.1448,
+ -0.5843,
+ 0.0037,
+ -0.0003,
+ 0.002,
+ 0.014,
+ -0.8609,
+ -0.0242,
+ 0.0035,
+ 0.0006,
+ 0.0018,
+ 0.0148,
+ -0.9519,
+ -0.0229,
+ -0.0005,
+ -0.0001,
+ 0.0031,
+ ],
+ "std": [
+ 0.0332,
+ 0.1283,
+ 0.1068,
+ 0.1185,
+ 0.1682,
+ 0.3075,
+ 0.0333,
+ 0.1285,
+ 0.1067,
+ 0.1184,
+ 0.1671,
+ 0.307,
+ 0.0099,
+ 0.0761,
+ 0.0799,
+ 0.0693,
+ 0.1329,
+ 0.239,
+ 0.0314,
+ 0.0963,
+ 0.1566,
+ 0.0793,
+ 0.3462,
+ 0.3367,
+ 0.0315,
+ 0.0962,
+ 0.1567,
+ 0.0793,
+ 0.3462,
+ 0.3366,
+ 0.0087,
+ 0.0664,
+ 0.0746,
+ 0.0634,
+ 0.0184,
+ 0.1343,
+ 0.0229,
+ 0.1183,
+ 0.1133,
+ 0.0988,
+ 0.0529,
+ 0.1959,
+ 0.0229,
+ 0.1181,
+ 0.1132,
+ 0.0986,
+ 0.0529,
+ 0.1959,
+ 0.0038,
+ 0.0401,
+ 0.052,
+ 0.0403,
+ 0.0072,
+ 0.081,
+ 0.0192,
+ 0.0719,
+ 0.0548,
+ 0.0739,
+ 0.0225,
+ 0.0626,
+ 0.019,
+ 0.0717,
+ 0.0548,
+ 0.0737,
+ 0.0225,
+ 0.0627,
+ 0.0351,
+ 0.1018,
+ 0.1728,
+ 0.1003,
+ 0.0388,
+ 0.1962,
+ 0.0822,
+ 0.2185,
+ 0.2142,
+ 0.2344,
+ 0.0683,
+ 0.1034,
+ 0.0823,
+ 0.2185,
+ 0.2142,
+ 0.2343,
+ 0.0684,
+ 0.1034,
+ 0.0327,
+ 0.1055,
+ 0.1656,
+ 0.1031,
+ 0.0439,
+ 0.2175,
+ 0.1751,
+ 0.2324,
+ 0.1924,
+ 0.2756,
+ 0.1695,
+ 0.1988,
+ 0.175,
+ 0.2322,
+ 0.1926,
+ 0.2751,
+ 0.1693,
+ 0.1987,
+ 0.4453,
+ 0.1619,
+ 0.2547,
+ 0.1981,
+ 0.0969,
+ 0.259,
+ 0.4461,
+ 0.1619,
+ 0.2544,
+ 0.198,
+ 0.0968,
+ 0.2587,
+ 0.127,
+ 0.3087,
+ 0.1633,
+ 0.2937,
+ 0.1664,
+ 0.3301,
+ 0.1272,
+ 0.3085,
+ 0.1634,
+ 0.2937,
+ 0.1665,
+ 0.3298,
+ 0.9774,
+ 0.8783,
+ 1.0809,
+ 1.5919,
+ 0.9995,
+ 1.411,
+ 0.9037,
+ 1.2517,
+ 1.1637,
+ 1.1538,
+ 0.6892,
+ 0.2422,
+ 0.6829,
+ 0.3238,
+ 0.1922,
+ 0.3409,
+ 0.6977,
+ 0.189,
+ 0.691,
+ 0.1656,
+ 0.1507,
+ 0.2075,
+ 0.0107,
+ 0.0082,
+ 0.0137,
+ ],
+}
+
+HUMANML3D_V2 = {
+ "mean": [
+ 0.9821,
+ -0.0788,
+ 0.015,
+ 0.0839,
+ 0.9104,
+ 0.1665,
+ 0.9822,
+ 0.0786,
+ -0.0151,
+ -0.0839,
+ 0.9105,
+ 0.1663,
+ 0.9939,
+ 0.0001,
+ 0.0,
+ -0.0001,
+ 0.9315,
+ -0.2305,
+ 0.979,
+ 0.0635,
+ -0.0518,
+ -0.0555,
+ 0.8076,
+ -0.3258,
+ 0.9791,
+ -0.0636,
+ 0.0519,
+ 0.0556,
+ 0.8076,
+ -0.3259,
+ 0.995,
+ -0.0001,
+ 0.0,
+ 0.0,
+ 0.9882,
+ -0.0333,
+ 0.9783,
+ 0.0129,
+ 0.1246,
+ -0.0376,
+ 0.9587,
+ 0.1688,
+ 0.9783,
+ -0.0129,
+ -0.1245,
+ 0.0376,
+ 0.9588,
+ 0.1691,
+ 0.9978,
+ -0.0,
+ -0.0001,
+ 0.0001,
+ 0.9959,
+ 0.0069,
+ 0.9956,
+ -0.0145,
+ 0.0064,
+ 0.0129,
+ 0.9949,
+ 0.0125,
+ 0.9956,
+ 0.0145,
+ -0.0064,
+ -0.0129,
+ 0.9949,
+ 0.0125,
+ 0.979,
+ 0.0,
+ 0.0003,
+ -0.0001,
+ 0.9745,
+ -0.0001,
+ 0.8793,
+ 0.3254,
+ -0.1431,
+ -0.3196,
+ 0.9096,
+ 0.0183,
+ 0.8793,
+ -0.3257,
+ 0.1431,
+ 0.3199,
+ 0.9095,
+ 0.0184,
+ 0.9799,
+ 0.0002,
+ -0.0002,
+ -0.0003,
+ 0.9679,
+ -0.0584,
+ 0.6333,
+ 0.6406,
+ -0.2583,
+ -0.6396,
+ 0.668,
+ 0.0205,
+ 0.6333,
+ -0.6411,
+ 0.2583,
+ 0.6401,
+ 0.668,
+ 0.0203,
+ 0.5877,
+ -0.2229,
+ -0.5616,
+ 0.1472,
+ 0.9108,
+ -0.1822,
+ 0.5881,
+ 0.2228,
+ 0.5614,
+ -0.1476,
+ 0.911,
+ -0.182,
+ 0.9152,
+ -0.0884,
+ -0.1288,
+ 0.1058,
+ 0.8751,
+ 0.0091,
+ 0.9154,
+ 0.0885,
+ 0.1288,
+ -0.106,
+ 0.8752,
+ 0.0094,
+ 0.1175,
+ 0.0753,
+ 0.4229,
+ 0.0269,
+ 0.1809,
+ -0.1599,
+ 0.5026,
+ -0.2562,
+ -0.1444,
+ -0.5831,
+ -0.0,
+ -0.0,
+ 0.0064,
+ 0.8612,
+ -0.036,
+ 0.0,
+ -0.0,
+ ],
+ "std": [
+ 0.0334,
+ 0.1284,
+ 0.1068,
+ 0.1184,
+ 0.1674,
+ 0.3071,
+ 0.0331,
+ 0.128,
+ 0.1066,
+ 0.1182,
+ 0.1673,
+ 0.3071,
+ 0.0098,
+ 0.076,
+ 0.0799,
+ 0.0691,
+ 0.1322,
+ 0.2384,
+ 0.0313,
+ 0.0961,
+ 0.1566,
+ 0.0793,
+ 0.3449,
+ 0.3368,
+ 0.0313,
+ 0.0961,
+ 0.1565,
+ 0.0792,
+ 0.3447,
+ 0.3366,
+ 0.0087,
+ 0.0664,
+ 0.0746,
+ 0.0634,
+ 0.0185,
+ 0.1343,
+ 0.0229,
+ 0.1183,
+ 0.1131,
+ 0.0987,
+ 0.0528,
+ 0.196,
+ 0.0229,
+ 0.1182,
+ 0.1131,
+ 0.0986,
+ 0.0526,
+ 0.1956,
+ 0.0038,
+ 0.0401,
+ 0.052,
+ 0.0404,
+ 0.0072,
+ 0.0809,
+ 0.0191,
+ 0.0719,
+ 0.0547,
+ 0.0739,
+ 0.0223,
+ 0.0624,
+ 0.0191,
+ 0.072,
+ 0.0547,
+ 0.0738,
+ 0.0223,
+ 0.0625,
+ 0.0351,
+ 0.102,
+ 0.173,
+ 0.1005,
+ 0.0393,
+ 0.1968,
+ 0.0823,
+ 0.2184,
+ 0.2142,
+ 0.2342,
+ 0.0684,
+ 0.1033,
+ 0.0823,
+ 0.2183,
+ 0.214,
+ 0.2341,
+ 0.0685,
+ 0.1033,
+ 0.0328,
+ 0.1053,
+ 0.166,
+ 0.1029,
+ 0.044,
+ 0.2173,
+ 0.1752,
+ 0.2324,
+ 0.1926,
+ 0.2758,
+ 0.1694,
+ 0.1989,
+ 0.1748,
+ 0.2318,
+ 0.1924,
+ 0.2747,
+ 0.1693,
+ 0.1988,
+ 0.4456,
+ 0.162,
+ 0.2545,
+ 0.1983,
+ 0.0967,
+ 0.2587,
+ 0.4455,
+ 0.1616,
+ 0.2542,
+ 0.1978,
+ 0.0964,
+ 0.2584,
+ 0.1268,
+ 0.3084,
+ 0.1637,
+ 0.2936,
+ 0.1664,
+ 0.3303,
+ 0.1266,
+ 0.3081,
+ 0.1635,
+ 0.2935,
+ 0.1662,
+ 0.33,
+ 0.9789,
+ 0.8786,
+ 1.081,
+ 1.593,
+ 1.0002,
+ 1.4113,
+ 0.9035,
+ 1.2523,
+ 1.1653,
+ 1.155,
+ 0.0332,
+ 0.0073,
+ 0.0151,
+ 0.1643,
+ 0.2451,
+ 0.001,
+ 0.1125,
+ ],
+}
diff --git a/genmo/network/utils.py b/genmo/network/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..dceb2264f37aa2c10d543bddd15cf7c246868a88
--- /dev/null
+++ b/genmo/network/utils.py
@@ -0,0 +1,39 @@
+import torch
+
+
+def load_and_freeze_llm(llm_version):
+ from transformers import T5EncoderModel, T5Tokenizer
+
+ tokenizer = T5Tokenizer.from_pretrained(llm_version)
+ model = T5EncoderModel.from_pretrained(llm_version)
+ # Freeze llm weights
+ model.eval()
+ for p in model.parameters():
+ p.requires_grad = False
+ return model, tokenizer
+
+
+def encode_text_batch(raw_text, text_encoder, tokenizer, device="cuda"):
+ # raw_text - list (batch_size length) of strings with input text prompts
+
+ with torch.no_grad():
+ max_text_len = 50
+
+ encoded = tokenizer.batch_encode_plus(
+ raw_text,
+ return_tensors="pt",
+ padding="max_length",
+ max_length=max_text_len,
+ truncation=True,
+ )
+ input_ids = encoded.input_ids.to(device)
+ attn_mask = encoded.attention_mask.to(device)
+
+ output = text_encoder(input_ids=input_ids, attention_mask=attn_mask)
+ encoded_text = output.last_hidden_state.detach()
+
+ encoded_text = encoded_text[:, :max_text_len]
+ attn_mask = attn_mask[:, :max_text_len]
+ encoded_text *= attn_mask.unsqueeze(-1)
+
+ return encoded_text
diff --git a/genmo/pipeline/genmo_pipeline.py b/genmo/pipeline/genmo_pipeline.py
new file mode 100644
index 0000000000000000000000000000000000000000..460cc78d2a5c82090a744098191bda94f022d6d1
--- /dev/null
+++ b/genmo/pipeline/genmo_pipeline.py
@@ -0,0 +1,641 @@
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from hydra.utils import instantiate
+from torch.cuda.amp import autocast
+
+from genmo.utils.net_utils import gaussian_smooth
+from genmo.utils.rotation_conversions import (
+ axis_angle_to_matrix,
+ matrix_to_axis_angle,
+ rotation_6d_to_matrix,
+)
+from third_party.GVHMR.hmr4d.model.gvhmr.utils.endecoder import EnDecoder
+from third_party.GVHMR.hmr4d.model.gvhmr.utils.postprocess import (
+ pp_static_joint,
+ pp_static_joint_cam,
+ process_ik,
+)
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import (
+ compute_transl_full_cam,
+ get_a_pred_cam,
+ project_to_bi01,
+)
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import (
+ get_static_joint_mask,
+ get_tgtcoord_rootparam,
+ rollout_local_transl_vel,
+)
+
+
+class Pipeline(nn.Module):
+ def __init__(self, args, args_denoiser3d, **kwargs):
+ super().__init__()
+ self.args = args
+ self.args_denoiser3d = args_denoiser3d
+ self.weights = args.weights # loss weights
+
+ # Networks
+ self.denoiser3d = instantiate(args_denoiser3d, _recursive_=False)
+ # Log.info(self.denoiser3d)
+
+ # Normalizer
+ self.endecoder: EnDecoder = instantiate(args.endecoder_opt, _recursive_=False)
+
+ self.denoiser3d.endecoder = self.endecoder
+
+ def forward(
+ self,
+ inputs,
+ train=False,
+ postproc=False,
+ static_cam=False,
+ global_step=0,
+ mode=None,
+ test_mode=None,
+ normalizer_stats=None,
+ ):
+ outputs = dict()
+
+ # Forward & output
+ model_output = self.denoiser3d(
+ inputs,
+ train=train,
+ postproc=postproc,
+ static_cam=static_cam,
+ mode=mode,
+ test_mode=test_mode,
+ normalizer_stats=normalizer_stats,
+ ) # pred_x, pred_cam, static_conf_logits
+ decode_dict = self.endecoder.decode(model_output["pred_x"]) # (B, L, C) -> dict
+ outputs.update({"model_output": model_output, "decode_dict": decode_dict})
+
+ # Post-processing``
+ if "gvhmr" in self.endecoder.feature_arr:
+ outputs["pred_smpl_params_incam"] = {
+ "body_pose": decode_dict["body_pose"], # (B, L, 63)
+ "betas": decode_dict["betas"], # (B, L, 10)
+ "global_orient": decode_dict["global_orient"], # (B, L, 3)
+ "transl": compute_transl_full_cam(
+ model_output["pred_cam"], inputs["bbx_xys"], inputs["K_fullimg"]
+ ),
+ }
+
+ if not train:
+ # if eval_gen_only:
+ # inputs["cam_angvel"] = torch.zeros(decode_dict["global_orient_gv"].shape[:2] + (6,), device=decode_dict["global_orient_gv"].device)
+ if "gvhmr" in self.endecoder.feature_arr:
+ if self.args.get("infer_version", 2) == 2:
+ pred_smpl_params_global = (
+ get_smpl_params_w_Rt_v2( # This function has for-loop
+ global_orient_gv=decode_dict["global_orient_gv"],
+ local_transl_vel=decode_dict["local_transl_vel"],
+ global_orient_c=decode_dict["global_orient"],
+ cam_angvel=inputs["cam_angvel"],
+ )
+ )
+ outputs["pred_smpl_params_global"] = {
+ "body_pose": decode_dict["body_pose"],
+ "betas": decode_dict["betas"],
+ **pred_smpl_params_global,
+ }
+ if "intermediate_decode_dict" in outputs:
+ intermediate_pred_smpl_params_global = []
+ for int_decode_dict in outputs["intermediate_decode_dict"]:
+ pred_smpl_params_global = get_smpl_params_w_Rt_v2(
+ global_orient_gv=int_decode_dict["global_orient_gv"],
+ local_transl_vel=int_decode_dict["local_transl_vel"],
+ global_orient_c=int_decode_dict["global_orient"],
+ cam_angvel=inputs["cam_angvel"],
+ )
+ pred_smpl_params_global = {
+ "body_pose": int_decode_dict["body_pose"],
+ "betas": int_decode_dict["betas"],
+ **pred_smpl_params_global,
+ }
+ intermediate_pred_smpl_params_global.append(
+ pred_smpl_params_global
+ )
+ outputs["intermediate_pred_smpl_params_global"] = (
+ intermediate_pred_smpl_params_global
+ )
+ elif "body_pose" in decode_dict:
+ outputs["pred_smpl_params_global"] = {
+ "body_pose": decode_dict["body_pose"],
+ "betas": decode_dict["betas"],
+ "global_orient": decode_dict["global_orient_w"],
+ "transl": decode_dict["transl_w"],
+ }
+
+ if "static_conf_logits" in model_output:
+ outputs["static_conf_logits"] = model_output["static_conf_logits"]
+
+ if (
+ postproc
+ and "gvhmr" in self.endecoder.feature_arr
+ and self.args.get("infer_version", 2) != 3
+ ): # apply post-processing
+ if static_cam: # extra post-processing to utilize static camera prior
+ if not bool(self.args.get("pp_ground", True)):
+ outputs["disable_pp_ground"] = True
+ outputs["pred_smpl_params_global"]["transl"] = pp_static_joint_cam(
+ outputs, self.endecoder
+ )
+ else:
+ if not bool(self.args.get("pp_ground", True)):
+ outputs["disable_pp_ground"] = True
+ outputs["pred_smpl_params_global"]["transl"] = pp_static_joint(
+ outputs, self.endecoder
+ )
+ body_pose = process_ik(outputs, self.endecoder)
+ decode_dict["body_pose"] = body_pose
+ outputs["pred_smpl_params_global"]["body_pose"] = body_pose
+ if "pred_smpl_params_incam" in outputs:
+ outputs["pred_smpl_params_incam"]["body_pose"] = body_pose
+
+ return outputs
+
+ # ========== Compute Loss ========== #
+ total_loss = 0
+ # mask = inputs["mask"]["valid"] # (B, L)
+
+ # 1. Simple loss: MSE
+ if self.weights.get("simple", 1.0) > 0.0:
+ pred_x = model_output["pred_x"][..., :151] # (B, L, C)
+ target_x = inputs["target_x"][..., :151] # (B, L, C)
+ target_x_mask = inputs["target_x_mask"][..., :151] # (B, L, C)
+
+ simple_loss = F.mse_loss(pred_x, target_x, reduction="none")
+
+ # Ensure all features are supervised
+ target_x_mask = target_x_mask.clone()
+ # Optional per-feature reweighting (helps prevent trajectory collapse on small fine-tunes).
+ gogv_mult = float(self.weights.get("gogv_mult", 1.0))
+ if gogv_mult != 1.0:
+ simple_loss[..., 142:148] = simple_loss[..., 142:148] * gogv_mult
+ tv_mult = float(self.weights.get("transl_vel_mult", 1.0))
+ if tv_mult != 1.0:
+ simple_loss[..., 148:151] = simple_loss[..., 148:151] * tv_mult
+ # Detailed logging: break down by feature type
+ # body_pose: 0:126, betas: 126:136, global_orient_c: 136:142, global_orient_gv: 142:148, local_transl_vel: 148:151
+ if train:
+ loss_bp = (simple_loss[..., :126] * target_x_mask[..., :126]).mean()
+ loss_betas = (simple_loss[..., 126:136] * target_x_mask[..., 126:136]).mean()
+ loss_go_c = (simple_loss[..., 136:142] * target_x_mask[..., 136:142]).mean()
+ loss_go_gv = (simple_loss[..., 142:148] * target_x_mask[..., 142:148]).mean()
+ loss_transl_vel = (simple_loss[..., 148:151] * target_x_mask[..., 148:151]).mean()
+ gogv_mask_frac = target_x_mask[..., 142:148].float().mean()
+ tv_mask_frac = target_x_mask[..., 148:151].float().mean()
+
+ # Debug trajectory collapse: measure target/pred translation-velocity magnitude in real units.
+ try:
+ pred_tv = self.endecoder.decode_translw(pred_x[..., 148:151].detach())
+ tgt_tv = self.endecoder.decode_translw(target_x[..., 148:151].detach())
+ pred_tv_abs = float(pred_tv.abs().mean().item())
+ tgt_tv_abs = float(tgt_tv.abs().mean().item())
+ pred_tv_zero_frac = float((pred_tv.abs().max(dim=-1)[0] < 1e-4).float().mean().item())
+ except Exception:
+ pred_tv_abs = float("nan")
+ tgt_tv_abs = float("nan")
+ pred_tv_zero_frac = float("nan")
+
+ # Debug rotation mismatch in degrees (more interpretable than MSE on 6D).
+ try:
+ mean = self.endecoder.stats_dict["gvhmr"]["mean"].to(pred_x)
+ std = self.endecoder.stats_dict["gvhmr"]["std"].to(pred_x)
+
+ def _rot6d_deg(a6d, b6d):
+ Ra = rotation_6d_to_matrix(a6d)
+ Rb = rotation_6d_to_matrix(b6d)
+ Rrel = Ra @ Rb.mT
+ tr = Rrel[..., 0, 0] + Rrel[..., 1, 1] + Rrel[..., 2, 2]
+ ang = torch.acos(torch.clamp((tr - 1.0) / 2.0, -1.0, 1.0))
+ return ang * (180.0 / torch.pi)
+
+ pred_go_c_6d = pred_x[..., 136:142].detach() * std[136:142] + mean[136:142]
+ tgt_go_c_6d = target_x[..., 136:142].detach() * std[136:142] + mean[136:142]
+ pred_go_gv_6d = pred_x[..., 142:148].detach() * std[142:148] + mean[142:148]
+ tgt_go_gv_6d = target_x[..., 142:148].detach() * std[142:148] + mean[142:148]
+
+ go_c_deg = _rot6d_deg(pred_go_c_6d, tgt_go_c_6d)
+ go_gv_deg = _rot6d_deg(pred_go_gv_6d, tgt_go_gv_6d)
+
+ go_c_deg_mean = float(go_c_deg.mean().item())
+ go_c_deg_max = float(go_c_deg.max().item())
+ go_gv_deg_mean = float(go_gv_deg.mean().item())
+ go_gv_deg_max = float(go_gv_deg.max().item())
+ except Exception:
+ go_c_deg_mean = float("nan")
+ go_c_deg_max = float("nan")
+ go_gv_deg_mean = float("nan")
+ go_gv_deg_max = float("nan")
+
+ try:
+ if "static_conf_logits" in model_output:
+ static_conf = model_output["static_conf_logits"].detach().sigmoid()
+ static_conf_mean = float(static_conf.mean().item())
+ static_conf_hi_frac = float((static_conf > 0.8).float().mean().item())
+ else:
+ static_conf_mean = float("nan")
+ static_conf_hi_frac = float("nan")
+ except Exception:
+ static_conf_mean = float("nan")
+ static_conf_hi_frac = float("nan")
+
+ from genmo.utils.pylogger import Log
+ Log.info(
+ f"[LossBreakdown] body_pose={loss_bp:.4f} betas={loss_betas:.4f} "
+ f"go_c={loss_go_c:.4f} go_gv={loss_go_gv:.4f} transl_vel={loss_transl_vel:.4f} "
+ f"gogv_mask={gogv_mask_frac:.2f} tv_mask={tv_mask_frac:.2f} "
+ f"go_c_deg(mean/max)={go_c_deg_mean:.2f}/{go_c_deg_max:.2f} "
+ f"go_gv_deg(mean/max)={go_gv_deg_mean:.2f}/{go_gv_deg_max:.2f} "
+ f"tv_abs_mean(pred)={pred_tv_abs:.4f} tv_abs_mean(tgt)={tgt_tv_abs:.4f} "
+ f"tv_zero_frac(pred)={pred_tv_zero_frac:.2f} "
+ f"static_conf_mean={static_conf_mean:.2f} static_conf_hi_frac={static_conf_hi_frac:.2f}"
+ )
+
+ simple_loss = (simple_loss * target_x_mask).mean()
+ total_loss += simple_loss * self.weights.get("simple", 1.0)
+ outputs["simple_loss"] = simple_loss
+
+ # 2. Extra loss
+ if "gvhmr" in self.endecoder.feature_arr:
+ extra_funcs = [
+ compute_extra_incam_loss,
+ compute_extra_global_loss,
+ ]
+ for extra_func in extra_funcs:
+ extra_loss, extra_loss_dict = extra_func(inputs, outputs, self, mode)
+ total_loss += extra_loss
+ outputs.update(extra_loss_dict)
+
+ outputs["loss"] = total_loss
+ return outputs
+
+
+def compute_extra_incam_loss(inputs, outputs, ppl, mode):
+ model_output = outputs["model_output"]
+ # decode_dict = outputs["decode_dict"]
+ endecoder = ppl.endecoder
+ weights = ppl.weights
+
+ # gen_only_losses = weights.get("gen_only_losses", "all")
+ # gen_only = inputs.get("gen_only", None)
+ # if weights.get("gen_only_no_reg_loss", False) and mode == "regression":
+ # gen_only_losses = []
+
+ extra_loss_dict = {}
+ extra_loss = 0
+ mask = inputs["mask"]["valid"].clone() # effective length mask
+ mask[inputs["mask"]["2d_only"]] = False
+ mask_reproj = ~inputs["mask"]["spv_incam_only"] # do not supervise reproj for 3DPW
+ mask_reproj_17 = torch.zeros_like(inputs["mask"]["valid"]).bool()
+ mask_reproj_17[inputs["mask"]["2d_only"]] = True
+
+ # Incam FK
+ # prediction
+ pred_c_j3d = endecoder.fk_v2(**outputs["pred_smpl_params_incam"])
+ pred_cr_j3d = pred_c_j3d - pred_c_j3d[:, :, :1] # (B, L, J, 3)
+ # pred_c_j17 = endecoder.smplx_model(**outputs["pred_smpl_params_incam"])[1]
+ # conf_c_j17 = inputs["det_kp2d_conf"]
+ # pred_c_j17[conf_c_j17 < 0.1] = 0.0
+
+ # gt
+ gt_c_j3d = endecoder.fk_v2(**inputs["smpl_params_c"]) # (B, L, J, 3)
+ gt_cr_j3d = gt_c_j3d - gt_c_j3d[:, :, :1] # (B, L, J, 3)
+
+ # Root aligned C-MPJPE Loss
+ if weights.cr_j3d > 0.0:
+ cr_j3d_loss = F.mse_loss(pred_cr_j3d, gt_cr_j3d, reduction="none")
+ # if (
+ # gen_only is not None
+ # and gen_only_losses != "all"
+ # and "cr_j3d" not in gen_only_losses
+ # ):
+ # cr_j3d_loss[gen_only] = 0
+ cr_j3d_loss = (cr_j3d_loss * mask[..., None, None]).mean()
+ extra_loss += cr_j3d_loss * weights.cr_j3d
+ extra_loss_dict["cr_j3d_loss"] = cr_j3d_loss
+
+ # Reprojection (to align with image)
+ if weights.transl_c > 0.0:
+ # pred_transl = decode_dict["transl"] # (B, L, 3)
+ # gt_transl = inputs["smpl_params_c"]["transl"]
+ # transl_c_loss = F.l1_loss(pred_transl, gt_transl, reduction="none")
+ # transl_c_loss = (transl_c_loss * mask[..., None]).mean()
+
+ # Instead of supervising transl, we convert gt to pred_cam (prevent divide 0)
+ pred_cam = model_output["pred_cam"] # (B, L, 3)
+ gt_transl = inputs["smpl_params_c"]["transl"] # (B, L, 3)
+ gt_pred_cam = get_a_pred_cam(
+ gt_transl, inputs["bbx_xys"], inputs["K_fullimg"]
+ ) # (B, L, 3)
+ gt_pred_cam[gt_pred_cam.isinf()] = -1 # this will be handled by valid_mask
+ # (compute_transl_full_cam(gt_pred_cam, inputs["bbx_xys"], inputs["K_fullimg"]) - gt_transl).abs().max()
+
+ # Skip gts that are not good during random construction
+ gt_j3d_z_min = inputs["gt_j3d"][..., 2].min(dim=-1)[0]
+ valid_mask = (
+ (gt_j3d_z_min > 0.3)
+ * (gt_pred_cam[..., 0] > 0.3)
+ * (gt_pred_cam[..., 0] < 5.0)
+ * (gt_pred_cam[..., 1] > -3.0)
+ * (gt_pred_cam[..., 1] < 3.0)
+ * (gt_pred_cam[..., 2] > -3.0)
+ * (gt_pred_cam[..., 2] < 3.0)
+ * (inputs["bbx_xys"][..., 2] > 0)
+ )[..., None]
+ transl_c_loss = F.mse_loss(pred_cam, gt_pred_cam, reduction="none")
+ # if (
+ # gen_only is not None
+ # and gen_only_losses != "all"
+ # and "transl_c" not in gen_only_losses
+ # ):
+ # transl_c_loss[gen_only] = 0
+ transl_c_loss = (transl_c_loss * mask[..., None] * valid_mask).mean()
+
+ extra_loss_dict["transl_c_loss"] = transl_c_loss
+ extra_loss += transl_c_loss * weights.transl_c
+
+ # Debug: if incam starts drifting/out-of-frame during fine-tune, this loss should pull it back.
+ if getattr(ppl, "training", False):
+ try:
+ from genmo.utils.pylogger import Log
+
+ pred_cam_abs = float(pred_cam.detach().abs().mean().item())
+ gt_pred_cam_abs = float(gt_pred_cam.detach().abs().mean().item())
+ valid_frac = float(valid_mask.float().mean().item())
+ Log.info(
+ f"[IncamDebug] transl_c_loss={float(transl_c_loss.item()):.4f} "
+ f"pred_cam_abs_mean={pred_cam_abs:.3f} gt_pred_cam_abs_mean={gt_pred_cam_abs:.3f} "
+ f"valid_frac={valid_frac:.2f}"
+ )
+ except Exception:
+ pass
+
+ if weights.j2d > 0.0:
+ # prevent divide 0 or small value to overflow(fp16)
+ reproj_z_thr = 0.3
+ pred_c_j3d_z0_mask = pred_c_j3d[..., 2].abs() <= reproj_z_thr
+ pred_c_j3d[pred_c_j3d_z0_mask] = reproj_z_thr
+ # pred_c_j17_z0_mask = pred_c_j17[..., 2].abs() <= reproj_z_thr
+ # pred_c_j17[pred_c_j17_z0_mask] = reproj_z_thr
+
+ gt_c_j3d_z0_mask = gt_c_j3d[..., 2].abs() <= reproj_z_thr
+ gt_c_j3d[gt_c_j3d_z0_mask] = reproj_z_thr
+
+ pred_j2d_01 = project_to_bi01(
+ pred_c_j3d, inputs["bbx_xys"], inputs["K_fullimg"]
+ )
+ # pred_j2d_17 = project_to_bi01(
+ # pred_c_j17, inputs["bbx_xys"], inputs["K_fullimg"]
+ # )
+ gt_j2d_01 = project_to_bi01(
+ gt_c_j3d, inputs["bbx_xys"], inputs["K_fullimg"]
+ ) # (B, L, J, 2)
+ # gt_kp2d_normed = inputs["gt_kp2d_normed"]
+ # valid_mask_j17 = inputs["valid_mask_j17"]
+ # pred_j2d_17[conf_c_j17 < 0.5] = 0.0
+ # gt_kp2d_normed[conf_c_j17 < 0.5] = 0.0
+
+ valid_mask = (
+ (gt_c_j3d[..., 2] > reproj_z_thr)
+ * (pred_c_j3d[..., 2] > reproj_z_thr) # Be safe
+ * (gt_j2d_01[..., 0] > 0.0)
+ * (gt_j2d_01[..., 0] < 1.0)
+ * (gt_j2d_01[..., 1] > 0.0)
+ * (gt_j2d_01[..., 1] < 1.0)
+ )[..., None]
+ valid_mask[~mask_reproj] = False # Do not supervise on 3dpw
+ # valid_mask_j17 = valid_mask_j17 & (pred_c_j17[..., 2] > reproj_z_thr)[..., None]
+
+ j2d_loss = F.mse_loss(pred_j2d_01, gt_j2d_01, reduction="none")
+ # j2d_17_loss = F.mse_loss(pred_j2d_17, gt_kp2d_normed, reduction="none")
+ # if (
+ # gen_only is not None
+ # and gen_only_losses != "all"
+ # and "j2d" not in gen_only_losses
+ # ):
+ # j2d_loss[gen_only] = 0
+ j2d_loss = (j2d_loss * mask[..., None, None] * valid_mask).mean()
+ # j2d_17_loss = (
+ # j2d_17_loss * mask_reproj_17[..., None, None] * valid_mask_j17
+ # ).mean()
+
+ extra_loss += j2d_loss * weights.j2d
+ extra_loss_dict["j2d_loss"] = j2d_loss
+ # extra_loss += j2d_17_loss * weights.j2d_17
+ # extra_loss_dict["j2d_17_loss"] = j2d_17_loss
+
+ if weights.cr_verts > 0:
+ # SMPL forward
+ pred_c_verts437, pred_c_j17 = endecoder.smplx_model(
+ **outputs["pred_smpl_params_incam"]
+ )
+ root_ = pred_c_j17[:, :, [11, 12], :].mean(-2, keepdim=True)
+ pred_cr_verts437 = pred_c_verts437 - root_
+
+ gt_cr_verts437 = inputs["gt_cr_verts437"] # (B, L, 437, 3)
+ cr_vert_loss = F.mse_loss(pred_cr_verts437, gt_cr_verts437, reduction="none")
+ # if (
+ # gen_only is not None
+ # and gen_only_losses != "all"
+ # and "cr_verts" not in gen_only_losses
+ # ):
+ # cr_vert_loss[gen_only] = 0
+ cr_vert_loss = (cr_vert_loss * mask[:, :, None, None]).mean()
+ extra_loss += cr_vert_loss * weights.cr_verts
+ extra_loss_dict["cr_vert_loss"] = cr_vert_loss
+
+ if weights.verts2d > 0:
+ gt_c_verts437 = inputs["gt_c_verts437"] # (B, L, 437, 3)
+
+ # prevent divide 0 or small value to overflow(fp16)
+ reproj_z_thr = 0.3
+ pred_c_verts437_z0_mask = pred_c_verts437[..., 2].abs() <= reproj_z_thr
+ pred_c_verts437[pred_c_verts437_z0_mask] = reproj_z_thr
+ gt_c_verts437_z0_mask = gt_c_verts437[..., 2].abs() <= reproj_z_thr
+ gt_c_verts437[gt_c_verts437_z0_mask] = reproj_z_thr
+
+ pred_verts2d_01 = project_to_bi01(
+ pred_c_verts437, inputs["bbx_xys"], inputs["K_fullimg"]
+ )
+ gt_verts2d_01 = project_to_bi01(
+ gt_c_verts437, inputs["bbx_xys"], inputs["K_fullimg"]
+ ) # (B, L, 437, 2)
+
+ valid_mask = (
+ (gt_c_verts437[..., 2] > reproj_z_thr)
+ * (pred_c_verts437[..., 2] > reproj_z_thr) # Be safe
+ * (gt_verts2d_01[..., 0] > 0.0)
+ * (gt_verts2d_01[..., 0] < 1.0)
+ * (gt_verts2d_01[..., 1] > 0.0)
+ * (gt_verts2d_01[..., 1] < 1.0)
+ )[..., None]
+ valid_mask[~mask_reproj] = False # Do not supervise on 3dpw
+ verts2d_loss = F.mse_loss(pred_verts2d_01, gt_verts2d_01, reduction="none")
+ # if (
+ # gen_only is not None
+ # and gen_only_losses != "all"
+ # and "verts2d" not in gen_only_losses
+ # ):
+ # verts2d_loss[gen_only] = 0
+ verts2d_loss = (verts2d_loss * mask[..., None, None] * valid_mask).mean()
+
+ extra_loss += verts2d_loss * weights.verts2d
+ extra_loss_dict["verts2d_loss"] = verts2d_loss
+
+ return extra_loss, extra_loss_dict
+
+
+def compute_extra_global_loss(inputs, outputs, ppl, mode):
+ decode_dict = outputs["decode_dict"]
+ endecoder = ppl.endecoder
+ weights = ppl.weights
+ args = ppl.args
+
+ extra_loss_dict = {}
+ extra_loss = 0
+ mask = inputs["mask"]["valid"].clone() # (B, L)
+ mask[inputs["mask"]["spv_incam_only"]] = False
+ mask[inputs["mask"]["2d_only"]] = False
+ mask_contact = mask.clone()
+ mask_contact[inputs["mask"]["invalid_contact"]] = False
+
+ # gen_only_losses = weights.get("gen_only_losses", "all")
+ # gen_only = inputs.get("gen_only", None)
+ # if weights.get("gen_only_no_reg_loss", False) and mode == "regression":
+ # gen_only_losses = []
+
+ if weights.transl_w > 0:
+ # compute pred_transl_w by rollout
+ gt_transl_w = inputs["smpl_params_w"]["transl"]
+ gt_global_orient_w = inputs["smpl_params_w"]["global_orient"]
+ local_transl_vel = decode_dict["local_transl_vel"]
+ # Roll out world translation using GT global-orient.
+ pred_transl_w = rollout_local_transl_vel(
+ local_transl_vel, gt_global_orient_w, gt_transl_w[:, [0]]
+ )
+
+ if args.get("transl_w_xz_only", False):
+ pred_ = pred_transl_w[..., [0, 2]]
+ gt_ = gt_transl_w[..., [0, 2]]
+ else:
+ pred_, gt_ = pred_transl_w, gt_transl_w
+ trans_w_loss = F.l1_loss(pred_, gt_, reduction="none")
+ # if (
+ # gen_only is not None
+ # and gen_only_losses != "all"
+ # and "transl_w" not in gen_only_losses
+ # ):
+ # trans_w_loss[gen_only] = 0
+ trans_w_loss = (trans_w_loss * mask[..., None]).mean()
+ extra_loss += trans_w_loss * weights.transl_w
+ extra_loss_dict["transl_w_loss"] = trans_w_loss
+
+ # Static-Conf loss
+ if weights.static_conf_bce > 0:
+ # Compute gt by thresholding velocity
+ vel_thr = args.static_conf.vel_thr
+ assert vel_thr > 0
+ joint_ids = [
+ 7,
+ 10,
+ 8,
+ 11,
+ 20,
+ 21,
+ ] # [L_Ankle, L_foot, R_Ankle, R_foot, L_wrist, R_wrist]
+ gt_w_j3d = endecoder.fk_v2(**inputs["smpl_params_w"]) # (B, L, J=22, 3)
+ static_gt = get_static_joint_mask(
+ gt_w_j3d, vel_thr=vel_thr, repeat_last=True
+ ) # (B, L, J)
+ static_gt = static_gt[:, :, joint_ids].float() # (B, L, J')
+ pred_static_conf_logits = outputs["model_output"]["static_conf_logits"]
+
+ static_conf_loss = F.binary_cross_entropy_with_logits(
+ pred_static_conf_logits, static_gt, reduction="none"
+ )
+ # if (
+ # gen_only is not None
+ # and gen_only_losses != "all"
+ # and "static_conf_bce" not in gen_only_losses
+ # ):
+ # static_conf_loss[gen_only] = 0
+ static_conf_loss = (static_conf_loss * mask_contact[..., None]).mean()
+ extra_loss += static_conf_loss * weights.static_conf_bce
+ extra_loss_dict["static_conf_loss"] = static_conf_loss
+
+ return extra_loss, extra_loss_dict
+
+
+@autocast(enabled=False)
+def get_smpl_params_w_Rt_v2(
+ global_orient_gv,
+ local_transl_vel,
+ global_orient_c,
+ cam_angvel,
+):
+ """Get global R,t in GV0(ay)
+ Args:
+ cam_angvel: (B, L, 6), defined as R @ R_{w2c}^{t} = R_{w2c}^{t+1}
+ """
+
+ # Get R_ct_to_c0 from cam_angvel
+ def as_identity(R):
+ is_I = matrix_to_axis_angle(R).norm(dim=-1) < 1e-5
+ R[is_I] = torch.eye(3)[None].expand(is_I.sum(), -1, -1).to(R)
+ return R
+
+ B = cam_angvel.shape[0]
+ R_t_to_tp1 = rotation_6d_to_matrix(cam_angvel) # (B, L, 3, 3)
+ R_t_to_tp1 = as_identity(R_t_to_tp1)
+
+ # Get R_c2gv
+ R_gv = axis_angle_to_matrix(global_orient_gv) # (B, L, 3, 3)
+ R_c = axis_angle_to_matrix(global_orient_c) # (B, L, 3, 3)
+
+ # Camera view direction in GV coordinate: Rc2gv @ [0,0,1]
+ R_c2gv = R_gv @ R_c.mT
+ view_axis_gv = R_c2gv[
+ :, :, :, 2
+ ] # (B, L, 3) Rc2gv is estimated, so the x-axis is not accurate, i.e. != 0
+
+ # Rotate axis use camera relative rotation
+ R_cnext2gv = R_c2gv @ R_t_to_tp1.mT
+ view_axis_gv_next = R_cnext2gv[..., 2]
+
+ vec1_xyz = view_axis_gv.clone()
+ vec1_xyz[..., 1] = 0
+ vec1_xyz = F.normalize(vec1_xyz, dim=-1)
+ vec2_xyz = view_axis_gv_next.clone()
+ vec2_xyz[..., 1] = 0
+ vec2_xyz = F.normalize(vec2_xyz, dim=-1)
+
+ aa_tp1_to_t = vec2_xyz.cross(vec1_xyz, dim=-1)
+ aa_tp1_to_t_angle = torch.acos(
+ torch.clamp((vec1_xyz * vec2_xyz).sum(dim=-1, keepdim=True), -1.0, 1.0)
+ )
+ aa_tp1_to_t = F.normalize(aa_tp1_to_t, dim=-1) * aa_tp1_to_t_angle
+
+ aa_tp1_to_t = gaussian_smooth(aa_tp1_to_t, dim=-2) # Smooth
+ R_tp1_to_t = axis_angle_to_matrix(aa_tp1_to_t).mT # (B, L, 3)
+
+ # Get R_t_to_0
+ R_t_to_0 = [torch.eye(3)[None].expand(B, -1, -1).to(R_t_to_tp1)]
+ for i in range(1, R_t_to_tp1.shape[1]):
+ R_t_to_0.append(R_t_to_0[-1] @ R_tp1_to_t[:, i])
+ R_t_to_0 = torch.stack(R_t_to_0, dim=1) # (B, L, 3, 3)
+ R_t_to_0 = as_identity(R_t_to_0)
+
+ global_orient = matrix_to_axis_angle(R_t_to_0 @ R_gv)
+
+ # Rollout to global transl
+ # Start from transl0, in gv0 -> flip y-axis of gv0
+ transl = rollout_local_transl_vel(local_transl_vel, global_orient)
+ global_orient, transl, _ = get_tgtcoord_rootparam(
+ global_orient, transl, tsf="any->ay"
+ )
+
+ smpl_params_w_Rt = {"global_orient": global_orient, "transl": transl}
+ return smpl_params_w_Rt
diff --git a/genmo/utils/eval_utils.py b/genmo/utils/eval_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..13e78e728725e1266c922b7ea664a857266f3fc3
--- /dev/null
+++ b/genmo/utils/eval_utils.py
@@ -0,0 +1,600 @@
+import numpy as np
+import torch
+from scipy.ndimage import gaussian_filter
+from scipy.signal import argrelextrema
+
+
+@torch.no_grad()
+def compute_camcoord_metrics(batch, pelvis_idxs=[1, 2], fps=30, mask=None):
+ """
+ Args:
+ batch (dict): {
+ "pred_j3d": (..., J, 3) tensor
+ "target_j3d":
+ "pred_verts":
+ "target_verts":
+ }
+ Returns:
+ cam_coord_metrics (dict): {
+ "pa_mpjpe": (..., ) numpy array
+ "mpjpe":
+ "pve":
+ "accel":
+ }
+ """
+ # All data is in camera coordinates
+ pred_j3d = batch["pred_j3d"].cpu() # (..., J, 3)
+ target_j3d = batch["target_j3d"].cpu()
+ pred_verts = batch["pred_verts"].cpu()
+ target_verts = batch["target_verts"].cpu()
+
+ if mask is not None:
+ mask = mask.cpu()
+ pred_j3d = pred_j3d[mask].clone()
+ target_j3d = target_j3d[mask].clone()
+ pred_verts = pred_verts[mask].clone()
+ target_verts = target_verts[mask].clone()
+ assert "mask" not in batch
+
+ # Align by pelvis
+ pred_j3d, target_j3d, pred_verts, target_verts = batch_align_by_pelvis(
+ [pred_j3d, target_j3d, pred_verts, target_verts], pelvis_idxs=pelvis_idxs
+ )
+
+ # Metrics
+ m2mm = 1000
+ S1_hat = batch_compute_similarity_transform_torch(pred_j3d, target_j3d)
+ pa_mpjpe = compute_jpe(S1_hat, target_j3d) * m2mm
+ mpjpe = compute_jpe(pred_j3d, target_j3d) * m2mm
+ pve = compute_jpe(pred_verts, target_verts) * m2mm
+ accel = compute_error_accel(joints_pred=pred_j3d, joints_gt=target_j3d, fps=fps)
+
+ camcoord_metrics = {
+ "pa_mpjpe": pa_mpjpe,
+ "mpjpe": mpjpe,
+ "pve": pve,
+ "accel": accel,
+ }
+ return camcoord_metrics
+
+
+@torch.no_grad()
+def compute_music_metrics(batch, mask=None):
+ """
+ Args:
+ batch (dict): {
+ "pred_j3d": (..., J, 3) tensor
+ "target_j3d":
+ "music_beats": (T,) numpy array
+ }
+ Returns:
+ music_metrics (dict): {
+ "PFC":
+ }
+ """
+ # All data is in global coordinates
+ pred_j3d_glob = batch["pred_j3d_glob"].cpu().numpy() # (..., J, 3)
+ # pred_j3d_glob = batch["target_j3d_glob"].cpu().numpy() # (..., J, 3)
+ up_dir = 1 # y is up
+ flat_dirs = [i for i in range(3) if i != up_dir]
+
+ DT = 1 / 30
+ assert pred_j3d_glob.ndim == 3
+
+ root_v = (
+ pred_j3d_glob[1:, 0, :] - pred_j3d_glob[:-1, 0, :]
+ ) / DT # root velocity (T-1, 3)
+ root_a = (root_v[1:, :] - root_v[:-1, :]) / DT # root acceleration (T-2, 3)
+
+ # clamp the up-direction of root acceleration
+ root_a[:, up_dir] = np.maximum(root_a[:, up_dir], 0) # (T-2, 3)
+ # l2 norm
+ root_a = np.linalg.norm(root_a, axis=-1) # (T-2,)
+ scaling = root_a.max()
+ root_a = root_a / scaling
+
+ foot_idx = [7, 10, 8, 11]
+ feet = pred_j3d_glob[:, foot_idx, :] # (T, 4, 3)
+ foot_v = np.linalg.norm(
+ feet[2:, :, flat_dirs] - feet[1:-1, :, flat_dirs], axis=-1
+ ) # horizontal velocity (T-2, 4)
+ foot_mins = np.zeros((len(foot_v), 2))
+ foot_mins[:, 0] = np.minimum(foot_v[:, 0], foot_v[:, 1])
+ foot_mins[:, 1] = np.minimum(foot_v[:, 2], foot_v[:, 3])
+ foot_v = np.maximum(foot_mins, 0)
+
+ foot_loss = (
+ foot_mins[:, 0] * foot_mins[:, 1] * root_a
+ ) # min leftv * min rightv * root_a (T-2,)
+ pfc = foot_loss.mean() * 10000
+
+ # compute Beat Align Score
+ motion_beats = compute_motion_beats(pred_j3d_glob)[0]
+ music_beats = compute_music_beats(batch["music_beats"])
+ ba = 0
+ for bb in music_beats:
+ ba += np.exp(-np.min((motion_beats - bb) ** 2) / 2 / 9)
+ bas = ba / len(music_beats)
+ return {
+ "PFC": pfc,
+ "BAS": bas,
+ }
+
+
+@torch.no_grad()
+def compute_global_metrics(batch, mask=None):
+ """Follow WHAM, the input has skipped invalid frames
+ Args:
+ batch (dict): {
+ "pred_j3d_glob": (F, J, 3) tensor
+ "target_j3d_glob":
+ "pred_verts_glob":
+ "target_verts_glob":
+ }
+ Returns:
+ global_metrics (dict): {
+ "wa2_mpjpe": (F, ) numpy array
+ "waa_mpjpe":
+ "rte":
+ "jitter":
+ "fs":
+ }
+ """
+ # All data is in global coordinates
+ pred_j3d_glob = batch["pred_j3d_glob"].cpu() # (..., J, 3)
+ target_j3d_glob = batch["target_j3d_glob"].cpu()
+ pred_verts_glob = batch["pred_verts_glob"].cpu()
+ target_verts_glob = batch["target_verts_glob"].cpu()
+ if mask is not None:
+ mask = mask.cpu()
+ pred_j3d_glob = pred_j3d_glob[mask].clone()
+ target_j3d_glob = target_j3d_glob[mask].clone()
+ pred_verts_glob = pred_verts_glob[mask].clone()
+ target_verts_glob = target_verts_glob[mask].clone()
+ assert "mask" not in batch
+
+ seq_length = pred_j3d_glob.shape[0]
+
+ # Use chunk to compare
+ chunk_length = 100
+ wa2_mpjpe, waa_mpjpe = [], []
+ for start in range(0, seq_length, chunk_length):
+ end = min(seq_length, start + chunk_length)
+
+ target_j3d = target_j3d_glob[start:end].clone().cpu()
+ pred_j3d = pred_j3d_glob[start:end].clone().cpu()
+
+ w_j3d = first_align_joints(target_j3d, pred_j3d)
+ wa_j3d = global_align_joints(target_j3d, pred_j3d)
+
+ wa2_mpjpe.append(compute_jpe(target_j3d, w_j3d))
+ waa_mpjpe.append(compute_jpe(target_j3d, wa_j3d))
+
+ # Metrics
+ m2mm = 1000
+ wa2_mpjpe = np.concatenate(wa2_mpjpe) * m2mm
+ waa_mpjpe = np.concatenate(waa_mpjpe) * m2mm
+
+ # Additional Metrics
+ rte = compute_rte(target_j3d_glob[:, 0].cpu(), pred_j3d_glob[:, 0].cpu()) * 1e2
+ jitter = compute_jitter(pred_j3d_glob, fps=30)
+ foot_sliding = compute_foot_sliding(target_verts_glob, pred_verts_glob) * m2mm
+
+ global_metrics = {
+ "wa2_mpjpe": wa2_mpjpe,
+ "waa_mpjpe": waa_mpjpe,
+ "rte": rte,
+ "jitter": jitter,
+ "fs": foot_sliding,
+ }
+ return global_metrics
+
+
+@torch.no_grad()
+def compute_camcoord_perjoint_metrics(batch, pelvis_idxs=[1, 2]):
+ """
+ Args:
+ batch (dict): {
+ "pred_j3d": (..., J, 3) tensor
+ "target_j3d":
+ }
+ Returns:
+ cam_coord_metrics (dict): {
+ "pa_mpjpe": (..., ) numpy array
+ "mpjpe":
+ "pve":
+ "accel":
+ }
+ """
+ # All data is in camera coordinates
+ pred_j3d = batch["pred_j3d"].cpu() # (..., J, 3)
+ target_j3d = batch["target_j3d"].cpu()
+ pred_verts = batch["pred_verts"].cpu()
+ target_verts = batch["target_verts"].cpu()
+
+ # Align by pelvis
+ pred_j3d, target_j3d, pred_verts, target_verts = batch_align_by_pelvis(
+ [pred_j3d, target_j3d, pred_verts, target_verts], pelvis_idxs=pelvis_idxs
+ )
+ # Metrics
+ m2mm = 1000
+ perjoint_mpjpe = compute_perjoint_jpe(pred_j3d, target_j3d) * m2mm
+
+ camcoord_perjoint_metrics = {
+ "mpjpe": perjoint_mpjpe,
+ }
+ return camcoord_perjoint_metrics
+
+
+# ===== Utilities =====
+
+
+def compute_jpe(S1, S2):
+ return torch.sqrt(((S1 - S2) ** 2).sum(dim=-1)).mean(dim=-1).numpy()
+
+
+def compute_perjoint_jpe(S1, S2):
+ return torch.sqrt(((S1 - S2) ** 2).sum(dim=-1)).numpy()
+
+
+def batch_align_by_pelvis(data_list, pelvis_idxs=[1, 2]):
+ """
+ Assumes data is given as [pred_j3d, target_j3d, pred_verts, target_verts].
+ Each data is in shape of (frames, num_points, 3)
+ Pelvis is notated as one / two joints indices.
+ Align all data to the corresponding pelvis location.
+ """
+
+ pred_j3d, target_j3d, pred_verts, target_verts = data_list
+
+ pred_pelvis = pred_j3d[:, pelvis_idxs].mean(dim=1, keepdims=True).clone()
+ target_pelvis = target_j3d[:, pelvis_idxs].mean(dim=1, keepdims=True).clone()
+
+ # Align to the pelvis
+ pred_j3d = pred_j3d - pred_pelvis
+ target_j3d = target_j3d - target_pelvis
+ pred_verts = pred_verts - pred_pelvis
+ target_verts = target_verts - target_pelvis
+
+ return (pred_j3d, target_j3d, pred_verts, target_verts)
+
+
+def batch_compute_similarity_transform_torch(S1, S2):
+ """
+ Computes a similarity transform (sR, t) that takes
+ a set of 3D points S1 (3 x N) closest to a set of 3D points S2,
+ where R is an 3x3 rotation matrix, t 3x1 translation, s scale.
+ i.e. solves the orthogonal Procrutes problem.
+ """
+ transposed = False
+ if S1.shape[0] != 3 and S1.shape[0] != 2:
+ S1 = S1.permute(0, 2, 1)
+ S2 = S2.permute(0, 2, 1)
+ transposed = True
+ assert S2.shape[1] == S1.shape[1]
+
+ # 1. Remove mean.
+ mu1 = S1.mean(axis=-1, keepdims=True)
+ mu2 = S2.mean(axis=-1, keepdims=True)
+
+ X1 = S1 - mu1
+ X2 = S2 - mu2
+
+ # 2. Compute variance of X1 used for scale.
+ var1 = torch.sum(X1**2, dim=1).sum(dim=1)
+
+ # 3. The outer product of X1 and X2.
+ K = X1.bmm(X2.permute(0, 2, 1))
+
+ # 4. Solution that Maximizes trace(R'K) is R=U*V', where U, V are
+ # singular vectors of K.
+ U, s, V = torch.svd(K)
+
+ # Construct Z that fixes the orientation of R to get det(R)=1.
+ Z = torch.eye(U.shape[1], device=S1.device).unsqueeze(0)
+ Z = Z.repeat(U.shape[0], 1, 1)
+ Z[:, -1, -1] *= torch.sign(torch.det(U.bmm(V.permute(0, 2, 1))))
+
+ # Construct R.
+ R = V.bmm(Z.bmm(U.permute(0, 2, 1)))
+
+ # 5. Recover scale.
+ scale = torch.cat([torch.trace(x).unsqueeze(0) for x in R.bmm(K)]) / var1
+
+ # 6. Recover translation.
+ t = mu2 - (scale.unsqueeze(-1).unsqueeze(-1) * (R.bmm(mu1)))
+
+ # 7. Error:
+ S1_hat = scale.unsqueeze(-1).unsqueeze(-1) * R.bmm(S1) + t
+
+ if transposed:
+ S1_hat = S1_hat.permute(0, 2, 1)
+
+ return S1_hat
+
+
+def batch_compute_scale_trans_torch(S1, S2):
+ """
+ Computes a similarity transform (sR, t) that takes
+ a set of 3D points S1 (3 x N) closest to a set of 3D points S2,
+ where R is an 3x3 rotation matrix, t 3x1 translation, s scale.
+ i.e. solves the orthogonal Procrutes problem.
+ """
+ transposed = False
+ if S1.shape[0] != 3 and S1.shape[0] != 2:
+ S1 = S1.permute(0, 2, 1)
+ S2 = S2.permute(0, 2, 1)
+ transposed = True
+ assert S2.shape[1] == S1.shape[1]
+
+ # 1. Remove mean.
+ mu1 = S1.mean(axis=-1, keepdims=True)
+ mu2 = S2.mean(axis=-1, keepdims=True)
+
+ X1 = S1 - mu1
+ X2 = S2 - mu2
+
+ # 2. Compute variance of X1 used for scale.
+ var1 = torch.sum(X1**2, dim=1).sum(dim=1)
+
+ # 3. The outer product of X1 and X2.
+ K = X1.bmm(X2.permute(0, 2, 1))
+
+ # 4. Solution that Maximizes trace(R'K) is R=U*V', where U, V are
+ # singular vectors of K.
+ U, s, V = torch.svd(K)
+
+ # Construct Z that fixes the orientation of R to get det(R)=1.
+ Z = torch.eye(U.shape[1], device=S1.device).unsqueeze(0)
+ Z = Z.repeat(U.shape[0], 1, 1)
+ Z[:, -1, -1] *= torch.sign(torch.det(U.bmm(V.permute(0, 2, 1))))
+
+ # Construct R.
+ R = V.bmm(Z.bmm(U.permute(0, 2, 1)))
+
+ # 5. Recover scale.
+ scale = torch.cat([torch.trace(x).unsqueeze(0) for x in R.bmm(K)]) / var1
+
+ # 6. Recover translation.
+ t = mu2 - (scale.unsqueeze(-1).unsqueeze(-1) * (R.bmm(mu1)))
+
+ return scale, t, R
+
+
+def compute_error_accel(joints_gt, joints_pred, valid_mask=None, fps=None):
+ """
+ Use [i-1, i, i+1] to compute acc at frame_i. The acceleration error:
+ 1/(n-2) \sum_{i=1}^{n-1} X_{i-1} - 2X_i + X_{i+1}
+ Note that for each frame that is not visible, three entries(-1, 0, +1) in the
+ acceleration error will be zero'd out.
+ Args:
+ joints_gt : (F, J, 3)
+ joints_pred : (F, J, 3)
+ valid_mask : (F)
+ Returns:
+ error_accel (F-2) when valid_mask is None, else (F'), F' <= F-2
+ """
+ # (F, J, 3) -> (F-2) per-joint
+ accel_gt = joints_gt[:-2] - 2 * joints_gt[1:-1] + joints_gt[2:]
+ accel_pred = joints_pred[:-2] - 2 * joints_pred[1:-1] + joints_pred[2:]
+ normed = np.linalg.norm(accel_pred - accel_gt, axis=-1).mean(axis=-1)
+ if fps is not None:
+ normed = normed * fps**2
+
+ if valid_mask is None:
+ new_vis = np.ones(len(normed), dtype=bool)
+ else:
+ invis = np.logical_not(valid_mask)
+ invis1 = np.roll(invis, -1)
+ invis2 = np.roll(invis, -2)
+ new_invis = np.logical_or(invis, np.logical_or(invis1, invis2))[:-2]
+ new_vis = np.logical_not(new_invis)
+ if new_vis.sum() == 0:
+ print("Warning!!! no valid acceleration error to compute.")
+
+ return normed[new_vis]
+
+
+def compute_rte(target_trans, pred_trans):
+ # Compute the global alignment
+ _, rot, trans = align_pcl(
+ target_trans[None, :], pred_trans[None, :], fixed_scale=True
+ )
+ pred_trans_hat = (
+ torch.einsum("tij,tnj->tni", rot, pred_trans[None, :]) + trans[None, :]
+ )[0]
+
+ # Compute the entire displacement of ground truth trajectory
+ disps, disp = [], 0
+ for p1, p2 in zip(target_trans, target_trans[1:]):
+ delta = (p2 - p1).norm(2, dim=-1)
+ disp += delta
+ disps.append(disp)
+
+ # Compute absolute root-translation-error (RTE)
+ rte = torch.norm(target_trans - pred_trans_hat, 2, dim=-1)
+
+ # Normalize it to the displacement
+ return (rte / disp).numpy()
+
+
+def compute_jitter(joints, fps=30):
+ """compute jitter of the motion
+ Args:
+ joints (N, J, 3).
+ fps (float).
+ Returns:
+ jitter (N-3).
+ """
+ pred_jitter = torch.norm(
+ (joints[3:] - 3 * joints[2:-1] + 3 * joints[1:-2] - joints[:-3]) * (fps**3),
+ dim=2,
+ ).mean(dim=-1)
+
+ return pred_jitter.cpu().numpy() / 10.0
+
+
+def compute_foot_sliding(target_verts, pred_verts, thr=1e-2):
+ """compute foot sliding error
+ The foot ground contact label is computed by the threshold of 1 cm/frame
+ Args:
+ target_verts (N, 6890, 3).
+ pred_verts (N, 6890, 3).
+ Returns:
+ error (N frames in contact).
+ """
+ assert target_verts.shape == pred_verts.shape
+ assert target_verts.shape[-2] == 6890
+
+ # Foot vertices idxs
+ foot_idxs = [3216, 3387, 6617, 6787]
+
+ # Compute contact label
+ foot_loc = target_verts[:, foot_idxs]
+ foot_disp = (foot_loc[1:] - foot_loc[:-1]).norm(2, dim=-1)
+ contact = foot_disp[:] < thr
+
+ pred_feet_loc = pred_verts[:, foot_idxs]
+ pred_disp = (pred_feet_loc[1:] - pred_feet_loc[:-1]).norm(2, dim=-1)
+
+ error = pred_disp[contact]
+
+ return error.cpu().numpy()
+
+
+def convert_joints22_to_24(joints22, ratio2220=0.3438, ratio2321=0.3345):
+ joints24 = torch.zeros(*joints22.shape[:-2], 24, 3).to(joints22.device)
+ joints24[..., :22, :] = joints22
+ joints24[..., 22, :] = joints22[..., 20, :] + ratio2220 * (
+ joints22[..., 20, :] - joints22[..., 18, :]
+ )
+ joints24[..., 23, :] = joints22[..., 21, :] + ratio2321 * (
+ joints22[..., 21, :] - joints22[..., 19, :]
+ )
+ return joints24
+
+
+def align_pcl(Y, X, weight=None, fixed_scale=False):
+ """align similarity transform to align X with Y using umeyama method
+ X' = s * R * X + t is aligned with Y
+ :param Y (*, N, 3) first trajectory
+ :param X (*, N, 3) second trajectory
+ :param weight (*, N, 1) optional weight of valid correspondences
+ :returns s (*, 1), R (*, 3, 3), t (*, 3)
+ """
+ *dims, N, _ = Y.shape
+ N = torch.ones(*dims, 1, 1) * N
+
+ if weight is not None:
+ Y = Y * weight
+ X = X * weight
+ N = weight.sum(dim=-2, keepdim=True) # (*, 1, 1)
+
+ # subtract mean
+ my = Y.sum(dim=-2) / N[..., 0] # (*, 3)
+ mx = X.sum(dim=-2) / N[..., 0]
+ y0 = Y - my[..., None, :] # (*, N, 3)
+ x0 = X - mx[..., None, :]
+
+ if weight is not None:
+ y0 = y0 * weight
+ x0 = x0 * weight
+
+ # correlation
+ C = torch.matmul(y0.transpose(-1, -2), x0) / N # (*, 3, 3)
+ U, D, Vh = torch.linalg.svd(C) # (*, 3, 3), (*, 3), (*, 3, 3)
+
+ S = torch.eye(3).reshape(*(1,) * (len(dims)), 3, 3).repeat(*dims, 1, 1)
+ neg = torch.det(U) * torch.det(Vh.transpose(-1, -2)) < 0
+ S[neg, 2, 2] = -1
+
+ R = torch.matmul(U, torch.matmul(S, Vh)) # (*, 3, 3)
+
+ D = torch.diag_embed(D) # (*, 3, 3)
+ if fixed_scale:
+ s = torch.ones(*dims, 1, device=Y.device, dtype=torch.float32)
+ else:
+ var = torch.sum(torch.square(x0), dim=(-1, -2), keepdim=True) / N # (*, 1, 1)
+ s = (
+ torch.diagonal(torch.matmul(D, S), dim1=-2, dim2=-1).sum(
+ dim=-1, keepdim=True
+ )
+ / var[..., 0]
+ ) # (*, 1)
+
+ t = my - s * torch.matmul(R, mx[..., None])[..., 0] # (*, 3)
+
+ return s, R, t
+
+
+def global_align_joints(gt_joints, pred_joints):
+ """
+ :param gt_joints (T, J, 3)
+ :param pred_joints (T, J, 3)
+ """
+ s_glob, R_glob, t_glob = align_pcl(
+ gt_joints.reshape(-1, 3), pred_joints.reshape(-1, 3)
+ )
+ pred_glob = (
+ s_glob * torch.einsum("ij,tnj->tni", R_glob, pred_joints) + t_glob[None, None]
+ )
+ return pred_glob
+
+
+def first_align_joints(gt_joints, pred_joints):
+ """
+ align the first two frames
+ :param gt_joints (T, J, 3)
+ :param pred_joints (T, J, 3)
+ """
+ # (1, 1), (1, 3, 3), (1, 3)
+ s_first, R_first, t_first = align_pcl(
+ gt_joints[:2].reshape(1, -1, 3), pred_joints[:2].reshape(1, -1, 3)
+ )
+ pred_first = (
+ s_first * torch.einsum("tij,tnj->tni", R_first, pred_joints) + t_first[:, None]
+ )
+ return pred_first
+
+
+def rearrange_by_mask(x, mask):
+ """
+ x (L, *)
+ mask (M,), M >= L
+ """
+ M = mask.size(0)
+ L = x.size(0)
+ if M == L:
+ return x
+ assert M > L
+ assert mask.sum() == L
+ x_rearranged = torch.zeros((M, *x.size()[1:]), dtype=x.dtype, device=x.device)
+ x_rearranged[mask] = x
+ return x_rearranged
+
+
+def as_np_array(d):
+ if isinstance(d, torch.Tensor):
+ return d.cpu().numpy()
+ elif isinstance(d, np.ndarray):
+ return d
+ else:
+ return np.array(d)
+
+
+def compute_motion_beats(keypoints):
+ keypoints = keypoints.reshape(-1, 24, 3)
+ kinetic_vel = np.mean(
+ np.sqrt(np.sum((keypoints[1:] - keypoints[:-1]) ** 2, axis=2)), axis=1
+ )
+ kinetic_vel = gaussian_filter(kinetic_vel, sigma=5)
+ motion_beats = argrelextrema(kinetic_vel, np.less)
+ return motion_beats
+
+
+def compute_music_beats(beats):
+ beats = beats.astype(bool)
+ beat_axis = np.arange(len(beats))
+ beat_axis = beat_axis[beats]
+
+ return beat_axis
diff --git a/genmo/utils/gather.py b/genmo/utils/gather.py
new file mode 100644
index 0000000000000000000000000000000000000000..0d0a8aa094ac7cc0239622a2f2d07afa5456da07
--- /dev/null
+++ b/genmo/utils/gather.py
@@ -0,0 +1,271 @@
+# Copyright (c) Facebook, Inc. and its affiliates. All Rights Reserved
+"""
+[Copied from detectron2]
+This file contains primitives for multi-gpu communication.
+This is useful when doing distributed training.
+"""
+
+import functools
+import logging
+import pickle
+
+import numpy as np
+import torch
+import torch.distributed as dist
+
+_LOCAL_PROCESS_GROUP = None
+"""
+A torch process group which only includes processes that on the same machine as the current process.
+This variable is set when processes are spawned by `launch()` in "engine/launch.py".
+"""
+
+
+def get_world_size() -> int:
+ if not dist.is_available():
+ return 1
+ if not dist.is_initialized():
+ return 1
+ return dist.get_world_size()
+
+
+def get_rank() -> int:
+ if not dist.is_available():
+ return 0
+ if not dist.is_initialized():
+ return 0
+ return dist.get_rank()
+
+
+def get_local_rank() -> int:
+ """
+ Returns:
+ The rank of the current process within the local (per-machine) process group.
+ """
+ if not dist.is_available():
+ return 0
+ if not dist.is_initialized():
+ return 0
+ assert _LOCAL_PROCESS_GROUP is not None
+ return dist.get_rank(group=_LOCAL_PROCESS_GROUP)
+
+
+def get_local_size() -> int:
+ """
+ Returns:
+ The size of the per-machine process group,
+ i.e. the number of processes per machine.
+ """
+ if not dist.is_available():
+ return 1
+ if not dist.is_initialized():
+ return 1
+ return dist.get_world_size(group=_LOCAL_PROCESS_GROUP)
+
+
+def is_main_process() -> bool:
+ return get_rank() == 0
+
+
+def synchronize():
+ """
+ Helper function to synchronize (barrier) among all processes when
+ using distributed training
+ """
+ if not dist.is_available():
+ return
+ if not dist.is_initialized():
+ return
+ world_size = dist.get_world_size()
+ if world_size == 1:
+ return
+ dist.barrier()
+
+
+@functools.lru_cache()
+def _get_global_gloo_group():
+ """
+ Return a process group based on gloo backend, containing all the ranks
+ The result is cached.
+ """
+ if dist.get_backend() == "nccl":
+ return dist.new_group(backend="gloo")
+ else:
+ return dist.group.WORLD
+
+
+def _serialize_to_tensor(data, group):
+ backend = dist.get_backend(group)
+ assert backend in ["gloo", "nccl"]
+ device = torch.device("cpu" if backend == "gloo" else "cuda")
+
+ buffer = pickle.dumps(data)
+ if len(buffer) > 1024**3:
+ logger = logging.getLogger(__name__)
+ logger.warning(
+ "Rank {} trying to all-gather {:.2f} GB of data on device {}".format(
+ get_rank(), len(buffer) / (1024**3), device
+ )
+ )
+ storage = torch.ByteStorage.from_buffer(buffer)
+ tensor = torch.ByteTensor(storage).to(device=device)
+ return tensor
+
+
+def _pad_to_largest_tensor(tensor, group):
+ """
+ Returns:
+ list[int]: size of the tensor, on each rank
+ Tensor: padded tensor that has the max size
+ """
+ world_size = dist.get_world_size(group=group)
+ assert world_size >= 1, (
+ "comm.gather/all_gather must be called from ranks within the given group!"
+ )
+ local_size = torch.tensor([tensor.numel()], dtype=torch.int64, device=tensor.device)
+ size_list = [
+ torch.zeros([1], dtype=torch.int64, device=tensor.device)
+ for _ in range(world_size)
+ ]
+ dist.all_gather(size_list, local_size, group=group)
+
+ size_list = [int(size.item()) for size in size_list]
+
+ max_size = max(size_list)
+
+ # we pad the tensor because torch all_gather does not support
+ # gathering tensors of different shapes
+ if local_size != max_size:
+ padding = torch.zeros(
+ (max_size - local_size,), dtype=torch.uint8, device=tensor.device
+ )
+ tensor = torch.cat((tensor, padding), dim=0)
+ return size_list, tensor
+
+
+def all_gather(data, group=None):
+ """
+ Run all_gather on arbitrary picklable data (not necessarily tensors).
+
+ Args:
+ data: any picklable object
+ group: a torch process group. By default, will use a group which
+ contains all ranks on gloo backend.
+
+ Returns:
+ list[data]: list of data gathered from each rank
+ """
+ if get_world_size() == 1:
+ return [data]
+ if group is None:
+ group = _get_global_gloo_group()
+ if dist.get_world_size(group) == 1:
+ return [data]
+
+ tensor = _serialize_to_tensor(data, group)
+
+ size_list, tensor = _pad_to_largest_tensor(tensor, group)
+ max_size = max(size_list)
+
+ # receiving Tensor from all ranks
+ tensor_list = [
+ torch.empty((max_size,), dtype=torch.uint8, device=tensor.device)
+ for _ in size_list
+ ]
+ dist.all_gather(tensor_list, tensor, group=group)
+
+ data_list = []
+ for size, tensor in zip(size_list, tensor_list):
+ buffer = tensor.cpu().numpy().tobytes()[:size]
+ data_list.append(pickle.loads(buffer))
+
+ return data_list
+
+
+def gather(data, dst=0, group=None):
+ """
+ Run gather on arbitrary picklable data (not necessarily tensors).
+
+ Args:
+ data: any picklable object
+ dst (int): destination rank
+ group: a torch process group. By default, will use a group which
+ contains all ranks on gloo backend.
+
+ Returns:
+ list[data]: on dst, a list of data gathered from each rank. Otherwise,
+ an empty list.
+ """
+ if get_world_size() == 1:
+ return [data]
+ if group is None:
+ group = _get_global_gloo_group()
+ if dist.get_world_size(group=group) == 1:
+ return [data]
+ rank = dist.get_rank(group=group)
+
+ tensor = _serialize_to_tensor(data, group)
+ size_list, tensor = _pad_to_largest_tensor(tensor, group)
+
+ # receiving Tensor from all ranks
+ if rank == dst:
+ max_size = max(size_list)
+ tensor_list = [
+ torch.empty((max_size,), dtype=torch.uint8, device=tensor.device)
+ for _ in size_list
+ ]
+ dist.gather(tensor, tensor_list, dst=dst, group=group)
+
+ data_list = []
+ for size, tensor in zip(size_list, tensor_list):
+ buffer = tensor.cpu().numpy().tobytes()[:size]
+ data_list.append(pickle.loads(buffer))
+ return data_list
+ else:
+ dist.gather(tensor, [], dst=dst, group=group)
+ return []
+
+
+def shared_random_seed():
+ """
+ Returns:
+ int: a random number that is the same across all workers.
+ If workers need a shared RNG, they can use this shared seed to
+ create one.
+
+ All workers must call this function, otherwise it will deadlock.
+ """
+ ints = np.random.randint(2**31)
+ all_ints = all_gather(ints)
+ return all_ints[0]
+
+
+def reduce_dict(input_dict, average=True):
+ """
+ Reduce the values in the dictionary from all processes so that process with rank
+ 0 has the reduced results.
+
+ Args:
+ input_dict (dict): inputs to be reduced. All the values must be scalar CUDA Tensor.
+ average (bool): whether to do average or sum
+
+ Returns:
+ a dict with the same keys as input_dict, after reduction.
+ """
+ world_size = get_world_size()
+ if world_size < 2:
+ return input_dict
+ with torch.no_grad():
+ names = []
+ values = []
+ # sort the keys so that they are consistent across processes
+ for k in sorted(input_dict.keys()):
+ names.append(k)
+ values.append(input_dict[k])
+ values = torch.stack(values, dim=0)
+ dist.reduce(values, dst=0)
+ if dist.get_rank() == 0 and average:
+ # only main process gets accumulated, so only divide by
+ # world_size in this case
+ values /= world_size
+ reduced_dict = {k: v for k, v in zip(names, values)}
+ return reduced_dict
diff --git a/genmo/utils/geo_transform.py b/genmo/utils/geo_transform.py
new file mode 100644
index 0000000000000000000000000000000000000000..0838a60c9a80e191cbc5d056acc74201bc16707d
--- /dev/null
+++ b/genmo/utils/geo_transform.py
@@ -0,0 +1,894 @@
+import cv2
+import numpy as np
+import torch
+import torch.nn.functional as F
+from einops import einsum
+
+import genmo.utils.matrix as matrix
+from genmo.utils.pylogger import Log
+from genmo.utils.rotation_conversions import (
+ euler_angles_to_matrix,
+ matrix_to_quaternion,
+ matrix_to_rotation_6d,
+ quaternion_to_axis_angle,
+)
+from genmo.utils.so3 import so3_exp_map, so3_log_map
+from third_party.GVHMR.hmr4d.utils.geo.quaternion import qbetween
+
+
+def homo_points(points):
+ """
+ Args:
+ points: (..., C)
+ Returns: (..., C+1), with 1 padded
+ """
+ return F.pad(points, [0, 1], value=1.0)
+
+
+def apply_Ts_on_seq_points(points, Ts):
+ """
+ perform translation matrix on related point
+ Args:
+ points: (..., N, 3)
+ Ts: (..., N, 4, 4)
+ Returns: (..., N, 3)
+ """
+ points = (
+ torch.torch.einsum("...ki,...i->...k", Ts[..., :3, :3], points) + Ts[..., :3, 3]
+ )
+ return points
+
+
+def apply_T_on_points(points, T):
+ """
+ Args:
+ points: (..., N, 3)
+ T: (..., 4, 4)
+ Returns: (..., N, 3)
+ """
+ points_T = (
+ torch.einsum("...ki,...ji->...jk", T[..., :3, :3], points) + T[..., None, :3, 3]
+ )
+ return points_T
+
+
+def T_transforms_points(T, points, pattern):
+ """manual mode of apply_T_on_points
+ T: (..., 4, 4)
+ points: (..., 3)
+ pattern: "... c d, ... d -> ... c"
+ """
+ return einsum(T, homo_points(points), pattern)[..., :3]
+
+
+def project_p2d(points, K=None, is_pinhole=True):
+ """
+ Args:
+ points: (..., (N), 3)
+ K: (..., 3, 3)
+ Returns: shape is similar to points but without z
+ """
+ points = points.clone()
+ if is_pinhole:
+ z = points[..., [-1]]
+ z.masked_fill_(z.abs() < 1e-6, 1e-6)
+ points_proj = points / z
+ else: # orthogonal
+ points_proj = F.pad(points[..., :2], (0, 1), value=1)
+
+ if K is not None:
+ # Handle N
+ if len(points_proj.shape) == len(K.shape):
+ p2d_h = torch.einsum("...ki,...ji->...jk", K, points_proj)
+ else:
+ p2d_h = torch.einsum("...ki,...i->...k", K, points_proj)
+ else:
+ p2d_h = points_proj[..., :2]
+
+ return p2d_h[..., :2]
+
+
+def gen_uv_from_HW(H, W, device="cpu"):
+ """Returns: (H, W, 2), as float. Note: uv not ij"""
+ grid_v, grid_u = torch.meshgrid(torch.arange(H), torch.arange(W))
+ return (
+ torch.stack(
+ [grid_u, grid_v],
+ dim=-1,
+ )
+ .float()
+ .to(device)
+ ) # (H, W, 2)
+
+
+def unproject_p2d(uv, z, K):
+ """we assume a pinhole camera for unprojection
+ uv: (B, N, 2)
+ z: (B, N, 1)
+ K: (B, 3, 3)
+ Returns: (B, N, 3)
+ """
+ xy_atz1 = (uv - K[:, None, :2, 2]) / K[:, None, [0, 1], [0, 1]] # (B, N, 2)
+ xyz = torch.cat([xy_atz1 * z, z], dim=-1)
+ return xyz
+
+
+def cvt_p2d_from_i_to_c(uv, K):
+ """
+ Args:
+ uv: (..., 2) or (..., N, 2)
+ K: (..., 3, 3)
+ Returns: the same shape as input uv
+ """
+ if len(uv.shape) == len(K.shape):
+ xy = (uv - K[..., None, :2, 2]) / K[..., None, [0, 1], [0, 1]]
+ else: # without N
+ xy = (uv - K[..., :2, 2]) / K[..., [0, 1], [0, 1]]
+ return xy
+
+
+def cvt_to_bi01_p2d(p2d, bbx_lurb):
+ """
+ p2d: (..., (N), 2)
+ bbx_lurb: (..., 4)
+ """
+ if len(p2d.shape) == len(bbx_lurb.shape) + 1:
+ bbx_lurb = bbx_lurb[..., None, :]
+
+ bbx_wh = bbx_lurb[..., 2:] - bbx_lurb[..., :2]
+ bi01_p2d = (p2d - bbx_lurb[..., :2]) / bbx_wh
+ return bi01_p2d
+
+
+def cvt_from_bi01_p2d(bi01_p2d, bbx_lurb):
+ """Use bbx_lurb to resize bi01_p2d to p2d (image-coordinates)
+ Args:
+ p2d: (..., 2) or (..., N, 2)
+ bbx_lurb: (..., 4)
+ Returns:
+ p2d: shape is the same as input
+ """
+ bbx_wh = bbx_lurb[..., 2:] - bbx_lurb[..., :2] # (..., 2)
+ if len(bi01_p2d.shape) == len(bbx_wh.shape) + 1:
+ p2d = (bi01_p2d * bbx_wh.unsqueeze(-2)) + bbx_lurb[..., None, :2]
+ else:
+ p2d = (bi01_p2d * bbx_wh) + bbx_lurb[..., :2]
+ return p2d
+
+
+def cvt_p2d_from_bi01_to_c(bi01, bbxs_lurb, Ks):
+ """
+ Args:
+ bi01: (..., (N), 2), value in range (0,1), the point in the bbx image
+ bbxs_lurb: (..., 4)
+ Ks: (..., 3, 3)
+ Returns:
+ c: (..., (N), 2)
+ """
+ i = cvt_from_bi01_p2d(bi01, bbxs_lurb)
+ c = cvt_p2d_from_i_to_c(i, Ks)
+ return c
+
+
+def cvt_p2d_from_pm1_to_i(p2d_pm1, bbx_xys):
+ """
+ Args:
+ p2d_pm1: (..., (N), 2), value in range (-1,1), the point in the bbx image
+ bbx_xys: (..., 3)
+ Returns:
+ p2d: (..., (N), 2)
+ """
+ return bbx_xys[..., :2] + p2d_pm1 * bbx_xys[..., [2]] / 2
+
+
+def uv2l_index(uv, W):
+ return uv[..., 0] + uv[..., 1] * W
+
+
+def l2uv_index(L, W):
+ v = torch.div(L, W, rounding_mode="floor")
+ u = L % W
+ return torch.stack([u, v], dim=-1)
+
+
+def transform_mat(R, t):
+ """
+ Args:
+ R: Bx3x3 array of a batch of rotation matrices
+ t: Bx3x(1) array of a batch of translation vectors
+ Returns:
+ T: Bx4x4 Transformation matrix
+ """
+ # No padding left or right, only add an extra row
+ if len(R.shape) > len(t.shape):
+ t = t[..., None]
+ return torch.cat([F.pad(R, [0, 0, 0, 1]), F.pad(t, [0, 0, 0, 1], value=1)], dim=-1)
+
+
+def axis_angle_to_matrix_exp_map(aa):
+ """use pytorch3d so3_exp_map
+ Args:
+ aa: (*, 3)
+ Returns:
+ R: (*, 3, 3)
+ """
+ print("Use pytorch3d.transforms.axis_angle_to_matrix instead!!!")
+ ori_shape = aa.shape[:-1]
+ return so3_exp_map(aa.reshape(-1, 3)).reshape(*ori_shape, 3, 3)
+
+
+def matrix_to_axis_angle_log_map(R):
+ """use pytorch3d so3_log_map
+ Args:
+ aa: (*, 3, 3)
+ Returns:
+ R: (*, 3)
+ """
+ print(
+ "WARINING! I met singularity problem with this function, use matrix_to_axis_angle instead!"
+ )
+ ori_shape = R.shape[:-2]
+ return so3_log_map(R.reshape(-1, 3, 3)).reshape(*ori_shape, 3)
+
+
+def matrix_to_axis_angle(R):
+ """use pytorch3d so3_log_map
+ Args:
+ aa: (*, 3, 3)
+ Returns:
+ R: (*, 3)
+ """
+ return quaternion_to_axis_angle(matrix_to_quaternion(R))
+
+
+def ransac_PnP(K, pts_2d, pts_3d, err_thr=10):
+ """solve pnp"""
+ dist_coeffs = np.zeros(shape=[8, 1], dtype="float64")
+
+ pts_2d = np.ascontiguousarray(pts_2d.astype(np.float64))
+ pts_3d = np.ascontiguousarray(pts_3d.astype(np.float64))
+ K = K.astype(np.float64)
+
+ try:
+ _, rvec, tvec, inliers = cv2.solvePnPRansac(
+ pts_3d,
+ pts_2d,
+ K,
+ dist_coeffs,
+ reprojectionError=err_thr,
+ iterationsCount=10000,
+ flags=cv2.SOLVEPNP_EPNP,
+ )
+
+ rotation = cv2.Rodrigues(rvec)[0]
+
+ pose = np.concatenate([rotation, tvec], axis=-1)
+ pose_homo = np.concatenate([pose, np.array([[0, 0, 0, 1]])], axis=0)
+
+ inliers = [] if inliers is None else inliers
+
+ return pose, pose_homo, inliers
+ except cv2.error:
+ print("CV ERROR")
+ return np.eye(4)[:3], np.eye(4), []
+
+
+def ransac_PnP_batch(K_raw, pts_2d, pts_3d, err_thr=10):
+ fit_R, fit_t = [], []
+ for b in range(K_raw.shape[0]):
+ pose, _, inliers = ransac_PnP(K_raw[b], pts_2d[b], pts_3d[b], err_thr=err_thr)
+ fit_R.append(pose[:3, :3])
+ fit_t.append(pose[:3, 3])
+ fit_R = np.stack(fit_R, axis=0)
+ fit_t = np.stack(fit_t, axis=0)
+ return fit_R, fit_t
+
+
+def get_nearby_points(points, query_verts, padding=0.0, p=1):
+ import pytorch3d.ops.knn as knn
+
+ """
+ points: (S, 3)
+ query_verts: (V, 3)
+ """
+ if p == 1:
+ max_xyz = query_verts.max(0)[0] + padding
+ min_xyz = query_verts.min(0)[0] - padding
+ idx = (
+ (
+ ((points - min_xyz) > 0).all(dim=-1)
+ * ((points - max_xyz) < 0).all(dim=-1)
+ )
+ .nonzero()
+ .squeeze(-1)
+ )
+ nearby_points = points[idx]
+ elif p == 2:
+ squared_dist, _, _ = knn.knn_points(
+ points[None], query_verts[None], K=1, return_nn=False
+ )
+ mask = squared_dist[0, :, 0] < padding**2 # (S,)
+ nearby_points = points[mask]
+
+ return nearby_points
+
+
+def unproj_bbx_to_fst(bbx_lurb, K, near_z=0.5, far_z=12.5):
+ B = bbx_lurb.size(0)
+ uv = bbx_lurb[:, [[0, 1], [2, 1], [2, 3], [0, 3], [0, 1], [2, 1], [2, 3], [0, 3]]]
+ if isinstance(near_z, float):
+ z = uv.new([near_z] * 4 + [far_z] * 4).reshape(1, 8, 1).repeat(B, 1, 1)
+ else:
+ z = torch.cat(
+ [
+ near_z[:, None, None].repeat(1, 4, 1),
+ far_z[:, None, None].repeat(1, 4, 1),
+ ],
+ dim=1,
+ )
+ c_frustum_points = unproject_p2d(uv, z, K) # (B, 8, 3)
+ return c_frustum_points
+
+
+def convert_bbx_xys_to_lurb(bbx_xys):
+ """
+ Args: bbx_xys (..., 3) -> bbx_lurb (..., 4)
+ """
+ size = bbx_xys[..., 2:]
+ center = bbx_xys[..., :2]
+ lurb = torch.cat([center - size / 2, center + size / 2], dim=-1)
+ return lurb
+
+
+def convert_lurb_to_bbx_xys(bbx_lurb):
+ """
+ Args: bbx_lurb (..., 4) -> bbx_xys (..., 3) be aware that it is squared
+ """
+ size = (bbx_lurb[..., 2:] - bbx_lurb[..., :2]).max(-1, keepdim=True)[0]
+ center = (bbx_lurb[..., :2] + bbx_lurb[..., 2:]) / 2
+ return torch.cat([center, size], dim=-1)
+
+
+def get_bbx_xys(
+ i_j2d, i_j2d_mask=None, bbx_ratio=[192, 256], do_augment=False, base_enlarge=1.2
+):
+ """
+ Args:
+ i_j2d: (B, L, J, 3) [x,y,c] or (B, L, J, 2) [x,y]
+ i_j2d_mask: (B, L, J) boolean mask indicating valid joints, if None use all joints
+ bbx_ratio: [width, height] ratio for the bounding box
+ do_augment: whether to apply random augmentation
+ base_enlarge: factor to enlarge the bounding box
+ Returns:
+ bbx_xys: (B, L, 3) [center_x, center_y, size]
+ """
+ # Apply mask if provided
+ if i_j2d_mask is not None:
+ # Create a masked version of i_j2d for min/max calculations
+ # For min calculation, set masked-out joints to large positive values
+ # For max calculation, set masked-out joints to large negative values
+ mask_expanded = i_j2d_mask.unsqueeze(-1) # (B, L, J, 1)
+
+ # Create copies for min and max calculations
+ i_j2d_for_min = i_j2d.clone()
+ i_j2d_for_max = i_j2d.clone()
+
+ # Set coordinates of masked joints appropriately
+ invalid_mask = ~mask_expanded.expand_as(i_j2d[..., :2])
+ i_j2d_for_min[..., :2][invalid_mask] = float(
+ "inf"
+ ) # For min, set to large positive
+ i_j2d_for_max[..., :2][invalid_mask] = float(
+ "-inf"
+ ) # For max, set to large negative
+
+ # Calculate min/max using the filtered joints
+ min_x = i_j2d_for_min[..., 0].min(-1)[0]
+ max_x = i_j2d_for_max[..., 0].max(-1)[0]
+ min_y = i_j2d_for_min[..., 1].min(-1)[0]
+ max_y = i_j2d_for_max[..., 1].max(-1)[0]
+ else:
+ # Use all joints
+ min_x = i_j2d[..., 0].min(-1)[0]
+ max_x = i_j2d[..., 0].max(-1)[0]
+ min_y = i_j2d[..., 1].min(-1)[0]
+ max_y = i_j2d[..., 1].max(-1)[0]
+
+ center_x = (min_x + max_x) / 2
+ center_y = (min_y + max_y) / 2
+
+ # Size
+ h = max_y - min_y # (B, L)
+ w = max_x - min_x # (B, L)
+
+ if True: # fit w and h into aspect-ratio
+ aspect_ratio = bbx_ratio[0] / bbx_ratio[1]
+ mask1 = w > aspect_ratio * h
+ h[mask1] = w[mask1] / aspect_ratio
+ mask2 = w < aspect_ratio * h
+ w[mask2] = h[mask2] * aspect_ratio
+
+ # apply a common factor to enlarge the bounding box
+ bbx_size = torch.max(h, w) * base_enlarge
+
+ if do_augment:
+ B, L = bbx_size.shape[:2]
+ device = bbx_size.device
+ if True:
+ scaleFactor = torch.rand((B, L), device=device) * 0.3 + 1.05 # 1.05~1.35
+ txFactor = torch.rand((B, L), device=device) * 1.6 - 0.8 # -0.8~0.8
+ tyFactor = torch.rand((B, L), device=device) * 1.6 - 0.8 # -0.8~0.8
+ else:
+ scaleFactor = torch.rand((B, 1), device=device) * 0.3 + 1.05 # 1.05~1.35
+ txFactor = torch.rand((B, 1), device=device) * 1.6 - 0.8 # -0.8~0.8
+ tyFactor = torch.rand((B, 1), device=device) * 1.6 - 0.8 # -0.8~0.8
+
+ raw_bbx_size = bbx_size / base_enlarge
+ bbx_size = raw_bbx_size * scaleFactor
+ center_x += raw_bbx_size / 2 * ((scaleFactor - 1) * txFactor)
+ center_y += raw_bbx_size / 2 * ((scaleFactor - 1) * tyFactor)
+
+ return torch.stack([center_x, center_y, bbx_size], dim=-1)
+
+
+def get_bbx_xys_from_xyxy(bbx_xyxy, base_enlarge=1.2):
+ """
+ Args:
+ bbx_xyxy: (N, 4) [x1, y1, x2, y2]
+ Returns:
+ bbx_xys: (N, 3) [center_x, center_y, size]
+ """
+
+ i_p2d = torch.stack([bbx_xyxy[:, [0, 1]], bbx_xyxy[:, [2, 3]]], dim=1) # (L, 2, 2)
+ bbx_xys = get_bbx_xys(i_p2d[None], base_enlarge=base_enlarge)[0]
+ return bbx_xys
+
+
+def normalize_kp2d(obs_kp2d, bbx_xys, clamp_scale_min=False):
+ """
+ Args:
+ obs_kp2d: (B, L, J, 3) [x, y, c]
+ bbx_xys: (B, L, 3)
+ Returns:
+ obs: (B, L, J, 3) [x, y, c]
+ """
+ obs_xy = obs_kp2d[..., :2] # (B, L, J, 2)
+ center = bbx_xys[..., :2]
+ scale = bbx_xys[..., [2]]
+
+ # Mark keypoints outside the bounding box as invisible
+ xy_max = center + scale / 2
+ xy_min = center - scale / 2
+ invisible_mask = (
+ (obs_xy[..., 0] < xy_min[..., None, 0])
+ + (obs_xy[..., 0] > xy_max[..., None, 0])
+ + (obs_xy[..., 1] < xy_min[..., None, 1])
+ + (obs_xy[..., 1] > xy_max[..., None, 1])
+ )
+ scale = scale.clamp(min=1e-2)
+ normalized_obs_xy = 2 * (obs_xy - center.unsqueeze(-2)) / scale.unsqueeze(-2)
+
+ if obs_kp2d.shape[-1] > 2:
+ obs_conf = obs_kp2d[..., 2] # (B, L, J)
+ obs_conf = obs_conf * ~invisible_mask
+ return torch.cat([normalized_obs_xy, obs_conf[..., None]], dim=-1)
+ else:
+ return normalized_obs_xy
+
+
+# ================== AZ/AY Transformations ================== #
+
+
+def compute_T_ayf2az(joints, inverse=False):
+ """
+ Args:
+ joints: (B, J, 3), in the start-frame, az-coordinate
+ Returns:
+ if inverse == False:
+ T_af2az: (B, 4, 4)
+ else :
+ T_az2af: (B, 4, 4)
+ """
+
+ t_ayf2az = joints[:, 0, :].detach().clone()
+ t_ayf2az[:, 2] = 0 # do not modify z
+
+ RL_xy_h = (
+ joints[:, 1, [0, 1]] - joints[:, 2, [0, 1]]
+ ) # (B, 2), hip point to left side
+ RL_xy_s = (
+ joints[:, 16, [0, 1]] - joints[:, 17, [0, 1]]
+ ) # (B, 2), shoulder point to left side
+ RL_xy = RL_xy_h + RL_xy_s
+ I_mask = (
+ RL_xy.pow(2).sum(-1) < 1e-4
+ ) # do not rotate, when can't decided the face direction
+ if I_mask.sum() > 0:
+ Log.warn("{} samples can't decide the face direction".format(I_mask.sum()))
+ x_dir = F.pad(F.normalize(RL_xy, 2, -1), (0, 1), value=0) # (B, 3)
+ y_dir = torch.zeros_like(x_dir)
+ y_dir[..., 2] = 1
+ z_dir = torch.cross(x_dir, y_dir, dim=-1)
+ R_ayf2az = torch.stack([x_dir, y_dir, z_dir], dim=-1) # (B, 3, 3)
+ R_ayf2az[I_mask] = torch.eye(3).to(R_ayf2az)
+
+ if inverse:
+ R_az2ayf = R_ayf2az.transpose(1, 2) # (B, 3, 3)
+ t_az2ayf = -einsum(R_ayf2az, t_ayf2az, "b i j , b i -> b j") # (B, 3)
+ return transform_mat(R_az2ayf, t_az2ayf)
+ else:
+ return transform_mat(R_ayf2az, t_ayf2az)
+
+
+def compute_T_ayfz2ay(joints, inverse=False):
+ """
+ Args:
+ joints: (B, J, 3), in the start-frame, ay-coordinate
+ Returns:
+ if inverse == False:
+ T_ayfz2ay: (B, 4, 4)
+ else :
+ T_ay2ayfz: (B, 4, 4)
+ """
+ t_ayfz2ay = joints[:, 0, :].detach().clone()
+ t_ayfz2ay[:, 1] = 0 # do not modify y
+
+ RL_xz_h = (
+ joints[:, 1, [0, 2]] - joints[:, 2, [0, 2]]
+ ) # (B, 2), hip point to left side
+ RL_xz_s = (
+ joints[:, 16, [0, 2]] - joints[:, 17, [0, 2]]
+ ) # (B, 2), shoulder point to left side
+ RL_xz = RL_xz_h + RL_xz_s
+ I_mask = (
+ RL_xz.pow(2).sum(-1) < 1e-4
+ ) # do not rotate, when can't decided the face direction
+ if I_mask.sum() > 0:
+ Log.warn("{} samples can't decide the face direction".format(I_mask.sum()))
+
+ x_dir = torch.zeros_like(t_ayfz2ay) # (B, 3)
+ x_dir[:, [0, 2]] = F.normalize(RL_xz, 2, -1)
+ y_dir = torch.zeros_like(x_dir)
+ y_dir[..., 1] = 1 # (B, 3)
+ z_dir = torch.cross(x_dir, y_dir, dim=-1)
+ R_ayfz2ay = torch.stack([x_dir, y_dir, z_dir], dim=-1) # (B, 3, 3)
+ R_ayfz2ay[I_mask] = torch.eye(3).to(R_ayfz2ay)
+
+ if inverse:
+ R_ay2ayfz = R_ayfz2ay.transpose(1, 2)
+ t_ay2ayfz = -einsum(R_ayfz2ay, t_ayfz2ay, "b i j , b i -> b j")
+ return transform_mat(R_ay2ayfz, t_ay2ayfz)
+ else:
+ return transform_mat(R_ayfz2ay, t_ayfz2ay)
+
+
+def compute_T_ay2ayrot(joints):
+ """
+ Args:
+ joints: (B, J, 3), in the start-frame, ay-coordinate
+ Returns:
+ T_ay2ayrot: (B, 4, 4)
+ """
+ t_ayrot2ay = joints[:, 0, :].detach().clone()
+ t_ayrot2ay[:, 1] = 0 # do not modify y
+
+ B = joints.shape[0]
+ euler_angle = torch.zeros((B, 3), device=joints.device)
+ yrot_angle = torch.rand((B,), device=joints.device) * 2 * torch.pi
+ euler_angle[:, 0] = yrot_angle
+ R_ay2ayrot = euler_angles_to_matrix(euler_angle, "YXZ") # (B, 3, 3)
+
+ R_ayrot2ay = R_ay2ayrot.transpose(1, 2)
+ t_ay2ayrot = -einsum(R_ayrot2ay, t_ayrot2ay, "b i j , b i -> b j")
+ return transform_mat(R_ay2ayrot, t_ay2ayrot)
+
+
+def compute_root_quaternion_ay(joints):
+ """
+ Args:
+ joints: (B, J, 3), in the start-frame, ay-coordinate
+ Returns:
+ root_quat: (B, 4) from z-axis to fz
+ """
+ joints_shape = joints.shape
+ joints = joints.reshape((-1,) + joints_shape[-2:])
+ t_ayfz2ay = joints[:, 0, :].detach().clone()
+ t_ayfz2ay[:, 1] = 0 # do not modify y
+
+ RL_xz_h = (
+ joints[:, 1, [0, 2]] - joints[:, 2, [0, 2]]
+ ) # (B, 2), hip point to left side
+ RL_xz_s = (
+ joints[:, 16, [0, 2]] - joints[:, 17, [0, 2]]
+ ) # (B, 2), shoulder point to left side
+ RL_xz = RL_xz_h + RL_xz_s
+ I_mask = (
+ RL_xz.pow(2).sum(-1) < 1e-4
+ ) # do not rotate, when can't decided the face direction
+ if I_mask.sum() > 0:
+ Log.warn("{} samples can't decide the face direction".format(I_mask.sum()))
+
+ x_dir = torch.zeros_like(t_ayfz2ay) # (B, 3)
+ x_dir[:, [0, 2]] = F.normalize(RL_xz, 2, -1)
+ y_dir = torch.zeros_like(x_dir)
+ y_dir[..., 1] = 1 # (B, 3)
+ z_dir = torch.cross(x_dir, y_dir, dim=-1)
+
+ z_dir[..., 2] += 1e-9
+ pos_z_vec = torch.tensor([0, 0, 1]).to(joints.device).float() # (3,)
+ root_quat = qbetween(pos_z_vec[None], z_dir) # (B, 4)
+ root_quat = root_quat.reshape(joints_shape[:-2] + (4,))
+ return root_quat
+
+
+# ================== Transformations between two sets of features ================== #
+
+
+def similarity_transform_batch(S1, S2):
+ """
+ Computes a similarity transform (sR, t) that solves the orthogonal Procrutes problem.
+ Args:
+ S1, S2: (*, L, 3)
+ """
+ assert S1.shape == S2.shape
+ S_shape = S1.shape
+ S1 = S1.reshape(-1, *S_shape[-2:])
+ S2 = S2.reshape(-1, *S_shape[-2:])
+
+ S1 = S1.transpose(-2, -1)
+ S2 = S2.transpose(-2, -1)
+
+ # --- The code is borrowed from WHAM ---
+ # 1. Remove mean.
+ mu1 = S1.mean(axis=-1, keepdims=True) # axis is along N, S1(B, 3, N)
+ mu2 = S2.mean(axis=-1, keepdims=True)
+
+ X1 = S1 - mu1
+ X2 = S2 - mu2
+
+ # 2. Compute variance of X1 used for scale.
+ var1 = torch.sum(X1**2, dim=1).sum(dim=1)
+
+ # 3. The outer product of X1 and X2.
+ K = X1.bmm(X2.permute(0, 2, 1))
+
+ # 4. Solution that Maximizes trace(R'K) is R=U*V', where U, V are
+ # singular vectors of K.
+ U, s, V = torch.svd(K)
+
+ # Construct Z that fixes the orientation of R to get det(R)=1.
+ Z = torch.eye(U.shape[1], device=S1.device).unsqueeze(0)
+ Z = Z.repeat(U.shape[0], 1, 1)
+ Z[:, -1, -1] *= torch.sign(torch.det(U.bmm(V.permute(0, 2, 1))))
+
+ # Construct R.
+ R = V.bmm(Z.bmm(U.permute(0, 2, 1)))
+
+ # 5. Recover scale.
+ scale = torch.cat([torch.trace(x).unsqueeze(0) for x in R.bmm(K)]) / var1
+
+ # 6. Recover translation.
+ t = mu2 - (scale.unsqueeze(-1).unsqueeze(-1) * (R.bmm(mu1)))
+
+ # -------
+ # reshape back
+ # sR = scale[:, None, None] * R
+ # sR = sR.reshape(*S_shape[:-2], 3, 3)
+ scale = scale.reshape(*S_shape[:-2], 1, 1)
+ R = R.reshape(*S_shape[:-2], 3, 3)
+ t = t.reshape(*S_shape[:-2], 3, 1)
+
+ return (scale, R), t
+
+
+def kabsch_algorithm_batch(X1, X2):
+ """
+ Computes a rigid transform (R, t)
+ Args:
+ X1, X2: (*, L, 3)
+ """
+ assert X1.shape == X2.shape
+ X_shape = X1.shape
+ X1 = X1.reshape(-1, *X_shape[-2:])
+ X2 = X2.reshape(-1, *X_shape[-2:])
+
+ # 1. 计算质心
+ centroid_X1 = torch.mean(X1, dim=-2, keepdim=True)
+ centroid_X2 = torch.mean(X2, dim=-2, keepdim=True)
+
+ # 2. 去中心化
+ X1_centered = X1 - centroid_X1
+ X2_centered = X2 - centroid_X2
+
+ # 3. 计算协方差矩阵
+ H = torch.matmul(X1_centered.transpose(-2, -1), X2_centered)
+
+ # 4. 奇异值分解
+ U, S, Vt = torch.linalg.svd(H)
+
+ # 5. 计算旋转矩阵
+ R = torch.matmul(Vt.transpose(-2, -1), U.transpose(-2, -1))
+
+ # 修正反射矩阵
+ d = (torch.det(R) < 0).unsqueeze(-1).unsqueeze(-1)
+ Vt = torch.where(d, -Vt, Vt)
+ R = torch.matmul(Vt.transpose(-2, -1), U.transpose(-2, -1))
+
+ # 6. 计算平移向量
+ t = centroid_X2.transpose(-2, -1) - torch.matmul(R, centroid_X1.transpose(-2, -1))
+
+ # -------
+ # reshape back
+ R = R.reshape(*X_shape[:-2], 3, 3)
+ t = t.reshape(*X_shape[:-2], 3, 1)
+
+ return R, t
+
+
+# ===== WHAM cam_angvel ===== #
+
+
+def compute_cam_angvel(R_w2c, padding_last=True):
+ """
+ R_w2c : (F, 3, 3)
+ """
+ # R @ R0 = R1, so R = R1 @ R0^T
+ cam_angvel = matrix_to_rotation_6d(
+ R_w2c[1:] @ R_w2c[:-1].transpose(-1, -2)
+ ) # (F-1, 6)
+ # cam_angvel = (cam_angvel - torch.tensor([[1, 0, 0, 0, 1, 0]])) * FPS
+ assert padding_last
+ cam_angvel = torch.cat([cam_angvel, cam_angvel[-1:]], dim=0) # (F, 6)
+ return cam_angvel.float()
+
+
+def compute_cam_tvel(t_w2c, padding_last=True):
+ """
+ t_w2c : (F, 3)
+ """
+ cam_tvel = t_w2c[1:] - t_w2c[:-1]
+ assert padding_last
+ cam_tvel = torch.cat([cam_tvel, cam_tvel[-1:]], dim=0) # (F, 3)
+ return cam_tvel.float()
+
+
+def compute_cam_tcw2_vel(T_w2c, padding_last=True):
+ """
+ T_w2c : (F, 4, 4)
+ """
+ T_c2w = T_w2c.inverse()
+ t_c2w = T_c2w[:, :3, 3]
+ cam_tvel = t_c2w[1:] - t_c2w[:-1]
+ assert padding_last
+ cam_tvel = torch.cat([cam_tvel, cam_tvel[-1:]], dim=0) # (F, 3)
+ return cam_tvel.float()
+
+
+def ransac_gravity_vec(xyz, num_iterations=100, threshold=0.05, verbose=False):
+ # xyz: (L, 3)
+ N = xyz.shape[0]
+ max_inliers = []
+ # best_model = None
+ norms = xyz.norm(dim=-1) # (L,)
+
+ for _ in range(num_iterations):
+ # random select a sample
+ sample_index = np.random.randint(N)
+ sample = xyz[sample_index] # (3,)
+
+ # compute the angle difference between all points and the sample
+ dot_product = (xyz * sample).sum(dim=-1) # (L,)
+ angles = dot_product / norms * norms[sample_index] # (L,)
+ angles = torch.clamp(angles, -1, 1) # prevent numerical errors
+ angles = torch.acos(angles)
+
+ # determine the inliers
+ inliers = xyz[angles < threshold]
+
+ if len(inliers) > len(max_inliers):
+ max_inliers = inliers
+ # best_model = sample
+ if len(max_inliers) == N:
+ break
+ if verbose:
+ print(f"Inliers: {len(max_inliers)} / {N}")
+ result = max_inliers.mean(dim=0)
+
+ return result, max_inliers
+
+
+def sequence_best_cammat(w_j3d, c_j3d, cam_rot):
+ # get best camera estimation along the sequence, requires static camera
+ # w_j3d: (L, J, 3)
+ # c_j3d: (L, J, 3)
+ # cam_rot: (L, 3, 3)
+
+ L, J, _ = w_j3d.shape
+
+ root_in_w = w_j3d[:, 0] # (L, 3)
+ root_in_c = c_j3d[:, 0] # (L, 3)
+ cam_mat = matrix.get_TRS(cam_rot, root_in_w) # (L, 4, 4)
+ cam_pos = matrix.get_position_from(-root_in_c[:, None], cam_mat)[:, 0] # (L, 3)
+ cam_mat = matrix.set_position(cam_mat, cam_pos) # (L, 4, 4)
+
+ w_j3d_expand = w_j3d[None].expand(L, -1, -1, -1) # (L, L, J, 3)
+ w_j3d_expand = w_j3d_expand.reshape(L, -1, 3) # (L, L*J, 3)
+
+ # get reproject error
+ w_j3d_expand_in_c = matrix.get_relative_position_to(
+ w_j3d_expand, cam_mat
+ ) # (L, L*J, 3)
+ w_j2d_expand_in_c = project_p2d(w_j3d_expand_in_c) # (L, L*J, 2)
+ w_j2d_expand_in_c = w_j2d_expand_in_c.reshape(L, L, J, 2) # (L, L, J, 2)
+ c_j2d = project_p2d(c_j3d) # (L, J, 2)
+ error = w_j2d_expand_in_c - c_j2d[None] # (L, L, J, 2)
+ error = error.norm(dim=-1).mean(dim=-1) # (L, L)
+ error = error.mean(dim=-1) # (L,)
+ ind = error.argmin()
+ return cam_mat[ind], ind
+
+
+def get_sequence_cammat(w_j3d, c_j3d, cam_rot):
+ # w_j3d: (L, J, 3)
+ # c_j3d: (L, J, 3)
+ # cam_rot: (L, 3, 3)
+
+ L, J, _ = w_j3d.shape
+
+ root_in_w = w_j3d[:, 0] # (L, 3)
+ root_in_c = c_j3d[:, 0] # (L, 3)
+ cam_mat = matrix.get_TRS(cam_rot, root_in_w) # (L, 4, 4)
+ cam_pos = matrix.get_position_from(-root_in_c[:, None], cam_mat)[:, 0] # (L, 3)
+ cam_mat = matrix.set_position(cam_mat, cam_pos) # (L, 4, 4)
+ return cam_mat
+
+
+def ransac_vec(vel, min_multiply=20, verbose=False):
+ # xyz: (L, 3)
+ # remove outlier velocity
+ N = vel.shape[0]
+ vel_1 = vel[None].expand(N, -1, -1) # (L, L, 3)
+ vel_2 = vel[:, None].expand(-1, N, -1) # (L, L, 3)
+ dist_mat = (vel_1 - vel_2).norm(dim=-1) # (L, L)
+ big_identity = torch.eye(N, device=vel.device) * 1e6
+ dist_mat_ = dist_mat + big_identity
+ threshold = dist_mat_.min() * min_multiply
+ inner_mask = dist_mat < threshold # (L, L)
+ inner_num = inner_mask.sum(dim=-1) # (L, )
+ ind = inner_num.argmax()
+ result = vel[inner_mask[ind]].mean(dim=0) # (3,)
+ if verbose:
+ print(inner_mask[ind].sum().item())
+
+ return result, inner_mask[ind]
+
+
+def as_identity(R):
+ is_I = matrix_to_axis_angle(R).norm(dim=-1) < 1e-5
+ R[is_I] = torch.eye(3)[None].expand(is_I.sum(), -1, -1).to(R)
+ return R
+
+
+def normalize_T_w2c(T_w2c):
+ if T_w2c.ndim == 2:
+ T_w2c = T_w2c[None]
+ L = T_w2c.shape[0]
+ device = T_w2c.device
+ norm_T_c2w = torch.eye(4)[None].repeat(L, 1, 1).to(device)
+
+ T_c2w = T_w2c.inverse()
+ R_c2w = as_identity(T_c2w[:, :3, :3])
+ t_c2w = T_c2w[:, :3, 3]
+
+ # align the first frame
+ R0_c2w = R_c2w[:1]
+ t0_c2w = t_c2w[:1]
+ norm_R_c2w = R0_c2w.mT @ R_c2w
+ norm_t_c2w = (R0_c2w.mT @ (t_c2w - t0_c2w)[..., None])[..., 0]
+ norm_T_c2w[:, :3, :3] = norm_R_c2w
+ norm_T_c2w[:, :3, 3] = norm_t_c2w
+ norm_T_w2c = norm_T_c2w.inverse()
+ norm_T_w2c[:, :3, :3] = as_identity(norm_T_w2c[:, :3, :3])
+ norm_T_w2c[:, 3, :3] = 0
+
+ return norm_T_w2c
diff --git a/genmo/utils/konia_transform.py b/genmo/utils/konia_transform.py
new file mode 100644
index 0000000000000000000000000000000000000000..5903aedaaf4db1fa5b63f48207a4f39aa32584c2
--- /dev/null
+++ b/genmo/utils/konia_transform.py
@@ -0,0 +1,1112 @@
+import enum
+import warnings
+from typing import Tuple
+
+import numpy as np
+import torch
+import torch.nn.functional as F
+
+__all__ = [
+ # functional api
+ "rad2deg",
+ "deg2rad",
+ "pol2cart",
+ "cart2pol",
+ "convert_points_from_homogeneous",
+ "convert_points_to_homogeneous",
+ "convert_affinematrix_to_homography",
+ "convert_affinematrix_to_homography3d",
+ "angle_axis_to_rotation_matrix",
+ "angle_axis_to_quaternion",
+ "rotation_matrix_to_angle_axis",
+ "rotation_matrix_to_quaternion",
+ "quaternion_to_angle_axis",
+ "quaternion_to_rotation_matrix",
+ "quaternion_log_to_exp",
+ "quaternion_exp_to_log",
+ "denormalize_pixel_coordinates",
+ "normalize_pixel_coordinates",
+ "normalize_quaternion",
+ "denormalize_pixel_coordinates3d",
+ "normalize_pixel_coordinates3d",
+]
+
+
+class QuaternionCoeffOrder(enum.Enum):
+ XYZW = "xyzw"
+ WXYZ = "wxyz"
+
+
+@torch.jit.script
+def torch_safe_atan2(y, x, eps: float = 1e-6):
+ y = y.clone()
+ if len(y.shape) == 0:
+ if y.abs() < eps and x.abs() < eps:
+ y += eps
+ else:
+ y[(y.abs() < eps) & (x.abs() < eps)] += eps
+ return torch.atan2(y, x)
+
+
+@torch.jit.script
+def rad2deg(tensor: torch.Tensor) -> torch.Tensor:
+ r"""Function that converts angles from radians to degrees.
+
+ Args:
+ tensor: Tensor of arbitrary shape.
+
+ Returns:
+ Tensor with same shape as input.
+
+ Example:
+ >>> input = torch.tensor(3.1415926535) * torch.rand(1, 3, 3)
+ >>> output = rad2deg(input)
+ """
+
+ pi = np.pi
+ if not isinstance(tensor, torch.Tensor):
+ raise TypeError("Input type is not a torch.Tensor. Got {}".format(type(tensor)))
+
+ return 180.0 * tensor / pi
+
+
+@torch.jit.script
+def deg2rad(tensor: torch.Tensor) -> torch.Tensor:
+ r"""Function that converts angles from degrees to radians.
+
+ Args:
+ tensor: Tensor of arbitrary shape.
+
+ Returns:
+ tensor with same shape as input.
+
+ Examples:
+ >>> input = 360. * torch.rand(1, 3, 3)
+ >>> output = deg2rad(input)
+ """
+
+ pi = np.pi
+ if not isinstance(tensor, torch.Tensor):
+ raise TypeError("Input type is not a torch.Tensor. Got {}".format(type(tensor)))
+
+ return tensor * pi / 180.0
+
+
+@torch.jit.script
+def pol2cart(rho: torch.Tensor, phi: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
+ r"""Function that converts polar coordinates to cartesian coordinates.
+
+ Args:
+ rho: Tensor of arbitrary shape.
+ phi: Tensor of same arbitrary shape.
+
+ Returns:
+ Tensor with same shape as input.
+
+ Example:
+ >>> rho = torch.rand(1, 3, 3)
+ >>> phi = torch.rand(1, 3, 3)
+ >>> x, y = pol2cart(rho, phi)
+ """
+ if not (isinstance(rho, torch.Tensor) & isinstance(phi, torch.Tensor)):
+ raise TypeError(
+ "Input type is not a torch.Tensor. Got {}, {}".format(type(rho), type(phi))
+ )
+
+ x = rho * torch.cos(phi)
+ y = rho * torch.sin(phi)
+ return x, y
+
+
+@torch.jit.script
+def cart2pol(
+ x: torch.Tensor, y: torch.Tensor, eps: float = 1.0e-8
+) -> Tuple[torch.Tensor, torch.Tensor]:
+ """Function that converts cartesian coordinates to polar coordinates.
+
+ Args:
+ rho: Tensor of arbitrary shape.
+ phi: Tensor of same arbitrary shape.
+ eps: To avoid division by zero.
+
+ Returns:
+ Tensor with same shape as input.
+
+ Example:
+ >>> x = torch.rand(1, 3, 3)
+ >>> y = torch.rand(1, 3, 3)
+ >>> rho, phi = cart2pol(x, y)
+ """
+ if not (isinstance(x, torch.Tensor) & isinstance(y, torch.Tensor)):
+ raise TypeError(
+ "Input type is not a torch.Tensor. Got {}, {}".format(type(x), type(y))
+ )
+
+ rho = torch.sqrt((x**2 + y**2).clamp_min(eps))
+ phi = torch_safe_atan2(y, x)
+ return rho, phi
+
+
+@torch.jit.script
+def convert_points_from_homogeneous(
+ points: torch.Tensor, eps: float = 1e-8
+) -> torch.Tensor:
+ r"""Function that converts points from homogeneous to Euclidean space.
+
+ Args:
+ points: the points to be transformed.
+ eps: to avoid division by zero.
+
+ Returns:
+ the points in Euclidean space.
+
+ Examples:
+ >>> input = torch.rand(2, 4, 3) # BxNx3
+ >>> output = convert_points_from_homogeneous(input) # BxNx2
+ """
+ if not isinstance(points, torch.Tensor):
+ raise TypeError("Input type is not a torch.Tensor. Got {}".format(type(points)))
+
+ if len(points.shape) < 2:
+ raise ValueError(
+ "Input must be at least a 2D tensor. Got {}".format(points.shape)
+ )
+
+ # we check for points at max_val
+ z_vec: torch.Tensor = points[..., -1:]
+
+ # set the results of division by zeror/near-zero to 1.0
+ # follow the convention of opencv:
+ # https://github.com/opencv/opencv/pull/14411/files
+ mask: torch.Tensor = torch.abs(z_vec) > eps
+ scale = torch.where(mask, 1.0 / (z_vec + eps), torch.ones_like(z_vec))
+
+ return scale * points[..., :-1]
+
+
+@torch.jit.script
+def convert_points_to_homogeneous(points: torch.Tensor) -> torch.Tensor:
+ r"""Function that converts points from Euclidean to homogeneous space.
+
+ Args:
+ points: the points to be transformed.
+
+ Returns:
+ the points in homogeneous coordinates.
+
+ Examples:
+ >>> input = torch.rand(2, 4, 3) # BxNx3
+ >>> output = convert_points_to_homogeneous(input) # BxNx4
+ """
+ if not isinstance(points, torch.Tensor):
+ raise TypeError("Input type is not a torch.Tensor. Got {}".format(type(points)))
+ if len(points.shape) < 2:
+ raise ValueError(
+ "Input must be at least a 2D tensor. Got {}".format(points.shape)
+ )
+
+ return torch.nn.functional.pad(points, [0, 1], "constant", 1.0)
+
+
+@torch.jit.script
+def _convert_affinematrix_to_homography_impl(A: torch.Tensor) -> torch.Tensor:
+ H: torch.Tensor = torch.nn.functional.pad(A, [0, 0, 0, 1], "constant", value=0.0)
+ H[..., -1, -1] += 1.0
+ return H
+
+
+@torch.jit.script
+def convert_affinematrix_to_homography(A: torch.Tensor) -> torch.Tensor:
+ r"""Function that converts batch of affine matrices.
+
+ Args:
+ A: the affine matrix with shape :math:`(B,2,3)`.
+
+ Returns:
+ the homography matrix with shape of :math:`(B,3,3)`.
+
+ Examples:
+ >>> input = torch.rand(2, 2, 3) # Bx2x3
+ >>> output = convert_affinematrix_to_homography(input) # Bx3x3
+ """
+ if not isinstance(A, torch.Tensor):
+ raise TypeError("Input type is not a torch.Tensor. Got {}".format(type(A)))
+ if not (len(A.shape) == 3 and A.shape[-2:] == (2, 3)):
+ raise ValueError("Input matrix must be a Bx2x3 tensor. Got {}".format(A.shape))
+ return _convert_affinematrix_to_homography_impl(A)
+
+
+@torch.jit.script
+def convert_affinematrix_to_homography3d(A: torch.Tensor) -> torch.Tensor:
+ r"""Function that converts batch of 3d affine matrices.
+
+ Args:
+ A: the affine matrix with shape :math:`(B,3,4)`.
+
+ Returns:
+ the homography matrix with shape of :math:`(B,4,4)`.
+
+ Examples:
+ >>> input = torch.rand(2, 3, 4) # Bx3x4
+ >>> output = convert_affinematrix_to_homography3d(input) # Bx4x4
+ """
+ if not isinstance(A, torch.Tensor):
+ raise TypeError("Input type is not a torch.Tensor. Got {}".format(type(A)))
+ if not (len(A.shape) == 3 and A.shape[-2:] == (3, 4)):
+ raise ValueError("Input matrix must be a Bx3x4 tensor. Got {}".format(A.shape))
+ return _convert_affinematrix_to_homography_impl(A)
+
+
+@torch.jit.script
+def _compute_rotation_matrix(angle_axis, theta2, eps: float = 1e-6):
+ # We want to be careful to only evaluate the square root if the
+ # norm of the angle_axis vector is greater than zero. Otherwise
+ # we get a division by zero.
+ k_one = 1.0
+ theta = torch.sqrt(theta2.clamp_min(eps))
+ wxyz = angle_axis / (theta + eps)
+ wx, wy, wz = torch.chunk(wxyz, 3, dim=1)
+ cos_theta = torch.cos(theta)
+ sin_theta = torch.sin(theta)
+
+ r00 = cos_theta + wx * wx * (k_one - cos_theta)
+ r10 = wz * sin_theta + wx * wy * (k_one - cos_theta)
+ r20 = -wy * sin_theta + wx * wz * (k_one - cos_theta)
+ r01 = wx * wy * (k_one - cos_theta) - wz * sin_theta
+ r11 = cos_theta + wy * wy * (k_one - cos_theta)
+ r21 = wx * sin_theta + wy * wz * (k_one - cos_theta)
+ r02 = wy * sin_theta + wx * wz * (k_one - cos_theta)
+ r12 = -wx * sin_theta + wy * wz * (k_one - cos_theta)
+ r22 = cos_theta + wz * wz * (k_one - cos_theta)
+ rotation_matrix = torch.cat([r00, r01, r02, r10, r11, r12, r20, r21, r22], dim=1)
+ return rotation_matrix.view(-1, 3, 3)
+
+
+@torch.jit.script
+def _compute_rotation_matrix_taylor(angle_axis):
+ rx, ry, rz = torch.chunk(angle_axis, 3, dim=1)
+ k_one = torch.ones_like(rx)
+ rotation_matrix = torch.cat([k_one, -rz, ry, rz, k_one, -rx, -ry, rx, k_one], dim=1)
+ return rotation_matrix.view(-1, 3, 3)
+
+
+@torch.jit.script
+def angle_axis_to_rotation_matrix(angle_axis: torch.Tensor) -> torch.Tensor:
+ r"""Convert 3d vector of axis-angle rotation to 3x3 rotation matrix.
+
+ Args:
+ angle_axis: tensor of 3d vector of axis-angle rotations.
+
+ Returns:
+ tensor of 3x3 rotation matrices.
+
+ Shape:
+ - Input: :math:`(N, 3)`
+ - Output: :math:`(N, 3, 3)`
+
+ Example:
+ >>> input = torch.rand(1, 3) # Nx3
+ >>> output = angle_axis_to_rotation_matrix(input) # Nx3x3
+ """
+ if not isinstance(angle_axis, torch.Tensor):
+ raise TypeError(
+ "Input type is not a torch.Tensor. Got {}".format(type(angle_axis))
+ )
+
+ if not angle_axis.shape[-1] == 3:
+ raise ValueError(
+ "Input size must be a (*, 3) tensor. Got {}".format(angle_axis.shape)
+ )
+
+ orig_shape = angle_axis.shape
+ angle_axis = angle_axis.reshape(-1, 3)
+
+ # stolen from ceres/rotation.h
+
+ _angle_axis = torch.unsqueeze(angle_axis, dim=1)
+ theta2 = torch.matmul(_angle_axis, _angle_axis.transpose(1, 2))
+ theta2 = torch.squeeze(theta2, dim=1)
+
+ # compute rotation matrices
+ rotation_matrix_normal = _compute_rotation_matrix(angle_axis, theta2)
+ rotation_matrix_taylor = _compute_rotation_matrix_taylor(angle_axis)
+
+ # create mask to handle both cases
+ eps = 1e-6
+ mask = (theta2 > eps).view(-1, 1, 1).to(theta2.device)
+ mask_pos = (mask).type_as(theta2)
+ mask_neg = (mask == torch.tensor(False)).type_as(theta2) # noqa
+
+ # create output pose matrix
+ batch_size = angle_axis.shape[0]
+ rotation_matrix = torch.eye(3).to(angle_axis.device).type_as(angle_axis)
+ rotation_matrix = rotation_matrix.view(1, 3, 3).repeat(batch_size, 1, 1)
+ # fill output matrix with masked values
+ rotation_matrix[..., :3, :3] = (
+ mask_pos * rotation_matrix_normal + mask_neg * rotation_matrix_taylor
+ )
+
+ rotation_matrix = rotation_matrix.view(orig_shape[:-1] + (3, 3))
+ return rotation_matrix # Nx3x3
+
+
+@torch.jit.script
+def safe_zero_division(
+ numerator: torch.Tensor, denominator: torch.Tensor, eps: float = 1.0e-6
+) -> torch.Tensor:
+ denominator = denominator.clone()
+ if len(denominator.shape) == 0:
+ if denominator.abs() < eps:
+ denominator += eps
+ else:
+ denominator[denominator.abs() < eps] += eps
+ return numerator / denominator
+
+
+@torch.jit.script
+def rotation_matrix_to_quaternion(
+ rotation_matrix: torch.Tensor,
+ eps: float = 1.0e-6,
+ order: QuaternionCoeffOrder = QuaternionCoeffOrder.WXYZ,
+) -> torch.Tensor:
+ r"""Convert 3x3 rotation matrix to 4d quaternion vector.
+
+ The quaternion vector has components in (w, x, y, z) or (x, y, z, w) format.
+
+ .. note::
+ The (x, y, z, w) order is going to be deprecated in favor of efficiency.
+
+ Args:
+ rotation_matrix: the rotation matrix to convert.
+ eps: small value to avoid zero division.
+ order: quaternion coefficient order. Note: 'xyzw' will be deprecated in favor of 'wxyz'.
+
+ Return:
+ the rotation in quaternion.
+
+ Shape:
+ - Input: :math:`(*, 3, 3)`
+ - Output: :math:`(*, 4)`
+
+ Example:
+ >>> input = torch.rand(4, 3, 3) # Nx3x3
+ >>> output = rotation_matrix_to_quaternion(input, eps=torch.finfo(input.dtype).eps,
+ ... order=QuaternionCoeffOrder.WXYZ) # Nx4
+ """
+ if not isinstance(rotation_matrix, torch.Tensor):
+ raise TypeError(
+ f"Input type is not a torch.Tensor. Got {type(rotation_matrix)}"
+ )
+
+ if not rotation_matrix.shape[-2:] == (3, 3):
+ raise ValueError(
+ f"Input size must be a (*, 3, 3) tensor. Got {rotation_matrix.shape}"
+ )
+
+ if not torch.jit.is_scripting():
+ if order.name not in QuaternionCoeffOrder.__members__.keys():
+ raise ValueError(
+ f"order must be one of {QuaternionCoeffOrder.__members__.keys()}"
+ )
+
+ if order == QuaternionCoeffOrder.XYZW:
+ warnings.warn(
+ "`XYZW` quaternion coefficient order is deprecated and"
+ " will be removed after > 0.6. "
+ "Please use `QuaternionCoeffOrder.WXYZ` instead."
+ )
+
+ m00, m01, m02 = (
+ rotation_matrix[..., 0, 0],
+ rotation_matrix[..., 0, 1],
+ rotation_matrix[..., 0, 2],
+ )
+ m10, m11, m12 = (
+ rotation_matrix[..., 1, 0],
+ rotation_matrix[..., 1, 1],
+ rotation_matrix[..., 1, 2],
+ )
+ m20, m21, m22 = (
+ rotation_matrix[..., 2, 0],
+ rotation_matrix[..., 2, 1],
+ rotation_matrix[..., 2, 2],
+ )
+
+ trace: torch.Tensor = m00 + m11 + m22
+
+ sq = torch.sqrt((trace + 1.0).clamp_min(eps)) * 2.0 # sq = 4 * qw.
+ qw = 0.25 * sq
+ qx = safe_zero_division(m21 - m12, sq)
+ qy = safe_zero_division(m02 - m20, sq)
+ qz = safe_zero_division(m10 - m01, sq)
+ if order == QuaternionCoeffOrder.XYZW:
+ trace_positive_cond = torch.stack((qx, qy, qz, qw), dim=-1)
+ trace_positive_cond = torch.stack((qw, qx, qy, qz), dim=-1)
+
+ sq = torch.sqrt((1.0 + m00 - m11 - m22).clamp_min(eps)) * 2.0 # sq = 4 * qx.
+ qw = safe_zero_division(m21 - m12, sq)
+ qx = 0.25 * sq
+ qy = safe_zero_division(m01 + m10, sq)
+ qz = safe_zero_division(m02 + m20, sq)
+ if order == QuaternionCoeffOrder.XYZW:
+ cond_1 = torch.stack((qx, qy, qz, qw), dim=-1)
+ cond_1 = torch.stack((qw, qx, qy, qz), dim=-1)
+
+ sq = torch.sqrt((1.0 + m11 - m00 - m22).clamp_min(eps)) * 2.0 # sq = 4 * qy.
+ qw = safe_zero_division(m02 - m20, sq)
+ qx = safe_zero_division(m01 + m10, sq)
+ qy = 0.25 * sq
+ qz = safe_zero_division(m12 + m21, sq)
+ if order == QuaternionCoeffOrder.XYZW:
+ cond_2 = torch.stack((qx, qy, qz, qw), dim=-1)
+ cond_2 = torch.stack((qw, qx, qy, qz), dim=-1)
+
+ sq = torch.sqrt((1.0 + m22 - m00 - m11).clamp_min(eps)) * 2.0 # sq = 4 * qz.
+ qw = safe_zero_division(m10 - m01, sq)
+ qx = safe_zero_division(m02 + m20, sq)
+ qy = safe_zero_division(m12 + m21, sq)
+ qz = 0.25 * sq
+ if order == QuaternionCoeffOrder.XYZW:
+ cond_3 = torch.stack((qx, qy, qz, qw), dim=-1)
+ cond_3 = torch.stack((qw, qx, qy, qz), dim=-1)
+
+ where_2 = torch.where((m11 > m22).unsqueeze(-1), cond_2, cond_3)
+ where_1 = torch.where(((m00 > m11) & (m00 > m22)).unsqueeze(-1), cond_1, where_2)
+
+ quaternion: torch.Tensor = torch.where(
+ (trace > 0.0).unsqueeze(-1), trace_positive_cond, where_1
+ )
+ return quaternion
+
+
+@torch.jit.script
+def normalize_quaternion(
+ quaternion: torch.Tensor, eps: float = 1.0e-12
+) -> torch.Tensor:
+ r"""Normalizes a quaternion.
+
+ The quaternion should be in (x, y, z, w) format.
+
+ Args:
+ quaternion: a tensor containing a quaternion to be normalized.
+ The tensor can be of shape :math:`(*, 4)`.
+ eps: small value to avoid division by zero.
+
+ Return:
+ the normalized quaternion of shape :math:`(*, 4)`.
+
+ Example:
+ >>> quaternion = torch.tensor((1., 0., 1., 0.))
+ >>> normalize_quaternion(quaternion)
+ tensor([0.7071, 0.0000, 0.7071, 0.0000])
+ """
+ if not isinstance(quaternion, torch.Tensor):
+ raise TypeError(
+ "Input type is not a torch.Tensor. Got {}".format(type(quaternion))
+ )
+
+ if not quaternion.shape[-1] == 4:
+ raise ValueError(
+ "Input must be a tensor of shape (*, 4). Got {}".format(quaternion.shape)
+ )
+ return F.normalize(quaternion, p=2.0, dim=-1, eps=eps)
+
+
+# based on:
+# https://github.com/matthew-brett/transforms3d/blob/8965c48401d9e8e66b6a8c37c65f2fc200a076fa/transforms3d/quaternions.py#L101
+# https://github.com/tensorflow/graphics/blob/master/tensorflow_graphics/geometry/transformation/rotation_matrix_3d.py#L247
+
+
+@torch.jit.script
+def quaternion_to_rotation_matrix(
+ quaternion: torch.Tensor, order: QuaternionCoeffOrder = QuaternionCoeffOrder.WXYZ
+) -> torch.Tensor:
+ r"""Converts a quaternion to a rotation matrix.
+
+ The quaternion should be in (x, y, z, w) or (w, x, y, z) format.
+
+ Args:
+ quaternion: a tensor containing a quaternion to be converted.
+ The tensor can be of shape :math:`(*, 4)`.
+ order: quaternion coefficient order. Note: 'xyzw' will be deprecated in favor of 'wxyz'.
+
+ Return:
+ the rotation matrix of shape :math:`(*, 3, 3)`.
+
+ Example:
+ >>> quaternion = torch.tensor((0., 0., 0., 1.))
+ >>> quaternion_to_rotation_matrix(quaternion, order=QuaternionCoeffOrder.WXYZ)
+ tensor([[-1., 0., 0.],
+ [ 0., -1., 0.],
+ [ 0., 0., 1.]])
+ """
+ if not isinstance(quaternion, torch.Tensor):
+ raise TypeError(f"Input type is not a torch.Tensor. Got {type(quaternion)}")
+
+ if not quaternion.shape[-1] == 4:
+ raise ValueError(
+ f"Input must be a tensor of shape (*, 4). Got {quaternion.shape}"
+ )
+
+ if not torch.jit.is_scripting():
+ if order.name not in QuaternionCoeffOrder.__members__.keys():
+ raise ValueError(
+ f"order must be one of {QuaternionCoeffOrder.__members__.keys()}"
+ )
+
+ if order == QuaternionCoeffOrder.XYZW:
+ warnings.warn(
+ "`XYZW` quaternion coefficient order is deprecated and"
+ " will be removed after > 0.6. "
+ "Please use `QuaternionCoeffOrder.WXYZ` instead."
+ )
+
+ # normalize the input quaternion
+ quaternion_norm: torch.Tensor = normalize_quaternion(quaternion)
+
+ # unpack the normalized quaternion components
+ if order == QuaternionCoeffOrder.XYZW:
+ x, y, z, w = (
+ quaternion_norm[..., 0],
+ quaternion_norm[..., 1],
+ quaternion_norm[..., 2],
+ quaternion_norm[..., 3],
+ )
+ else:
+ w, x, y, z = (
+ quaternion_norm[..., 0],
+ quaternion_norm[..., 1],
+ quaternion_norm[..., 2],
+ quaternion_norm[..., 3],
+ )
+
+ # compute the actual conversion
+ tx: torch.Tensor = 2.0 * x
+ ty: torch.Tensor = 2.0 * y
+ tz: torch.Tensor = 2.0 * z
+ twx: torch.Tensor = tx * w
+ twy: torch.Tensor = ty * w
+ twz: torch.Tensor = tz * w
+ txx: torch.Tensor = tx * x
+ txy: torch.Tensor = ty * x
+ txz: torch.Tensor = tz * x
+ tyy: torch.Tensor = ty * y
+ tyz: torch.Tensor = tz * y
+ tzz: torch.Tensor = tz * z
+ one: torch.Tensor = torch.tensor(1.0)
+
+ matrix: torch.Tensor = torch.stack(
+ (
+ one - (tyy + tzz),
+ txy - twz,
+ txz + twy,
+ txy + twz,
+ one - (txx + tzz),
+ tyz - twx,
+ txz - twy,
+ tyz + twx,
+ one - (txx + tyy),
+ ),
+ dim=-1,
+ ).view(quaternion.shape[:-1] + (3, 3))
+
+ # if len(quaternion.shape) == 1:
+ # matrix = torch.squeeze(matrix, dim=0)
+ return matrix
+
+
+@torch.jit.script
+def quaternion_to_angle_axis(
+ quaternion: torch.Tensor,
+ eps: float = 1.0e-6,
+ order: QuaternionCoeffOrder = QuaternionCoeffOrder.WXYZ,
+) -> torch.Tensor:
+ """Convert quaternion vector to angle axis of rotation.
+
+ The quaternion should be in (x, y, z, w) or (w, x, y, z) format.
+
+ Adapted from ceres C++ library: ceres-solver/include/ceres/rotation.h
+
+ Args:
+ quaternion: tensor with quaternions.
+ order: quaternion coefficient order. Note: 'xyzw' will be deprecated in favor of 'wxyz'.
+
+ Return:
+ tensor with angle axis of rotation.
+
+ Shape:
+ - Input: :math:`(*, 4)` where `*` means, any number of dimensions
+ - Output: :math:`(*, 3)`
+
+ Example:
+ >>> quaternion = torch.rand(2, 4) # Nx4
+ >>> angle_axis = quaternion_to_angle_axis(quaternion) # Nx3
+ """
+
+ if not quaternion.shape[-1] == 4:
+ raise ValueError(
+ f"Input must be a tensor of shape Nx4 or 4. Got {quaternion.shape}"
+ )
+
+ if not torch.jit.is_scripting():
+ if order.name not in QuaternionCoeffOrder.__members__.keys():
+ raise ValueError(
+ f"order must be one of {QuaternionCoeffOrder.__members__.keys()}"
+ )
+
+ if order == QuaternionCoeffOrder.XYZW:
+ warnings.warn(
+ "`XYZW` quaternion coefficient order is deprecated and"
+ " will be removed after > 0.6. "
+ "Please use `QuaternionCoeffOrder.WXYZ` instead."
+ )
+ # unpack input and compute conversion
+ q1: torch.Tensor = torch.tensor([])
+ q2: torch.Tensor = torch.tensor([])
+ q3: torch.Tensor = torch.tensor([])
+ cos_theta: torch.Tensor = torch.tensor([])
+
+ if order == QuaternionCoeffOrder.XYZW:
+ q1 = quaternion[..., 0]
+ q2 = quaternion[..., 1]
+ q3 = quaternion[..., 2]
+ cos_theta = quaternion[..., 3]
+ else:
+ cos_theta = quaternion[..., 0]
+ q1 = quaternion[..., 1]
+ q2 = quaternion[..., 2]
+ q3 = quaternion[..., 3]
+
+ sin_squared_theta: torch.Tensor = q1 * q1 + q2 * q2 + q3 * q3
+
+ sin_theta: torch.Tensor = torch.sqrt((sin_squared_theta).clamp_min(eps))
+ two_theta: torch.Tensor = 2.0 * torch.where(
+ cos_theta < 0.0,
+ torch_safe_atan2(-sin_theta, -cos_theta),
+ torch_safe_atan2(sin_theta, cos_theta),
+ )
+
+ k_pos: torch.Tensor = safe_zero_division(two_theta, sin_theta, eps)
+ k_neg: torch.Tensor = 2.0 * torch.ones_like(sin_theta)
+ k: torch.Tensor = torch.where(sin_squared_theta > 0.0, k_pos, k_neg)
+
+ angle_axis: torch.Tensor = torch.zeros_like(quaternion)[..., :3]
+ angle_axis[..., 0] += q1 * k
+ angle_axis[..., 1] += q2 * k
+ angle_axis[..., 2] += q3 * k
+ return angle_axis
+
+
+@torch.jit.script
+def rotation_matrix_to_angle_axis(rotation_matrix: torch.Tensor) -> torch.Tensor:
+ r"""Convert 3x3 rotation matrix to Rodrigues vector.
+
+ Args:
+ rotation_matrix: rotation matrix.
+
+ Returns:
+ Rodrigues vector transformation.
+
+ Shape:
+ - Input: :math:`(N, 3, 3)`
+ - Output: :math:`(N, 3)`
+
+ Example:
+ >>> input = torch.rand(2, 3, 3) # Nx3x3
+ >>> output = rotation_matrix_to_angle_axis(input) # Nx3
+ """
+ if not isinstance(rotation_matrix, torch.Tensor):
+ raise TypeError(
+ f"Input type is not a torch.Tensor. Got {type(rotation_matrix)}"
+ )
+
+ if not rotation_matrix.shape[-2:] == (3, 3):
+ raise ValueError(
+ f"Input size must be a (*, 3, 3) tensor. Got {rotation_matrix.shape}"
+ )
+ quaternion: torch.Tensor = rotation_matrix_to_quaternion(
+ rotation_matrix, order=QuaternionCoeffOrder.WXYZ
+ )
+ return quaternion_to_angle_axis(quaternion, order=QuaternionCoeffOrder.WXYZ)
+
+
+@torch.jit.script
+def quaternion_log_to_exp(
+ quaternion: torch.Tensor,
+ eps: float = 1.0e-6,
+ order: QuaternionCoeffOrder = QuaternionCoeffOrder.WXYZ,
+) -> torch.Tensor:
+ r"""Applies exponential map to log quaternion.
+
+ The quaternion should be in (x, y, z, w) or (w, x, y, z) format.
+
+ Args:
+ quaternion: a tensor containing a quaternion to be converted.
+ The tensor can be of shape :math:`(*, 3)`.
+ order: quaternion coefficient order. Note: 'xyzw' will be deprecated in favor of 'wxyz'.
+
+ Return:
+ the quaternion exponential map of shape :math:`(*, 4)`.
+
+ Example:
+ >>> quaternion = torch.tensor((0., 0., 0.))
+ >>> quaternion_log_to_exp(quaternion, eps=torch.finfo(quaternion.dtype).eps,
+ ... order=QuaternionCoeffOrder.WXYZ)
+ tensor([1., 0., 0., 0.])
+ """
+ if not isinstance(quaternion, torch.Tensor):
+ raise TypeError(f"Input type is not a torch.Tensor. Got {type(quaternion)}")
+
+ if not quaternion.shape[-1] == 3:
+ raise ValueError(
+ f"Input must be a tensor of shape (*, 3). Got {quaternion.shape}"
+ )
+
+ if not torch.jit.is_scripting():
+ if order.name not in QuaternionCoeffOrder.__members__.keys():
+ raise ValueError(
+ f"order must be one of {QuaternionCoeffOrder.__members__.keys()}"
+ )
+
+ if order == QuaternionCoeffOrder.XYZW:
+ warnings.warn(
+ "`XYZW` quaternion coefficient order is deprecated and"
+ " will be removed after > 0.6. "
+ "Please use `QuaternionCoeffOrder.WXYZ` instead."
+ )
+
+ # compute quaternion norm
+ norm_q: torch.Tensor = torch.norm(quaternion, p=2, dim=-1, keepdim=True).clamp(
+ min=eps
+ )
+
+ # compute scalar and vector
+ quaternion_vector: torch.Tensor = quaternion * torch.sin(norm_q) / norm_q
+ quaternion_scalar: torch.Tensor = torch.cos(norm_q)
+
+ # compose quaternion and return
+ quaternion_exp: torch.Tensor = torch.tensor([])
+ if order == QuaternionCoeffOrder.XYZW:
+ quaternion_exp = torch.cat((quaternion_vector, quaternion_scalar), dim=-1)
+ else:
+ quaternion_exp = torch.cat((quaternion_scalar, quaternion_vector), dim=-1)
+
+ return quaternion_exp
+
+
+@torch.jit.script
+def quaternion_exp_to_log(
+ quaternion: torch.Tensor,
+ eps: float = 1.0e-6,
+ order: QuaternionCoeffOrder = QuaternionCoeffOrder.WXYZ,
+) -> torch.Tensor:
+ r"""Applies the log map to a quaternion.
+
+ The quaternion should be in (x, y, z, w) format.
+
+ Args:
+ quaternion: a tensor containing a quaternion to be converted.
+ The tensor can be of shape :math:`(*, 4)`.
+ eps: A small number for clamping.
+ order: quaternion coefficient order. Note: 'xyzw' will be deprecated in favor of 'wxyz'.
+
+ Return:
+ the quaternion log map of shape :math:`(*, 3)`.
+
+ Example:
+ >>> quaternion = torch.tensor((1., 0., 0., 0.))
+ >>> quaternion_exp_to_log(quaternion, eps=torch.finfo(quaternion.dtype).eps,
+ ... order=QuaternionCoeffOrder.WXYZ)
+ tensor([0., 0., 0.])
+ """
+ if not isinstance(quaternion, torch.Tensor):
+ raise TypeError(f"Input type is not a torch.Tensor. Got {type(quaternion)}")
+
+ if not quaternion.shape[-1] == 4:
+ raise ValueError(
+ f"Input must be a tensor of shape (*, 4). Got {quaternion.shape}"
+ )
+
+ if not torch.jit.is_scripting():
+ if order.name not in QuaternionCoeffOrder.__members__.keys():
+ raise ValueError(
+ f"order must be one of {QuaternionCoeffOrder.__members__.keys()}"
+ )
+
+ if order == QuaternionCoeffOrder.XYZW:
+ warnings.warn(
+ "`XYZW` quaternion coefficient order is deprecated and"
+ " will be removed after > 0.6. "
+ "Please use `QuaternionCoeffOrder.WXYZ` instead."
+ )
+
+ # unpack quaternion vector and scalar
+ quaternion_vector: torch.Tensor = torch.tensor([])
+ quaternion_scalar: torch.Tensor = torch.tensor([])
+
+ if order == QuaternionCoeffOrder.XYZW:
+ quaternion_vector = quaternion[..., 0:3]
+ quaternion_scalar = quaternion[..., 3:4]
+ else:
+ quaternion_scalar = quaternion[..., 0:1]
+ quaternion_vector = quaternion[..., 1:4]
+
+ # compute quaternion norm
+ norm_q: torch.Tensor = torch.norm(
+ quaternion_vector, p=2, dim=-1, keepdim=True
+ ).clamp(min=eps)
+
+ # apply log map
+ quaternion_log: torch.Tensor = (
+ quaternion_vector
+ * torch.acos(torch.clamp(quaternion_scalar, min=-1.0 + eps, max=1.0 - eps))
+ / norm_q
+ )
+
+ return quaternion_log
+
+
+# based on:
+# https://github.com/facebookresearch/QuaterNet/blob/master/common/quaternion.py#L138
+
+
+@torch.jit.script
+def angle_axis_to_quaternion(
+ angle_axis: torch.Tensor,
+ eps: float = 1.0e-6,
+ order: QuaternionCoeffOrder = QuaternionCoeffOrder.WXYZ,
+) -> torch.Tensor:
+ r"""Convert an angle axis to a quaternion.
+
+ The quaternion vector has components in (x, y, z, w) or (w, x, y, z) format.
+
+ Adapted from ceres C++ library: ceres-solver/include/ceres/rotation.h
+
+ Args:
+ angle_axis: tensor with angle axis.
+ order: quaternion coefficient order. Note: 'xyzw' will be deprecated in favor of 'wxyz'.
+
+ Return:
+ tensor with quaternion.
+
+ Shape:
+ - Input: :math:`(*, 3)` where `*` means, any number of dimensions
+ - Output: :math:`(*, 4)`
+
+ Example:
+ >>> angle_axis = torch.rand(2, 3) # Nx3
+ >>> quaternion = angle_axis_to_quaternion(angle_axis, order=QuaternionCoeffOrder.WXYZ) # Nx4
+ """
+
+ if not angle_axis.shape[-1] == 3:
+ raise ValueError(
+ f"Input must be a tensor of shape Nx3 or 3. Got {angle_axis.shape}"
+ )
+
+ if not torch.jit.is_scripting():
+ if order.name not in QuaternionCoeffOrder.__members__.keys():
+ raise ValueError(
+ f"order must be one of {QuaternionCoeffOrder.__members__.keys()}"
+ )
+
+ if order == QuaternionCoeffOrder.XYZW:
+ warnings.warn(
+ "`XYZW` quaternion coefficient order is deprecated and"
+ " will be removed after > 0.6. "
+ "Please use `QuaternionCoeffOrder.WXYZ` instead."
+ )
+
+ # unpack input and compute conversion
+ a0: torch.Tensor = angle_axis[..., 0:1]
+ a1: torch.Tensor = angle_axis[..., 1:2]
+ a2: torch.Tensor = angle_axis[..., 2:3]
+ theta_squared: torch.Tensor = a0 * a0 + a1 * a1 + a2 * a2
+
+ theta: torch.Tensor = torch.sqrt((theta_squared).clamp_min(eps))
+ half_theta: torch.Tensor = theta * 0.5
+
+ mask: torch.Tensor = theta_squared > 0.0
+ ones: torch.Tensor = torch.ones_like(half_theta)
+
+ k_neg: torch.Tensor = 0.5 * ones
+ k_pos: torch.Tensor = safe_zero_division(torch.sin(half_theta), theta, eps)
+ k: torch.Tensor = torch.where(mask, k_pos, k_neg)
+ w: torch.Tensor = torch.where(mask, torch.cos(half_theta), ones)
+
+ quaternion: torch.Tensor = torch.zeros(
+ size=angle_axis.shape[:-1] + (4,),
+ dtype=angle_axis.dtype,
+ device=angle_axis.device,
+ )
+ if order == QuaternionCoeffOrder.XYZW:
+ quaternion[..., 0:1] = a0 * k
+ quaternion[..., 1:2] = a1 * k
+ quaternion[..., 2:3] = a2 * k
+ quaternion[..., 3:4] = w
+ else:
+ quaternion[..., 1:2] = a0 * k
+ quaternion[..., 2:3] = a1 * k
+ quaternion[..., 3:4] = a2 * k
+ quaternion[..., 0:1] = w
+ return quaternion
+
+
+# based on:
+# https://github.com/ClementPinard/SfmLearner-Pytorch/blob/master/inverse_warp.py#L65-L71
+
+
+@torch.jit.script
+def normalize_pixel_coordinates(
+ pixel_coordinates: torch.Tensor, height: int, width: int, eps: float = 1e-8
+) -> torch.Tensor:
+ r"""Normalize pixel coordinates between -1 and 1.
+
+ Normalized, -1 if on extreme left, 1 if on extreme right (x = w-1).
+
+ Args:
+ pixel_coordinates: the grid with pixel coordinates. Shape can be :math:`(*, 2)`.
+ width: the maximum width in the x-axis.
+ height: the maximum height in the y-axis.
+ eps: safe division by zero.
+
+ Return:
+ the normalized pixel coordinates.
+ """
+ if pixel_coordinates.shape[-1] != 2:
+ raise ValueError(
+ "Input pixel_coordinates must be of shape (*, 2). Got {}".format(
+ pixel_coordinates.shape
+ )
+ )
+ # compute normalization factor
+ hw: torch.Tensor = torch.stack(
+ [
+ torch.tensor(
+ width, device=pixel_coordinates.device, dtype=pixel_coordinates.dtype
+ ),
+ torch.tensor(
+ height, device=pixel_coordinates.device, dtype=pixel_coordinates.dtype
+ ),
+ ]
+ )
+
+ factor: torch.Tensor = torch.tensor(
+ 2.0, device=pixel_coordinates.device, dtype=pixel_coordinates.dtype
+ ) / (hw - 1).clamp(eps)
+
+ return factor * pixel_coordinates - 1
+
+
+@torch.jit.script
+def denormalize_pixel_coordinates(
+ pixel_coordinates: torch.Tensor, height: int, width: int, eps: float = 1e-8
+) -> torch.Tensor:
+ r"""Denormalize pixel coordinates.
+
+ The input is assumed to be -1 if on extreme left, 1 if on extreme right (x = w-1).
+
+ Args:
+ pixel_coordinates: the normalized grid coordinates. Shape can be :math:`(*, 2)`.
+ width: the maximum width in the x-axis.
+ height: the maximum height in the y-axis.
+ eps: safe division by zero.
+
+ Return:
+ the denormalized pixel coordinates.
+ """
+ if pixel_coordinates.shape[-1] != 2:
+ raise ValueError(
+ "Input pixel_coordinates must be of shape (*, 2). Got {}".format(
+ pixel_coordinates.shape
+ )
+ )
+ # compute normalization factor
+ hw: torch.Tensor = (
+ torch.stack([torch.tensor(width), torch.tensor(height)])
+ .to(pixel_coordinates.device)
+ .to(pixel_coordinates.dtype)
+ )
+
+ factor: torch.Tensor = torch.tensor(2.0) / (hw - 1).clamp(eps)
+
+ return torch.tensor(1.0) / factor * (pixel_coordinates + 1)
+
+
+@torch.jit.script
+def normalize_pixel_coordinates3d(
+ pixel_coordinates: torch.Tensor,
+ depth: int,
+ height: int,
+ width: int,
+ eps: float = 1e-8,
+) -> torch.Tensor:
+ r"""Normalize pixel coordinates between -1 and 1.
+
+ Normalized, -1 if on extreme left, 1 if on extreme right (x = w-1).
+
+ Args:
+ pixel_coordinates: the grid with pixel coordinates. Shape can be :math:`(*, 3)`.
+ depth: the maximum depth in the z-axis.
+ height: the maximum height in the y-axis.
+ width: the maximum width in the x-axis.
+ eps: safe division by zero.
+
+ Return:
+ the normalized pixel coordinates.
+ """
+ if pixel_coordinates.shape[-1] != 3:
+ raise ValueError(
+ "Input pixel_coordinates must be of shape (*, 3). Got {}".format(
+ pixel_coordinates.shape
+ )
+ )
+ # compute normalization factor
+ dhw: torch.Tensor = (
+ torch.stack([torch.tensor(depth), torch.tensor(width), torch.tensor(height)])
+ .to(pixel_coordinates.device)
+ .to(pixel_coordinates.dtype)
+ )
+
+ factor: torch.Tensor = torch.tensor(2.0) / (dhw - 1).clamp(eps)
+
+ return factor * pixel_coordinates - 1
+
+
+@torch.jit.script
+def denormalize_pixel_coordinates3d(
+ pixel_coordinates: torch.Tensor,
+ depth: int,
+ height: int,
+ width: int,
+ eps: float = 1e-8,
+) -> torch.Tensor:
+ r"""Denormalize pixel coordinates.
+
+ The input is assumed to be -1 if on extreme left, 1 if on extreme right (x = w-1).
+
+ Args:
+ pixel_coordinates: the normalized grid coordinates. Shape can be :math:`(*, 3)`.
+ depth: the maximum depth in the x-axis.
+ height: the maximum height in the y-axis.
+ width: the maximum width in the x-axis.
+ eps: safe division by zero.
+
+ Return:
+ the denormalized pixel coordinates.
+ """
+ if pixel_coordinates.shape[-1] != 3:
+ raise ValueError(
+ "Input pixel_coordinates must be of shape (*, 3). Got {}".format(
+ pixel_coordinates.shape
+ )
+ )
+ # compute normalization factor
+ dhw: torch.Tensor = (
+ torch.stack([torch.tensor(depth), torch.tensor(width), torch.tensor(height)])
+ .to(pixel_coordinates.device)
+ .to(pixel_coordinates.dtype)
+ )
+
+ factor: torch.Tensor = torch.tensor(2.0) / (dhw - 1).clamp(eps)
+
+ return torch.tensor(1.0) / factor * (pixel_coordinates + 1)
diff --git a/genmo/utils/math.py b/genmo/utils/math.py
new file mode 100644
index 0000000000000000000000000000000000000000..68d9f16acbf152451ad8b124d2681d15b5f9737c
--- /dev/null
+++ b/genmo/utils/math.py
@@ -0,0 +1,86 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the BSD-style license found in the
+# LICENSE file in the root directory of this source tree.
+
+# pyre-unsafe
+
+import math
+from typing import Tuple
+
+import torch
+
+DEFAULT_ACOS_BOUND: float = 1.0 - 1e-4
+
+
+def acos_linear_extrapolation(
+ x: torch.Tensor,
+ bounds: Tuple[float, float] = (-DEFAULT_ACOS_BOUND, DEFAULT_ACOS_BOUND),
+) -> torch.Tensor:
+ """
+ Implements `arccos(x)` which is linearly extrapolated outside `x`'s original
+ domain of `(-1, 1)`. This allows for stable backpropagation in case `x`
+ is not guaranteed to be strictly within `(-1, 1)`.
+
+ More specifically::
+
+ bounds=(lower_bound, upper_bound)
+ if lower_bound <= x <= upper_bound:
+ acos_linear_extrapolation(x) = acos(x)
+ elif x <= lower_bound: # 1st order Taylor approximation
+ acos_linear_extrapolation(x)
+ = acos(lower_bound) + dacos/dx(lower_bound) * (x - lower_bound)
+ else: # x >= upper_bound
+ acos_linear_extrapolation(x)
+ = acos(upper_bound) + dacos/dx(upper_bound) * (x - upper_bound)
+
+ Args:
+ x: Input `Tensor`.
+ bounds: A float 2-tuple defining the region for the
+ linear extrapolation of `acos`.
+ The first/second element of `bound`
+ describes the lower/upper bound that defines the lower/upper
+ extrapolation region, i.e. the region where
+ `x <= bound[0]`/`bound[1] <= x`.
+ Note that all elements of `bound` have to be within (-1, 1).
+ Returns:
+ acos_linear_extrapolation: `Tensor` containing the extrapolated `arccos(x)`.
+ """
+
+ lower_bound, upper_bound = bounds
+
+ if lower_bound > upper_bound:
+ raise ValueError("lower bound has to be smaller or equal to upper bound.")
+
+ if lower_bound <= -1.0 or upper_bound >= 1.0:
+ raise ValueError("Both lower bound and upper bound have to be within (-1, 1).")
+
+ # init an empty tensor and define the domain sets
+ acos_extrap = torch.empty_like(x)
+ x_upper = x >= upper_bound
+ x_lower = x <= lower_bound
+ x_mid = (~x_upper) & (~x_lower)
+
+ # acos calculation for upper_bound < x < lower_bound
+ acos_extrap[x_mid] = torch.acos(x[x_mid])
+ # the linear extrapolation for x >= upper_bound
+ acos_extrap[x_upper] = _acos_linear_approximation(x[x_upper], upper_bound)
+ # the linear extrapolation for x <= lower_bound
+ acos_extrap[x_lower] = _acos_linear_approximation(x[x_lower], lower_bound)
+
+ return acos_extrap
+
+
+def _acos_linear_approximation(x: torch.Tensor, x0: float) -> torch.Tensor:
+ """
+ Calculates the 1st order Taylor expansion of `arccos(x)` around `x0`.
+ """
+ return (x - x0) * _dacos_dx(x0) + math.acos(x0)
+
+
+def _dacos_dx(x: float) -> float:
+ """
+ Calculates the derivative of `arccos(x)` w.r.t. `x`.
+ """
+ return (-1.0) / math.sqrt(1.0 - x * x)
diff --git a/genmo/utils/matrix.py b/genmo/utils/matrix.py
new file mode 100644
index 0000000000000000000000000000000000000000..ec1ec3755ba2370dda5eaa9249e2124b1d1880d7
--- /dev/null
+++ b/genmo/utils/matrix.py
@@ -0,0 +1,1717 @@
+import copy
+import math
+from typing import List, Optional
+
+import numpy as np
+import torch
+
+
+def identity_mat(x=None, device="cpu", is_numpy=False):
+ if x is not None:
+ if isinstance(x, torch.Tensor):
+ mat = torch.eye(4, device=device)
+ mat = mat.repeat(x.shape[:-2] + (1, 1))
+ elif isinstance(x, np.ndarray):
+ mat = np.eye(4, dtype=np.float32)
+ if x is not None:
+ for _ in range(len(x.shape) - 2):
+ mat = mat[None]
+ mat = np.tile(mat, x.shape[:-2] + (1, 1))
+ else:
+ raise ValueError
+ else:
+ # (4, 4)
+ if is_numpy:
+ mat = np.eye(4, dtype=np.float32)
+ else:
+ mat = torch.eye(4, device=device)
+
+ return mat
+
+
+def vec2mat(vec):
+ """_summary_
+
+ Args:
+ vec (tensor): [12], pos, forward, up and right
+
+ Returns:
+ mat_world(tensor): [4, 4]
+ """
+ # Assume bs = 1
+ v = np.tile(np.array([[0, 0, 0, 1]]), (1, 1))
+ if isinstance(vec, torch.Tensor):
+ v = torch.tensor(
+ v,
+ device=vec.device,
+ dtype=vec.dtype,
+ )
+ pos = vec[:3]
+ forward = vec[3:6]
+ up = vec[6:9]
+ right = vec[9:12]
+
+ if isinstance(vec, torch.Tensor):
+ mat_world = torch.stack([right, up, forward, pos], dim=-1)
+ mat_world = torch.cat([mat_world, v], dim=-2)
+ elif isinstance(vec, np.ndarray):
+ mat_world = np.stack([right, up, forward, pos], axis=-1)
+ mat_world = np.concatenate([mat_world, v], axis=-2)
+ else:
+ raise ValueError
+ mat_world = normalized_matrix(mat_world)
+ return mat_world
+
+
+def mat2vec(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [4, 4]
+
+ Returns:
+ vec (tensor): [12], pos, forward, up and right
+ """
+ # Assume bs = 1
+ pos = mat[:-1, 3]
+ forward = normalized(mat[:-1, 2])
+ up = normalized(mat[:-1, 1])
+ right = normalized(mat[:-1, 0])
+ if isinstance(mat, torch.Tensor):
+ vec = torch.cat((pos, forward, up, right))
+ elif isinstance(mat, np.ndarray):
+ vec = np.concatenate((pos, forward, up, right))
+ else:
+ raise ValueError
+
+ return vec
+
+
+def vec2mat_batch(vec):
+ """_summary_
+
+ Args:
+ vec (tensor): [B, 12], pos, forward, up and right
+
+ Returns:
+ mat_world(tensor): [B, 4, 4]
+ """
+ # Assume bs = 1
+
+ v = np.tile(np.array([[0, 0, 0, 1]], dtype=np.float32), (vec.shape[0], 1, 1))
+ if isinstance(vec, torch.Tensor):
+ v = torch.tensor(
+ v,
+ device=vec.device,
+ dtype=vec.dtype,
+ )
+ pos = vec[..., :3]
+ forward = vec[..., 3:6]
+ up = vec[..., 6:9]
+ right = vec[..., 9:12]
+ if isinstance(vec, torch.Tensor):
+ mat_world = torch.stack([right, up, forward, pos], dim=-1)
+ mat_world = torch.cat([mat_world, v], dim=-2)
+ elif isinstance(vec, np.ndarray):
+ mat_world = np.stack([right, up, forward, pos], axis=-1)
+ mat_world = np.concatenate([mat_world, v], axis=-2)
+ else:
+ raise ValueError
+
+ mat_world = normalized_matrix(mat_world)
+ return mat_world
+
+
+def rotmat2tan_norm(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 3, 3]
+
+ Returns:
+ vec (tensor): [B, 6], tan norm
+ """
+ if isinstance(mat, np.ndarray):
+ tan = np.zeros_like(mat[..., 2])
+ norm = np.zeros_like(mat[..., 0])
+ elif isinstance(mat, torch.Tensor):
+ tan = torch.zeros_like(mat[..., 2])
+ norm = torch.zeros_like(mat[..., 0])
+ else:
+ raise ValueError
+ tan[...] = mat[..., 2, ::-1]
+ tan[..., -1] *= -1
+ norm[...] = mat[..., 0, ::-1]
+ norm[..., -1] *= -1
+ if isinstance(mat, np.ndarray):
+ tan_norm = np.concatenate((tan, norm), axis=-1)
+ elif isinstance(mat, torch.Tensor):
+ tan_norm = torch.cat((tan, norm), dim=-1)
+ else:
+ raise ValueError
+ return tan_norm
+
+
+def mat2tan_norm(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 4, 4]
+
+ Returns:
+ vec (tensor): [B, 6], tan norm
+ """
+ rot_mat = mat[..., :-1, :-1]
+ return rotmat2tan_norm(rot_mat)
+
+
+def rotmat2tan_norm(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 3, 3]
+
+ Returns:
+ vec (tensor): [B, 6], tan norm
+ """
+ if isinstance(mat, np.ndarray):
+ tan = np.zeros_like(mat[..., 2])
+ norm = np.zeros_like(mat[..., 0])
+ tan[...] = mat[..., 2, ::-1]
+ norm[...] = mat[..., 0, ::-1]
+ elif isinstance(mat, torch.Tensor):
+ tan = torch.zeros_like(mat[..., 2])
+ norm = torch.zeros_like(mat[..., 0])
+ tan[...] = torch.flip(mat[..., 2], dims=[-1])
+ norm[...] = torch.flip(mat[..., 0], dims=[-1])
+ else:
+ raise ValueError
+ tan[..., -1] *= -1
+ norm[..., -1] *= -1
+ if isinstance(mat, np.ndarray):
+ tan_norm = np.concatenate((tan, norm), axis=-1)
+ elif isinstance(mat, torch.Tensor):
+ tan_norm = torch.cat((tan, norm), dim=-1)
+ else:
+ raise ValueError
+ return tan_norm
+
+
+def tan_norm2rotmat(tan_norm):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 6]
+
+ Returns:
+ vec (tensor): [B, 3]
+ """
+ tan = copy.deepcopy(tan_norm[..., :3])
+ norm = copy.deepcopy(tan_norm[..., 3:])
+ tan[..., -1] *= -1
+ norm[..., -1] *= -1
+ if isinstance(tan_norm, np.ndarray):
+ rotmat = np.zeros(tan_norm.shape[:-1] + (3, 3))
+ tan = tan[..., ::-1]
+ norm = norm[..., ::-1]
+ other = np.cross(tan, norm)
+ elif isinstance(tan_norm, torch.Tensor):
+ rotmat = torch.zeros(tan_norm.shape[:-1] + (3, 3), device=tan_norm.device)
+ tan = torch.flip(tan, dims=[-1])
+ norm = torch.flip(norm, dims=[-1])
+ other = torch.cross(tan, norm)
+ else:
+ raise ValueError
+ rotmat[..., 2, :] = tan
+ rotmat[..., 0, :] = norm
+ rotmat[..., 1, :] = other
+ return rotmat
+
+
+def rotmat332vec_batch(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 3, 3]
+
+ Returns:
+ vec (tensor): [B, 6], forward, up, right
+ """
+ # Assume bs = 1
+ mat = normalized_matrix(mat)
+ forward = mat[..., :, 2]
+ up = mat[..., :, 1]
+ right = mat[..., :, 0]
+ if isinstance(mat, torch.Tensor):
+ vec = torch.cat((forward, up, right), dim=-1)
+ elif isinstance(mat, np.ndarray):
+ vec = np.concatenate((forward, up, right), axis=-1)
+ else:
+ raise ValueError
+ return vec
+
+
+def rotmat2vec_batch(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 4, 4]
+
+ Returns:
+ vec (tensor): [B, 9], forward, up, right
+ """
+ # Assume bs = 1
+ mat = normalized_matrix(mat)
+ forward = mat[..., :-1, 2]
+ up = mat[..., :-1, 1]
+ right = mat[..., :-1, 0]
+ if isinstance(mat, torch.Tensor):
+ vec = torch.cat((forward, up, right), dim=-1)
+ elif isinstance(mat, np.ndarray):
+ vec = np.concatenate((forward, up, right), axis=-1)
+ else:
+ raise ValueError
+ return vec
+
+
+def mat2vec_batch(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 4, 4]
+
+ Returns:
+ vec (tensor): [B, 12], pos, forward, up and right
+ """
+ # Assume bs = 1
+ mat = normalized_matrix(mat)
+ pos = mat[..., :-1, 3]
+ forward = mat[..., :-1, 2]
+ up = mat[..., :-1, 1]
+ right = mat[..., :-1, 0]
+ if isinstance(mat, torch.Tensor):
+ vec = torch.cat((pos, forward, up, right), dim=-1)
+ elif isinstance(mat, np.ndarray):
+ vec = np.concatenate((pos, forward, up, right), axis=-1)
+ else:
+ raise ValueError
+ return vec
+
+
+def mat2pose_batch(mat, returnvel=True):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 4, 4]
+
+ Returns:
+ vec (tensor): [B, 12], pos, forward, up, zeros
+ """
+ # Assume bs = 1
+ mat = normalized_matrix(mat)
+ pos = mat[..., :-1, 3]
+ forward = mat[..., :-1, 2]
+ up = mat[..., :-1, 1]
+ if isinstance(mat, torch.Tensor):
+ if returnvel:
+ vel = torch.zeros_like(up)
+ vec = torch.cat((pos, forward, up, vel), dim=-1)
+ else:
+ vec = torch.cat((pos, forward, up), dim=-1)
+ elif isinstance(mat, np.ndarray):
+ if returnvel:
+ vel = np.zeros_like(up)
+ vec = np.concatenate((pos, forward, up, vel), axis=-1)
+ else:
+ vec = np.concatenate((pos, forward, up), axis=-1)
+ else:
+ raise ValueError
+ return vec
+
+
+def get_mat_BinA(matCtoA, matCtoB):
+ """
+ given matrix of the same object in two coordinate A and B,
+ return matrix B in the coordinate of A
+
+ Args:
+ matCtoA (tensor): [4, 4] world matrix
+ matCtoB (tensor): [4, 4] world matrix
+ """
+ if isinstance(matCtoA, torch.Tensor):
+ matCtoB_inv = torch.inverse(matCtoB)
+ elif isinstance(matCtoA, np.ndarray):
+ matCtoB_inv = np.linalg.inv(matCtoB)
+ else:
+ raise ValueError
+ matCtoB_inv = normalized_matrix(matCtoB_inv)
+ if isinstance(matCtoA, torch.Tensor):
+ mat_BtoA = torch.matmul(matCtoA, matCtoB_inv)
+ elif isinstance(matCtoA, np.ndarray):
+ mat_BtoA = np.matmul(matCtoA, matCtoB_inv)
+ mat_BtoA = normalized_matrix(mat_BtoA)
+ return mat_BtoA
+
+
+def get_mat_BtoA(matA, matB):
+ """
+ return matrix B in the coordinate of A
+
+ Args:
+ matA (tensor): [4, 4] world matrix
+ matB (tensor): [4, 4] world matrix
+ """
+ if isinstance(matA, torch.Tensor):
+ matA_inv = torch.inverse(matA)
+ elif isinstance(matA, np.ndarray):
+ matA_inv = np.linalg.inv(matA)
+ else:
+ raise ValueError
+ matA_inv = normalized_matrix(matA_inv)
+ if isinstance(matA, torch.Tensor):
+ mat_BtoA = torch.matmul(matA_inv, matB)
+ elif isinstance(matA, np.ndarray):
+ mat_BtoA = np.matmul(matA_inv, matB)
+ mat_BtoA = normalized_matrix(mat_BtoA)
+ return mat_BtoA
+
+
+def get_mat_BfromA(matA, matBtoA):
+ """
+ return world matrix B given matrix A and mat B realtive to A
+
+ Args:
+ matA (_type_): [4, 4] world matrix
+ matBtoA (_type_): [4, 4] matrix B relative to A
+ """
+ if isinstance(matA, torch.Tensor):
+ matB = torch.matmul(matA, matBtoA)
+ if isinstance(matA, np.ndarray):
+ matB = np.matmul(matA, matBtoA)
+ matB = normalized_matrix(matB)
+ return matB
+
+
+def get_relative_position_to(pos, mat):
+ """_summary_
+
+ Args:
+ pos (_type_): [N, M, 3] or [N, 3]
+ mat (_type_): [N, 4, 4] or [4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ if isinstance(mat, torch.Tensor):
+ mat_inv = torch.inverse(mat)
+ elif isinstance(mat, np.ndarray):
+ mat_inv = np.linalg.inv(mat)
+ else:
+ raise ValueError
+ mat_inv = normalized_matrix(mat_inv)
+ if isinstance(mat, torch.Tensor):
+ rot_pos = torch.matmul(mat_inv[..., :-1, :-1], pos.transpose(-1, -2)).transpose(
+ -1, -2
+ )
+ elif isinstance(mat, np.ndarray):
+ rot_pos = np.matmul(mat_inv[..., :-1, :-1], pos.swapaxes(-1, -2)).swapaxes(
+ -1, -2
+ )
+ world_pos = rot_pos + mat_inv[..., None, :-1, 3]
+ return world_pos
+
+
+def get_rotation(mat):
+ """_summary_
+
+ Args:
+ mat (_type_): [..., 4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ return mat[..., :-1, :-1]
+
+
+def set_rotation(mat, rotmat):
+ """_summary_
+
+ Args:
+ mat (_type_): [..., 4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ mat[..., :-1, :-1] = rotmat
+ return mat
+
+
+def set_position(mat, pos):
+ """_summary_
+
+ Args:
+ mat (_type_): [..., 4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ mat[..., :-1, 3] = pos
+ return mat
+
+
+def get_position(mat):
+ """_summary_
+
+ Args:
+ mat (_type_): [..., 4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ return mat[..., :-1, 3]
+
+
+def get_position_from(pos, mat):
+ """_summary_
+
+ Args:
+ pos (_type_): [N, M, 3] or [N, 3]
+ mat (_type_): [N, 4, 4] or [4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ if isinstance(mat, torch.Tensor):
+ rot_pos = torch.matmul(mat[..., :-1, :-1], pos.transpose(-1, -2)).transpose(
+ -1, -2
+ )
+ elif isinstance(mat, np.ndarray):
+ rot_pos = np.matmul(mat[..., :-1, :-1], pos.swapaxes(-1, -2)).swapaxes(-1, -2)
+ else:
+ raise ValueError
+
+ world_pos = rot_pos + mat[..., None, :-1, 3]
+ return world_pos
+
+
+def get_position_from_rotmat(pos, mat):
+ """_summary_
+
+ Args:
+ pos (_type_): [N, M, 3] or [N, 3]
+ mat (_type_): [N, 4, 4] or [4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ if isinstance(mat, torch.Tensor):
+ rot_pos = torch.matmul(mat, pos.transpose(-1, -2)).transpose(-1, -2)
+ elif isinstance(mat, np.ndarray):
+ rot_pos = np.matmul(mat, pos.swapaxes(-1, -2)).swapaxes(-1, -2)
+ else:
+ raise ValueError
+ return rot_pos
+
+
+def get_relative_direction_to(dir, mat):
+ """_summary_
+
+ Args:
+ dir (_type_): [N, M, 3] or [N, 3]
+ mat (_type_): [N, 4, 4] or [4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ if isinstance(mat, torch.Tensor):
+ mat_inv = torch.inverse(mat)
+ elif isinstance(mat, np.ndarray):
+ mat_inv = np.linalg.inv(mat)
+ else:
+ raise ValueError
+ mat_inv = normalized_matrix(mat_inv)
+ rot_mat_inv = mat_inv[..., :3, :3]
+ if isinstance(mat, torch.Tensor):
+ rel_dir = torch.matmul(rot_mat_inv, dir.transpose(-1, -2))
+ return rel_dir.transpose(-1, -2)
+ elif isinstance(mat, np.ndarray):
+ rel_dir = np.matmul(rot_mat_inv, dir.swapaxes(-1, -2))
+ return rel_dir.swapaxes(-1, -2)
+ else:
+ raise ValueError
+ return
+
+
+def get_direction_from(dir, mat):
+ """_summary_
+
+ Args:
+ dir (_type_): [N, M, 3] or [N, 3]
+ mat (_type_): [N, 4, 4] or [4, 4]
+
+ Returns:
+ tensor: [N, M, 3] or [N, 3]
+ """
+ rot_mat = mat[..., :3, :3]
+ if isinstance(mat, torch.Tensor):
+ world_dir = torch.matmul(rot_mat, dir.transpose(-1, -2))
+ return world_dir.transpose(-1, -2)
+ elif isinstance(mat, np.ndarray):
+ world_dir = np.matmul(rot_mat, dir.swapaxes(-1, -2))
+ return world_dir.swapaxes(-1, -2)
+ else:
+ raise ValueError
+ return
+
+
+def get_coord_vis(pos, rot_mat, scale=1.0):
+ forward = rot_mat[..., :, 2]
+ up = rot_mat[..., :, 1]
+ right = rot_mat[..., :, 0]
+ return pos + right * scale, pos + up * scale, pos + forward * scale
+
+
+def project_vec(vec):
+ """_summary_
+
+ Args:
+ vec (tensor): [*, 12], pos, forward, up and right
+
+ Returns:
+ proj_vec (tensor): [*, 4], posx, posz, forwardx, forwardz
+ """
+ posx = vec[..., 0:1]
+ posz = vec[..., 2:3]
+ forwardx = vec[..., 3:4]
+ forwardz = vec[..., 5:6]
+ if isinstance(vec, torch.Tensor):
+ proj_vec = torch.cat((posx, posz, forwardx, forwardz), dim=-1)
+ elif isinstance(vec, np.ndarray):
+ proj_vec = np.concatenate((posx, posz, forwardx, forwardz), axis=-1)
+ else:
+ raise ValueError
+
+ return proj_vec
+
+
+def xz2xyz(vec):
+ x = vec[..., 0:1]
+ z = vec[..., 1:2]
+ if isinstance(vec, torch.Tensor):
+ y = torch.zeros(vec.shape[:-1] + (1,), device=vec.device)
+ xyz_vec = torch.cat((x, y, z), dim=-1)
+ elif isinstance(vec, np.ndarray):
+ y = np.zeros(vec.shape[:-1] + (1,))
+ xyz_vec = np.concatenate((x, y, z), axis=-1)
+ else:
+ raise ValueError
+
+ return xyz_vec
+
+
+def normalized(vec):
+ if isinstance(vec, torch.Tensor):
+ norm_vec = vec / (vec.norm(2, dim=-1, keepdim=True) + 1e-9)
+ elif isinstance(vec, np.ndarray):
+ norm_vec = vec / (np.linalg.norm(vec, ord=2, axis=-1, keepdims=True) + 1e-9)
+ else:
+ raise ValueError
+
+ return norm_vec
+
+
+def normalized_matrix(mat):
+ if mat.shape[-1] == 4:
+ rot_mat = mat[..., :-1, :-1]
+ else:
+ rot_mat = mat
+ if isinstance(mat, torch.Tensor):
+ rot_mat_norm = rot_mat / (rot_mat.norm(2, dim=-2, keepdim=True) + 1e-9)
+ norm_mat = torch.zeros_like(mat)
+ elif isinstance(mat, np.ndarray):
+ rot_mat_norm = rot_mat / (
+ np.linalg.norm(rot_mat, ord=2, axis=-2, keepdims=True) + 1e-9
+ )
+ norm_mat = np.zeros_like(mat)
+ else:
+ raise ValueError
+ if mat.shape[-1] == 4:
+ norm_mat[..., :-1, :-1] = rot_mat_norm
+ norm_mat[..., :-1, -1] = mat[..., :-1, -1]
+ norm_mat[..., -1, -1] = 1.0
+ else:
+ norm_mat = rot_mat_norm
+ return norm_mat
+
+
+def get_rot_mat_from_forward(forward):
+ """_summary_
+
+ Args:
+ forward (tensor): [N, M, 3]
+
+ Returns:
+ mat (tensor): [N, M, 3, 3]
+ """
+ if isinstance(forward, torch.Tensor):
+ mat = torch.eye(3, device=forward.device).repeat(forward.shape[:-1] + (1, 1))
+ right = torch.zeros_like(forward)
+ elif isinstance(forward, np.ndarray):
+ mat = np.eye(3, dtype=np.float32)
+ for _ in range(len(forward.shape) - 1):
+ mat = mat[None]
+ mat = np.tile(mat, forward.shape[:-1] + (1, 1))
+ right = np.zeros_like(forward)
+ else:
+ raise ValueError
+
+ right[..., 0] = forward[..., 2]
+ right[..., 1] = 0.0
+ right[..., 2] = -forward[..., 0]
+ # right = torch.cross(mat[..., 1], forward) # cannot backward
+
+ mat[..., 2] = normalized(forward)
+ right = normalized(right)
+ mat[..., 0] = right
+ return mat
+
+
+def get_rot_mat_from_forward_up(forward, up):
+ """_summary_
+
+ Args:
+ forward (tensor): [N, M, 3]
+ up (tensor): [N, M, 3]
+
+ Returns:
+ mat (tensor): [N, M, 3, 3]
+ """
+ if isinstance(forward, torch.Tensor):
+ mat = torch.eye(3, device=forward.device).repeat(forward.shape[:-1] + (1, 1))
+ right = torch.cross(up, forward)
+ elif isinstance(forward, np.ndarray):
+ mat = np.eye(3, dtype=np.float32)
+ for _ in range(len(forward.shape) - 1):
+ mat = mat[None]
+ mat = np.tile(mat, forward.shape[:-1] + (1, 1))
+ right = np.cross(up, forward)
+ else:
+ raise ValueError
+
+ right = normalized(right)
+ mat[..., 2] = normalized(forward)
+ mat[..., 1] = normalized(up)
+ mat[..., 0] = right
+ return mat
+
+
+def get_rot_mat_from_pose_vec(vec):
+ """_summary_
+
+ Args:
+ vec (tensor): [N, M, 6]
+
+ Returns:
+ mat (tensor): [N, M, 3, 3]
+ """
+ forward = vec[..., :3]
+ up = vec[..., 3:6]
+ return get_rot_mat_from_forward_up(forward, up)
+
+
+def get_TRS(rot_mat, pos):
+ """_summary_
+
+ Args:
+ rot_mat (tensor): [N, 3, 3]
+ pos (tensor): [N, 3]
+
+ Returns:
+ mat (tensor): [N, 4, 4]
+ """
+ if isinstance(rot_mat, torch.Tensor):
+ mat = torch.eye(4, device=pos.device).repeat(pos.shape[:-1] + (1, 1))
+ elif isinstance(rot_mat, np.ndarray):
+ mat = np.eye(4, dtype=np.float32)
+ for _ in range(len(pos.shape) - 1):
+ mat = mat[None]
+ mat = np.tile(mat, pos.shape[:-1] + (1, 1))
+ else:
+ raise ValueError
+ mat[..., :3, :3] = rot_mat
+ mat[..., :3, 3] = pos
+ mat = normalized_matrix(mat)
+ return mat
+
+
+def xzvec2mat(vec):
+ """_summary_
+
+ Args:
+ vec (tensor): [N, 4]
+
+ Returns:
+ mat (tensor): [N, 4, 4]
+ """
+ vec_shape = vec.shape[:-1]
+ if isinstance(vec, torch.Tensor):
+ pos = torch.zeros(vec_shape + (3,))
+ forward = torch.zeros(vec_shape + (3,))
+ elif isinstance(vec, np.ndarray):
+ pos = np.zeros(vec_shape + (3,))
+ forward = np.zeros(vec_shape + (3,))
+ else:
+ raise ValueError
+
+ pos[..., 0] = vec[..., 0]
+ pos[..., 2] = vec[..., 1]
+ forward[..., 0] = vec[..., 2]
+ forward[..., 2] = vec[..., 3]
+ rot_mat = get_rot_mat_from_forward(forward)
+ mat = get_TRS(rot_mat, pos)
+ return mat
+
+
+def distance(vec1, vec2):
+ return ((vec1 - vec2) ** 2).sum() ** 0.5
+
+
+def get_relative_pose_from_vec(pose, root, N):
+ root_p_mat = xzvec2mat(root)
+ pose = pose.reshape(-1, N, 12)
+ pose[..., :3] = get_position_from(pose[..., :3], root_p_mat)
+ pose[..., 3:6] = get_direction_from(pose[..., 3:6], root_p_mat)
+ pose[..., 6:9] = get_direction_from(pose[..., 6:9], root_p_mat)
+ pose[..., 9:] = get_direction_from(pose[..., 9:], root_p_mat)
+ pos = pose[..., 0, :3]
+ rot = pose[..., 3:9].reshape(-1, N * 6)
+ pose = np.concatenate((pos, rot), axis=-1)
+ return pose
+
+
+def get_forward_from_pos(pos):
+ """_summary_
+
+ Args:
+ pos (N, J, 3): joints positions of each frame
+
+ Returns:
+ _type_: _description_
+ """
+
+ pos_y_vec = torch.tensor([0, 1, 0], dtype=torch.float32).to(pos.device)
+ face_joint_indx = [2, 1, 17, 16]
+ r_hip, l_hip, r_sdr, l_sdr = (
+ face_joint_indx # use hip and shoulder to get the cross vector
+ )
+ cross_hip = pos[..., 0, r_hip, :] - pos[..., 0, l_hip, :]
+ cross_sdr = pos[..., 0, r_sdr, :] - pos[..., 0, l_sdr, :]
+ cross_vec = cross_hip + cross_sdr # (3, )
+ forward_vec = torch.cross(pos_y_vec, cross_vec, dim=-1)
+ forward_vec = normalized(forward_vec)
+ return forward_vec
+
+
+def project_point_along_ray(p, ray, keepnorm=False):
+ """_summary_
+
+ Args:
+ p (*, 3): point positions
+ ray (*, 3): ray direction
+ keepnorm: False -> project point on the ray,
+ True -> project point on the ray and keep the point length
+
+ Returns:
+ _type_: _description_
+ """
+ ray = normalized(ray)
+ if keepnorm:
+ new_p = ray * p.norm(dim=-1, keepdim=True)
+ else:
+ dot_product = torch.sum(p * ray, dim=-1, keepdim=True)
+ new_p = dot_product * ray
+ return new_p
+
+
+def solve_point_along_ray_with_constraint(c, ray, p, constraint="x"):
+ """_summary_
+
+ Args:
+ c (*,): constraint value
+ ray (*, 3): ray direction
+ p (*, 3): start point of the ray
+
+ Returns:
+ _type_: _description_
+ """
+ ray = normalized(ray)
+ if constraint == "x":
+ ind = 0
+ elif constraint == "y":
+ ind = 1
+ elif constraint == "z":
+ ind = 2
+ else:
+ raise ValueError
+ t = (c - p[..., ind]) / ray[..., ind]
+ out_p = ray * t[..., None] + p
+
+ return out_p
+
+
+def calc_cosine(vec1, vec2, return_angle=False):
+ """_summary_
+
+ Args:
+ vec1 (*, 3): vector
+ vec2 (*, 3): vector
+ return_angle: True -> return angle, False -> return cosine
+
+ Returns:
+ _type_: _description_
+ """
+ vec1 = normalized(vec1)
+ vec2 = normalized(vec2)
+ cosine = torch.sum(vec1 * vec2, dim=-1)
+ if return_angle:
+ return torch.acos(cosine)
+ return cosine
+
+
+############################################
+#
+# quaternion assumes xyzw
+#
+############################################
+
+
+def quat_xyzw2wxyz(quat):
+ new_quat = torch.cat([quat[..., 3:4], quat[..., :3]], dim=-1)
+ return new_quat
+
+
+def quat_wxyz2xyzw(quat):
+ new_quat = torch.cat([quat[..., 1:4], quat[..., :1]], dim=-1)
+ return new_quat
+
+
+def quat_mul(a, b):
+ """
+ quaternion multiplication
+ """
+ x1, y1, z1, w1 = a[..., 0], a[..., 1], a[..., 2], a[..., 3]
+ x2, y2, z2, w2 = b[..., 0], b[..., 1], b[..., 2], b[..., 3]
+
+ w = w1 * w2 - x1 * x2 - y1 * y2 - z1 * z2
+ x = w1 * x2 + x1 * w2 + y1 * z2 - z1 * y2
+ y = w1 * y2 + y1 * w2 + z1 * x2 - x1 * z2
+ z = w1 * z2 + z1 * w2 + x1 * y2 - y1 * x2
+
+ return torch.stack([x, y, z, w], dim=-1)
+
+
+def quat_pos(x):
+ """
+ make all the real part of the quaternion positive
+ """
+ q = x
+ z = (q[..., 3:] < 0).float()
+ q = (1 - 2 * z) * q
+ return q
+
+
+def quat_abs(x):
+ """
+ quaternion norm (unit quaternion represents a 3D rotation, which has norm of 1)
+ """
+ x = x.norm(p=2, dim=-1)
+ return x
+
+
+def quat_unit(x):
+ """
+ normalized quaternion with norm of 1
+ """
+ norm = quat_abs(x).unsqueeze(-1)
+ return x / (norm.clamp(min=1e-4))
+
+
+def quat_conjugate(x):
+ """
+ quaternion with its imaginary part negated
+ """
+ return torch.cat([-x[..., :3], x[..., 3:]], dim=-1)
+
+
+def quat_real(x):
+ """
+ real component of the quaternion
+ """
+ return x[..., 3]
+
+
+def quat_imaginary(x):
+ """
+ imaginary components of the quaternion
+ """
+ return x[..., :3]
+
+
+def quat_norm_check(x):
+ """
+ verify that a quaternion has norm 1
+ """
+ assert bool((abs(x.norm(p=2, dim=-1) - 1) < 1e-3).all()), (
+ "the quaternion is has non-1 norm: {}".format(abs(x.norm(p=2, dim=-1) - 1))
+ )
+ assert bool((x[..., 3] >= 0).all()), "the quaternion has negative real part"
+
+
+def quat_normalize(q):
+ """
+ Construct 3D rotation from quaternion (the quaternion needs not to be normalized).
+ """
+ q = quat_unit(quat_pos(q)) # normalized to positive and unit quaternion
+ return q
+
+
+def quat_from_xyz(xyz):
+ """
+ Construct 3D rotation from the imaginary component
+ """
+ w = (1.0 - xyz.norm()).unsqueeze(-1)
+ assert bool((w >= 0).all()), "xyz has its norm greater than 1"
+ return torch.cat([xyz, w], dim=-1)
+
+
+def quat_identity(shape: List[int]):
+ """
+ Construct 3D identity rotation given shape
+ """
+ w = torch.ones(shape + (1,))
+ xyz = torch.zeros(shape + (3,))
+ q = torch.cat([xyz, w], dim=-1)
+ return quat_normalize(q)
+
+
+def tgm_quat_from_angle_axis(angle, axis, degree: bool = False):
+ """Create a 3D rotation from angle and axis of rotation. The rotation is counter-clockwise
+ along the axis.
+
+ The rotation can be interpreted as a_R_b where frame "b" is the new frame that
+ gets rotated counter-clockwise along the axis from frame "a"
+
+ :param angle: angle of rotation
+ :type angle: Tensor
+ :param axis: axis of rotation
+ :type axis: Tensor
+ :param degree: put True here if the angle is given by degree
+ :type degree: bool, optional, default=False
+ """
+ if degree:
+ angle = angle / 180.0 * math.pi
+ theta = (angle / 2).unsqueeze(-1)
+ axis = axis / (axis.norm(p=2, dim=-1, keepdim=True).clamp(min=1e-4))
+ xyz = axis * theta.sin()
+ w = theta.cos()
+ return quat_normalize(torch.cat([w, xyz], dim=-1))
+
+
+def quat_from_rotation_matrix(m):
+ """
+ Construct a 3D rotation from a valid 3x3 rotation matrices.
+ Reference can be found here:
+ http://www.cg.info.hiroshima-cu.ac.jp/~miyazaki/knowledge/teche52.html
+
+ :param m: 3x3 orthogonal rotation matrices.
+ :type m: Tensor
+
+ :rtype: Tensor
+ """
+ m = m.unsqueeze(0)
+ diag0 = m[..., 0, 0]
+ diag1 = m[..., 1, 1]
+ diag2 = m[..., 2, 2]
+
+ # Math stuff.
+ w = (((diag0 + diag1 + diag2 + 1.0) / 4.0).clamp(0.0, None)) ** 0.5
+ x = (((diag0 - diag1 - diag2 + 1.0) / 4.0).clamp(0.0, None)) ** 0.5
+ y = (((-diag0 + diag1 - diag2 + 1.0) / 4.0).clamp(0.0, None)) ** 0.5
+ z = (((-diag0 - diag1 + diag2 + 1.0) / 4.0).clamp(0.0, None)) ** 0.5
+
+ # Only modify quaternions where w > x, y, z.
+ c0 = (w >= x) & (w >= y) & (w >= z)
+ x[c0] *= (m[..., 2, 1][c0] - m[..., 1, 2][c0]).sign()
+ y[c0] *= (m[..., 0, 2][c0] - m[..., 2, 0][c0]).sign()
+ z[c0] *= (m[..., 1, 0][c0] - m[..., 0, 1][c0]).sign()
+
+ # Only modify quaternions where x > w, y, z
+ c1 = (x >= w) & (x >= y) & (x >= z)
+ w[c1] *= (m[..., 2, 1][c1] - m[..., 1, 2][c1]).sign()
+ y[c1] *= (m[..., 1, 0][c1] + m[..., 0, 1][c1]).sign()
+ z[c1] *= (m[..., 0, 2][c1] + m[..., 2, 0][c1]).sign()
+
+ # Only modify quaternions where y > w, x, z.
+ c2 = (y >= w) & (y >= x) & (y >= z)
+ w[c2] *= (m[..., 0, 2][c2] - m[..., 2, 0][c2]).sign()
+ x[c2] *= (m[..., 1, 0][c2] + m[..., 0, 1][c2]).sign()
+ z[c2] *= (m[..., 2, 1][c2] + m[..., 1, 2][c2]).sign()
+
+ # Only modify quaternions where z > w, x, y.
+ c3 = (z >= w) & (z >= x) & (z >= y)
+ w[c3] *= (m[..., 1, 0][c3] - m[..., 0, 1][c3]).sign()
+ x[c3] *= (m[..., 2, 0][c3] + m[..., 0, 2][c3]).sign()
+ y[c3] *= (m[..., 2, 1][c3] + m[..., 1, 2][c3]).sign()
+
+ return quat_normalize(torch.stack([x, y, z, w], dim=-1)).squeeze(0)
+
+
+def quat_mul_norm(x, y):
+ """
+ Combine two set of 3D rotations together using \**\* operator. The shape needs to be
+ broadcastable
+ """
+ return quat_normalize(quat_mul(x, y))
+
+
+def quat_rotate(rot, vec):
+ """
+ Rotate a 3D vector with the 3D rotation
+ """
+ other_q = torch.cat([vec, torch.zeros_like(vec[..., :1])], dim=-1)
+ return quat_imaginary(quat_mul(quat_mul(rot, other_q), quat_conjugate(rot)))
+
+
+def quat_inverse(x):
+ """
+ The inverse of the rotation
+ """
+ return quat_conjugate(x)
+
+
+def quat_identity_like(x):
+ """
+ Construct identity 3D rotation with the same shape
+ """
+ return quat_identity(x.shape[:-1])
+
+
+def quat_angle_axis(x):
+ """
+ The (angle, axis) representation of the rotation. The axis is normalized to unit length.
+ The angle is guaranteed to be between [0, pi].
+ """
+ s = 2 * (x[..., 3] ** 2) - 1
+ angle = s.clamp(-1, 1).arccos() # just to be safe
+ axis = x[..., :3]
+ axis /= axis.norm(p=2, dim=-1, keepdim=True).clamp(min=1e-4)
+ return angle, axis
+
+
+def quat_yaw_rotation(x, z_up: bool = True):
+ """
+ Yaw rotation (rotation along z-axis)
+ """
+ q = x
+ if z_up:
+ q = torch.cat([torch.zeros_like(q[..., 0:2]), q[..., 2:3], q[..., 3:]], dim=-1)
+ else:
+ q = torch.cat(
+ [
+ torch.zeros_like(q[..., 0:1]),
+ q[..., 1:2],
+ torch.zeros_like(q[..., 2:3]),
+ q[..., 3:4],
+ ],
+ dim=-1,
+ )
+ return quat_normalize(q)
+
+
+def transform_from_rotation_translation(
+ r: Optional[torch.Tensor] = None, t: Optional[torch.Tensor] = None
+):
+ """
+ Construct a transform from a quaternion and 3D translation. Only one of them can be None.
+ """
+ assert r is not None or t is not None, "rotation and translation can't be all None"
+ if r is None:
+ assert t is not None
+ r = quat_identity(list(t.shape))
+ if t is None:
+ t = torch.zeros(list(r.shape) + [3])
+ return torch.cat([r, t], dim=-1)
+
+
+def transform_identity(shape: List[int]):
+ """
+ Identity transformation with given shape
+ """
+ r = quat_identity(shape)
+ t = torch.zeros(shape + [3])
+ return transform_from_rotation_translation(r, t)
+
+
+def transform_rotation(x):
+ """Get rotation from transform"""
+ return x[..., :4]
+
+
+def transform_translation(x):
+ """Get translation from transform"""
+ return x[..., 4:]
+
+
+def transform_inverse(x):
+ """
+ Inverse transformation
+ """
+ inv_so3 = quat_inverse(transform_rotation(x))
+ return transform_from_rotation_translation(
+ r=inv_so3, t=quat_rotate(inv_so3, -transform_translation(x))
+ )
+
+
+def transform_identity_like(x):
+ """
+ identity transformation with the same shape
+ """
+ return transform_identity(x.shape)
+
+
+def transform_mul(x, y):
+ """
+ Combine two transformation together
+ """
+ z = transform_from_rotation_translation(
+ r=quat_mul_norm(transform_rotation(x), transform_rotation(y)),
+ t=quat_rotate(transform_rotation(x), transform_translation(y))
+ + transform_translation(x),
+ )
+ return z
+
+
+def transform_apply(rot, vec):
+ """
+ Transform a 3D vector
+ """
+ assert isinstance(vec, torch.Tensor)
+ return quat_rotate(transform_rotation(rot), vec) + transform_translation(rot)
+
+
+def rot_matrix_det(x):
+ """
+ Return the determinant of the 3x3 matrix. The shape of the tensor will be as same as the
+ shape of the matrix
+ """
+ a, b, c = x[..., 0, 0], x[..., 0, 1], x[..., 0, 2]
+ d, e, f = x[..., 1, 0], x[..., 1, 1], x[..., 1, 2]
+ g, h, i = x[..., 2, 0], x[..., 2, 1], x[..., 2, 2]
+ t1 = a * (e * i - f * h)
+ t2 = b * (d * i - f * g)
+ t3 = c * (d * h - e * g)
+ return t1 - t2 + t3
+
+
+def rot_matrix_integrity_check(x):
+ """
+ Verify that a rotation matrix has a determinant of one and is orthogonal
+ """
+ det = rot_matrix_det(x)
+ assert bool((abs(det - 1) < 1e-3).all()), "the matrix has non-one determinant"
+ rtr = x @ x.permute(torch.arange(x.dim() - 2), -1, -2)
+ rtr_gt = rtr.zeros_like()
+ rtr_gt[..., 0, 0] = 1
+ rtr_gt[..., 1, 1] = 1
+ rtr_gt[..., 2, 2] = 1
+ assert bool(((rtr - rtr_gt) < 1e-3).all()), "the matrix is not orthogonal"
+
+
+def rot_matrix_from_quaternion(q):
+ """
+ Construct rotation matrix from quaternion
+ """
+ # Shortcuts for individual elements (using wikipedia's convention)
+ qi, qj, qk, qr = q[..., 0], q[..., 1], q[..., 2], q[..., 3]
+
+ # Set individual elements
+ R00 = 1.0 - 2.0 * (qj**2 + qk**2)
+ R01 = 2 * (qi * qj - qk * qr)
+ R02 = 2 * (qi * qk + qj * qr)
+ R10 = 2 * (qi * qj + qk * qr)
+ R11 = 1.0 - 2.0 * (qi**2 + qk**2)
+ R12 = 2 * (qj * qk - qi * qr)
+ R20 = 2 * (qi * qk - qj * qr)
+ R21 = 2 * (qj * qk + qi * qr)
+ R22 = 1.0 - 2.0 * (qi**2 + qj**2)
+
+ R0 = torch.stack([R00, R01, R02], dim=-1)
+ R1 = torch.stack([R10, R11, R12], dim=-1)
+ R2 = torch.stack([R20, R21, R22], dim=-1)
+
+ R = torch.stack([R0, R1, R2], dim=-2)
+
+ return R
+
+
+def euclidean_to_rotation_matrix(x):
+ """
+ Get the rotation matrix on the top-left corner of a Euclidean transformation matrix
+ """
+ return x[..., :3, :3]
+
+
+def euclidean_integrity_check(x):
+ euclidean_to_rotation_matrix(x) # check 3d-rotation matrix
+ assert bool((x[..., 3, :3] == 0).all()), "the last row is illegal"
+ assert bool((x[..., 3, 3] == 1).all()), "the last row is illegal"
+
+
+def euclidean_translation(x):
+ """
+ Get the translation vector located at the last column of the matrix
+ """
+ return x[..., :3, 3]
+
+
+def euclidean_inverse(x):
+ """
+ Compute the matrix that represents the inverse rotation
+ """
+ s = x.zeros_like()
+ irot = quat_inverse(quat_from_rotation_matrix(x))
+ s[..., :3, :3] = irot
+ s[..., :3, 4] = quat_rotate(irot, -euclidean_translation(x))
+ return s
+
+
+def euclidean_to_transform(transformation_matrix):
+ """
+ Construct a transform from a Euclidean transformation matrix
+ """
+ return transform_from_rotation_translation(
+ r=quat_from_rotation_matrix(
+ m=euclidean_to_rotation_matrix(transformation_matrix)
+ ),
+ t=euclidean_translation(transformation_matrix),
+ )
+
+
+def to_torch(x, dtype=torch.float, device="cuda:0", requires_grad=False):
+ return torch.tensor(x, dtype=dtype, device=device, requires_grad=requires_grad)
+
+
+def quat_mul(a, b):
+ assert a.shape == b.shape
+ shape = a.shape
+ a = a.reshape(-1, 4)
+ b = b.reshape(-1, 4)
+
+ x1, y1, z1, w1 = a[:, 0], a[:, 1], a[:, 2], a[:, 3]
+ x2, y2, z2, w2 = b[:, 0], b[:, 1], b[:, 2], b[:, 3]
+ ww = (z1 + x1) * (x2 + y2)
+ yy = (w1 - y1) * (w2 + z2)
+ zz = (w1 + y1) * (w2 - z2)
+ xx = ww + yy + zz
+ qq = 0.5 * (xx + (z1 - x1) * (x2 - y2))
+ w = qq - ww + (z1 - y1) * (y2 - z2)
+ x = qq - xx + (x1 + w1) * (x2 + w2)
+ y = qq - yy + (w1 - x1) * (y2 + z2)
+ z = qq - zz + (z1 + y1) * (w2 - x2)
+
+ quat = torch.stack([x, y, z, w], dim=-1).view(shape)
+
+ return quat
+
+
+def normalize(x, eps: float = 1e-9):
+ return x / x.norm(p=2, dim=-1).clamp(min=eps, max=None).unsqueeze(-1)
+
+
+def quat_apply(a, b):
+ shape = b.shape
+ a = a.reshape(-1, 4)
+ b = b.reshape(-1, 3)
+ xyz = a[:, :3]
+ t = xyz.cross(b, dim=-1) * 2
+ return (b + a[:, 3:] * t + xyz.cross(t, dim=-1)).view(shape)
+
+
+def quat_rotate(q, v):
+ shape = q.shape
+ q_w = q[:, -1]
+ q_vec = q[:, :3]
+ a = v * (2.0 * q_w**2 - 1.0).unsqueeze(-1)
+ b = torch.cross(q_vec, v, dim=-1) * q_w.unsqueeze(-1) * 2.0
+ c = (
+ q_vec
+ * torch.bmm(q_vec.view(shape[0], 1, 3), v.view(shape[0], 3, 1)).squeeze(-1)
+ * 2.0
+ )
+ return a + b + c
+
+
+def quat_rotate_inverse(q, v):
+ shape = q.shape
+ q_w = q[:, -1]
+ q_vec = q[:, :3]
+ a = v * (2.0 * q_w**2 - 1.0).unsqueeze(-1)
+ b = torch.cross(q_vec, v, dim=-1) * q_w.unsqueeze(-1) * 2.0
+ c = (
+ q_vec
+ * torch.bmm(q_vec.view(shape[0], 1, 3), v.view(shape[0], 3, 1)).squeeze(-1)
+ * 2.0
+ )
+ return a - b + c
+
+
+def quat_conjugate(a):
+ shape = a.shape
+ a = a.reshape(-1, 4)
+ return torch.cat((-a[:, :3], a[:, -1:]), dim=-1).view(shape)
+
+
+def quat_unit(a):
+ return normalize(a)
+
+
+def quat_from_angle_axis(angle, axis):
+ theta = (angle / 2).unsqueeze(-1)
+ xyz = normalize(axis) * torch.sin(theta.clone())
+ w = torch.cos(theta.clone())
+ return quat_unit(torch.cat([xyz, w], dim=-1))
+
+
+def normalize_angle(x):
+ return torch.atan2(torch.sin(x.clone()), torch.cos(x.clone()))
+
+
+def tf_inverse(q, t):
+ q_inv = quat_conjugate(q)
+ return q_inv, -quat_apply(q_inv, t)
+
+
+def tf_apply(q, t, v):
+ return quat_apply(q, v) + t
+
+
+def tf_vector(q, v):
+ return quat_apply(q, v)
+
+
+def tf_combine(q1, t1, q2, t2):
+ return quat_mul(q1, q2), quat_apply(q1, t2) + t1
+
+
+def get_basis_vector(q, v):
+ return quat_rotate(q, v)
+
+
+def get_axis_params(value, axis_idx, x_value=0.0, dtype=float, n_dims=3):
+ """construct arguments to `Vec` according to axis index."""
+ zs = np.zeros((n_dims,))
+ assert axis_idx < n_dims, "the axis dim should be within the vector dimensions"
+ zs[axis_idx] = 1.0
+ params = np.where(zs == 1.0, value, zs)
+ params[0] = x_value
+ return list(params.astype(dtype))
+
+
+def copysign(a, b):
+ # type: (float, Tensor) -> Tensor
+ a = torch.tensor(a, device=b.device, dtype=torch.float).repeat(b.shape[0])
+ return torch.abs(a) * torch.sign(b)
+
+
+def get_euler_xyz(q):
+ qx, qy, qz, qw = 0, 1, 2, 3
+ # roll (x-axis rotation)
+ sinr_cosp = 2.0 * (q[:, qw] * q[:, qx] + q[:, qy] * q[:, qz])
+ cosr_cosp = (
+ q[:, qw] * q[:, qw]
+ - q[:, qx] * q[:, qx]
+ - q[:, qy] * q[:, qy]
+ + q[:, qz] * q[:, qz]
+ )
+ roll = torch.atan2(sinr_cosp, cosr_cosp)
+
+ # pitch (y-axis rotation)
+ sinp = 2.0 * (q[:, qw] * q[:, qy] - q[:, qz] * q[:, qx])
+ pitch = torch.where(
+ torch.abs(sinp) >= 1, copysign(np.pi / 2.0, sinp), torch.asin(sinp)
+ )
+
+ # yaw (z-axis rotation)
+ siny_cosp = 2.0 * (q[:, qw] * q[:, qz] + q[:, qx] * q[:, qy])
+ cosy_cosp = (
+ q[:, qw] * q[:, qw]
+ + q[:, qx] * q[:, qx]
+ - q[:, qy] * q[:, qy]
+ - q[:, qz] * q[:, qz]
+ )
+ yaw = torch.atan2(siny_cosp, cosy_cosp)
+
+ return roll % (2 * np.pi), pitch % (2 * np.pi), yaw % (2 * np.pi)
+
+
+def quat_from_euler_xyz(roll, pitch, yaw):
+ cy = torch.cos(yaw * 0.5)
+ sy = torch.sin(yaw * 0.5)
+ cr = torch.cos(roll * 0.5)
+ sr = torch.sin(roll * 0.5)
+ cp = torch.cos(pitch * 0.5)
+ sp = torch.sin(pitch * 0.5)
+
+ qw = cy * cr * cp + sy * sr * sp
+ qx = cy * sr * cp - sy * cr * sp
+ qy = cy * cr * sp + sy * sr * cp
+ qz = sy * cr * cp - cy * sr * sp
+
+ return torch.stack([qx, qy, qz, qw], dim=-1)
+
+
+def torch_rand_float(lower, upper, shape, device):
+ # type: (float, float, Tuple[int, int], str) -> Tensor
+ return (upper - lower) * torch.rand(*shape, device=device) + lower
+
+
+def torch_random_dir_2(shape, device):
+ # type: (Tuple[int, int], str) -> Tensor
+ angle = torch_rand_float(-np.pi, np.pi, shape, device).squeeze(-1)
+ return torch.stack([torch.cos(angle), torch.sin(angle)], dim=-1)
+
+
+def tensor_clamp(t, min_t, max_t):
+ return torch.max(torch.min(t, max_t), min_t)
+
+
+def scale(x, lower, upper):
+ return 0.5 * (x + 1.0) * (upper - lower) + lower
+
+
+def unscale(x, lower, upper):
+ return (2.0 * x - upper - lower) / (upper - lower)
+
+
+def unscale_np(x, lower, upper):
+ return (2.0 * x - upper - lower) / (upper - lower)
+
+
+def quat_to_angle_axis(q):
+ # type: (Tensor) -> Tuple[Tensor, Tensor]
+ # computes axis-angle representation from quaternion q
+ # q must be normalized
+ min_theta = 1e-5
+ qx, qy, qz, qw = 0, 1, 2, 3
+
+ sin_theta = torch.sqrt(1 - q[..., qw] * q[..., qw])
+ angle = 2 * torch.acos(q[..., qw])
+ angle = normalize_angle(angle)
+ sin_theta_expand = sin_theta.unsqueeze(-1)
+ axis = q[..., qx:qw] / sin_theta_expand
+
+ mask = torch.abs(sin_theta) > min_theta
+ default_axis = torch.zeros_like(axis)
+ default_axis[..., -1] = 1
+
+ angle = torch.where(mask, angle, torch.zeros_like(angle))
+ mask_expand = mask.unsqueeze(-1)
+ axis = torch.where(mask_expand, axis, default_axis)
+ return angle, axis
+
+
+def angle_axis_to_exp_map(angle, axis):
+ # type: (Tensor, Tensor) -> Tensor
+ # compute exponential map from axis-angle
+ angle_expand = angle.unsqueeze(-1)
+ exp_map = angle_expand * axis
+ return exp_map
+
+
+def quat_to_exp_map(q):
+ # type: (Tensor) -> Tensor
+ # compute exponential map from quaternion
+ # q must be normalized
+ angle, axis = quat_to_angle_axis(q)
+ exp_map = angle_axis_to_exp_map(angle, axis)
+ return exp_map
+
+
+def quat_to_tan_norm(q):
+ # type: (Tensor) -> Tensor
+ # represents a rotation using the tangent and normal vectors
+ ref_tan = torch.zeros_like(q[..., 0:3])
+ ref_tan[..., 0] = 1
+ tan = quat_rotate(q, ref_tan)
+
+ ref_norm = torch.zeros_like(q[..., 0:3])
+ ref_norm[..., -1] = 1
+ norm = quat_rotate(q, ref_norm)
+
+ norm_tan = torch.cat([tan, norm], dim=len(tan.shape) - 1)
+ return norm_tan
+
+
+def euler_xyz_to_exp_map(roll, pitch, yaw):
+ # type: (Tensor, Tensor, Tensor) -> Tensor
+ q = quat_from_euler_xyz(roll, pitch, yaw)
+ exp_map = quat_to_exp_map(q)
+ return exp_map
+
+
+def exp_map_to_angle_axis(exp_map):
+ min_theta = 1e-5
+
+ angle = torch.norm(exp_map.clone(), dim=-1) + 1e-6
+ angle_exp = torch.unsqueeze(angle, dim=-1)
+ axis = exp_map.clone() / angle_exp.clone()
+ angle = normalize_angle(angle)
+
+ default_axis = torch.zeros_like(exp_map)
+ default_axis[..., -1] = 1
+
+ mask = torch.abs(angle) > min_theta
+ angle = torch.where(mask, angle, torch.zeros_like(angle))
+ mask_expand = mask.unsqueeze(-1)
+ axis = torch.where(mask_expand, axis, default_axis)
+
+ return angle, axis
+
+
+def exp_map_to_quat(exp_map):
+ angle, axis = exp_map_to_angle_axis(exp_map)
+ q = quat_from_angle_axis(angle, axis)
+ return q
+
+
+def slerp(q0, q1, t):
+ # type: (Tensor, Tensor, Tensor) -> Tensor
+ cos_half_theta = torch.sum(q0 * q1, dim=-1)
+
+ neg_mask = cos_half_theta < 0
+ q1 = q1.clone()
+ q1[neg_mask] = -q1[neg_mask]
+ cos_half_theta = torch.abs(cos_half_theta)
+ cos_half_theta = torch.unsqueeze(cos_half_theta, dim=-1)
+
+ half_theta = torch.acos(cos_half_theta)
+ sin_half_theta = torch.sqrt(1.0 - cos_half_theta * cos_half_theta)
+
+ ratioA = torch.sin((1 - t) * half_theta) / sin_half_theta
+ ratioB = torch.sin(t * half_theta) / sin_half_theta
+
+ new_q = ratioA * q0 + ratioB * q1
+
+ new_q = torch.where(torch.abs(sin_half_theta) < 0.001, 0.5 * q0 + 0.5 * q1, new_q)
+ new_q = torch.where(torch.abs(cos_half_theta) >= 1, q0, new_q)
+
+ return new_q
+
+
+def calc_heading_vec(q, head_ind=0):
+ # type: (Tensor, int) -> Tensor
+ # calculate heading direction from quaternion
+ # the heading is the direction vector
+ # q must be normalized
+ ref_dir = torch.zeros_like(q[..., 0:3])
+ ref_dir[..., head_ind] = 1
+ rot_dir = quat_rotate(q, ref_dir)
+
+ return rot_dir
+
+
+def calc_heading(q, head_ind=0, gravity_axis="z"):
+ # type: (Tensor, int, str) -> Tensor
+ # calculate heading direction from quaternion
+ # the heading is the direction on the xy plane
+ # q must be normalized
+ ref_dir = torch.zeros_like(q[..., 0:3])
+ ref_dir[..., head_ind] = 1
+ # ref_dir[..., 0] = 1
+ shape = ref_dir.shape[:-1]
+ q = q.reshape((-1, 4))
+ ref_dir = ref_dir.reshape(-1, 3)
+ rot_dir = quat_rotate(q, ref_dir)
+ rot_dir = rot_dir.reshape(shape + (3,))
+ if gravity_axis == "z":
+ heading = torch.atan2(rot_dir[..., 1], rot_dir[..., 0])
+ elif gravity_axis == "y":
+ heading = torch.atan2(rot_dir[..., 0], rot_dir[..., 2])
+ elif gravity_axis == "x":
+ heading = torch.atan2(rot_dir[..., 2], rot_dir[..., 1])
+ return heading
+
+
+def calc_heading_quat(q, head_ind=0, gravity_axis="z"):
+ # type: (Tensor, int, str) -> Tensor
+ # calculate heading rotation from quaternion
+ # the heading is the direction on the xy plane
+ # q must be normalized
+ heading = calc_heading(q, head_ind, gravity_axis=gravity_axis)
+ axis = torch.zeros_like(q[..., 0:3])
+ if gravity_axis == "z":
+ g_axis = 2
+ elif gravity_axis == "y":
+ g_axis = 1
+ elif gravity_axis == "x":
+ g_axis = 0
+ axis[..., g_axis] = 1
+
+ heading_q = quat_from_angle_axis(heading, axis)
+ return heading_q
+
+
+def calc_heading_quat_inv(q, head_ind=0):
+ # type: (Tensor, int) -> Tensor
+ # calculate heading rotation from quaternion
+ # the heading is the direction on the xy plane
+ # q must be normalized
+ heading = calc_heading(q, head_ind)
+ axis = torch.zeros_like(q[..., 0:3])
+ axis[..., 2] = 1
+
+ heading_q = quat_from_angle_axis(-heading, axis)
+ return heading_q
+
+
+def forward_kinematics(mat, parent):
+ """_summary_
+
+ Args:
+ mat ([..., N, 3, 3]): _description_
+ parent (): _description_
+ """
+ if isinstance(mat, torch.Tensor):
+ rotations = torch.eye(mat.shape[-1], device=mat.device)
+ rotations = rotations.repeat(mat.shape[:-2] + (1, 1))
+ else:
+ rotations = np.eye(mat.shape[-1], dtype=np.float32)
+ rotations = np.tile(rotations, mat.shape[:-2] + (1, 1))
+ for i in range(mat.shape[-3]):
+ if parent[i] != -1:
+ if isinstance(mat, torch.Tensor):
+ # this way make gradient flow
+ new_mat = get_mat_BfromA(
+ rotations[..., parent[i], :, :], mat[..., i, :, :]
+ )
+ rotations = torch.cat(
+ (
+ rotations[..., :i, :, :],
+ new_mat[..., None, :, :],
+ rotations[..., i + 1 :, :, :],
+ ),
+ dim=-3,
+ )
+ else:
+ rotations[..., i, :, :] = get_mat_BfromA(
+ rotations[..., parent[i], :, :], mat[..., i, :, :]
+ )
+ else:
+ if isinstance(mat, torch.Tensor):
+ # this way make gradient flow
+ rotations = torch.cat(
+ (mat[..., : i + 1, :, :], rotations[..., i + 1 :, :, :]), dim=-3
+ )
+ else:
+ rotations[..., i, :, :] = mat[..., i, :, :]
+ return rotations
diff --git a/genmo/utils/net_utils.py b/genmo/utils/net_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..6435bd7e207521f4d8255c183f2ec2e29282c691
--- /dev/null
+++ b/genmo/utils/net_utils.py
@@ -0,0 +1,196 @@
+from pathlib import Path
+
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from einops import rearrange, repeat
+from pytorch_lightning.utilities.memory import recursive_detach
+from scipy.ndimage._filters import _gaussian_kernel1d
+
+from genmo.utils.pylogger import Log
+
+
+def load_pretrained_model(model, ckpt_path):
+ """
+ Load ckpt to model with strategy
+ """
+ assert Path(ckpt_path).exists()
+ # use model's own load_pretrained_model method
+ if hasattr(model, "load_pretrained_model"):
+ ckpt = model.load_pretrained_model(ckpt_path)
+ else:
+ Log.info(f"Loading ckpt: {ckpt_path}")
+ ckpt = torch.load(ckpt_path, "cpu")
+ model.load_state_dict(ckpt, strict=True)
+ return ckpt
+
+
+def find_last_ckpt_path(dirpath):
+ """
+ Assume ckpt is named as e{}* or last*, following the convention of pytorch-lightning.
+ """
+ assert dirpath is not None
+ dirpath = Path(dirpath)
+ assert dirpath.exists()
+ # Priority 1: last.ckpt
+ auto_last_ckpt_path = dirpath / "last.ckpt"
+ if auto_last_ckpt_path.exists():
+ return auto_last_ckpt_path
+
+ # Priority 2
+ model_paths = []
+ for p in sorted(list(dirpath.glob("*.ckpt"))):
+ if "last" in p.name:
+ continue
+ model_paths.append(p)
+ if len(model_paths) > 0:
+ return model_paths[-1]
+ else:
+ Log.info("No checkpoint found, set model_path to None")
+ return None
+
+
+def get_resume_ckpt_path(resume_mode, ckpt_dir=None):
+ if Path(resume_mode).exists(): # This is a path
+ return resume_mode
+ assert resume_mode == "last"
+ return find_last_ckpt_path(ckpt_dir)
+
+
+def select_state_dict_by_prefix(state_dict, prefix, new_prefix=""):
+ """
+ For each weight that start with {old_prefix}, remove the {old_prefic} and form a new state_dict.
+ Args:
+ state_dict: dict
+ prefix: str
+ new_prefix: str, if exists, the new key will be {new_prefix} + {old_key[len(prefix):]}
+ Returns:
+ state_dict_new: dict
+ """
+ state_dict_new = {}
+ for k in list(state_dict.keys()):
+ if k.startswith(prefix):
+ new_key = new_prefix + k[len(prefix) :]
+ state_dict_new[new_key] = state_dict[k]
+ return state_dict_new
+
+
+def detach_to_cpu(in_dict):
+ return recursive_detach(in_dict, to_cpu=True)
+
+
+def to_cuda(data):
+ """Move data in the batch to cuda(), carefully handle data that is not tensor"""
+ if isinstance(data, torch.Tensor):
+ return data.cuda()
+ elif isinstance(data, dict):
+ return {k: to_cuda(v) for k, v in data.items()}
+ elif isinstance(data, list):
+ return [to_cuda(v) for v in data]
+ else:
+ return data
+
+
+def get_valid_mask(max_len, valid_len, device="cpu"):
+ mask = torch.zeros(max_len, dtype=torch.bool).to(device)
+ mask[:valid_len] = True
+ return mask
+
+
+def length_to_mask(lengths, max_len):
+ """
+ Returns: (B, max_len)
+ """
+ mask = torch.arange(max_len, device=lengths.device).expand(
+ len(lengths), max_len
+ ) < lengths.unsqueeze(1)
+ return mask
+
+
+def repeat_to_max_len(x, max_len, dim=0):
+ """Repeat last frame to max_len along dim"""
+ assert isinstance(x, torch.Tensor)
+ if x.shape[dim] == max_len:
+ return x
+ elif x.shape[dim] < max_len:
+ x = x.clone()
+ x = x.transpose(0, dim)
+ x = torch.cat([x, repeat(x[-1:], "b ... -> (b r) ...", r=max_len - x.shape[0])])
+ x = x.transpose(0, dim)
+ return x
+ else:
+ raise ValueError(f"Unexpected length v.s. max_len: {x.shape[0]} v.s. {max_len}")
+
+
+def repeat_to_max_len_dict(x_dict, max_len, dim=0):
+ for k, v in x_dict.items():
+ x_dict[k] = repeat_to_max_len(v, max_len, dim=dim)
+ return x_dict
+
+
+class Transpose(nn.Module):
+ def __init__(self, dim1, dim2):
+ super(Transpose, self).__init__()
+ self.dim1 = dim1
+ self.dim2 = dim2
+
+ def forward(self, x):
+ return x.transpose(self.dim1, self.dim2)
+
+
+class GaussianSmooth(nn.Module):
+ def __init__(self, sigma=3, dim=-1):
+ super(GaussianSmooth, self).__init__()
+ kernel_smooth = _gaussian_kernel1d(
+ sigma=sigma, order=0, radius=int(4 * sigma + 0.5)
+ )
+ kernel_smooth = torch.from_numpy(kernel_smooth).float()[None, None] # (1, 1, K)
+ self.register_buffer("kernel_smooth", kernel_smooth, persistent=False)
+ self.dim = dim
+
+ def forward(self, x):
+ """x (..., f, ...) f at dim"""
+ rad = self.kernel_smooth.size(-1) // 2
+
+ x = x.transpose(self.dim, -1)
+ x_shape = x.shape[:-1]
+ x = rearrange(x, "... f -> (...) 1 f") # (NB, 1, f)
+ x = F.pad(x[None], (rad, rad, 0, 0), mode="replicate")[0]
+ x = F.conv1d(x, self.kernel_smooth)
+ x = x.squeeze(1).reshape(*x_shape, -1) # (..., f)
+ x = x.transpose(-1, self.dim)
+ return x
+
+
+def gaussian_smooth(x, sigma=3, dim=-1):
+ kernel_smooth = _gaussian_kernel1d(
+ sigma=sigma, order=0, radius=int(4 * sigma + 0.5)
+ )
+ kernel_smooth = (
+ torch.from_numpy(kernel_smooth).float()[None, None].to(x)
+ ) # (1, 1, K)
+ rad = kernel_smooth.size(-1) // 2
+
+ x = x.transpose(dim, -1)
+ x_shape = x.shape[:-1]
+ x = rearrange(x, "... f -> (...) 1 f") # (NB, 1, f)
+ x = F.pad(x[None], (rad, rad, 0, 0), mode="replicate")[0]
+ x = F.conv1d(x, kernel_smooth)
+ x = x.squeeze(1).reshape(*x_shape, -1) # (..., f)
+ x = x.transpose(-1, dim)
+ return x
+
+
+def moving_average_smooth(x, window_size=5, dim=-1):
+ kernel_smooth = torch.ones(window_size).float() / window_size
+ kernel_smooth = kernel_smooth[None, None].to(x) # (1, 1, window_size)
+ rad = kernel_smooth.size(-1) // 2
+
+ x = x.transpose(dim, -1)
+ x_shape = x.shape[:-1]
+ x = rearrange(x, "... f -> (...) 1 f") # (NB, 1, f)
+ x = F.pad(x[None], (rad, rad, 0, 0), mode="replicate")[0]
+ x = F.conv1d(x, kernel_smooth)
+ x = x.squeeze(1).reshape(*x_shape, -1) # (..., f)
+ x = x.transpose(-1, dim)
+ return x
diff --git a/genmo/utils/pylogger.py b/genmo/utils/pylogger.py
new file mode 100644
index 0000000000000000000000000000000000000000..1552bc52fdd28e16646e2e557e8d3055f363dd6a
--- /dev/null
+++ b/genmo/utils/pylogger.py
@@ -0,0 +1,57 @@
+import logging
+from time import time
+
+import torch
+from colorlog import ColoredFormatter
+
+from third_party.GVHMR.hmr4d.utils.pylogger import Log
+
+
+def timer(sync_cuda=False, mem=False, loop=1):
+ """
+ Args:
+ func: function
+ sync_cuda: bool, whether to synchronize cuda
+ mem: bool, whether to log memory
+ """
+
+ def decorator(func):
+ def wrapper(*args, **kwargs):
+ if mem:
+ start_mem = torch.cuda.memory_allocated() / 1024**2
+ if sync_cuda:
+ torch.cuda.synchronize()
+
+ start = Log.time()
+ for _ in range(loop):
+ result = func(*args, **kwargs)
+
+ if sync_cuda:
+ torch.cuda.synchronize()
+ if loop == 1:
+ message = f"{func.__name__} took {Log.time() - start:.3f} s."
+ else:
+ message = f"{func.__name__} took {(Log.time() - start) / loop:.3f} s. (loop={loop})"
+
+ if mem:
+ end_mem = torch.cuda.memory_allocated() / 1024**2
+ end_max_mem = torch.cuda.max_memory_allocated() / 1024**2
+ message += f" Start_Mem {start_mem:.1f} Max {end_max_mem:.1f} MB"
+ Log.info(message)
+
+ return result
+
+ return wrapper
+
+ return decorator
+
+
+def timed(fn):
+ """example usage: timed(lambda: model(inp))"""
+ start = torch.cuda.Event(enable_timing=True)
+ end = torch.cuda.Event(enable_timing=True)
+ start.record()
+ result = fn()
+ end.record()
+ torch.cuda.synchronize()
+ return result, start.elapsed_time(end) / 1000
diff --git a/genmo/utils/rotation_conversions.py b/genmo/utils/rotation_conversions.py
new file mode 100644
index 0000000000000000000000000000000000000000..a6ae3845cc6d4a3720c6456a0e9e8340a7a4b004
--- /dev/null
+++ b/genmo/utils/rotation_conversions.py
@@ -0,0 +1,551 @@
+# This code is based on https://github.com/Mathux/ACTOR.git
+# Copyright (c) Facebook, Inc. and its affiliates. All rights reserved.
+# Check PYTORCH3D_LICENCE before use
+
+import functools
+from typing import Optional
+
+import torch
+import torch.nn.functional as F
+
+"""
+The transformation matrices returned from the functions in this file assume
+the points on which the transformation will be applied are column vectors.
+i.e. the R matrix is structured as
+
+ R = [
+ [Rxx, Rxy, Rxz],
+ [Ryx, Ryy, Ryz],
+ [Rzx, Rzy, Rzz],
+ ] # (3, 3)
+
+This matrix can be applied to column vectors by post multiplication
+by the points e.g.
+
+ points = [[0], [1], [2]] # (3 x 1) xyz coordinates of a point
+ transformed_points = R * points
+
+To apply the same matrix to points which are row vectors, the R matrix
+can be transposed and pre multiplied by the points:
+
+e.g.
+ points = [[0, 1, 2]] # (1 x 3) xyz coordinates of a point
+ transformed_points = points * R.transpose(1, 0)
+"""
+
+
+def quaternion_to_matrix(quaternions):
+ """
+ Convert rotations given as quaternions to rotation matrices.
+
+ Args:
+ quaternions: quaternions with real part first,
+ as tensor of shape (..., 4).
+
+ Returns:
+ Rotation matrices as tensor of shape (..., 3, 3).
+ """
+ r, i, j, k = torch.unbind(quaternions, -1)
+ two_s = 2.0 / (quaternions * quaternions).sum(-1)
+
+ o = torch.stack(
+ (
+ 1 - two_s * (j * j + k * k),
+ two_s * (i * j - k * r),
+ two_s * (i * k + j * r),
+ two_s * (i * j + k * r),
+ 1 - two_s * (i * i + k * k),
+ two_s * (j * k - i * r),
+ two_s * (i * k - j * r),
+ two_s * (j * k + i * r),
+ 1 - two_s * (i * i + j * j),
+ ),
+ -1,
+ )
+ return o.reshape(quaternions.shape[:-1] + (3, 3))
+
+
+def _copysign(a, b):
+ """
+ Return a tensor where each element has the absolute value taken from the,
+ corresponding element of a, with sign taken from the corresponding
+ element of b. This is like the standard copysign floating-point operation,
+ but is not careful about negative 0 and NaN.
+
+ Args:
+ a: source tensor.
+ b: tensor whose signs will be used, of the same shape as a.
+
+ Returns:
+ Tensor of the same shape as a with the signs of b.
+ """
+ signs_differ = (a < 0) != (b < 0)
+ return torch.where(signs_differ, -a, a)
+
+
+def _sqrt_positive_part(x):
+ """
+ Returns torch.sqrt(torch.max(0, x))
+ but with a zero subgradient where x is 0.
+ """
+ ret = torch.zeros_like(x)
+ positive_mask = x > 0
+ ret[positive_mask] = torch.sqrt(x[positive_mask])
+ return ret
+
+
+def matrix_to_quaternion(matrix):
+ """
+ Convert rotations given as rotation matrices to quaternions.
+
+ Args:
+ matrix: Rotation matrices as tensor of shape (..., 3, 3).
+
+ Returns:
+ quaternions with real part first, as tensor of shape (..., 4).
+ """
+ if matrix.size(-1) != 3 or matrix.size(-2) != 3:
+ raise ValueError(f"Invalid rotation matrix shape f{matrix.shape}.")
+ m00 = matrix[..., 0, 0]
+ m11 = matrix[..., 1, 1]
+ m22 = matrix[..., 2, 2]
+ o0 = 0.5 * _sqrt_positive_part(1 + m00 + m11 + m22)
+ x = 0.5 * _sqrt_positive_part(1 + m00 - m11 - m22)
+ y = 0.5 * _sqrt_positive_part(1 - m00 + m11 - m22)
+ z = 0.5 * _sqrt_positive_part(1 - m00 - m11 + m22)
+ o1 = _copysign(x, matrix[..., 2, 1] - matrix[..., 1, 2])
+ o2 = _copysign(y, matrix[..., 0, 2] - matrix[..., 2, 0])
+ o3 = _copysign(z, matrix[..., 1, 0] - matrix[..., 0, 1])
+ return torch.stack((o0, o1, o2, o3), -1)
+
+
+def _axis_angle_rotation(axis: str, angle):
+ """
+ Return the rotation matrices for one of the rotations about an axis
+ of which Euler angles describe, for each value of the angle given.
+
+ Args:
+ axis: Axis label "X" or "Y or "Z".
+ angle: any shape tensor of Euler angles in radians
+
+ Returns:
+ Rotation matrices as tensor of shape (..., 3, 3).
+ """
+
+ cos = torch.cos(angle)
+ sin = torch.sin(angle)
+ one = torch.ones_like(angle)
+ zero = torch.zeros_like(angle)
+
+ if axis == "X":
+ R_flat = (one, zero, zero, zero, cos, -sin, zero, sin, cos)
+ if axis == "Y":
+ R_flat = (cos, zero, sin, zero, one, zero, -sin, zero, cos)
+ if axis == "Z":
+ R_flat = (cos, -sin, zero, sin, cos, zero, zero, zero, one)
+
+ return torch.stack(R_flat, -1).reshape(angle.shape + (3, 3))
+
+
+def euler_angles_to_matrix(euler_angles, convention: str):
+ """
+ Convert rotations given as Euler angles in radians to rotation matrices.
+
+ Args:
+ euler_angles: Euler angles in radians as tensor of shape (..., 3).
+ convention: Convention string of three uppercase letters from
+ {"X", "Y", and "Z"}.
+
+ Returns:
+ Rotation matrices as tensor of shape (..., 3, 3).
+ """
+ if euler_angles.dim() == 0 or euler_angles.shape[-1] != 3:
+ raise ValueError("Invalid input euler angles.")
+ if len(convention) != 3:
+ raise ValueError("Convention must have 3 letters.")
+ if convention[1] in (convention[0], convention[2]):
+ raise ValueError(f"Invalid convention {convention}.")
+ for letter in convention:
+ if letter not in ("X", "Y", "Z"):
+ raise ValueError(f"Invalid letter {letter} in convention string.")
+ matrices = map(_axis_angle_rotation, convention, torch.unbind(euler_angles, -1))
+ return functools.reduce(torch.matmul, matrices)
+
+
+def _angle_from_tan(
+ axis: str, other_axis: str, data, horizontal: bool, tait_bryan: bool
+):
+ """
+ Extract the first or third Euler angle from the two members of
+ the matrix which are positive constant times its sine and cosine.
+
+ Args:
+ axis: Axis label "X" or "Y or "Z" for the angle we are finding.
+ other_axis: Axis label "X" or "Y or "Z" for the middle axis in the
+ convention.
+ data: Rotation matrices as tensor of shape (..., 3, 3).
+ horizontal: Whether we are looking for the angle for the third axis,
+ which means the relevant entries are in the same row of the
+ rotation matrix. If not, they are in the same column.
+ tait_bryan: Whether the first and third axes in the convention differ.
+
+ Returns:
+ Euler Angles in radians for each matrix in dataset as a tensor
+ of shape (...).
+ """
+
+ i1, i2 = {"X": (2, 1), "Y": (0, 2), "Z": (1, 0)}[axis]
+ if horizontal:
+ i2, i1 = i1, i2
+ even = (axis + other_axis) in ["XY", "YZ", "ZX"]
+ if horizontal == even:
+ return torch.atan2(data[..., i1], data[..., i2])
+ if tait_bryan:
+ return torch.atan2(-data[..., i2], data[..., i1])
+ return torch.atan2(data[..., i2], -data[..., i1])
+
+
+def _index_from_letter(letter: str):
+ if letter == "X":
+ return 0
+ if letter == "Y":
+ return 1
+ if letter == "Z":
+ return 2
+
+
+def matrix_to_euler_angles(matrix, convention: str):
+ """
+ Convert rotations given as rotation matrices to Euler angles in radians.
+
+ Args:
+ matrix: Rotation matrices as tensor of shape (..., 3, 3).
+ convention: Convention string of three uppercase letters.
+
+ Returns:
+ Euler angles in radians as tensor of shape (..., 3).
+ """
+ if len(convention) != 3:
+ raise ValueError("Convention must have 3 letters.")
+ if convention[1] in (convention[0], convention[2]):
+ raise ValueError(f"Invalid convention {convention}.")
+ for letter in convention:
+ if letter not in ("X", "Y", "Z"):
+ raise ValueError(f"Invalid letter {letter} in convention string.")
+ if matrix.size(-1) != 3 or matrix.size(-2) != 3:
+ raise ValueError(f"Invalid rotation matrix shape f{matrix.shape}.")
+ i0 = _index_from_letter(convention[0])
+ i2 = _index_from_letter(convention[2])
+ tait_bryan = i0 != i2
+ if tait_bryan:
+ central_angle = torch.asin(
+ matrix[..., i0, i2] * (-1.0 if i0 - i2 in [-1, 2] else 1.0)
+ )
+ else:
+ central_angle = torch.acos(matrix[..., i0, i0])
+
+ o = (
+ _angle_from_tan(
+ convention[0], convention[1], matrix[..., i2], False, tait_bryan
+ ),
+ central_angle,
+ _angle_from_tan(
+ convention[2], convention[1], matrix[..., i0, :], True, tait_bryan
+ ),
+ )
+ return torch.stack(o, -1)
+
+
+def random_quaternions(
+ n: int, dtype: Optional[torch.dtype] = None, device=None, requires_grad=False
+):
+ """
+ Generate random quaternions representing rotations,
+ i.e. versors with nonnegative real part.
+
+ Args:
+ n: Number of quaternions in a batch to return.
+ dtype: Type to return.
+ device: Desired device of returned tensor. Default:
+ uses the current device for the default tensor type.
+ requires_grad: Whether the resulting tensor should have the gradient
+ flag set.
+
+ Returns:
+ Quaternions as tensor of shape (N, 4).
+ """
+ o = torch.randn((n, 4), dtype=dtype, device=device, requires_grad=requires_grad)
+ s = (o * o).sum(1)
+ o = o / _copysign(torch.sqrt(s), o[:, 0])[:, None]
+ return o
+
+
+def random_rotations(
+ n: int, dtype: Optional[torch.dtype] = None, device=None, requires_grad=False
+):
+ """
+ Generate random rotations as 3x3 rotation matrices.
+
+ Args:
+ n: Number of rotation matrices in a batch to return.
+ dtype: Type to return.
+ device: Device of returned tensor. Default: if None,
+ uses the current device for the default tensor type.
+ requires_grad: Whether the resulting tensor should have the gradient
+ flag set.
+
+ Returns:
+ Rotation matrices as tensor of shape (n, 3, 3).
+ """
+ quaternions = random_quaternions(
+ n, dtype=dtype, device=device, requires_grad=requires_grad
+ )
+ return quaternion_to_matrix(quaternions)
+
+
+def random_rotation(
+ dtype: Optional[torch.dtype] = None, device=None, requires_grad=False
+):
+ """
+ Generate a single random 3x3 rotation matrix.
+
+ Args:
+ dtype: Type to return
+ device: Device of returned tensor. Default: if None,
+ uses the current device for the default tensor type
+ requires_grad: Whether the resulting tensor should have the gradient
+ flag set
+
+ Returns:
+ Rotation matrix as tensor of shape (3, 3).
+ """
+ return random_rotations(1, dtype, device, requires_grad)[0]
+
+
+def standardize_quaternion(quaternions):
+ """
+ Convert a unit quaternion to a standard form: one in which the real
+ part is non negative.
+
+ Args:
+ quaternions: Quaternions with real part first,
+ as tensor of shape (..., 4).
+
+ Returns:
+ Standardized quaternions as tensor of shape (..., 4).
+ """
+ return torch.where(quaternions[..., 0:1] < 0, -quaternions, quaternions)
+
+
+def quaternion_raw_multiply(a, b):
+ """
+ Multiply two quaternions.
+ Usual torch rules for broadcasting apply.
+
+ Args:
+ a: Quaternions as tensor of shape (..., 4), real part first.
+ b: Quaternions as tensor of shape (..., 4), real part first.
+
+ Returns:
+ The product of a and b, a tensor of quaternions shape (..., 4).
+ """
+ aw, ax, ay, az = torch.unbind(a, -1)
+ bw, bx, by, bz = torch.unbind(b, -1)
+ ow = aw * bw - ax * bx - ay * by - az * bz
+ ox = aw * bx + ax * bw + ay * bz - az * by
+ oy = aw * by - ax * bz + ay * bw + az * bx
+ oz = aw * bz + ax * by - ay * bx + az * bw
+ return torch.stack((ow, ox, oy, oz), -1)
+
+
+def quaternion_multiply(a, b):
+ """
+ Multiply two quaternions representing rotations, returning the quaternion
+ representing their composition, i.e. the versor with nonnegative real part.
+ Usual torch rules for broadcasting apply.
+
+ Args:
+ a: Quaternions as tensor of shape (..., 4), real part first.
+ b: Quaternions as tensor of shape (..., 4), real part first.
+
+ Returns:
+ The product of a and b, a tensor of quaternions of shape (..., 4).
+ """
+ ab = quaternion_raw_multiply(a, b)
+ return standardize_quaternion(ab)
+
+
+def quaternion_invert(quaternion):
+ """
+ Given a quaternion representing rotation, get the quaternion representing
+ its inverse.
+
+ Args:
+ quaternion: Quaternions as tensor of shape (..., 4), with real part
+ first, which must be versors (unit quaternions).
+
+ Returns:
+ The inverse, a tensor of quaternions of shape (..., 4).
+ """
+
+ return quaternion * quaternion.new_tensor([1, -1, -1, -1])
+
+
+def quaternion_apply(quaternion, point):
+ """
+ Apply the rotation given by a quaternion to a 3D point.
+ Usual torch rules for broadcasting apply.
+
+ Args:
+ quaternion: Tensor of quaternions, real part first, of shape (..., 4).
+ point: Tensor of 3D points of shape (..., 3).
+
+ Returns:
+ Tensor of rotated points of shape (..., 3).
+ """
+ if point.size(-1) != 3:
+ raise ValueError(f"Points are not in 3D, f{point.shape}.")
+ real_parts = point.new_zeros(point.shape[:-1] + (1,))
+ point_as_quaternion = torch.cat((real_parts, point), -1)
+ out = quaternion_raw_multiply(
+ quaternion_raw_multiply(quaternion, point_as_quaternion),
+ quaternion_invert(quaternion),
+ )
+ return out[..., 1:]
+
+
+def axis_angle_to_matrix(axis_angle):
+ """
+ Convert rotations given as axis/angle to rotation matrices.
+
+ Args:
+ axis_angle: Rotations given as a vector in axis angle form,
+ as a tensor of shape (..., 3), where the magnitude is
+ the angle turned anticlockwise in radians around the
+ vector's direction.
+
+ Returns:
+ Rotation matrices as tensor of shape (..., 3, 3).
+ """
+ return quaternion_to_matrix(axis_angle_to_quaternion(axis_angle))
+
+
+def matrix_to_axis_angle(matrix):
+ """
+ Convert rotations given as rotation matrices to axis/angle.
+
+ Args:
+ matrix: Rotation matrices as tensor of shape (..., 3, 3).
+
+ Returns:
+ Rotations given as a vector in axis angle form, as a tensor
+ of shape (..., 3), where the magnitude is the angle
+ turned anticlockwise in radians around the vector's
+ direction.
+ """
+ return quaternion_to_axis_angle(matrix_to_quaternion(matrix))
+
+
+def axis_angle_to_quaternion(axis_angle):
+ """
+ Convert rotations given as axis/angle to quaternions.
+
+ Args:
+ axis_angle: Rotations given as a vector in axis angle form,
+ as a tensor of shape (..., 3), where the magnitude is
+ the angle turned anticlockwise in radians around the
+ vector's direction.
+
+ Returns:
+ quaternions with real part first, as tensor of shape (..., 4).
+ """
+ angles = torch.norm(axis_angle, p=2, dim=-1, keepdim=True)
+ half_angles = 0.5 * angles
+ eps = 1e-6
+ small_angles = angles.abs() < eps
+ sin_half_angles_over_angles = torch.empty_like(angles)
+ sin_half_angles_over_angles[~small_angles] = (
+ torch.sin(half_angles[~small_angles]) / angles[~small_angles]
+ )
+ # for x small, sin(x/2) is about x/2 - (x/2)^3/6
+ # so sin(x/2)/x is about 1/2 - (x*x)/48
+ sin_half_angles_over_angles[small_angles] = (
+ 0.5 - (angles[small_angles] * angles[small_angles]) / 48
+ )
+ quaternions = torch.cat(
+ [torch.cos(half_angles), axis_angle * sin_half_angles_over_angles], dim=-1
+ )
+ return quaternions
+
+
+def quaternion_to_axis_angle(quaternions):
+ """
+ Convert rotations given as quaternions to axis/angle.
+
+ Args:
+ quaternions: quaternions with real part first,
+ as tensor of shape (..., 4).
+
+ Returns:
+ Rotations given as a vector in axis angle form, as a tensor
+ of shape (..., 3), where the magnitude is the angle
+ turned anticlockwise in radians around the vector's
+ direction.
+ """
+ norms = torch.norm(quaternions[..., 1:], p=2, dim=-1, keepdim=True)
+ half_angles = torch.atan2(norms, quaternions[..., :1])
+ angles = 2 * half_angles
+ eps = 1e-6
+ small_angles = angles.abs() < eps
+ sin_half_angles_over_angles = torch.empty_like(angles)
+ sin_half_angles_over_angles[~small_angles] = (
+ torch.sin(half_angles[~small_angles]) / angles[~small_angles]
+ )
+ # for x small, sin(x/2) is about x/2 - (x/2)^3/6
+ # so sin(x/2)/x is about 1/2 - (x*x)/48
+ sin_half_angles_over_angles[small_angles] = (
+ 0.5 - (angles[small_angles] * angles[small_angles]) / 48
+ )
+ return quaternions[..., 1:] / sin_half_angles_over_angles
+
+
+def rotation_6d_to_matrix(d6: torch.Tensor) -> torch.Tensor:
+ """
+ Converts 6D rotation representation by Zhou et al. [1] to rotation matrix
+ using Gram--Schmidt orthogonalisation per Section B of [1].
+ Args:
+ d6: 6D rotation representation, of size (*, 6)
+
+ Returns:
+ batch of rotation matrices of size (*, 3, 3)
+
+ [1] Zhou, Y., Barnes, C., Lu, J., Yang, J., & Li, H.
+ On the Continuity of Rotation Representations in Neural Networks.
+ IEEE Conference on Computer Vision and Pattern Recognition, 2019.
+ Retrieved from http://arxiv.org/abs/1812.07035
+ """
+
+ a1, a2 = d6[..., :3], d6[..., 3:]
+ b1 = F.normalize(a1, dim=-1)
+ b2 = a2 - (b1 * a2).sum(-1, keepdim=True) * b1
+ b2 = F.normalize(b2, dim=-1)
+ b3 = torch.cross(b1, b2, dim=-1)
+ return torch.stack((b1, b2, b3), dim=-2)
+
+
+def matrix_to_rotation_6d(matrix: torch.Tensor) -> torch.Tensor:
+ """
+ Converts rotation matrices to 6D rotation representation by Zhou et al. [1]
+ by dropping the last row. Note that 6D representation is not unique.
+ Args:
+ matrix: batch of rotation matrices of size (*, 3, 3)
+
+ Returns:
+ 6D rotation representation, of size (*, 6)
+
+ [1] Zhou, Y., Barnes, C., Lu, J., Yang, J., & Li, H.
+ On the Continuity of Rotation Representations in Neural Networks.
+ IEEE Conference on Computer Vision and Pattern Recognition, 2019.
+ Retrieved from http://arxiv.org/abs/1812.07035
+ """
+ return matrix[..., :2, :].clone().reshape(*matrix.size()[:-2], 6)
diff --git a/genmo/utils/seq_utils.py b/genmo/utils/seq_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..686ec2682de8254b4766c23ac958c2ff0fd5f330
--- /dev/null
+++ b/genmo/utils/seq_utils.py
@@ -0,0 +1,196 @@
+import numpy as np
+import torch
+
+# def get_frame_id_list_from_mask(mask):
+# """
+# Args:
+# mask (F,), bool.
+# Return:
+# frame_id_list: List of frame_ids.
+# """
+# frame_id_list = []
+# i = 0
+# while i < len(mask):
+# if not mask[i]:
+# i += 1
+# else:
+# j = i
+# while j < len(mask) and mask[j]:
+# j += 1
+# frame_id_list.append(torch.arange(i, j))
+# i = j
+
+# return frame_id_list
+
+
+# From GPT
+def get_frame_id_list_from_mask(mask):
+ # batch=64, 0.13s
+ """
+ Vectorized approach to get frame id list from a boolean mask.
+
+ Args:
+ mask (F,), bool tensor: Mask array where `True` indicates a frame to be processed.
+
+ Returns:
+ frame_id_list: List of torch.Tensors, each tensor containing continuous indices where mask is True.
+ """
+ # Find the indices where the mask changes from False to True and vice versa
+ padded_mask = torch.cat(
+ [
+ torch.tensor([False], device=mask.device),
+ mask,
+ torch.tensor([False], device=mask.device),
+ ]
+ )
+ diffs = torch.diff(padded_mask.int())
+ starts = (diffs == 1).nonzero(as_tuple=False).squeeze()
+ ends = (diffs == -1).nonzero(as_tuple=False).squeeze()
+ if starts.numel() == 0:
+ return []
+ if starts.numel() == 1:
+ starts = starts.reshape(-1)
+ ends = ends.reshape(-1)
+
+ # Create list of ranges
+ frame_id_list = [torch.arange(start, end) for start, end in zip(starts, ends)]
+ return frame_id_list
+
+
+def get_batch_frame_id_lists_from_mask_BLC(masks):
+ # batch=64, 0.10s
+ """
+ 处理三维掩码数组,为每个批次和通道提取连续True区段的索引列表。
+
+ 参数:
+ masks (B, L, C), 布尔张量:每个元素代表一个掩码,True表示需要处理的帧。
+
+ 返回:
+ batch_frame_id_lists: 对应于每个批次和每个通道的帧id列表的嵌套列表。
+ """
+ B, L, C = masks.size()
+ # 在序列长度两端添加一个False
+ padded_masks = torch.cat(
+ [
+ torch.zeros((B, 1, C), dtype=torch.bool, device=masks.device),
+ masks,
+ torch.zeros((B, 1, C), dtype=torch.bool, device=masks.device),
+ ],
+ dim=1,
+ )
+ # 计算差分来找到True区段的起始和结束点
+ diffs = torch.diff(padded_masks.int(), dim=1)
+ starts = (diffs == 1).nonzero(as_tuple=True)
+ ends = (diffs == -1).nonzero(as_tuple=True)
+
+ # 初始化返回列表
+ batch_frame_id_lists = [[[] for _ in range(C)] for _ in range(B)]
+ for b in range(B):
+ for c in range(C):
+ batch_start = starts[0][(starts[0] == b) & (starts[2] == c)]
+ batch_end = ends[0][(ends[0] == b) & (ends[2] == c)]
+ # 确保start和end都是1维张量
+ batch_frame_id_lists[b][c] = [
+ torch.arange(start.item(), end.item())
+ for start, end in zip(batch_start, batch_end)
+ ]
+
+ return batch_frame_id_lists
+
+
+def get_frame_id_list_from_frame_id(frame_id):
+ mask = torch.zeros(frame_id[-1] + 1, dtype=torch.bool)
+ mask[frame_id] = True
+ frame_id_list = get_frame_id_list_from_mask(mask)
+ return frame_id_list
+
+
+def rearrange_by_mask(x, mask):
+ """
+ x (L, *)
+ mask (M,), M >= L
+ """
+ M = mask.size(0)
+ L = x.size(0)
+ if M == L:
+ return x
+ assert M > L
+ assert mask.sum() == L
+ x_rearranged = torch.zeros((M, *x.size()[1:]), dtype=x.dtype, device=x.device)
+ x_rearranged[mask] = x
+ return x_rearranged
+
+
+def frame_id_to_mask(frame_id, max_len):
+ mask = torch.zeros(max_len, dtype=torch.bool)
+ mask[frame_id] = True
+ return mask
+
+
+def mask_to_frame_id(mask):
+ frame_id = torch.where(mask)[0]
+ return frame_id
+
+
+def linear_interpolate_frame_ids(data, frame_id_list):
+ data = data.clone()
+ for i, invalid_frame_ids in enumerate(frame_id_list):
+ # interplate between prev, next
+ # if at beginning or end, use the same value
+ if invalid_frame_ids[0] - 1 < 0 or invalid_frame_ids[-1] + 1 >= len(data):
+ if invalid_frame_ids[0] - 1 < 0:
+ data[invalid_frame_ids] = data[invalid_frame_ids[-1] + 1].clone()
+ else:
+ data[invalid_frame_ids] = data[invalid_frame_ids[0] - 1].clone()
+ else:
+ prev = data[invalid_frame_ids[0] - 1]
+ next = data[invalid_frame_ids[-1] + 1]
+ data[invalid_frame_ids] = (
+ torch.linspace(0, 1, len(invalid_frame_ids) + 2)[1:-1][:, None]
+ * (next - prev)[None]
+ + prev[None]
+ )
+ return data
+
+
+def linear_interpolate(data, N_middle_frames):
+ """
+ Args:
+ data: (2, C)
+ Returns:
+ data_interpolated: (1+N+1, C)
+ """
+ prev = data[0]
+ next = data[1]
+ middle = (
+ torch.linspace(0, 1, N_middle_frames + 2)[1:-1][:, None] * (next - prev)[None]
+ + prev[None]
+ ) # (N, C)
+ data_interpolated = torch.cat(
+ [data[0][None], middle, data[1][None]], dim=0
+ ) # (1+N+1, C)
+ return data_interpolated
+
+
+def find_top_k_span(mask, k=3):
+ """
+ Args:
+ mask: (L,)
+ Return:
+ topk_span: List of tuple, usage: [start, end)
+ """
+ if isinstance(mask, np.ndarray):
+ mask = torch.from_numpy(mask)
+ if mask.sum() == 0:
+ return []
+ mask = mask.clone().float()
+ mask = torch.cat([mask.new([0]), mask, mask.new([0])])
+ diff = mask[1:] - mask[:-1]
+ start = torch.where(diff == 1)[0]
+ end = torch.where(diff == -1)[0]
+ assert len(start) == len(end)
+ span_lengths = end - start
+ span_lengths, idx = span_lengths.sort(descending=True)
+ start = start[idx]
+ end = end[idx]
+ return list(zip(start.tolist(), end.tolist()))[:k]
diff --git a/genmo/utils/so3.py b/genmo/utils/so3.py
new file mode 100644
index 0000000000000000000000000000000000000000..71b0db700905612eef2c980d151ff7374fe3ed6c
--- /dev/null
+++ b/genmo/utils/so3.py
@@ -0,0 +1,270 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the BSD-style license found in the
+# LICENSE file in the root directory of this source tree.
+
+# pyre-unsafe
+
+import warnings
+from typing import Tuple
+
+import torch
+
+from genmo.utils.math import acos_linear_extrapolation
+from genmo.utils.rotation_conversions import axis_angle_to_matrix, matrix_to_axis_angle
+
+
+def so3_relative_angle(
+ R1: torch.Tensor,
+ R2: torch.Tensor,
+ cos_angle: bool = False,
+ cos_bound: float = 1e-4,
+ eps: float = 1e-4,
+) -> torch.Tensor:
+ """
+ Calculates the relative angle (in radians) between pairs of
+ rotation matrices `R1` and `R2` with `angle = acos(0.5 * (Trace(R1 R2^T)-1))`
+
+ .. note::
+ This corresponds to a geodesic distance on the 3D manifold of rotation
+ matrices.
+
+ Args:
+ R1: Batch of rotation matrices of shape `(minibatch, 3, 3)`.
+ R2: Batch of rotation matrices of shape `(minibatch, 3, 3)`.
+ cos_angle: If==True return cosine of the relative angle rather than
+ the angle itself. This can avoid the unstable calculation of `acos`.
+ cos_bound: Clamps the cosine of the relative rotation angle to
+ [-1 + cos_bound, 1 - cos_bound] to avoid non-finite outputs/gradients
+ of the `acos` call. Note that the non-finite outputs/gradients
+ are returned when the angle is requested (i.e. `cos_angle==False`)
+ and the rotation angle is close to 0 or π.
+ eps: Tolerance for the valid trace check of the relative rotation matrix
+ in `so3_rotation_angle`.
+ Returns:
+ Corresponding rotation angles of shape `(minibatch,)`.
+ If `cos_angle==True`, returns the cosine of the angles.
+
+ Raises:
+ ValueError if `R1` or `R2` is of incorrect shape.
+ ValueError if `R1` or `R2` has an unexpected trace.
+ """
+ R12 = torch.bmm(R1, R2.permute(0, 2, 1))
+ return so3_rotation_angle(R12, cos_angle=cos_angle, cos_bound=cos_bound, eps=eps)
+
+
+def so3_rotation_angle(
+ R: torch.Tensor,
+ eps: float = 1e-4,
+ cos_angle: bool = False,
+ cos_bound: float = 1e-4,
+) -> torch.Tensor:
+ """
+ Calculates angles (in radians) of a batch of rotation matrices `R` with
+ `angle = acos(0.5 * (Trace(R)-1))`. The trace of the
+ input matrices is checked to be in the valid range `[-1-eps,3+eps]`.
+ The `eps` argument is a small constant that allows for small errors
+ caused by limited machine precision.
+
+ Args:
+ R: Batch of rotation matrices of shape `(minibatch, 3, 3)`.
+ eps: Tolerance for the valid trace check.
+ cos_angle: If==True return cosine of the rotation angles rather than
+ the angle itself. This can avoid the unstable
+ calculation of `acos`.
+ cos_bound: Clamps the cosine of the rotation angle to
+ [-1 + cos_bound, 1 - cos_bound] to avoid non-finite outputs/gradients
+ of the `acos` call. Note that the non-finite outputs/gradients
+ are returned when the angle is requested (i.e. `cos_angle==False`)
+ and the rotation angle is close to 0 or π.
+
+ Returns:
+ Corresponding rotation angles of shape `(minibatch,)`.
+ If `cos_angle==True`, returns the cosine of the angles.
+
+ Raises:
+ ValueError if `R` is of incorrect shape.
+ ValueError if `R` has an unexpected trace.
+ """
+
+ N, dim1, dim2 = R.shape
+ if dim1 != 3 or dim2 != 3:
+ raise ValueError("Input has to be a batch of 3x3 Tensors.")
+
+ rot_trace = R[:, 0, 0] + R[:, 1, 1] + R[:, 2, 2]
+
+ if ((rot_trace < -1.0 - eps) + (rot_trace > 3.0 + eps)).any():
+ raise ValueError("A matrix has trace outside valid range [-1-eps,3+eps].")
+
+ # phi ... rotation angle
+ phi_cos = (rot_trace - 1.0) * 0.5
+
+ if cos_angle:
+ return phi_cos
+ else:
+ if cos_bound > 0.0:
+ bound = 1.0 - cos_bound
+ return acos_linear_extrapolation(phi_cos, (-bound, bound))
+ else:
+ return torch.acos(phi_cos)
+
+
+def so3_exp_map(log_rot: torch.Tensor, eps: float = 0.0001) -> torch.Tensor:
+ """
+ Convert a batch of logarithmic representations of rotation matrices `log_rot`
+ to a batch of 3x3 rotation matrices using Rodrigues formula [1].
+
+ In the logarithmic representation, each rotation matrix is represented as
+ a 3-dimensional vector (`log_rot`) who's l2-norm and direction correspond
+ to the magnitude of the rotation angle and the axis of rotation respectively.
+
+ The conversion has a singularity around `log(R) = 0`
+ which is handled by clamping controlled with the `eps` argument.
+
+ Args:
+ log_rot: Batch of vectors of shape `(minibatch, 3)`.
+ eps: A float constant handling the conversion singularity.
+
+ Returns:
+ Batch of rotation matrices of shape `(minibatch, 3, 3)`.
+
+ Raises:
+ ValueError if `log_rot` is of incorrect shape.
+
+ [1] https://en.wikipedia.org/wiki/Rodrigues%27_rotation_formula
+ """
+ return _so3_exp_map(log_rot, eps=eps)[0]
+
+
+def so3_exponential_map(log_rot: torch.Tensor, eps: float = 0.0001) -> torch.Tensor:
+ warnings.warn(
+ """so3_exponential_map is deprecated,
+ Use so3_exp_map instead.
+ so3_exponential_map will be removed in future releases.""",
+ PendingDeprecationWarning,
+ )
+
+ return so3_exp_map(log_rot, eps)
+
+
+def _so3_exp_map(
+ log_rot: torch.Tensor, eps: float = 0.0001
+) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]:
+ """
+ A helper function that computes the so3 exponential map and,
+ apart from the rotation matrix, also returns intermediate variables
+ that can be re-used in other functions.
+ """
+ _, dim = log_rot.shape
+ if dim != 3:
+ raise ValueError("Input tensor shape has to be Nx3.")
+
+ nrms = (log_rot * log_rot).sum(1)
+ # phis ... rotation angles
+ rot_angles = torch.clamp(nrms, eps).sqrt()
+ skews = hat(log_rot)
+ skews_square = torch.bmm(skews, skews)
+
+ R = axis_angle_to_matrix(log_rot)
+
+ return R, rot_angles, skews, skews_square
+
+
+def so3_log_map(
+ R: torch.Tensor, eps: float = 0.0001, cos_bound: float = 1e-4
+) -> torch.Tensor:
+ """
+ Convert a batch of 3x3 rotation matrices `R`
+ to a batch of 3-dimensional matrix logarithms of rotation matrices
+ The conversion has a singularity around `(R=I)`.
+
+ Args:
+ R: batch of rotation matrices of shape `(minibatch, 3, 3)`.
+ eps: (unused, for backward compatibility)
+ cos_bound: (unused, for backward compatibility)
+
+ Returns:
+ Batch of logarithms of input rotation matrices
+ of shape `(minibatch, 3)`.
+ """
+
+ N, dim1, dim2 = R.shape
+ if dim1 != 3 or dim2 != 3:
+ raise ValueError("Input has to be a batch of 3x3 Tensors.")
+
+ return matrix_to_axis_angle(R)
+
+
+def hat_inv(h: torch.Tensor) -> torch.Tensor:
+ """
+ Compute the inverse Hat operator [1] of a batch of 3x3 matrices.
+
+ Args:
+ h: Batch of skew-symmetric matrices of shape `(minibatch, 3, 3)`.
+
+ Returns:
+ Batch of 3d vectors of shape `(minibatch, 3, 3)`.
+
+ Raises:
+ ValueError if `h` is of incorrect shape.
+ ValueError if `h` not skew-symmetric.
+
+ [1] https://en.wikipedia.org/wiki/Hat_operator
+ """
+
+ N, dim1, dim2 = h.shape
+ if dim1 != 3 or dim2 != 3:
+ raise ValueError("Input has to be a batch of 3x3 Tensors.")
+
+ ss_diff = torch.abs(h + h.permute(0, 2, 1)).max()
+
+ HAT_INV_SKEW_SYMMETRIC_TOL = 1e-5
+ if float(ss_diff) > HAT_INV_SKEW_SYMMETRIC_TOL:
+ raise ValueError("One of input matrices is not skew-symmetric.")
+
+ x = h[:, 2, 1]
+ y = h[:, 0, 2]
+ z = h[:, 1, 0]
+
+ v = torch.stack((x, y, z), dim=1)
+
+ return v
+
+
+def hat(v: torch.Tensor) -> torch.Tensor:
+ """
+ Compute the Hat operator [1] of a batch of 3D vectors.
+
+ Args:
+ v: Batch of vectors of shape `(minibatch , 3)`.
+
+ Returns:
+ Batch of skew-symmetric matrices of shape
+ `(minibatch, 3 , 3)` where each matrix is of the form:
+ `[ 0 -v_z v_y ]
+ [ v_z 0 -v_x ]
+ [ -v_y v_x 0 ]`
+
+ Raises:
+ ValueError if `v` is of incorrect shape.
+
+ [1] https://en.wikipedia.org/wiki/Hat_operator
+ """
+
+ N, dim = v.shape
+ if dim != 3:
+ raise ValueError("Input vectors have to be 3-dimensional.")
+
+ h = torch.zeros((N, 3, 3), dtype=v.dtype, device=v.device)
+
+ x, y, z = v.unbind(1)
+
+ h[:, 0, 1] = -z
+ h[:, 0, 2] = y
+ h[:, 1, 0] = z
+ h[:, 1, 2] = -x
+ h[:, 2, 0] = -y
+ h[:, 2, 1] = x
+
+ return h
diff --git a/genmo/utils/tools.py b/genmo/utils/tools.py
new file mode 100644
index 0000000000000000000000000000000000000000..71664a8461c79b6cbfd8df41025c0e54741982a5
--- /dev/null
+++ b/genmo/utils/tools.py
@@ -0,0 +1,230 @@
+import datetime
+import glob
+import importlib
+import itertools
+import os
+import os.path as osp
+import subprocess
+import time
+
+import numpy as np
+
+
+class AverageMeter(object):
+ def __init__(self, avg=None, count=1):
+ self.reset()
+ if avg is not None:
+ self.val = avg
+ self.avg = avg
+ self.count = count
+ self.sum = avg * count
+
+ def __repr__(self) -> str:
+ return f"{self.avg: .4f}"
+
+ def reset(self):
+ self.val = 0
+ self.avg = 0
+ self.sum = 0
+ self.count = 0
+
+ def update(self, val, n=1):
+ if n > 0:
+ self.val = val
+ self.sum += val * n
+ self.count += n
+ self.avg = self.sum / self.count
+
+
+def worker_init_fn(worker_id):
+ os.environ["worker_id"] = str(worker_id)
+ np.random.seed(np.random.get_state()[1][0] + worker_id * 7)
+
+
+def find_last_version(folder, prefix="version_", cp="last"):
+ version_folders = glob.glob(f"{folder}/{prefix}*")
+ if cp is not None:
+ if cp == "last":
+ suffix = "last.ckpt"
+ elif cp == "best":
+ suffix = "*best*.ckpt"
+ elif cp.isdigit():
+ suffix = f"*{int(cp):07d}.ckpt"
+ else:
+ suffix = f"{cp}.ckpt"
+ version_folders = [
+ x for x in version_folders if len(glob.glob(f"{x}/**/{suffix}")) > 0
+ ]
+ version_numbers = sorted(
+ [int(osp.basename(x)[len(prefix) :]) for x in version_folders]
+ )
+ if len(version_numbers) == 0:
+ return None
+ last_version = version_numbers[-1]
+ return last_version
+
+
+def get_eta_str(cur_iter, total_iter, time_per_iter):
+ eta = time_per_iter * (total_iter - cur_iter - 1)
+ return convert_sec_to_time(eta)
+
+
+def convert_sec_to_time(secs):
+ return str(datetime.timedelta(seconds=round(secs)))
+
+
+def concat_lists(list_of_lists):
+ return list(itertools.chain.from_iterable(list_of_lists))
+
+
+def find_consecutive_runs(x, min_len=1):
+ """Find runs of consecutive items in an array."""
+
+ # ensure array
+ x = np.asanyarray(x)
+ if x.ndim != 1:
+ raise ValueError("only 1D array supported")
+ n = x.shape[0]
+
+ # handle empty array
+ if n == 0:
+ return np.array([]), np.array([]), np.array([])
+
+ else:
+ # find run starts
+ loc_run_start = np.empty(n, dtype=bool)
+ loc_run_start[0] = True
+ np.not_equal(x[:-1], x[1:] - 1, out=loc_run_start[1:])
+ run_starts = np.nonzero(loc_run_start)[0]
+
+ # find run lengths
+ run_lengths = np.diff(np.append(run_starts, n))
+ ind = run_lengths >= min_len
+ run_starts = run_starts[ind]
+ run_lengths = run_lengths[ind]
+
+ # find run values
+ run_values = [
+ x[start : start + length] for start, length in zip(run_starts, run_lengths)
+ ]
+ # assert np.allclose(np.concatenate(run_values), x)
+
+ return run_values, run_starts, run_lengths
+
+
+def get_checkpoint_path(checkpoint_dir, cp, return_name=False):
+ if cp == "last": # use last epoch
+ cp_name = "last.ckpt"
+ elif cp == "best": # use best epoch
+ cp_name = osp.basename(sorted(glob.glob(f"{checkpoint_dir}/*best*.ckpt"))[-1])
+ else:
+ cp_name = osp.basename(sorted(glob.glob(f"{checkpoint_dir}/{cp}.ckpt"))[-1])
+ cp_path = f"{checkpoint_dir}/{cp_name}"
+ if return_name:
+ return cp_path, cp_name
+ return cp_path
+
+
+def subprocess_run(cmd, ignore_err=False, **kwargs):
+ try:
+ result = subprocess.run(cmd, **kwargs)
+ except subprocess.CalledProcessError as err:
+ print("####### subprocess-run error message ######")
+ print(f"{err} {err.stderr.decode('utf8')}")
+ if result.returncode != 0:
+ if not ignore_err:
+ raise Exception("error in subprocess_run!")
+ return result
+
+
+def import_type_from_str(s):
+ module_name, type_name = s.rsplit(".", 1)
+ module = importlib.import_module(module_name)
+ type_to_import = getattr(module, type_name)
+ return type_to_import
+
+
+def build_object_from_dict(d, type_field="type", **add_kwargs):
+ d = d.copy()
+ _type = import_type_from_str(d.pop(type_field))
+ return _type(**d, **add_kwargs)
+
+
+def write_list_to_file(filename, string_list):
+ with open(filename, "w") as file:
+ for item in string_list:
+ file.write(item + "\n")
+
+
+def are_arrays_equal(array1, array2, sort=False):
+ if array1 is None or array2 is None:
+ return False
+ # if array1 == array2:
+ # return True
+ if len(array1) != len(array2):
+ return False
+
+ # Sort both arrays
+ if sort:
+ array1 = sorted(array1)
+ array2 = sorted(array2)
+
+ # Compare each element
+ for i in range(len(array1)):
+ if array1[i] != array2[i]:
+ return False
+
+ return True
+
+
+def load_ema_weights_from_checkpoint(model, checkpoint):
+ ema_params = checkpoint["optimizer_states"][0]["ema"]
+ for param, ema_param in zip(model.parameters(), ema_params):
+ param.data.copy_(ema_param.data)
+ return
+
+
+def rsync_file_from_remote(fname, remote_dir, local_dir, hostname):
+ remote_fname = fname.replace(local_dir, f"{remote_dir}/./")
+ cmd = f"rsync -avzP -m --relative {hostname}:{remote_fname} {local_dir}/"
+ subprocess_run(cmd, shell=True)
+ return
+
+
+# Global variable for timing indentation level
+timer_indent_level = 0
+
+
+# Context manager for timing
+class Timer:
+ def __init__(self, name="", enabled=True, show_rank=False, rank_zero_only=True):
+ self.name = name
+ self.start_time = None
+ self.enabled = enabled
+ if "LOCAL_RANK" in os.environ:
+ self.rank = int(os.environ["LOCAL_RANK"])
+ else:
+ self.rank = 0
+ self.show_rank = show_rank
+ self.rank_zero_only = rank_zero_only
+
+ def __enter__(self):
+ if (not self.enabled) or (self.rank_zero_only and self.rank != 0):
+ return self
+ global timer_indent_level
+ self.start_time = time.perf_counter()
+ self.current_indent = timer_indent_level # Capture current indent level
+ timer_indent_level += 1 # Increment global indent level for next call
+ return self
+
+ def __exit__(self, exc_type, exc_val, exc_tb):
+ if exc_type:
+ return False # Re-raise the exception
+ if (not self.enabled) or (self.rank_zero_only and self.rank != 0):
+ return self
+ global timer_indent_level
+ elapsed_time = time.perf_counter() - self.start_time
+ indent = " " * self.current_indent # 4 spaces per indent level
+ rank_str = f"[rank{self.rank}] " if self.show_rank else ""
+ print(f"{indent}{rank_str}[{self.name}] time: {elapsed_time:.4f} seconds")
+ timer_indent_level -= 1 # Decrement global indent level after finishing
diff --git a/genmo/utils/torch_transform.py b/genmo/utils/torch_transform.py
new file mode 100644
index 0000000000000000000000000000000000000000..28bbf55dee87ff6849a4b7415dbcc243d353c5e1
--- /dev/null
+++ b/genmo/utils/torch_transform.py
@@ -0,0 +1,413 @@
+import numpy as np
+import torch
+
+if __name__ != "__main__":
+ from .konia_transform import (
+ angle_axis_to_quaternion,
+ angle_axis_to_rotation_matrix,
+ quaternion_to_angle_axis,
+ quaternion_to_rotation_matrix,
+ rotation_matrix_to_angle_axis,
+ rotation_matrix_to_quaternion,
+ )
+else:
+ from konia_transform import (
+ angle_axis_to_quaternion,
+ angle_axis_to_rotation_matrix,
+ quaternion_to_rotation_matrix,
+ rotation_matrix_to_angle_axis,
+ rotation_matrix_to_quaternion,
+ )
+
+
+def normalize(x, eps: float = 1e-9):
+ return x / x.norm(p=2, dim=-1).clamp(min=eps, max=None).unsqueeze(-1)
+
+
+@torch.jit.script
+def quat_mul(a, b):
+ assert a.shape == b.shape
+ shape = a.shape
+ a = a.reshape(-1, 4)
+ b = b.reshape(-1, 4)
+
+ w1, x1, y1, z1 = a[:, 0], a[:, 1], a[:, 2], a[:, 3]
+ w2, x2, y2, z2 = b[:, 0], b[:, 1], b[:, 2], b[:, 3]
+ ww = (z1 + x1) * (x2 + y2)
+ yy = (w1 - y1) * (w2 + z2)
+ zz = (w1 + y1) * (w2 - z2)
+ xx = ww + yy + zz
+ qq = 0.5 * (xx + (z1 - x1) * (x2 - y2))
+ w = qq - ww + (z1 - y1) * (y2 - z2)
+ x = qq - xx + (x1 + w1) * (x2 + w2)
+ y = qq - yy + (w1 - x1) * (y2 + z2)
+ z = qq - zz + (z1 + y1) * (w2 - x2)
+ return torch.stack([w, x, y, z], dim=-1).view(shape)
+
+
+@torch.jit.script
+def quat_conjugate(a):
+ shape = a.shape
+ a = a.reshape(-1, 4)
+ return torch.cat((a[:, 0:1], -a[:, 1:]), dim=-1).view(shape)
+
+
+@torch.jit.script
+def quat_apply(a, b):
+ shape = b.shape
+ a = a.reshape(-1, 4)
+ b = b.reshape(-1, 3)
+ xyz = a[:, 1:].clone()
+ t = xyz.cross(b, dim=-1) * 2
+ return (b + a[:, 0:1].clone() * t + xyz.cross(t, dim=-1)).view(shape)
+
+
+@torch.jit.script
+def quat_angle(a, eps: float = 1e-6):
+ shape = a.shape
+ a = a.reshape(-1, 4)
+ s = 2 * (a[:, 0] ** 2) - 1
+ s = s.clamp(-1 + eps, 1 - eps)
+ s = s.acos()
+ return s.view(shape[:-1])
+
+
+@torch.jit.script
+def quat_angle_diff(quat1, quat2):
+ return quat_angle(quat_mul(quat1, quat_conjugate(quat2)))
+
+
+@torch.jit.script
+def torch_safe_atan2(y, x, eps: float = 1e-8):
+ y = y.clone()
+ y[(y.abs() < eps) & (x.abs() < eps)] += eps
+ return torch.atan2(y, x)
+
+
+@torch.jit.script
+def ypr_euler_from_quat(
+ q, handle_singularity: bool = False, eps: float = 1e-6, singular_eps: float = 1e-6
+):
+ """
+ convert quaternion to yaw-pitch-roll euler angles
+ """
+ yaw_atany = 2 * (q[..., 0] * q[..., 3] + q[..., 1] * q[..., 2])
+ yaw_atanx = 1 - 2 * (q[..., 2] * q[..., 2] + q[..., 3] * q[..., 3])
+ roll_atany = 2 * (q[..., 0] * q[..., 1] + q[..., 2] * q[..., 3])
+ roll_atanx = 1 - 2 * (q[..., 1] * q[..., 1] + q[..., 2] * q[..., 2])
+ yaw = torch_safe_atan2(yaw_atany, yaw_atanx, eps)
+ pitch = torch.asin(
+ torch.clamp(
+ 2 * (q[..., 0] * q[..., 2] - q[..., 1] * q[..., 3]),
+ min=-1 + eps,
+ max=1 - eps,
+ )
+ )
+ roll = torch_safe_atan2(roll_atany, roll_atanx, eps)
+
+ if handle_singularity:
+ """ handle two special cases """
+ test = q[..., 0] * q[..., 2] - q[..., 1] * q[..., 3]
+ # north pole, pitch ~= 90 degrees
+ np_ind = test > 0.5 - singular_eps
+ if torch.any(np_ind):
+ # print('ypr_euler_from_quat singularity -- north pole!')
+ roll[np_ind] = 0.0
+ pitch[np_ind].clamp_max_(0.5 * np.pi)
+ yaw_atany = q[..., 3][np_ind]
+ yaw_atanx = q[..., 0][np_ind]
+ yaw[np_ind] = 2 * torch_safe_atan2(yaw_atany, yaw_atanx, eps)
+ # south pole, pitch ~= -90 degrees
+ sp_ind = test < -0.5 + singular_eps
+ if torch.any(sp_ind):
+ # print('ypr_euler_from_quat singularity -- south pole!')
+ roll[sp_ind] = 0.0
+ pitch[sp_ind].clamp_min_(-0.5 * np.pi)
+ yaw_atany = q[..., 3][sp_ind]
+ yaw_atanx = q[..., 0][sp_ind]
+ yaw[sp_ind] = 2 * torch_safe_atan2(yaw_atany, yaw_atanx, eps)
+
+ return torch.stack([roll, pitch, yaw], dim=-1)
+
+
+@torch.jit.script
+def quat_from_ypr_euler(angles):
+ """
+ convert yaw-pitch-roll euler angles to quaternion
+ """
+ half_ang = angles * 0.5
+ sin = torch.sin(half_ang)
+ cos = torch.cos(half_ang)
+ q = torch.stack(
+ [
+ cos[..., 0] * cos[..., 1] * cos[..., 2]
+ + sin[..., 0] * sin[..., 1] * sin[..., 2],
+ sin[..., 0] * cos[..., 1] * cos[..., 2]
+ - cos[..., 0] * sin[..., 1] * sin[..., 2],
+ cos[..., 0] * sin[..., 1] * cos[..., 2]
+ + sin[..., 0] * cos[..., 1] * sin[..., 2],
+ cos[..., 0] * cos[..., 1] * sin[..., 2]
+ - sin[..., 0] * sin[..., 1] * cos[..., 2],
+ ],
+ dim=-1,
+ )
+ return q
+
+
+def quat_between_two_vec(v1, v2, eps: float = 1e-6):
+ """
+ quaternion for rotating v1 to v2
+ """
+ orig_shape = v1.shape
+ v1 = v1.reshape(-1, 3)
+ v2 = v2.reshape(-1, 3)
+ dot = (v1 * v2).sum(-1)
+ cross = torch.cross(v1, v2, dim=-1)
+ out = torch.cat([(1 + dot).unsqueeze(-1), cross], dim=-1)
+ # handle v1 & v2 with same direction
+ sind = dot > 1 - eps
+ out[sind] = torch.tensor([1.0, 0.0, 0.0, 0.0], device=v1.device)
+ # handle v1 & v2 with opposite direction
+ nind = dot < -1 + eps
+ if torch.any(nind):
+ vx = torch.tensor([1.0, 0.0, 0.0], device=v1.device)
+ vxdot = (v1 * vx).sum(-1).abs()
+ nxind = nind & (vxdot < 1 - eps)
+ if torch.any(nxind):
+ out[nxind] = angle_axis_to_quaternion(
+ normalize(torch.cross(vx.expand_as(v1[nxind]), v1[nxind], dim=-1))
+ * np.pi
+ )
+ # handle v1 & v2 with opposite direction and they are parallel to x axis
+ pind = nind & (vxdot >= 1 - eps)
+ if torch.any(pind):
+ vy = torch.tensor([0.0, 1.0, 0.0], device=v1.device)
+ out[pind] = angle_axis_to_quaternion(
+ normalize(torch.cross(vy.expand_as(v1[pind]), v1[pind], dim=-1)) * np.pi
+ )
+ # normalize and reshape
+ out = normalize(out).view(orig_shape[:-1] + (4,))
+ return out
+
+
+@torch.jit.script
+def get_yaw(q, eps: float = 1e-6):
+ yaw_atany = 2 * (q[..., 0] * q[..., 3] + q[..., 1] * q[..., 2])
+ yaw_atanx = 1 - 2 * (q[..., 2] * q[..., 2] + q[..., 3] * q[..., 3])
+ yaw = torch_safe_atan2(yaw_atany, yaw_atanx, eps)
+ return yaw
+
+
+@torch.jit.script
+def get_yaw_q(q):
+ yaw = get_yaw(q)
+ angle_axis = torch.cat(
+ [torch.zeros(yaw.shape + (2,), device=q.device), yaw.unsqueeze(-1)], dim=-1
+ )
+ heading_q = angle_axis_to_quaternion(angle_axis)
+ return heading_q
+
+
+@torch.jit.script
+def get_heading(q, eps: float = 1e-6):
+ heading_atany = q[..., 3]
+ heading_atanx = q[..., 0]
+ heading = 2 * torch_safe_atan2(heading_atany, heading_atanx, eps)
+ return heading
+
+
+def get_heading_q(q):
+ q_new = q.clone()
+ q_new[..., 1] = 0
+ q_new[..., 2] = 0
+ q_new = normalize(q_new)
+ return q_new
+
+
+def get_y_heading_q(q):
+ q_new = q.clone()
+ q_new[..., 1] = 0
+ q_new[..., 3] = 0
+ q_new = normalize(q_new)
+ return q_new
+
+
+@torch.jit.script
+def heading_to_vec(h_theta):
+ v = torch.stack([torch.cos(h_theta), torch.sin(h_theta)], dim=-1)
+ return v
+
+
+@torch.jit.script
+def vec_to_heading(h_vec):
+ h_theta = torch_safe_atan2(h_vec[..., 1], h_vec[..., 0])
+ return h_theta
+
+
+@torch.jit.script
+def heading_to_quat(h_theta):
+ angle_axis = torch.cat(
+ [
+ torch.zeros(h_theta.shape + (2,), device=h_theta.device),
+ h_theta.unsqueeze(-1),
+ ],
+ dim=-1,
+ )
+ heading_q = angle_axis_to_quaternion(angle_axis)
+ return heading_q
+
+
+def deheading_quat(q, heading_q=None):
+ if heading_q is None:
+ heading_q = get_heading_q(q)
+ dq = quat_mul(quat_conjugate(heading_q), q)
+ return dq
+
+
+@torch.jit.script
+def rotmat_to_rot6d(mat):
+ rot6d = torch.cat([mat[..., 0], mat[..., 1]], dim=-1)
+ return rot6d
+
+
+@torch.jit.script
+def rot6d_to_rotmat(rot6d, eps: float = 1e-8):
+ a1 = rot6d[..., :3].clone()
+ a2 = rot6d[..., 3:].clone()
+ ind = torch.norm(a1, dim=-1) < eps
+ a1[ind] = torch.tensor([1.0, 0.0, 0.0], device=a1.device)
+ b1 = normalize(a1)
+
+ b2 = normalize(a2 - (b1 * a2).sum(dim=-1).unsqueeze(-1) * b1)
+ ind = torch.norm(b2, dim=-1) < eps
+ b2[ind] = torch.tensor([0.0, 1.0, 0.0], device=b2.device)
+
+ b3 = torch.cross(b1, b2, dim=-1)
+ mat = torch.stack([b1, b2, b3], dim=-1)
+ return mat
+
+
+@torch.jit.script
+def angle_axis_to_rot6d(aa):
+ return rotmat_to_rot6d(angle_axis_to_rotation_matrix(aa))
+
+
+@torch.jit.script
+def rot6d_to_angle_axis(rot6d):
+ return rotation_matrix_to_angle_axis(rot6d_to_rotmat(rot6d))
+
+
+@torch.jit.script
+def quat_to_rot6d(q):
+ return rotmat_to_rot6d(quaternion_to_rotation_matrix(q))
+
+
+@torch.jit.script
+def rot6d_to_quat(rot6d):
+ return rotation_matrix_to_quaternion(rot6d_to_rotmat(rot6d))
+
+
+@torch.jit.script
+def make_transform(rot, trans, rot_type: str = "rotmat"):
+ if rot_type == "axis_angle":
+ rot = angle_axis_to_rotation_matrix(rot)
+ elif rot_type == "6d":
+ rot = rot6d_to_rotmat(rot)
+ transform = torch.eye(4).to(trans.device).repeat(rot.shape[:-2] + (1, 1))
+ transform[..., :3, :3] = rot
+ transform[..., :3, 3] = trans
+ return transform
+
+
+@torch.jit.script
+def transform_trans(transform_mat, trans):
+ trans = torch.cat((trans, torch.ones_like(trans[..., :1])), dim=-1)[..., None, :]
+ while len(transform_mat.shape) < len(trans.shape):
+ transform_mat = transform_mat.unsqueeze(-3)
+ trans_new = torch.matmul(trans, transform_mat.transpose(-2, -1))[..., 0, :3]
+ return trans_new
+
+
+@torch.jit.script
+def transform_rot(transform_mat, rot):
+ rot_qmat = angle_axis_to_rotation_matrix(rot)
+ while len(transform_mat.shape) < len(rot_qmat.shape):
+ transform_mat = transform_mat.unsqueeze(-3)
+ rot_qmat_new = torch.matmul(transform_mat[..., :3, :3], rot_qmat)
+ rot_new = rotation_matrix_to_angle_axis(rot_qmat_new)
+ return rot_new
+
+
+@torch.jit.script
+def inverse_transform(transform_mat):
+ transform_inv = torch.zeros_like(transform_mat)
+ transform_inv[..., :3, :3] = transform_mat[..., :3, :3].transpose(-2, -1)
+ transform_inv[..., :3, 3] = -torch.matmul(
+ transform_mat[..., :3, 3].unsqueeze(-2), transform_mat[..., :3, :3]
+ ).squeeze(-2)
+ transform_inv[..., 3, 3] = 1.0
+ return transform_inv
+
+
+def batch_compute_similarity_transform_torch(S1, S2):
+ """
+ Computes a similarity transform (sR, t) that takes
+ a set of 3D points S1 (3 x N) closest to a set of 3D points S2,
+ where R is an 3x3 rotation matrix, t 3x1 translation, s scale.
+ i.e. solves the orthogonal Procrutes problem.
+ """
+ if len(S1.shape) > 3:
+ orig_shape = S1.shape
+ S1 = S1.reshape(-1, *S1.shape[-2:])
+ S2 = S2.reshape(-1, *S2.shape[-2:])
+ else:
+ orig_shape = None
+
+ transposed = False
+ if S1.shape[0] != 3 and S1.shape[0] != 2:
+ S1 = S1.permute(0, 2, 1)
+ S2 = S2.permute(0, 2, 1)
+ transposed = True
+ assert S2.shape[1] == S1.shape[1]
+
+ # 1. Remove mean.
+ mu1 = S1.mean(axis=-1, keepdims=True)
+ mu2 = S2.mean(axis=-1, keepdims=True)
+
+ X1 = S1 - mu1
+ X2 = S2 - mu2
+
+ # 2. Compute variance of X1 used for scale.
+ var1 = torch.sum(X1**2, dim=1).sum(dim=1)
+
+ # 3. The outer product of X1 and X2.
+ K = X1.bmm(X2.permute(0, 2, 1))
+
+ # 4. Solution that Maximizes trace(R'K) is R=U*V', where U, V are
+ # singular vectors of K.
+ U, s, V = torch.svd(K)
+
+ # Construct Z that fixes the orientation of R to get det(R)=1.
+ Z = torch.eye(U.shape[1], device=S1.device).unsqueeze(0)
+ Z = Z.repeat(U.shape[0], 1, 1)
+ Z[:, -1, -1] *= torch.sign(torch.det(U.bmm(V.permute(0, 2, 1))))
+
+ # Construct R.
+ R = V.bmm(Z.bmm(U.permute(0, 2, 1)))
+
+ # 5. Recover scale.
+ scale = torch.cat([torch.trace(x).unsqueeze(0) for x in R.bmm(K)]) / var1
+
+ # 6. Recover translation.
+ t = mu2 - (scale.unsqueeze(-1).unsqueeze(-1) * (R.bmm(mu1)))
+
+ # 7. Error:
+ S1_hat = scale.unsqueeze(-1).unsqueeze(-1) * R.bmm(S1) + t
+
+ if transposed:
+ S1_hat = S1_hat.permute(0, 2, 1)
+
+ if orig_shape is not None:
+ S1_hat = S1_hat.reshape(orig_shape)
+
+ return S1_hat
diff --git a/genmo/utils/video_io_utils.py b/genmo/utils/video_io_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..41e26de8ae26482a38b44749fe0d13eece7d7640
--- /dev/null
+++ b/genmo/utils/video_io_utils.py
@@ -0,0 +1,245 @@
+import os
+import shutil
+from pathlib import Path
+
+import cv2
+import ffmpeg
+import imageio
+import imageio.v3 as iio
+import numpy as np
+import torch
+from tqdm import tqdm
+
+
+def get_video_lwh(video_path):
+ L, H, W, _ = iio.improps(video_path, plugin="pyav").shape
+ return L, W, H
+
+
+def read_video_np(video_path, start_frame=0, end_frame=-1, scale=1.0):
+ """
+ Args:
+ video_path: str
+ Returns:
+ frames: np.array, (N, H, W, 3) RGB, uint8
+ """
+ # If video path not exists, an error will be raised by ffmpegs
+ filter_args = []
+ should_check_length = False
+
+ # 1. Trim
+ if not (start_frame == 0 and end_frame == -1):
+ if end_frame == -1:
+ filter_args.append(("trim", f"start_frame={start_frame}"))
+ else:
+ should_check_length = True
+ filter_args.append(
+ ("trim", f"start_frame={start_frame}:end_frame={end_frame}")
+ )
+
+ # 2. Scale
+ if scale != 1.0:
+ filter_args.append(("scale", f"iw*{scale}:ih*{scale}"))
+
+ # Excute then check
+ frames = iio.imread(video_path, plugin="pyav", filter_sequence=filter_args)
+ if should_check_length:
+ assert len(frames) == end_frame - start_frame
+
+ return frames
+
+
+def get_video_reader(video_path):
+ return iio.imiter(video_path, plugin="pyav")
+
+
+def read_images_np(image_paths, verbose=False):
+ """
+ Args:
+ image_paths: list of str
+ Returns:
+ images: np.array, (N, H, W, 3) RGB, uint8
+ """
+ if verbose:
+ images = [
+ cv2.imread(str(img_path))[..., ::-1] for img_path in tqdm(image_paths)
+ ]
+ else:
+ images = [cv2.imread(str(img_path))[..., ::-1] for img_path in image_paths]
+ images = np.stack(images, axis=0)
+ return images
+
+
+def save_video(images, video_path, fps=30, crf=17):
+ """
+ Args:
+ images: (N, H, W, 3) RGB, uint8
+ crf: 17 is visually lossless, 23 is default, +6 results in half the bitrate
+ 0 is lossless, https://trac.ffmpeg.org/wiki/Encode/H.264#crf
+ """
+ if isinstance(images, torch.Tensor):
+ images = images.cpu().numpy().astype(np.uint8)
+ elif isinstance(images, list):
+ images = np.array(images).astype(np.uint8)
+
+ with iio.imopen(video_path, "w", plugin="pyav") as writer:
+ writer.init_video_stream("libx264", fps=fps)
+ writer._video_stream.options = {"crf": str(crf)}
+ writer.write(images)
+
+
+class _CompatWriter:
+ def __init__(self, writer, use_append):
+ self._writer = writer
+ self._use_append = use_append
+
+ def write_frame(self, frame):
+ if self._use_append:
+ self._writer.append_data(frame)
+ else:
+ self._writer.write_frame(frame)
+
+ def close(self):
+ self._writer.close()
+
+
+def _open_pyav_writer(video_path, fps, crf):
+ writer = iio.imopen(video_path, "w", plugin="pyav")
+ writer.init_video_stream("libx264", fps=fps)
+ writer._video_stream.options = {"crf": str(crf)}
+ try:
+ time_base = writer._video_stream.codec_context.time_base
+ except Exception:
+ time_base = None
+ if time_base is None:
+ writer.close()
+ raise RuntimeError("pyav stream missing time_base")
+ return writer
+
+
+def get_writer(video_path, fps=30, crf=17):
+ """remember to .close()"""
+ try:
+ writer = _open_pyav_writer(video_path, fps, crf)
+ return _CompatWriter(writer, use_append=False)
+ except Exception:
+ # Fallback for environments where pyav fails to set time_base.
+ writer = imageio.get_writer(
+ video_path,
+ fps=fps,
+ format="FFMPEG",
+ mode="I",
+ codec="libx264",
+ macro_block_size=1,
+ ffmpeg_params=["-crf", str(crf)],
+ )
+ return _CompatWriter(writer, use_append=True)
+
+
+def copy_file(video_path, out_video_path, overwrite=True):
+ if not overwrite and Path(out_video_path).exists():
+ return
+ shutil.copy(video_path, out_video_path)
+
+
+def concat_videos(cfg, out_video_path: str, in_video_paths=None):
+ # if len(in_video_paths) < 2:
+ # raise ValueError("At least two video paths are required for merging.")
+ # in_video_paths = [cfg.video1_path, cfg.text1_video_path, cfg.video2_path]
+ if in_video_paths is None:
+ in_video_paths = [
+ cfg.paths.incam_video1,
+ cfg.text1_video_path,
+ cfg.paths.incam_video2,
+ ]
+
+ # Get the size of the first video to use as target size
+ probe = ffmpeg.probe(in_video_paths[0])
+ video_stream = next(
+ (stream for stream in probe["streams"] if stream["codec_type"] == "video"), None
+ )
+ target_size = (int(video_stream["width"]), int(video_stream["height"]))
+
+ # Resize and pad all videos to match the target size
+ temp_paths = [resize_and_pad_video(path, target_size) for path in in_video_paths]
+
+ try:
+ # Create inputs from the resized videos
+ inputs = [ffmpeg.input(path) for path in temp_paths]
+ merged_video = ffmpeg.concat(*inputs)
+ output = ffmpeg.output(merged_video, out_video_path)
+ ffmpeg.run(output, overwrite_output=True, quiet=True)
+ finally:
+ # Clean up temporary files
+ for path in temp_paths:
+ if os.path.exists(path):
+ os.unlink(path)
+
+
+def resize_and_pad_video(video_path, target_size):
+ """
+ Resize and pad a video to match the target size.
+
+ Args:
+ video_path: Path to the input video
+ target_size: Tuple of (width, height) for the target size
+
+ Returns:
+ Path to the resized and padded temporary video
+ """
+ import os
+ import tempfile
+
+ target_width, target_height = target_size
+
+ # Create a temporary file for the resized video
+ temp_file = tempfile.NamedTemporaryFile(suffix=".mp4", delete=False)
+ temp_path = temp_file.name
+ temp_file.close()
+
+ # Get video info
+ probe = ffmpeg.probe(video_path)
+ video_stream = next(
+ (stream for stream in probe["streams"] if stream["codec_type"] == "video"), None
+ )
+ width = int(video_stream["width"])
+ height = int(video_stream["height"])
+
+ # Calculate scaling to maintain aspect ratio
+ if width / height > target_width / target_height:
+ # Width is the limiting factor
+ scale_w = target_width
+ scale_h = -1 # Maintain aspect ratio
+ else:
+ # Height is the limiting factor
+ scale_w = -1 # Maintain aspect ratio
+ scale_h = target_height
+
+ # Resize and pad
+ stream = ffmpeg.input(video_path)
+ stream = ffmpeg.filter(stream, "scale", scale_w, scale_h)
+ stream = ffmpeg.filter(
+ stream, "pad", target_width, target_height, "(ow-iw)/2", "(oh-ih)/2"
+ )
+ stream = ffmpeg.output(stream, temp_path)
+ ffmpeg.run(stream, quiet=True, overwrite_output=True)
+
+ return temp_path
+
+
+def merge_videos_horizontal(in_video_paths: list, out_video_path: str):
+ if len(in_video_paths) < 2:
+ raise ValueError("At least two video paths are required for merging.")
+ inputs = [ffmpeg.input(path) for path in in_video_paths]
+ merged_video = ffmpeg.filter(inputs, "hstack", inputs=len(inputs))
+ output = ffmpeg.output(merged_video, out_video_path)
+ ffmpeg.run(output, overwrite_output=True, quiet=True)
+
+
+def merge_videos_vertical(in_video_paths: list, out_video_path: str):
+ if len(in_video_paths) < 2:
+ raise ValueError("At least two video paths are required for merging.")
+ inputs = [ffmpeg.input(path) for path in in_video_paths]
+ merged_video = ffmpeg.filter(inputs, "vstack", inputs=len(inputs))
+ output = ffmpeg.output(merged_video, out_video_path)
+ ffmpeg.run(output, overwrite_output=True, quiet=True)
diff --git a/genmo/utils/vis/cv2_utils.py b/genmo/utils/vis/cv2_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..e9923a8b5ed95e4b39b9aa4c759cc3c1180c9f2c
--- /dev/null
+++ b/genmo/utils/vis/cv2_utils.py
@@ -0,0 +1,160 @@
+import cv2
+import numpy as np
+
+from third_party.GVHMR.hmr4d.utils.wis3d_utils import get_colors_by_conf
+
+
+def to_numpy(x):
+ if isinstance(x, np.ndarray):
+ return x.copy()
+ elif isinstance(x, list):
+ return np.array(x)
+ return x.clone().cpu().numpy()
+
+
+def draw_bbx_xys_on_image(bbx_xys, image, conf=True):
+ assert isinstance(bbx_xys, np.ndarray)
+ assert isinstance(image, np.ndarray)
+ image = image.copy()
+ lu_point = (bbx_xys[:2] - bbx_xys[2:] / 2).astype(int)
+ rd_point = (bbx_xys[:2] + bbx_xys[2:] / 2).astype(int)
+ color = (255, 178, 102) if conf else (128, 128, 128) # orange or gray
+ image = cv2.rectangle(image, lu_point, rd_point, color, 2)
+ return image
+
+
+def draw_bbx_xys_on_image_batch(bbx_xys_batch, image_batch, conf=None):
+ """conf: if provided, list of bool"""
+ use_conf = conf is not None
+ bbx_xys_batch = to_numpy(bbx_xys_batch)
+ assert len(bbx_xys_batch) == len(image_batch)
+ image_batch_out = []
+ for i in range(len(bbx_xys_batch)):
+ if use_conf:
+ image_batch_out.append(
+ draw_bbx_xys_on_image(bbx_xys_batch[i], image_batch[i], conf[i])
+ )
+ else:
+ image_batch_out.append(
+ draw_bbx_xys_on_image(bbx_xys_batch[i], image_batch[i])
+ )
+ return image_batch_out
+
+
+def draw_bbx_xyxy_on_image(bbx_xys, image, conf=True):
+ bbx_xys = to_numpy(bbx_xys)
+ image = to_numpy(image)
+ color = (255, 178, 102) if conf else (128, 128, 128) # orange or gray
+ image = cv2.rectangle(
+ image,
+ (int(bbx_xys[0]), int(bbx_xys[1])),
+ (int(bbx_xys[2]), int(bbx_xys[3])),
+ color,
+ 2,
+ )
+ return image
+
+
+def draw_bbx_xyxy_on_image_batch(bbx_xyxy_batch, image_batch, mask=None, conf=None):
+ """
+ Args:
+ conf: if provided, list of bool, mutually exclusive with mask
+ mask: whether to draw, historically used
+ """
+ if mask is not None:
+ assert conf is None
+ if conf is not None:
+ assert mask is None
+ use_conf = conf is not None
+ bbx_xyxy_batch = to_numpy(bbx_xyxy_batch)
+ image_batch = to_numpy(image_batch)
+ assert len(bbx_xyxy_batch) == len(image_batch)
+ image_batch_out = []
+ for i in range(len(bbx_xyxy_batch)):
+ if use_conf:
+ image_batch_out.append(
+ draw_bbx_xyxy_on_image(bbx_xyxy_batch[i], image_batch[i], conf[i])
+ )
+ else:
+ if mask is None or mask[i]:
+ image_batch_out.append(
+ draw_bbx_xyxy_on_image(bbx_xyxy_batch[i], image_batch[i])
+ )
+ else:
+ image_batch_out.append(image_batch[i])
+ return image_batch_out
+
+
+def draw_kpts(frame, keypoints, color=(0, 255, 0), thickness=2):
+ frame_ = frame.copy()
+ for x, y in keypoints:
+ cv2.circle(frame_, (int(x), int(y)), thickness, color, -1)
+ return frame_
+
+
+def draw_kpts_with_conf(frame, kp2d, conf, thickness=2):
+ """
+ Args:
+ kp2d: (J, 2),
+ conf: (J,)
+ """
+ frame_ = frame.copy()
+ conf = conf.reshape(-1)
+ colors = get_colors_by_conf(conf) # (J, 3)
+ colors = colors[:, [2, 1, 0]].int().numpy().tolist()
+ for j in range(kp2d.shape[0]):
+ x, y = kp2d[j, :2]
+ c = colors[j]
+ cv2.circle(frame_, (int(x), int(y)), thickness, c, -1)
+ return frame_
+
+
+def draw_kpts_with_conf_batch(frames, kp2d_batch, conf_batch, thickness=2):
+ """
+ Args:
+ kp2d_batch: (B, J, 2),
+ conf_batch: (B, J)
+ """
+ assert len(frames) == len(kp2d_batch)
+ assert len(frames) == len(conf_batch)
+ frames_ = []
+ for i in range(len(frames)):
+ frames_.append(
+ draw_kpts_with_conf(frames[i], kp2d_batch[i], conf_batch[i], thickness)
+ )
+ return frames_
+
+
+def draw_coco17_skeleton(img, keypoints, conf_thr=0):
+ use_conf_thr = True if keypoints.shape[1] == 3 else False
+ img = img.copy()
+ # fmt:off
+ coco_skel = [[15, 13], [13, 11], [16, 14], [14, 12], [11, 12], [5, 11], [6, 12], [5, 6], [5, 7], [6, 8], [7, 9], [8, 10], [1, 2], [0, 1], [0, 2], [1, 3], [2, 4], [3, 5], [4, 6]]
+ # fmt:on
+ for bone in coco_skel:
+ if use_conf_thr:
+ kp1 = keypoints[bone[0]][:2].astype(int)
+ kp2 = keypoints[bone[1]][:2].astype(int)
+ kp1_c = keypoints[bone[0]][2]
+ kp2_c = keypoints[bone[1]][2]
+ if kp1_c > conf_thr and kp2_c > conf_thr:
+ img = cv2.line(img, (kp1[0], kp1[1]), (kp2[0], kp2[1]), (0, 255, 0), 4)
+ if kp1_c > conf_thr:
+ img = cv2.circle(img, (kp1[0], kp1[1]), 6, (0, 255, 0), -1)
+ if kp2_c > conf_thr:
+ img = cv2.circle(img, (kp2[0], kp2[1]), 6, (0, 255, 0), -1)
+
+ else:
+ kp1 = keypoints[bone[0]][:2].astype(int)
+ kp2 = keypoints[bone[1]][:2].astype(int)
+ img = cv2.line(img, (kp1[0], kp1[1]), (kp2[0], kp2[1]), (0, 255, 0), 4)
+ return img
+
+
+def draw_coco17_skeleton_batch(imgs, keypoints_batch, conf_thr=0):
+ assert len(imgs) == len(keypoints_batch)
+ keypoints_batch = to_numpy(keypoints_batch)
+ imgs_out = []
+ for i in range(len(imgs)):
+ imgs_out.append(draw_coco17_skeleton(imgs[i], keypoints_batch[i], conf_thr))
+ return imgs_out
diff --git a/genmo/utils/vis/o3d_render.py b/genmo/utils/vis/o3d_render.py
new file mode 100644
index 0000000000000000000000000000000000000000..08f0b5604bdfc90a843feac3280e1723434c4e89
--- /dev/null
+++ b/genmo/utils/vis/o3d_render.py
@@ -0,0 +1,241 @@
+import open3d as o3d
+import open3d.visualization.gui as gui
+import open3d.visualization.rendering as rendering
+import torch
+
+from .renderer_tools import checkerboard_geometry
+
+
+class Settings:
+ UNLIT = "defaultUnlit"
+ LIT = "defaultLit"
+ NORMALS = "normals"
+ DEPTH = "depth"
+ LINE = "unlitLine"
+ Transparency = "defaultLitTransparency"
+ LitSSR = "defaultLitSSR"
+
+ DEFAULT_PROFILE_NAME = "Bright day with sun at +Y [default]"
+ POINT_CLOUD_PROFILE_NAME = "Cloudy day (no direct sun)"
+ CUSTOM_PROFILE_NAME = "Custom"
+ LIGHTING_PROFILES = {
+ DEFAULT_PROFILE_NAME: {
+ "ibl_intensity": 45000,
+ "sun_intensity": 45000,
+ "sun_dir": [0.577, -0.577, -0.577],
+ # "ibl_rotation":
+ "use_ibl": True,
+ "use_sun": True,
+ },
+ "Bright day with sun at -Y": {
+ "ibl_intensity": 45000,
+ "sun_intensity": 45000,
+ "sun_dir": [0.577, 0.577, 0.577],
+ # "ibl_rotation":[]
+ "use_ibl": True,
+ "use_sun": True,
+ },
+ "Bright day with sun at +Z": {
+ "ibl_intensity": 45000,
+ "sun_intensity": 45000,
+ "sun_dir": [0.577, 0.577, -0.577],
+ # "ibl_rotation":
+ "use_ibl": True,
+ "use_sun": True,
+ },
+ "Less Bright day with sun at +Y": {
+ "ibl_intensity": 35000,
+ "sun_intensity": 50000,
+ "sun_dir": [0.577, -0.577, -0.577],
+ # "ibl_rotation":
+ "use_ibl": True,
+ "use_sun": True,
+ },
+ "Less Bright day with sun at -Y": {
+ "ibl_intensity": 35000,
+ "sun_intensity": 50000,
+ "sun_dir": [0.577, 0.577, 0.577],
+ # "ibl_rotation":
+ "use_ibl": True,
+ "use_sun": True,
+ },
+ "Less Bright day with sun at +Z": {
+ "ibl_intensity": 35000,
+ "sun_intensity": 50000,
+ "sun_dir": [0.577, 0.577, -0.577],
+ # "ibl_rotation":
+ "use_ibl": True,
+ "use_sun": True,
+ },
+ POINT_CLOUD_PROFILE_NAME: {
+ "ibl_intensity": 60000,
+ "sun_intensity": 50000,
+ "use_ibl": True,
+ "use_sun": False,
+ # "ibl_rotation":
+ },
+ }
+
+ DEFAULT_MATERIAL_NAME = "Polished ceramic [default]"
+ PREFAB = {
+ DEFAULT_MATERIAL_NAME: {
+ "metallic": 0.0,
+ "roughness": 0.7,
+ "reflectance": 0.5,
+ "clearcoat": 0.2,
+ "clearcoat_roughness": 0.2,
+ "anisotropy": 0.0,
+ },
+ "Metal (rougher)": {
+ "metallic": 1.0,
+ "roughness": 0.5,
+ "reflectance": 0.9,
+ "clearcoat": 0.0,
+ "clearcoat_roughness": 0.0,
+ "anisotropy": 0.0,
+ },
+ "Metal (smoother)": {
+ "metallic": 1.0,
+ "roughness": 0.3,
+ "reflectance": 0.9,
+ "clearcoat": 0.0,
+ "clearcoat_roughness": 0.0,
+ "anisotropy": 0.0,
+ },
+ "Plastic": {
+ "metallic": 0.0,
+ "roughness": 0.5,
+ "reflectance": 0.5,
+ "clearcoat": 0.5,
+ "clearcoat_roughness": 0.2,
+ "anisotropy": 0.0,
+ },
+ "Glazed ceramic": {
+ "metallic": 0.0,
+ "roughness": 0.5,
+ "reflectance": 0.9,
+ "clearcoat": 1.0,
+ "clearcoat_roughness": 0.1,
+ "anisotropy": 0.0,
+ },
+ "Clay": {
+ "metallic": 0.0,
+ "roughness": 1.0,
+ "reflectance": 0.5,
+ "clearcoat": 0.1,
+ "clearcoat_roughness": 0.287,
+ "anisotropy": 0.0,
+ },
+ "Transparency": {
+ "metallic": 0.0,
+ "roughness": 0.0,
+ "reflectance": 0.0,
+ "clearcoat": 1.0,
+ "clearcoat_roughness": 0.0,
+ "anisotropy": 0.0,
+ },
+ }
+
+ def __init__(self):
+ # self.mouse_model = gui.SceneWidget.Controls.ROTATE_CAMERA
+ self.prefab = Settings.DEFAULT_MATERIAL_NAME
+ self.bg_color = gui.Color(1, 1, 1)
+ self.show_skybox = True
+ self.show_ground_plane = True
+ self.show_axes = False
+ self.use_ibl = True
+ self.use_sun = True
+ self.new_ibl_name = None # clear to None after loading
+ self.ibl_intensity = 45000
+ self.sun_intensity = 45000
+ self.sun_dir = [0.577, -0.577, -0.577]
+ self.sun_color = gui.Color(1, 1, 1)
+
+ self.apply_material = True # clear to False after processing
+ self._materials = {
+ Settings.LIT: rendering.MaterialRecord(),
+ Settings.UNLIT: rendering.MaterialRecord(),
+ Settings.NORMALS: rendering.MaterialRecord(),
+ Settings.DEPTH: rendering.MaterialRecord(),
+ Settings.LINE: rendering.MaterialRecord(),
+ Settings.Transparency: rendering.MaterialRecord(),
+ Settings.LitSSR: rendering.MaterialRecord(),
+ }
+ self._materials[Settings.LIT].base_color = [0.9, 0.9, 0.9, 1.0]
+ self._materials[Settings.LIT].shader = Settings.LIT
+ self._materials[Settings.UNLIT].base_color = [0.9, 0.9, 0.9, 1.0]
+ self._materials[Settings.UNLIT].shader = Settings.UNLIT
+ self._materials[Settings.LINE].base_color = [0.9, 0.9, 0.9, 1.0]
+ self._materials[Settings.LINE].shader = Settings.LINE
+ self._materials[Settings.LINE].line_width = 3
+ self._materials[Settings.Transparency].base_color = [0.467, 0.467, 0.467, 0.2]
+ self._materials[Settings.Transparency].base_color = [0.9, 0.9, 0.9, 0.5]
+ self._materials[Settings.Transparency].shader = Settings.Transparency
+ self._materials[Settings.Transparency].thickness = 1.0
+ self._materials[Settings.Transparency].transmission = 1.0
+ self._materials[Settings.Transparency].absorption_distance = 10
+ self._materials[Settings.Transparency].absorption_color = [0.5, 0.5, 0.5]
+ self._materials[Settings.LitSSR].base_color = [0.467, 0.467, 0.467, 0.2]
+ self._materials[Settings.LitSSR].shader = Settings.LitSSR
+ self._materials[Settings.LitSSR].thickness = 1.0
+ self._materials[Settings.LitSSR].transmission = 1.0
+ self._materials[Settings.LitSSR].absorption_distance = 10
+ self._materials[Settings.LitSSR].absorption_color = [0.5, 0.5, 0.5]
+ self._materials[Settings.NORMALS].shader = Settings.NORMALS
+ self._materials[Settings.DEPTH].shader = Settings.DEPTH
+
+ # Conveniently, assigning from self._materials[...] assigns a reference,
+ # not a copy, so if we change the property of a material, then switch
+ # to another one, then come back, the old setting will still be there.
+ self.material = self._materials[Settings.LIT]
+
+ def set_material(self, name):
+ self.material = self._materials[name]
+ self.apply_material = True
+
+ def apply_material_prefab(self, name):
+ # assert (self.material.shader == Settings.LIT)
+ self.prefab = name
+ for key, val in Settings.PREFAB[name].items():
+ setattr(self.material, "base_" + key, val)
+
+ def apply_lighting_profile(self, name):
+ """
+ It takes a string as an argument, and then sets the attributes of the object to the values in the
+ dictionary that is associated with that string
+
+ Args:
+ name: The name of the lighting profile.
+ """
+ profile = Settings.LIGHTING_PROFILES[name]
+ for key, val in profile.items():
+ setattr(self, key, val)
+
+
+def get_ground(length, center_x, center_z):
+ length, center_x, center_z = map(float, (length, center_x, center_z))
+ v, f, vc, fc = map(
+ torch.from_numpy,
+ checkerboard_geometry(length=length, c1=center_x, c2=center_z, up="y"),
+ )
+ # v, f, vc = v.to(device), f.to(device), vc.to(device)
+ ground_geometry = [v, f, vc]
+ return ground_geometry
+
+
+def create_meshes(verts, faces, colors):
+ """
+ :param verts (B, V, 3)
+ :param faces (B, F, 3)
+ :param colors (B, V, 3)
+ """
+ mesh = o3d.geometry.TriangleMesh(
+ vertices=o3d.utility.Vector3dVector(verts.cpu().numpy()),
+ triangles=o3d.utility.Vector3iVector(faces.cpu().numpy()),
+ )
+ mesh.compute_vertex_normals()
+ if len(colors.shape) == 1:
+ colors = colors[None, :].repeat(len(verts), 1)
+ mesh.vertex_colors = o3d.utility.Vector3dVector(colors.cpu().numpy())
+
+ return mesh
diff --git a/genmo/utils/vis/renderer.py b/genmo/utils/vis/renderer.py
new file mode 100644
index 0000000000000000000000000000000000000000..61f661420db288e6d541786d5fc37c40e537e0e7
--- /dev/null
+++ b/genmo/utils/vis/renderer.py
@@ -0,0 +1,568 @@
+import numpy as np
+import torch
+from PIL import Image
+from pytorch3d.structures import Meshes
+from pytorch3d.structures.meshes import join_meshes_as_scene
+
+from genmo.utils.rotation_conversions import axis_angle_to_matrix
+from genmo.utils.vis.renderer_tools import checkerboard_geometry
+
+try:
+ from pytorch3d.renderer import (
+ Materials,
+ MeshRasterizer,
+ MeshRenderer,
+ PerspectiveCameras,
+ PointLights,
+ RasterizationSettings,
+ SoftPhongShader,
+ TexturesVertex,
+ )
+ from pytorch3d.renderer.cameras import look_at_rotation
+except ImportError:
+ print("pytorch3d 3d renderer not loaded!")
+
+
+colors_str_map = {
+ "gray": [0.8, 0.8, 0.8],
+ "green": [39, 194, 128],
+}
+
+
+def overlay_image_onto_background(image, mask, bbox, background):
+ if isinstance(image, torch.Tensor):
+ image = image.detach().cpu().numpy()
+ if isinstance(mask, torch.Tensor):
+ mask = mask.detach().cpu().numpy()
+
+ out_image = background.copy()
+ bbox = bbox[0].int().cpu().numpy().copy()
+ roi_image = out_image[bbox[1] : bbox[3], bbox[0] : bbox[2]]
+
+ roi_image[mask] = image[mask]
+ out_image[bbox[1] : bbox[3], bbox[0] : bbox[2]] = roi_image
+
+ return out_image
+
+
+def update_intrinsics_from_bbox(K_org, bbox):
+ device, dtype = K_org.device, K_org.dtype
+
+ K = torch.zeros((K_org.shape[0], 4, 4)).to(device=device, dtype=dtype)
+ K[:, :3, :3] = K_org.clone()
+ K[:, 2, 2] = 0
+ K[:, 2, -1] = 1
+ K[:, -1, 2] = 1
+
+ image_sizes = []
+ for idx, bbox in enumerate(bbox):
+ left, upper, right, lower = bbox
+ cx, cy = K[idx, 0, 2], K[idx, 1, 2]
+
+ new_cx = cx - left
+ new_cy = cy - upper
+ new_height = max(lower - upper, 1)
+ new_width = max(right - left, 1)
+ new_cx = new_width - new_cx
+ new_cy = new_height - new_cy
+
+ K[idx, 0, 2] = new_cx
+ K[idx, 1, 2] = new_cy
+ image_sizes.append((int(new_height), int(new_width)))
+
+ return K, image_sizes
+
+
+def perspective_projection(x3d, K, R=None, T=None):
+ if R is not None:
+ x3d = torch.matmul(R, x3d.transpose(1, 2)).transpose(1, 2)
+ if T is not None:
+ x3d = x3d + T.transpose(1, 2)
+
+ x2d = torch.div(x3d, x3d[..., 2:])
+ x2d = torch.matmul(K, x2d.transpose(-1, -2)).transpose(-1, -2)[..., :2]
+ return x2d
+
+
+def compute_bbox_from_points(X, img_w, img_h, scaleFactor=1.2):
+ left = torch.clamp(X.min(1)[0][:, 0], min=0, max=img_w)
+ right = torch.clamp(X.max(1)[0][:, 0], min=0, max=img_w)
+ top = torch.clamp(X.min(1)[0][:, 1], min=0, max=img_h)
+ bottom = torch.clamp(X.max(1)[0][:, 1], min=0, max=img_h)
+
+ cx = (left + right) / 2
+ cy = (top + bottom) / 2
+ width = right - left
+ height = bottom - top
+
+ new_left = torch.clamp(cx - width / 2 * scaleFactor, min=0, max=img_w - 1)
+ new_right = torch.clamp(cx + width / 2 * scaleFactor, min=1, max=img_w)
+ new_top = torch.clamp(cy - height / 2 * scaleFactor, min=0, max=img_h - 1)
+ new_bottom = torch.clamp(cy + height / 2 * scaleFactor, min=1, max=img_h)
+
+ bbox = (
+ torch.stack(
+ (
+ new_left.detach(),
+ new_top.detach(),
+ new_right.detach(),
+ new_bottom.detach(),
+ )
+ )
+ .int()
+ .float()
+ .T
+ )
+
+ return bbox
+
+
+class Renderer:
+ def __init__(
+ self,
+ width,
+ height,
+ focal_length=None,
+ device="cuda",
+ faces=None,
+ K=None,
+ bin_size=None,
+ max_faces_per_bin=None,
+ max_points_per_bin=None,
+ ):
+ """set bin_size to 0 for no binning"""
+ self.width = width
+ self.height = height
+ self.bin_size = bin_size
+ self.max_faces_per_bin = max_faces_per_bin
+ self.max_points_per_bin = max_points_per_bin
+ assert (focal_length is not None) ^ (K is not None), (
+ "focal_length and K are mutually exclusive"
+ )
+
+ self.device = device
+ if faces is not None:
+ if isinstance(faces, np.ndarray):
+ faces = torch.from_numpy((faces).astype("int"))
+ self.faces = faces.unsqueeze(0).to(self.device)
+
+ self.initialize_camera_params(focal_length, K)
+ self.lights = PointLights(device=device, location=[[0.0, 0.0, -10.0]])
+ self.create_renderer()
+
+ def create_renderer(self):
+ raster_kwargs = dict(
+ image_size=self.image_sizes[0],
+ blur_radius=1e-5,
+ bin_size=self.bin_size,
+ )
+ if self.max_faces_per_bin is not None:
+ raster_kwargs["max_faces_per_bin"] = self.max_faces_per_bin
+ if self.max_points_per_bin is not None:
+ raster_kwargs["max_points_per_bin"] = self.max_points_per_bin
+
+ # PyTorch3D has changed RasterizationSettings kwargs across versions.
+ # Try the most capable signature first, then gracefully drop unsupported args.
+ raster_settings = None
+ for key_to_drop in (None, "max_points_per_bin", "max_faces_per_bin"):
+ try_kwargs = dict(raster_kwargs)
+ if key_to_drop is not None:
+ try_kwargs.pop(key_to_drop, None)
+ try:
+ raster_settings = RasterizationSettings(**try_kwargs)
+ break
+ except TypeError:
+ continue
+ if raster_settings is None:
+ raster_settings = RasterizationSettings(
+ image_size=self.image_sizes[0],
+ blur_radius=1e-5,
+ bin_size=self.bin_size,
+ )
+
+ self.renderer = MeshRenderer(
+ rasterizer=MeshRasterizer(
+ raster_settings=raster_settings,
+ ),
+ shader=SoftPhongShader(
+ device=self.device,
+ lights=self.lights,
+ ),
+ )
+
+ def create_camera(self, R=None, T=None):
+ if R is not None:
+ self.R = R.clone().view(1, 3, 3).to(self.device)
+ if T is not None:
+ self.T = T.clone().view(1, 3).to(self.device)
+
+ return PerspectiveCameras(
+ device=self.device,
+ R=self.R.mT,
+ T=self.T,
+ K=self.K_full,
+ image_size=self.image_sizes,
+ in_ndc=False,
+ )
+
+ def initialize_camera_params(self, focal_length, K):
+ # Extrinsics
+ self.R = (
+ torch.diag(torch.tensor([1, 1, 1])).float().to(self.device).unsqueeze(0)
+ )
+
+ self.T = torch.tensor([0, 0, 0]).unsqueeze(0).float().to(self.device)
+
+ # Intrinsics
+ if K is not None:
+ self.K = K.float().reshape(1, 3, 3).to(self.device)
+ else:
+ assert focal_length is not None, "focal_length or K should be provided"
+ self.K = (
+ torch.tensor(
+ [
+ [focal_length, 0, self.width / 2],
+ [0, focal_length, self.height / 2],
+ [0, 0, 1],
+ ]
+ )
+ .float()
+ .reshape(1, 3, 3)
+ .to(self.device)
+ )
+ self.bboxes = torch.tensor([[0, 0, self.width, self.height]]).float()
+ self.K_full, self.image_sizes = update_intrinsics_from_bbox(self.K, self.bboxes)
+ self.cameras = self.create_camera()
+
+ def set_intrinsic(self, K):
+ self.K = K.reshape(1, 3, 3)
+
+ def set_ground(self, length, center_x, center_z):
+ device = self.device
+ length, center_x, center_z = map(float, (length, center_x, center_z))
+ v, f, vc, fc = map(
+ torch.from_numpy,
+ checkerboard_geometry(length=length, c1=center_x, c2=center_z, up="y"),
+ )
+ v, f, vc = v.to(device), f.to(device), vc.to(device)
+ self.ground_geometry = [v, f, vc]
+
+ def update_bbox(self, x3d, scale=2.0, mask=None):
+ """Update bbox of cameras from the given 3d points
+
+ x3d: input 3D keypoints (or vertices), (num_frames, num_points, 3)
+ """
+
+ if x3d.size(-1) != 3:
+ x2d = x3d.unsqueeze(0)
+ else:
+ x2d = perspective_projection(
+ x3d.unsqueeze(0), self.K, self.R, self.T.reshape(1, 3, 1)
+ )
+
+ if mask is not None:
+ x2d = x2d[:, ~mask]
+
+ bbox = compute_bbox_from_points(x2d, self.width, self.height, scale)
+ self.bboxes = bbox
+
+ self.K_full, self.image_sizes = update_intrinsics_from_bbox(self.K, bbox)
+ self.cameras = self.create_camera()
+ self.create_renderer()
+
+ def reset_bbox(
+ self,
+ ):
+ bbox = torch.zeros((1, 4)).float().to(self.device)
+ bbox[0, 2] = self.width
+ bbox[0, 3] = self.height
+ self.bboxes = bbox
+
+ self.K_full, self.image_sizes = update_intrinsics_from_bbox(self.K, bbox)
+ self.cameras = self.create_camera()
+ self.create_renderer()
+
+ def render_mesh(self, vertices, background=None, colors=[0.8, 0.8, 0.8], VI=50):
+ self.update_bbox(vertices[::VI], scale=1.2)
+ vertices = vertices.unsqueeze(0)
+
+ if isinstance(colors, torch.Tensor):
+ # per-vertex color
+ verts_features = colors.to(device=vertices.device, dtype=vertices.dtype)
+ colors = [0.8, 0.8, 0.8]
+ else:
+ # Accept either [0..1] floats or [0..255] uint8-like colors.
+ # Don't key off `colors[0]` because valid RGB like green [0,255,0] would fail.
+ try:
+ if max(colors) > 1:
+ colors = [c / 255.0 for c in colors]
+ except Exception:
+ pass
+ verts_features = (
+ torch.tensor(colors)
+ .reshape(1, 1, 3)
+ .to(device=vertices.device, dtype=vertices.dtype)
+ )
+ verts_features = verts_features.repeat(1, vertices.shape[1], 1)
+ textures = TexturesVertex(verts_features=verts_features)
+
+ mesh = Meshes(
+ verts=vertices,
+ faces=self.faces,
+ textures=textures,
+ )
+
+ materials = Materials(device=self.device, specular_color=(colors,), shininess=0)
+
+ results = torch.flip(
+ self.renderer(
+ mesh, materials=materials, cameras=self.cameras, lights=self.lights
+ ),
+ [1, 2],
+ )
+ image = results[0, ..., :3] * 255
+ mask = results[0, ..., -1] > 1e-3
+
+ if background is None:
+ background = np.ones((self.height, self.width, 3)).astype(np.uint8) * 255
+
+ image = overlay_image_onto_background(
+ image, mask, self.bboxes, background.copy()
+ )
+ self.reset_bbox()
+ return image
+
+ def render_with_ground(
+ self, verts, colors, cameras, lights, faces=None, opacity=1.0
+ ):
+ """
+ :param verts (N, V, 3), potential multiple people
+ :param colors (N, 3) or (N, V, 3)
+ :param faces (N, F, 3), optional, otherwise self.faces is used will be used
+ """
+ # Sanity check of input verts, colors and faces: (B, V, 3), (B, F, 3), (B, V, 3)
+ N, V, _ = verts.shape
+ if faces is None:
+ faces = self.faces.clone().expand(N, -1, -1)
+ else:
+ assert len(faces.shape) == 3, "faces should have shape of (N, F, 3)"
+
+ assert len(colors.shape) in [2, 3]
+ if len(colors.shape) == 2:
+ assert len(colors) == N, "colors of shape 2 should be (N, 3)"
+ colors = colors[:, None]
+ colors = colors.expand(N, V, -1)[..., :3]
+
+ # (V, 3), (F, 3), (V, 3)
+ gv, gf, gc = self.ground_geometry
+ verts = list(torch.unbind(verts, dim=0)) + [gv]
+ faces = list(torch.unbind(faces, dim=0)) + [gf]
+ colors = list(torch.unbind(colors, dim=0)) + [gc[..., :3]]
+ mesh = create_meshes(verts, faces, colors)
+
+ materials = Materials(device=self.device, shininess=0)
+
+ results = self.renderer(
+ mesh, cameras=cameras, lights=lights, materials=materials
+ )
+ image = (results[0, ..., :3].cpu().numpy() * 255).astype(np.uint8)
+
+ return image
+
+ def render_with_ground_timeline(
+ self, verts_list, colors, cameras, lights, faces=None
+ ):
+ """
+ :param verts (N, V, 3), potential multiple people
+ :param colors (N, 3) or (N, V, 3)
+ :param faces (N, F, 3), optional, otherwise self.faces is used will be used
+ """
+ # Sanity check of input verts, colors and faces: (B, V, 3), (B, F, 3), (B, V, 3)
+ N, V, _ = verts_list[0].shape
+ if faces is None:
+ faces = self.faces.clone().expand(N, -1, -1)
+ else:
+ assert len(faces.shape) == 3, "faces should have shape of (N, F, 3)"
+ final_img = Image.new("RGBA", (self.width, self.height))
+ t_weights = torch.tensor([t / len(verts_list) for t in range(len(verts_list))])
+ # t_weights = (t_weights) / torch.sum(t_weights)
+ import ipdb
+
+ ipdb.set_trace()
+ torch.save(
+ {
+ "verts_list": verts_list,
+ "colors": colors,
+ "cameras": cameras,
+ "lights": lights,
+ "faces": faces,
+ "ground_geometry": self.ground_geometry,
+ },
+ "tmp.pth",
+ )
+ for t, verts in enumerate(verts_list):
+ N, V, _ = verts.shape
+
+ assert len(colors.shape) in [2, 3]
+ if len(colors.shape) == 2:
+ assert len(colors) == N, "colors of shape 2 should be (N, 3)"
+ colors = colors[:, None]
+ colors = colors.expand(N, V, -1)[..., :3]
+
+ # (V, 3), (F, 3), (V, 3)
+ gv, gf, gc = self.ground_geometry
+ verts = list(torch.unbind(verts, dim=0)) + [gv]
+ faces_list = list(torch.unbind(faces, dim=0)) + [gf]
+ colors_list = list(torch.unbind(colors, dim=0)) + [gc[..., :3]]
+ mesh = create_meshes(verts, faces_list, colors_list)
+
+ materials = Materials(device=self.device, shininess=0)
+ results = self.renderer(
+ mesh, cameras=cameras, lights=lights, materials=materials
+ )
+ # image = (results[0, ..., :3].cpu().numpy() * 255).astype(np.uint8)
+ image = results[0, ..., :4].cpu().numpy() * 255
+ image[..., 3] *= int(t_weights[t].item() * 255)
+ image = image.astype(np.uint8)
+ image = Image.fromarray(image, "RGBA")
+ # image.putalpha(int(t_weights[t].item() * 255))
+ final_img = Image.alpha_composite(final_img, image)
+ # tmp_list.append(image)
+ return final_img
+
+
+def create_meshes(verts, faces, colors):
+ """
+ :param verts (B, V, 3)
+ :param faces (B, F, 3)
+ :param colors (B, V, 3)
+ """
+ textures = TexturesVertex(verts_features=colors)
+ meshes = Meshes(verts=verts, faces=faces, textures=textures)
+ return join_meshes_as_scene(meshes)
+
+
+def get_global_cameras(verts, device="cuda", distance=5, position=(-5.0, 5.0, 0.0)):
+ """This always put object at the center of view"""
+ positions = torch.tensor([position]).repeat(len(verts), 1)
+ targets = verts.mean(1)
+
+ directions = targets - positions
+ directions = directions / torch.norm(directions, dim=-1).unsqueeze(-1) * distance
+ positions = targets - directions
+
+ rotation = look_at_rotation(positions, targets).mT
+ translation = -(rotation @ positions.unsqueeze(-1)).squeeze(-1)
+
+ lights = PointLights(device=device, location=[position])
+ return rotation, translation, lights
+
+
+def get_global_cameras_static(
+ verts,
+ beta=4.0,
+ cam_height_degree=30,
+ target_center_height=1.0,
+ use_long_axis=False,
+ vec_rot=45,
+ device="cuda",
+):
+ L, V, _ = verts.shape
+
+ # Compute target trajectory, denote as center + scale
+ targets = verts.mean(1) # (L, 3)
+ targets[:, 1] = 0 # project to xz-plane
+ target_center = targets.mean(0) # (3,)
+ target_scale, target_idx = torch.norm(targets - target_center, dim=-1).max(0)
+
+ # a 45 degree vec from longest axis
+ if use_long_axis:
+ long_vec = targets[target_idx] - target_center # (x, 0, z)
+ long_vec = long_vec / torch.norm(long_vec)
+ R = axis_angle_to_matrix(torch.tensor([0, np.pi / 4, 0])).to(long_vec)
+ vec = R @ long_vec
+ else:
+ vec_rad = vec_rot / 180 * np.pi
+ vec = torch.tensor([np.sin(vec_rad), 0, np.cos(vec_rad)]).float()
+ vec = vec / torch.norm(vec)
+
+ # Compute camera position (center + scale * vec * beta) + y=4
+ target_scale = max(target_scale, 1.0) * beta
+ position = target_center + vec * target_scale
+ position[1] = (
+ target_scale * np.tan(np.pi * cam_height_degree / 180) + target_center_height
+ )
+
+ # Compute camera rotation and translation
+ positions = position.unsqueeze(0).repeat(L, 1)
+ target_centers = target_center.unsqueeze(0).repeat(L, 1)
+ target_centers[:, 1] = target_center_height
+ rotation = look_at_rotation(positions, target_centers).mT
+ translation = -(rotation @ positions.unsqueeze(-1)).squeeze(-1)
+
+ lights = PointLights(device=device, location=[position.tolist()])
+ return rotation, translation, lights
+
+
+def get_global_cameras_static_v2(
+ verts,
+ beta=4.0,
+ cam_height_degree=30,
+ target_center_height=1.0,
+ use_long_axis=False,
+ vec_rot=45,
+ device="cuda",
+):
+ L, V, _ = verts.shape
+
+ # Compute target trajectory, denote as center + scale
+ targets = verts.mean(1) # (L, 3)
+ targets[:, 1] = 0 # project to xz-plane
+ target_center = targets.mean(0) # (3,)
+ target_scale, target_idx = torch.norm(targets - target_center, dim=-1).max(0)
+
+ # a 45 degree vec from longest axis
+ if use_long_axis:
+ long_vec = targets[target_idx] - target_center # (x, 0, z)
+ long_vec = long_vec / torch.norm(long_vec)
+ R = axis_angle_to_matrix(torch.tensor([0, np.pi / 4, 0])).to(long_vec)
+ vec = R @ long_vec
+ else:
+ vec_rad = vec_rot / 180 * np.pi
+ vec = torch.tensor([np.sin(vec_rad), 0, np.cos(vec_rad)]).float()
+ vec = vec / torch.norm(vec)
+
+ # Compute camera position (center + scale * vec * beta) + y=4
+ target_scale = max(target_scale, 1.0) * beta
+ position = target_center + vec * target_scale
+ position[1] = (
+ target_scale * np.tan(np.pi * cam_height_degree / 180) + target_center_height
+ )
+
+ # Compute camera rotation and translation
+ # positions = position.unsqueeze(0).repeat(L, 1)
+ # target_centers = target_center.unsqueeze(0).repeat(L, 1)
+ target_center[1] = target_center_height
+ # rotation = look_at_rotation(positions, target_centers).mT
+ # translation = -(rotation @ positions.unsqueeze(-1)).squeeze(-1)
+
+ # lights = PointLights(device=device, location=[position.tolist()])
+ # return rotation, translation, lights
+ up = torch.tensor([0, 1, 0])
+ return position, target_center, up
+
+
+def get_ground_params_from_points(root_points, vert_points):
+ """xz-plane is the ground plane
+ Args:
+ root_points: (L, 3), to decide center
+ vert_points: (L, V, 3), to decide scale
+ """
+ root_max = root_points.max(0)[0] # (3,)
+ root_min = root_points.min(0)[0] # (3,)
+ cx, _, cz = (root_max + root_min) / 2.0
+
+ vert_max = vert_points.reshape(-1, 3).max(0)[0] # (L, 3)
+ vert_min = vert_points.reshape(-1, 3).min(0)[0] # (L, 3)
+ scale = (vert_max - vert_min)[[0, 2]].max()
+ return float(scale), float(cx), float(cz)
diff --git a/genmo/utils/vis/renderer_tools.py b/genmo/utils/vis/renderer_tools.py
new file mode 100644
index 0000000000000000000000000000000000000000..313f48e78da06a81509cbf70217a7d5a81925ca2
--- /dev/null
+++ b/genmo/utils/vis/renderer_tools.py
@@ -0,0 +1,841 @@
+import math
+import os
+
+import cv2
+import numpy as np
+import torch
+from PIL import Image
+
+
+def read_image(path, scale=1):
+ im = Image.open(path)
+ if scale == 1:
+ return np.array(im)
+ W, H = im.size
+ w, h = int(scale * W), int(scale * H)
+ return np.array(im.resize((w, h), Image.ANTIALIAS))
+
+
+def transform_torch3d(T_c2w):
+ """
+ :param T_c2w (*, 4, 4)
+ returns (*, 3, 3), (*, 3)
+ """
+ R1 = torch.tensor(
+ [
+ [-1.0, 0.0, 0.0],
+ [0.0, -1.0, 0.0],
+ [0.0, 0.0, 1.0],
+ ],
+ device=T_c2w.device,
+ )
+ R2 = torch.tensor(
+ [
+ [1.0, 0.0, 0.0],
+ [0.0, -1.0, 0.0],
+ [0.0, 0.0, -1.0],
+ ],
+ device=T_c2w.device,
+ )
+ cam_R, cam_t = T_c2w[..., :3, :3], T_c2w[..., :3, 3]
+ cam_R = torch.einsum("...ij,jk->...ik", cam_R, R1)
+ cam_t = torch.einsum("ij,...j->...i", R2, cam_t)
+ return cam_R, cam_t
+
+
+def transform_pyrender(T_c2w):
+ """
+ :param T_c2w (*, 4, 4)
+ """
+ T_vis = torch.tensor(
+ [
+ [1.0, 0.0, 0.0, 0.0],
+ [0.0, -1.0, 0.0, 0.0],
+ [0.0, 0.0, -1.0, 0.0],
+ [0.0, 0.0, 0.0, 1.0],
+ ],
+ device=T_c2w.device,
+ )
+ return torch.einsum(
+ "...ij,jk->...ik", torch.einsum("ij,...jk->...ik", T_vis, T_c2w), T_vis
+ )
+
+
+def smpl_to_geometry(verts, faces, vis_mask=None, track_ids=None):
+ """
+ :param verts (B, T, V, 3)
+ :param faces (F, 3)
+ :param vis_mask (optional) (B, T) visibility of each person
+ :param track_ids (optional) (B,)
+ returns list of T verts (B, V, 3), faces (F, 3), colors (B, 3)
+ where B is different depending on the visibility of the people
+ """
+ B, T = verts.shape[:2]
+ device = verts.device
+
+ # (B, 3)
+ colors = (
+ track_to_colors(track_ids)
+ if track_ids is not None
+ else torch.ones(B, 3, device) * 0.5
+ )
+
+ # list T (B, V, 3), T (B, 3), T (F, 3)
+ return filter_visible_meshes(verts, colors, faces, vis_mask)
+
+
+def filter_visible_meshes(verts, colors, faces, vis_mask=None, vis_opacity=False):
+ """
+ :param verts (B, T, V, 3)
+ :param colors (B, 3)
+ :param faces (F, 3)
+ :param vis_mask (optional tensor, default None) (B, T) ternary mask
+ -1 if not in frame
+ 0 if temporarily occluded
+ 1 if visible
+ :param vis_opacity (optional bool, default False)
+ if True, make occluded people alpha=0.5, otherwise alpha=1
+ returns a list of T lists verts (Bi, V, 3), colors (Bi, 4), faces (F, 3)
+ """
+ # import ipdb; ipdb.set_trace()
+ B, T = verts.shape[:2]
+ faces = [faces for t in range(T)]
+ if vis_mask is None:
+ verts = [verts[:, t] for t in range(T)]
+ colors = [colors for t in range(T)]
+ return verts, colors, faces
+
+ # render occluded and visible, but not removed
+ vis_mask = vis_mask >= 0
+ if vis_opacity:
+ alpha = 0.5 * (vis_mask[..., None] + 1)
+ else:
+ alpha = (vis_mask[..., None] >= 0).float()
+ vert_list = [verts[vis_mask[:, t], t] for t in range(T)]
+ colors = [
+ torch.cat([colors[vis_mask[:, t]], alpha[vis_mask[:, t], t]], dim=-1)
+ for t in range(T)
+ ]
+ bounds = get_bboxes(verts, vis_mask)
+ return vert_list, colors, faces, bounds
+
+
+def get_bboxes(verts, vis_mask):
+ """
+ return bb_min, bb_max, and mean for each track (B, 3) over entire trajectory
+ :param verts (B, T, V, 3)
+ :param vis_mask (B, T)
+ """
+ B, T, *_ = verts.shape
+ bb_min, bb_max, mean = [], [], []
+ for b in range(B):
+ v = verts[b, vis_mask[b, :T]] # (Tb, V, 3)
+ bb_min.append(v.amin(dim=(0, 1)))
+ bb_max.append(v.amax(dim=(0, 1)))
+ mean.append(v.mean(dim=(0, 1)))
+ bb_min = torch.stack(bb_min, dim=0)
+ bb_max = torch.stack(bb_max, dim=0)
+ mean = torch.stack(mean, dim=0)
+ # point to a track that's long and close to the camera
+ zs = mean[:, 2]
+ counts = vis_mask[:, :T].sum(dim=-1) # (B,)
+ mask = counts < 0.8 * T
+ zs[mask] = torch.inf
+ sel = torch.argmin(zs)
+ return bb_min.amin(dim=0), bb_max.amax(dim=0), mean[sel]
+
+
+def track_to_colors(track_ids):
+ """
+ :param track_ids (B)
+ """
+ color_map = torch.from_numpy(get_colors()).to(track_ids)
+ return color_map[track_ids] / 255 # (B, 3)
+
+
+def get_colors():
+ # color_file = os.path.abspath(os.path.join(__file__, "../colors_phalp.txt"))
+ color_file = os.path.abspath(os.path.join(__file__, "../colors.txt"))
+ RGB_tuples = np.vstack(
+ [
+ np.loadtxt(color_file, skiprows=0),
+ # np.loadtxt(color_file, skiprows=1),
+ np.random.uniform(0, 255, size=(10000, 3)),
+ [[0, 0, 0]],
+ ]
+ )
+ b = np.where(RGB_tuples == 0)
+ RGB_tuples[b] = 1
+ return RGB_tuples.astype(np.float32)
+
+
+def checkerboard_geometry(
+ length=12.0,
+ color0=[0.8, 0.9, 0.9],
+ color1=[0.6, 0.7, 0.7],
+ tile_width=0.5,
+ alpha=1.0,
+ up="y",
+ c1=0.0,
+ c2=0.0,
+):
+ assert up == "y" or up == "z"
+ color0 = np.array(color0 + [alpha])
+ color1 = np.array(color1 + [alpha])
+ num_rows = num_cols = max(2, int(length / tile_width))
+ radius = float(num_rows * tile_width) / 2.0
+ vertices = []
+ vert_colors = []
+ faces = []
+ face_colors = []
+ for i in range(num_rows):
+ for j in range(num_cols):
+ u0, v0 = j * tile_width - radius, i * tile_width - radius
+ us = np.array([u0, u0, u0 + tile_width, u0 + tile_width])
+ vs = np.array([v0, v0 + tile_width, v0 + tile_width, v0])
+ zs = np.zeros(4)
+ if up == "y":
+ cur_verts = np.stack([us, zs, vs], axis=-1) # (4, 3)
+ cur_verts[:, 0] += c1
+ cur_verts[:, 2] += c2
+ else:
+ cur_verts = np.stack([us, vs, zs], axis=-1) # (4, 3)
+ cur_verts[:, 0] += c1
+ cur_verts[:, 1] += c2
+
+ cur_faces = np.array(
+ [[0, 1, 3], [1, 2, 3], [0, 3, 1], [1, 3, 2]], dtype=np.int64
+ )
+ cur_faces += 4 * (i * num_cols + j) # the number of previously added verts
+ use_color0 = (i % 2 == 0 and j % 2 == 0) or (i % 2 == 1 and j % 2 == 1)
+ cur_color = color0 if use_color0 else color1
+ cur_colors = np.array([cur_color, cur_color, cur_color, cur_color])
+
+ vertices.append(cur_verts)
+ faces.append(cur_faces)
+ vert_colors.append(cur_colors)
+ face_colors.append(cur_colors)
+
+ vertices = np.concatenate(vertices, axis=0).astype(np.float32)
+ vert_colors = np.concatenate(vert_colors, axis=0).astype(np.float32)
+ faces = np.concatenate(faces, axis=0).astype(np.float32)
+ face_colors = np.concatenate(face_colors, axis=0).astype(np.float32)
+
+ return vertices, faces, vert_colors, face_colors
+
+
+def camera_marker_geometry(radius, height, up):
+ assert up == "y" or up == "z"
+ if up == "y":
+ vertices = np.array(
+ [
+ [-radius, -radius, 0],
+ [radius, -radius, 0],
+ [radius, radius, 0],
+ [-radius, radius, 0],
+ [0, 0, height],
+ ]
+ )
+ else:
+ vertices = np.array(
+ [
+ [-radius, 0, -radius],
+ [radius, 0, -radius],
+ [radius, 0, radius],
+ [-radius, 0, radius],
+ [0, -height, 0],
+ ]
+ )
+
+ faces = np.array(
+ [
+ [0, 3, 1],
+ [1, 3, 2],
+ [0, 1, 4],
+ [1, 2, 4],
+ [2, 3, 4],
+ [3, 0, 4],
+ ]
+ )
+
+ face_colors = np.array(
+ [
+ [1.0, 1.0, 1.0, 1.0],
+ [1.0, 1.0, 1.0, 1.0],
+ [0.0, 1.0, 0.0, 1.0],
+ [1.0, 0.0, 0.0, 1.0],
+ [0.0, 1.0, 0.0, 1.0],
+ [1.0, 0.0, 0.0, 1.0],
+ ]
+ )
+ return vertices, faces, face_colors
+
+
+def vis_keypoints(
+ keypts_list,
+ img_size,
+ radius=6,
+ thickness=3,
+ kpt_score_thr=0.3,
+ dataset="TopDownCocoDataset",
+):
+ """
+ Visualize keypoints
+ From ViTPose/mmpose/apis/inference.py
+ """
+ palette = np.array(
+ [
+ [255, 128, 0],
+ [255, 153, 51],
+ [255, 178, 102],
+ [230, 230, 0],
+ [255, 153, 255],
+ [153, 204, 255],
+ [255, 102, 255],
+ [255, 51, 255],
+ [102, 178, 255],
+ [51, 153, 255],
+ [255, 153, 153],
+ [255, 102, 102],
+ [255, 51, 51],
+ [153, 255, 153],
+ [102, 255, 102],
+ [51, 255, 51],
+ [0, 255, 0],
+ [0, 0, 255],
+ [255, 0, 0],
+ [255, 255, 255],
+ ]
+ )
+
+ if dataset in (
+ "TopDownCocoDataset",
+ "BottomUpCocoDataset",
+ "TopDownOCHumanDataset",
+ "AnimalMacaqueDataset",
+ ):
+ # show the results
+ skeleton = [
+ [15, 13],
+ [13, 11],
+ [16, 14],
+ [14, 12],
+ [11, 12],
+ [5, 11],
+ [6, 12],
+ [5, 6],
+ [5, 7],
+ [6, 8],
+ [7, 9],
+ [8, 10],
+ [1, 2],
+ [0, 1],
+ [0, 2],
+ [1, 3],
+ [2, 4],
+ [3, 5],
+ [4, 6],
+ ]
+
+ pose_link_color = palette[
+ [0, 0, 0, 0, 7, 7, 7, 9, 9, 9, 9, 9, 16, 16, 16, 16, 16, 16, 16]
+ ]
+ pose_kpt_color = palette[
+ [16, 16, 16, 16, 16, 9, 9, 9, 9, 9, 9, 0, 0, 0, 0, 0, 0]
+ ]
+
+ elif dataset == "TopDownCocoWholeBodyDataset":
+ # show the results
+ skeleton = [
+ [15, 13],
+ [13, 11],
+ [16, 14],
+ [14, 12],
+ [11, 12],
+ [5, 11],
+ [6, 12],
+ [5, 6],
+ [5, 7],
+ [6, 8],
+ [7, 9],
+ [8, 10],
+ [1, 2],
+ [0, 1],
+ [0, 2],
+ [1, 3],
+ [2, 4],
+ [3, 5],
+ [4, 6],
+ [15, 17],
+ [15, 18],
+ [15, 19],
+ [16, 20],
+ [16, 21],
+ [16, 22],
+ [91, 92],
+ [92, 93],
+ [93, 94],
+ [94, 95],
+ [91, 96],
+ [96, 97],
+ [97, 98],
+ [98, 99],
+ [91, 100],
+ [100, 101],
+ [101, 102],
+ [102, 103],
+ [91, 104],
+ [104, 105],
+ [105, 106],
+ [106, 107],
+ [91, 108],
+ [108, 109],
+ [109, 110],
+ [110, 111],
+ [112, 113],
+ [113, 114],
+ [114, 115],
+ [115, 116],
+ [112, 117],
+ [117, 118],
+ [118, 119],
+ [119, 120],
+ [112, 121],
+ [121, 122],
+ [122, 123],
+ [123, 124],
+ [112, 125],
+ [125, 126],
+ [126, 127],
+ [127, 128],
+ [112, 129],
+ [129, 130],
+ [130, 131],
+ [131, 132],
+ ]
+
+ pose_link_color = palette[
+ [0, 0, 0, 0, 7, 7, 7, 9, 9, 9, 9, 9, 16, 16, 16, 16, 16, 16, 16]
+ + [16, 16, 16, 16, 16, 16]
+ + [0, 0, 0, 0, 4, 4, 4, 4, 8, 8, 8, 8, 12, 12, 12, 12, 16, 16, 16, 16]
+ + [0, 0, 0, 0, 4, 4, 4, 4, 8, 8, 8, 8, 12, 12, 12, 12, 16, 16, 16, 16]
+ ]
+ pose_kpt_color = palette[
+ [16, 16, 16, 16, 16, 9, 9, 9, 9, 9, 9, 0, 0, 0, 0, 0, 0]
+ + [0, 0, 0, 0, 0, 0]
+ + [19] * (68 + 42)
+ ]
+
+ elif dataset == "TopDownAicDataset":
+ skeleton = [
+ [2, 1],
+ [1, 0],
+ [0, 13],
+ [13, 3],
+ [3, 4],
+ [4, 5],
+ [8, 7],
+ [7, 6],
+ [6, 9],
+ [9, 10],
+ [10, 11],
+ [12, 13],
+ [0, 6],
+ [3, 9],
+ ]
+
+ pose_link_color = palette[[9, 9, 9, 9, 9, 9, 16, 16, 16, 16, 16, 0, 7, 7]]
+ pose_kpt_color = palette[[9, 9, 9, 9, 9, 9, 16, 16, 16, 16, 16, 16, 0, 0]]
+
+ elif dataset == "TopDownMpiiDataset":
+ skeleton = [
+ [0, 1],
+ [1, 2],
+ [2, 6],
+ [6, 3],
+ [3, 4],
+ [4, 5],
+ [6, 7],
+ [7, 8],
+ [8, 9],
+ [8, 12],
+ [12, 11],
+ [11, 10],
+ [8, 13],
+ [13, 14],
+ [14, 15],
+ ]
+
+ pose_link_color = palette[[16, 16, 16, 16, 16, 16, 7, 7, 0, 9, 9, 9, 9, 9, 9]]
+ pose_kpt_color = palette[[16, 16, 16, 16, 16, 16, 7, 7, 0, 0, 9, 9, 9, 9, 9, 9]]
+
+ elif dataset == "TopDownMpiiTrbDataset":
+ skeleton = [
+ [12, 13],
+ [13, 0],
+ [13, 1],
+ [0, 2],
+ [1, 3],
+ [2, 4],
+ [3, 5],
+ [0, 6],
+ [1, 7],
+ [6, 7],
+ [6, 8],
+ [7, 9],
+ [8, 10],
+ [9, 11],
+ [14, 15],
+ [16, 17],
+ [18, 19],
+ [20, 21],
+ [22, 23],
+ [24, 25],
+ [26, 27],
+ [28, 29],
+ [30, 31],
+ [32, 33],
+ [34, 35],
+ [36, 37],
+ [38, 39],
+ ]
+
+ pose_link_color = palette[[16] * 14 + [19] * 13]
+ pose_kpt_color = palette[[16] * 14 + [0] * 26]
+
+ elif dataset in ("OneHand10KDataset", "FreiHandDataset", "PanopticDataset"):
+ skeleton = [
+ [0, 1],
+ [1, 2],
+ [2, 3],
+ [3, 4],
+ [0, 5],
+ [5, 6],
+ [6, 7],
+ [7, 8],
+ [0, 9],
+ [9, 10],
+ [10, 11],
+ [11, 12],
+ [0, 13],
+ [13, 14],
+ [14, 15],
+ [15, 16],
+ [0, 17],
+ [17, 18],
+ [18, 19],
+ [19, 20],
+ ]
+
+ pose_link_color = palette[
+ [0, 0, 0, 0, 4, 4, 4, 4, 8, 8, 8, 8, 12, 12, 12, 12, 16, 16, 16, 16]
+ ]
+ pose_kpt_color = palette[
+ [0, 0, 0, 0, 0, 4, 4, 4, 4, 8, 8, 8, 8, 12, 12, 12, 12, 16, 16, 16, 16]
+ ]
+
+ elif dataset == "InterHand2DDataset":
+ skeleton = [
+ [0, 1],
+ [1, 2],
+ [2, 3],
+ [4, 5],
+ [5, 6],
+ [6, 7],
+ [8, 9],
+ [9, 10],
+ [10, 11],
+ [12, 13],
+ [13, 14],
+ [14, 15],
+ [16, 17],
+ [17, 18],
+ [18, 19],
+ [3, 20],
+ [7, 20],
+ [11, 20],
+ [15, 20],
+ [19, 20],
+ ]
+
+ pose_link_color = palette[
+ [0, 0, 0, 4, 4, 4, 8, 8, 8, 12, 12, 12, 16, 16, 16, 0, 4, 8, 12, 16]
+ ]
+ pose_kpt_color = palette[
+ [0, 0, 0, 0, 4, 4, 4, 4, 8, 8, 8, 8, 12, 12, 12, 12, 16, 16, 16, 16, 0]
+ ]
+
+ elif dataset == "Face300WDataset":
+ # show the results
+ skeleton = []
+
+ pose_link_color = palette[[]]
+ pose_kpt_color = palette[[19] * 68]
+ kpt_score_thr = 0
+
+ elif dataset == "FaceAFLWDataset":
+ # show the results
+ skeleton = []
+
+ pose_link_color = palette[[]]
+ pose_kpt_color = palette[[19] * 19]
+ kpt_score_thr = 0
+
+ elif dataset == "FaceCOFWDataset":
+ # show the results
+ skeleton = []
+
+ pose_link_color = palette[[]]
+ pose_kpt_color = palette[[19] * 29]
+ kpt_score_thr = 0
+
+ elif dataset == "FaceWFLWDataset":
+ # show the results
+ skeleton = []
+
+ pose_link_color = palette[[]]
+ pose_kpt_color = palette[[19] * 98]
+ kpt_score_thr = 0
+
+ elif dataset == "AnimalHorse10Dataset":
+ skeleton = [
+ [0, 1],
+ [1, 12],
+ [12, 16],
+ [16, 21],
+ [21, 17],
+ [17, 11],
+ [11, 10],
+ [10, 8],
+ [8, 9],
+ [9, 12],
+ [2, 3],
+ [3, 4],
+ [5, 6],
+ [6, 7],
+ [13, 14],
+ [14, 15],
+ [18, 19],
+ [19, 20],
+ ]
+
+ pose_link_color = palette[[4] * 10 + [6] * 2 + [6] * 2 + [7] * 2 + [7] * 2]
+ pose_kpt_color = palette[
+ [4, 4, 6, 6, 6, 6, 6, 6, 4, 4, 4, 4, 4, 7, 7, 7, 4, 4, 7, 7, 7, 4]
+ ]
+
+ elif dataset == "AnimalFlyDataset":
+ skeleton = [
+ [1, 0],
+ [2, 0],
+ [3, 0],
+ [4, 3],
+ [5, 4],
+ [7, 6],
+ [8, 7],
+ [9, 8],
+ [11, 10],
+ [12, 11],
+ [13, 12],
+ [15, 14],
+ [16, 15],
+ [17, 16],
+ [19, 18],
+ [20, 19],
+ [21, 20],
+ [23, 22],
+ [24, 23],
+ [25, 24],
+ [27, 26],
+ [28, 27],
+ [29, 28],
+ [30, 3],
+ [31, 3],
+ ]
+
+ pose_link_color = palette[[0] * 25]
+ pose_kpt_color = palette[[0] * 32]
+
+ elif dataset == "AnimalLocustDataset":
+ skeleton = [
+ [1, 0],
+ [2, 1],
+ [3, 2],
+ [4, 3],
+ [6, 5],
+ [7, 6],
+ [9, 8],
+ [10, 9],
+ [11, 10],
+ [13, 12],
+ [14, 13],
+ [15, 14],
+ [17, 16],
+ [18, 17],
+ [19, 18],
+ [21, 20],
+ [22, 21],
+ [24, 23],
+ [25, 24],
+ [26, 25],
+ [28, 27],
+ [29, 28],
+ [30, 29],
+ [32, 31],
+ [33, 32],
+ [34, 33],
+ ]
+
+ pose_link_color = palette[[0] * 26]
+ pose_kpt_color = palette[[0] * 35]
+
+ elif dataset == "AnimalZebraDataset":
+ skeleton = [[1, 0], [2, 1], [3, 2], [4, 2], [5, 7], [6, 7], [7, 2], [8, 7]]
+
+ pose_link_color = palette[[0] * 8]
+ pose_kpt_color = palette[[0] * 9]
+
+ elif dataset in "AnimalPoseDataset":
+ skeleton = [
+ [0, 1],
+ [0, 2],
+ [1, 3],
+ [0, 4],
+ [1, 4],
+ [4, 5],
+ [5, 7],
+ [6, 7],
+ [5, 8],
+ [8, 12],
+ [12, 16],
+ [5, 9],
+ [9, 13],
+ [13, 17],
+ [6, 10],
+ [10, 14],
+ [14, 18],
+ [6, 11],
+ [11, 15],
+ [15, 19],
+ ]
+
+ pose_link_color = palette[[0] * 20]
+ pose_kpt_color = palette[[0] * 20]
+ else:
+ NotImplementedError()
+
+ img_w, img_h = img_size
+ img = 255 * np.ones((img_h, img_w, 3), dtype=np.uint8)
+ img = imshow_keypoints(
+ img,
+ keypts_list,
+ skeleton,
+ kpt_score_thr,
+ pose_kpt_color,
+ pose_link_color,
+ radius,
+ thickness,
+ )
+ alpha = 255 * (img != 255).any(axis=-1, keepdims=True).astype(np.uint8)
+ return np.concatenate([img, alpha], axis=-1)
+
+
+def imshow_keypoints(
+ img,
+ pose_result,
+ skeleton=None,
+ kpt_score_thr=0.3,
+ pose_kpt_color=None,
+ pose_link_color=None,
+ radius=4,
+ thickness=1,
+ show_keypoint_weight=False,
+):
+ """Draw keypoints and links on an image.
+ From ViTPose/mmpose/core/visualization/image.py
+
+ Args:
+ img (H, W, 3) array
+ pose_result (list[kpts]): The poses to draw. Each element kpts is
+ a set of K keypoints as an Kx3 numpy.ndarray, where each
+ keypoint is represented as x, y, score.
+ kpt_score_thr (float, optional): Minimum score of keypoints
+ to be shown. Default: 0.3.
+ pose_kpt_color (np.array[Nx3]`): Color of N keypoints. If None,
+ the keypoint will not be drawn.
+ pose_link_color (np.array[Mx3]): Color of M links. If None, the
+ links will not be drawn.
+ thickness (int): Thickness of lines.
+ show_keypoint_weight (bool): If True, opacity indicates keypoint score
+ """
+ img_h, img_w, _ = img.shape
+ idcs = [0, 16, 15, 18, 17, 5, 2, 6, 3, 7, 4, 12, 9, 13, 10, 14, 11]
+ for kpts in pose_result:
+ kpts = np.array(kpts, copy=False)[idcs]
+
+ # draw each point on image
+ if pose_kpt_color is not None:
+ assert len(pose_kpt_color) == len(kpts)
+ for kid, kpt in enumerate(kpts):
+ x_coord, y_coord, kpt_score = int(kpt[0]), int(kpt[1]), kpt[2]
+ if kpt_score > kpt_score_thr:
+ color = tuple(int(c) for c in pose_kpt_color[kid])
+ if show_keypoint_weight:
+ img_copy = img.copy()
+ cv2.circle(
+ img_copy, (int(x_coord), int(y_coord)), radius, color, -1
+ )
+ transparency = max(0, min(1, kpt_score))
+ cv2.addWeighted(
+ img_copy, transparency, img, 1 - transparency, 0, dst=img
+ )
+ else:
+ cv2.circle(img, (int(x_coord), int(y_coord)), radius, color, -1)
+
+ # draw links
+ if skeleton is not None and pose_link_color is not None:
+ assert len(pose_link_color) == len(skeleton)
+ for sk_id, sk in enumerate(skeleton):
+ pos1 = (int(kpts[sk[0], 0]), int(kpts[sk[0], 1]))
+ pos2 = (int(kpts[sk[1], 0]), int(kpts[sk[1], 1]))
+ if (
+ pos1[0] > 0
+ and pos1[0] < img_w
+ and pos1[1] > 0
+ and pos1[1] < img_h
+ and pos2[0] > 0
+ and pos2[0] < img_w
+ and pos2[1] > 0
+ and pos2[1] < img_h
+ and kpts[sk[0], 2] > kpt_score_thr
+ and kpts[sk[1], 2] > kpt_score_thr
+ ):
+ color = tuple(int(c) for c in pose_link_color[sk_id])
+ if show_keypoint_weight:
+ img_copy = img.copy()
+ X = (pos1[0], pos2[0])
+ Y = (pos1[1], pos2[1])
+ mX = np.mean(X)
+ mY = np.mean(Y)
+ length = ((Y[0] - Y[1]) ** 2 + (X[0] - X[1]) ** 2) ** 0.5
+ angle = math.degrees(math.atan2(Y[0] - Y[1], X[0] - X[1]))
+ stickwidth = 2
+ polygon = cv2.ellipse2Poly(
+ (int(mX), int(mY)),
+ (int(length / 2), int(stickwidth)),
+ int(angle),
+ 0,
+ 360,
+ 1,
+ )
+ cv2.fillConvexPoly(img_copy, polygon, color)
+ transparency = max(
+ 0, min(1, 0.5 * (kpts[sk[0], 2] + kpts[sk[1], 2]))
+ )
+ cv2.addWeighted(
+ img_copy, transparency, img, 1 - transparency, 0, dst=img
+ )
+ else:
+ cv2.line(img, pos1, pos2, color, thickness=thickness)
+
+ return img
diff --git a/genmo/utils/vis/renderer_utils.py b/genmo/utils/vis/renderer_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..f90226057d32761783cd70ac71eed002767d711e
--- /dev/null
+++ b/genmo/utils/vis/renderer_utils.py
@@ -0,0 +1,41 @@
+import numpy as np
+from tqdm import tqdm
+
+from genmo.utils.vis.renderer import Renderer
+
+
+def simple_render_mesh(render_dict):
+ """Render an camera-space mesh, blank background"""
+ width, height, focal_length = render_dict["whf"]
+ faces = render_dict["faces"]
+ verts = render_dict["verts"]
+
+ renderer = Renderer(width, height, focal_length, device="cuda", faces=faces)
+ outputs = []
+ for i in tqdm(range(len(verts)), desc="Rendering"):
+ img = renderer.render_mesh(verts[i].cuda(), colors=[0.8, 0.8, 0.8])
+ outputs.append(img)
+ outputs = np.stack(outputs, axis=0)
+ return outputs
+
+
+def simple_render_mesh_background(render_dict, VI=50, colors=[0.8, 0.8, 0.8]):
+ """Render an camera-space mesh, blank background"""
+ K = render_dict["K"]
+ faces = render_dict["faces"]
+ verts = render_dict["verts"]
+ background = render_dict["background"]
+ N_frames = len(verts)
+ if len(background.shape) == 3:
+ background = [background] * N_frames
+ height, width = background[0].shape[:2]
+
+ renderer = Renderer(width, height, device="cuda", faces=faces, K=K)
+ outputs = []
+ for i in tqdm(range(len(verts)), desc="Rendering"):
+ img = renderer.render_mesh(
+ verts[i].cuda(), colors=colors, background=background[i], VI=VI
+ )
+ outputs.append(img)
+ outputs = np.stack(outputs, axis=0)
+ return outputs
diff --git a/genmo/utils/vis/rich_logger.py b/genmo/utils/vis/rich_logger.py
new file mode 100644
index 0000000000000000000000000000000000000000..25cdb7e71c8d9402df8087d0c7330e2e9aa63bc2
--- /dev/null
+++ b/genmo/utils/vis/rich_logger.py
@@ -0,0 +1,39 @@
+import rich
+import rich.syntax
+import rich.tree
+from omegaconf import DictConfig, OmegaConf
+from pytorch_lightning.utilities import rank_zero_only
+
+from genmo.utils.pylogger import Log
+
+
+@rank_zero_only
+def print_cfg(cfg: DictConfig, use_rich: bool = False):
+ if use_rich:
+ print_order = ("data", "model", "callbacks", "logger", "pl_trainer")
+ style = "dim"
+ tree = rich.tree.Tree("CONFIG", style=style, guide_style=style)
+
+ # add fields from `print_order` to queue
+ # add all the other fields to queue (not specified in `print_order`)
+ queue = []
+ for field in print_order:
+ queue.append(field) if field in cfg else Log.warn(
+ f"Field '{field}' not found in config. Skipping."
+ )
+ for field in cfg:
+ if field not in queue:
+ queue.append(field)
+
+ # generate config tree from queue
+ for field in queue:
+ branch = tree.add(field, style=style, guide_style=style)
+ config_group = cfg[field]
+ if isinstance(config_group, DictConfig):
+ branch_content = OmegaConf.to_yaml(config_group, resolve=False)
+ else:
+ branch_content = str(config_group)
+ branch.add(rich.syntax.Syntax(branch_content, "yaml"))
+ rich.print(tree)
+ else:
+ Log.info(OmegaConf.to_yaml(cfg, resolve=False))
diff --git a/genmo/utils/vis/vis.py b/genmo/utils/vis/vis.py
new file mode 100644
index 0000000000000000000000000000000000000000..e8806148ea329098a84de45e66b121f2d26c0066
--- /dev/null
+++ b/genmo/utils/vis/vis.py
@@ -0,0 +1,530 @@
+import os
+import os.path as osp
+import platform
+import random
+import subprocess
+
+import cv2 as cv
+import numpy as np
+import seaborn as sns
+from PIL import ImageColor
+
+FFMPEG_PATH = "/usr/bin/ffmpeg" if osp.exists("/usr/bin/ffmpeg") else "ffmpeg"
+font_files = {
+ "Windows": "C:/Windows/Fonts/arial.ttf",
+ "Linux": "/usr/share/fonts/truetype/lato/Lato-Regular.ttf",
+ "Darwin": "/System/Library/Fonts/Supplemental/Arial.ttf",
+}
+
+
+def get_video_width_height(video_file):
+ vcap = cv.VideoCapture(video_file)
+ img_w = int(vcap.get(3))
+ img_h = int(vcap.get(4))
+ return img_w, img_h
+
+
+def get_video_num_fr(video_file):
+ vcap = cv.VideoCapture(video_file)
+ num_fr = int(vcap.get(cv.CAP_PROP_FRAME_COUNT))
+ return num_fr
+
+
+def get_video_fps(video_file):
+ vcap = cv.VideoCapture(video_file)
+ fps = vcap.get(cv.CAP_PROP_FPS)
+ return fps
+
+
+def rescale_video(video_path, out_path, width=-1, height=-1, verbose=True):
+ os.makedirs(osp.dirname(out_path), exist_ok=True)
+ cmd = [
+ FFMPEG_PATH,
+ "-y",
+ "-i",
+ video_path,
+ "-vf",
+ f"scale={width}:{height}",
+ out_path,
+ ]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+
+
+def crop_video(
+ video_path, out_path, width="iw", height="ih", x=0, y=0, crop_str=None, verbose=True
+):
+ os.makedirs(osp.dirname(out_path), exist_ok=True)
+ if crop_str is None:
+ crop_str = f"{width}:{height}:{x}:{y}"
+ cmd = [FFMPEG_PATH, "-y", "-i", video_path, "-vf", f"crop={crop_str}", out_path]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+
+
+def clip_video(video_path, out_path, start, end, verbose=True):
+ os.makedirs(osp.dirname(out_path), exist_ok=True)
+ cmd = [
+ FFMPEG_PATH,
+ "-y",
+ "-ss",
+ f"{start}",
+ "-i",
+ video_path,
+ "-t",
+ f"{end - start}",
+ out_path,
+ ]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+
+
+def text_video(
+ video_path,
+ out_path,
+ text,
+ x=10,
+ y=30,
+ verbose=True,
+ text_color="white",
+ text_size=60,
+):
+ os.makedirs(osp.dirname(out_path), exist_ok=True)
+ font_file = font_files[platform.system()]
+ draw_str = f"drawtext=fontsize={text_size}:fontfile={font_file}:fontcolor={text_color}:text='{text}':x={x}:y={y}"
+ cmd = [FFMPEG_PATH, "-y", "-i", video_path, "-vf", draw_str, out_path]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+
+
+def make_gif(
+ video_path,
+ out_path,
+ t_start="00:00:00",
+ t_end=None,
+ width=600,
+ height=-1,
+ verbose=True,
+):
+ os.makedirs(osp.dirname(out_path), exist_ok=True)
+ cmd = (
+ [FFMPEG_PATH, "-y", "-i", video_path, "-ss", t_start]
+ + (["-t", t_end] if t_end is not None else [])
+ + [
+ "-vf",
+ f"fps=30,scale={width}:{height}:flags=lanczos,split[s0][s1];[s0]palettegen[p];[s1][p]paletteuse",
+ "-loop",
+ "0",
+ out_path,
+ ]
+ )
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+
+
+def images_to_video(
+ img_dir, out_path, img_fmt="%06d.jpg", fps=30, crf=25, verbose=True
+):
+ os.makedirs(osp.dirname(out_path), exist_ok=True)
+ cmd = [
+ FFMPEG_PATH,
+ "-y",
+ "-r",
+ f"{fps}",
+ "-f",
+ "image2",
+ "-start_number",
+ "0",
+ "-i",
+ f"{img_dir}/{img_fmt}",
+ "-vcodec",
+ "libx264",
+ "-vf",
+ "pad=ceil(iw/2)*2:ceil(ih/2)*2",
+ "-crf",
+ f"{crf}",
+ "-pix_fmt",
+ "yuv420p",
+ out_path,
+ ]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ p = subprocess.run(cmd)
+ if p.returncode != 0:
+ raise Exception("Something went wrong during images_to_video!")
+
+
+def video_to_images(video_path, out_path, img_fmt="%06d.jpg", fps=30, verbose=True):
+ os.makedirs(out_path, exist_ok=True)
+ cmd = [FFMPEG_PATH, "-i", video_path, "-r", f"{fps}", f"{out_path}/{img_fmt}"]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ p = subprocess.run(cmd)
+ if p.returncode != 0:
+ raise Exception("Something went wrong during video_to_images!")
+
+
+def hstack_videos(
+ video1_path,
+ video2_path,
+ out_path,
+ crf=25,
+ verbose=True,
+ text1=None,
+ text2=None,
+ text_color="white",
+ text_size=60,
+):
+ if not (text1 is None or text2 is None):
+ write_text = True
+ tmp_file = f"{osp.splitext(out_path)[0]}_tmp.mp4"
+ else:
+ write_text = False
+
+ os.makedirs(osp.dirname(out_path), exist_ok=True)
+ cmd = [
+ FFMPEG_PATH,
+ "-y",
+ "-i",
+ video1_path,
+ "-i",
+ video2_path,
+ "-filter_complex",
+ "hstack,format=yuv420p",
+ "-vcodec",
+ "libx264",
+ "-crf",
+ f"{crf}",
+ tmp_file if write_text else out_path,
+ ]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+
+ if write_text:
+ font_file = font_files[platform.system()]
+ draw_str = (
+ f"drawtext=fontsize={text_size}:fontfile={font_file}:fontcolor={text_color}:text='{text1}':x=(w-text_w)/4:y=20,"
+ f"drawtext=fontsize={text_size}:fontfile={font_file}:fontcolor={text_color}:text='{text2}':x=3*(w-text_w)/4:y=20"
+ )
+ cmd = [
+ FFMPEG_PATH,
+ "-i",
+ tmp_file,
+ "-y",
+ "-vf",
+ draw_str,
+ "-c:a",
+ "copy",
+ out_path,
+ ]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+ os.remove(tmp_file)
+
+
+def vstack_videos(
+ video1_path,
+ video2_path,
+ out_path,
+ crf=25,
+ verbose=True,
+ text1=None,
+ text2=None,
+ text_color="white",
+ text_size=60,
+):
+ if not (text1 is None or text2 is None):
+ write_text = True
+ tmp_file = f"{osp.splitext(out_path)[0]}_tmp.mp4"
+ else:
+ write_text = False
+
+ os.makedirs(osp.dirname(out_path), exist_ok=True)
+ cmd = [
+ FFMPEG_PATH,
+ "-y",
+ "-i",
+ video1_path,
+ "-i",
+ video2_path,
+ "-filter_complex",
+ "vstack,format=yuv420p",
+ "-vcodec",
+ "libx264",
+ "-crf",
+ f"{crf}",
+ tmp_file if write_text else out_path,
+ ]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+
+ if write_text:
+ font_file = font_files[platform.system()]
+ draw_str = (
+ f"drawtext=fontsize={text_size}:fontfile={font_file}:fontcolor={text_color}:text='{text1}':x=10:y=20,"
+ f"drawtext=fontsize={text_size}:fontfile={font_file}:fontcolor={text_color}:text='{text2}':x=10:y=h/2+20"
+ )
+ cmd = [
+ FFMPEG_PATH,
+ "-i",
+ tmp_file,
+ "-y",
+ "-vf",
+ draw_str,
+ "-c:a",
+ "copy",
+ out_path,
+ ]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+ os.remove(tmp_file)
+
+
+def vstack_video_arr(
+ video_arr,
+ out_path,
+ crf=25,
+ verbose=True,
+ text_arr=None,
+ text_color="white",
+ text_size=60,
+):
+ assert len(video_arr) > 1
+ tmp_file1 = f"{osp.splitext(out_path)[0]}_tmp1.mp4"
+ tmp_file2 = f"{osp.splitext(out_path)[0]}_tmp2.mp4"
+
+ height = np.array([get_video_width_height(x)[1] for x in video_arr])
+ start_h = np.concatenate([np.array([0]), np.cumsum(height)[:-1]])
+
+ os.makedirs(osp.dirname(out_path), exist_ok=True)
+
+ for i in range(1, len(video_arr)):
+ prev_video = video_arr[0] if i == 1 else tmp_file1
+ cmd = [
+ FFMPEG_PATH,
+ "-y",
+ "-i",
+ prev_video,
+ "-i",
+ video_arr[i],
+ "-filter_complex",
+ "vstack,format=yuv420p",
+ "-vcodec",
+ "libx264",
+ "-crf",
+ f"{crf}",
+ tmp_file2,
+ ]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+ tmp_file1, tmp_file2 = tmp_file2, tmp_file1
+
+ if text_arr is not None:
+ font_file = font_files[platform.system()]
+ draw_str = ",".join(
+ [
+ f"drawtext=fontsize={text_size}:fontfile={font_file}:fontcolor={text_color}:text='{x}':x=10:y={h}+20"
+ for h, x in zip(start_h, text_arr)
+ ]
+ )
+ cmd = [
+ FFMPEG_PATH,
+ "-i",
+ tmp_file1,
+ "-y",
+ "-vf",
+ draw_str,
+ "-c:a",
+ "copy",
+ out_path,
+ ]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+ else:
+ os.rename(tmp_file1, out_path)
+
+ if os.path.exists(tmp_file1):
+ os.remove(tmp_file1)
+ if os.path.exists(tmp_file2):
+ os.remove(tmp_file2)
+
+
+def hstack_video_arr(
+ video_arr,
+ out_path,
+ crf=25,
+ verbose=True,
+ text_arr=None,
+ text_color="white",
+ text_size=60,
+):
+ assert len(video_arr) > 1
+ tmp_file1 = f"{osp.splitext(out_path)[0]}_tmp1.mp4"
+ tmp_file2 = f"{osp.splitext(out_path)[0]}_tmp2.mp4"
+
+ width = np.array([get_video_width_height(x)[0] for x in video_arr])
+ start_w = np.concatenate([np.array([0]), np.cumsum(width)[:-1]])
+
+ os.makedirs(osp.dirname(out_path), exist_ok=True)
+
+ for i in range(1, len(video_arr)):
+ prev_video = video_arr[0] if i == 1 else tmp_file1
+ cmd = [
+ FFMPEG_PATH,
+ "-y",
+ "-i",
+ prev_video,
+ "-i",
+ video_arr[i],
+ "-filter_complex",
+ "hstack,format=yuv420p",
+ "-vcodec",
+ "libx264",
+ "-crf",
+ f"{crf}",
+ tmp_file2,
+ ]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+ tmp_file1, tmp_file2 = tmp_file2, tmp_file1
+
+ if text_arr is not None:
+ font_file = font_files[platform.system()]
+ draw_str = ",".join(
+ [
+ f"drawtext=fontsize={text_size}:fontfile={font_file}:fontcolor={text_color}:text='{x}':x={w}+10:y=20"
+ for w, x in zip(start_w, text_arr)
+ ]
+ )
+ cmd = [
+ FFMPEG_PATH,
+ "-i",
+ tmp_file1,
+ "-y",
+ "-vf",
+ draw_str,
+ "-c:a",
+ "copy",
+ out_path,
+ ]
+ if not verbose:
+ cmd += ["-hide_banner", "-loglevel", "error"]
+ subprocess.run(cmd)
+ else:
+ os.rename(tmp_file1, out_path)
+
+ if os.path.exists(tmp_file1):
+ os.remove(tmp_file1)
+ if os.path.exists(tmp_file2):
+ os.remove(tmp_file2)
+
+
+def make_checker_board_texture(
+ color1="black", color2="white", width=10, height=10, n_tile=15, to_bgr=False
+):
+ c1 = np.asarray(ImageColor.getcolor(color1, "RGB")).astype(np.uint8)
+ c2 = np.asarray(ImageColor.getcolor(color2, "RGB")).astype(np.uint8)
+ if to_bgr:
+ c1 = c1[[2, 1, 0]]
+ c2 = c2[[2, 1, 0]]
+ hw = width // 2
+ hh = height // 2
+ c1_block = np.tile(c1, (hh, hw, 1))
+ c2_block = np.tile(c2, (hh, hw, 1))
+ tex = np.block([[[c1_block], [c2_block]], [[c2_block], [c1_block]]])
+ tex = np.tile(tex, (n_tile, n_tile, 1))
+ return tex
+
+
+def resize_bbox(bbox, scale):
+ x1, y1, x2, y2 = bbox[..., 0], bbox[..., 1], bbox[..., 2], bbox[..., 3]
+ h, w = y2 - y1, x2 - x1
+ cx, cy = x1 + 0.5 * w, y1 + 0.5 * h
+ h_new, w_new = h * scale, w * scale
+ x1_new, x2_new = cx - 0.5 * w_new, cx + 0.5 * w_new
+ y1_new, y2_new = cy - 0.5 * h_new, cy + 0.5 * h_new
+ bbox_new = np.stack([x1_new, y1_new, x2_new, y2_new], axis=-1)
+ return bbox_new
+
+
+def nparray_to_vtk_matrix(array):
+ """Convert a numpy.ndarray to a vtk.vtkMatrix4x4"""
+ import vtk
+
+ matrix = vtk.vtkMatrix4x4()
+ for i in range(array.shape[0]):
+ for j in range(array.shape[1]):
+ matrix.SetElement(i, j, array[i, j])
+ return matrix
+
+
+def vtk_matrix_to_nparray(matrix):
+ """Convert a numpy.ndarray to a vtk.vtkMatrix4x4"""
+ array = np.zeros([4, 4])
+ for i in range(array.shape[0]):
+ for j in range(array.shape[1]):
+ array[i, j] = matrix.GetElement(i, j)
+ return array
+
+
+def random_color(seed):
+ """Random a color according to the input seed."""
+ random.seed(seed)
+ colors = sns.color_palette()
+ color = random.choice(colors)
+ return color
+
+
+def draw_tracks(
+ img, bbox, idx, score, thickness=2, font_scale=0.4, text_height=10, text_width=15
+):
+ # taken from mmtracking
+ x1, y1, x2, y2 = bbox.astype(np.int32)
+
+ # bbox
+ bbox_color = random_color(idx)
+ bbox_color = [int(255 * _c) for _c in bbox_color][::-1]
+ cv.rectangle(img, (x1, y1), (x2, y2), bbox_color, thickness=thickness)
+
+ # id
+ text = str(idx)
+ width = len(text) * text_width
+ img[y1 : y1 + text_height, x1 : x1 + width, :] = bbox_color
+ cv.putText(
+ img,
+ str(idx),
+ (x1, y1 + text_height - 2),
+ cv.FONT_HERSHEY_COMPLEX,
+ font_scale,
+ color=(0, 0, 0),
+ )
+
+ # score
+ text = "{:.02f}".format(score)
+ width = len(text) * text_width
+ img[y1 - text_height : y1, x1 : x1 + width, :] = bbox_color
+ cv.putText(
+ img, text, (x1, y1 - 2), cv.FONT_HERSHEY_COMPLEX, font_scale, color=(0, 0, 0)
+ )
+ return img
+
+
+def draw_keypoints(img, keypoints, confidence, size=4, color=(255, 0, 255)):
+ for kp, conf in zip(keypoints, confidence):
+ if conf > 0.2:
+ cv.circle(
+ img, np.round(kp).astype(int).tolist(), size, color=color, thickness=-1
+ )
+ return img
diff --git a/genmo/utils/vis/vis_scenepic.py b/genmo/utils/vis/vis_scenepic.py
new file mode 100644
index 0000000000000000000000000000000000000000..91ee7edbeea6658d68b78928f6f2b1083b778bc7
--- /dev/null
+++ b/genmo/utils/vis/vis_scenepic.py
@@ -0,0 +1,465 @@
+import os
+import sys
+
+import numpy as np
+import scenepic as sp
+import torch
+from matplotlib import cm
+
+from genmo.utils.torch_transform import quat_between_two_vec, quaternion_to_angle_axis
+from genmo.utils.vis.vis import make_checker_board_texture
+from third_party.GVHMR.hmr4d.utils.smplx_utils import SMPL
+
+sys.path.append(os.path.join(os.getcwd()))
+
+
+class SkeletonActor:
+ def __init__(
+ self,
+ scene,
+ name,
+ joint_parents,
+ joint_color="Yellow",
+ bone_color="Green",
+ joint_constr_color="Cyan",
+ joint_radius=0.06,
+ bone_radius=0.04,
+ joint_constr_radius=0.1,
+ ):
+ self.scene = scene
+ self.name = name
+ self.joint_parents = joint_parents
+ self.joint_radius = joint_radius
+ self.joint_color = getattr(sp.Colors, joint_color, sp.Colors.Green)
+ self.bone_color = getattr(sp.Colors, bone_color, sp.Colors.Green)
+ self.joint_constr_color = getattr(
+ sp.Colors, joint_constr_color, sp.Colors.Green
+ )
+ self.bone_radius = bone_radius
+ self.joint_meshes = []
+ self.joint_constr_meshes = []
+ self.bone_meshes = []
+ self.bone_pairs = []
+
+ self.floor_img = scene.create_image(image_id="floor")
+ self.floor_img.from_numpy(make_checker_board_texture("#81C6EB", "#D4F1F7"))
+ self.floor_mesh = scene.create_mesh(texture_id="floor", layer_id="floor")
+ self.floor_mesh.add_image(transform=sp.Transforms.Scale(20))
+
+ for j, pa in enumerate(self.joint_parents):
+ # joint
+ joint_mesh = scene.create_mesh(f"{name}_joint{j}", layer_id=f"{self.name}")
+ joint_mesh.add_sphere(
+ color=self.joint_color, transform=sp.Transforms.scale(joint_radius)
+ )
+ self.joint_meshes.append(joint_mesh)
+ # joint constraints
+ joint_constr_mesh = scene.create_mesh(
+ f"{name}_joint_constr{j}", layer_id=f"{self.name}"
+ )
+ joint_constr_mesh.add_sphere(
+ color=self.joint_constr_color,
+ transform=sp.Transforms.scale(joint_constr_radius),
+ )
+ self.joint_constr_meshes.append(joint_constr_mesh)
+ # bone
+ if pa >= 0:
+ bone_mesh = scene.create_mesh(
+ f"{name}_bone{j}", layer_id=f"{self.name}"
+ )
+ bone_mesh.add_cone(
+ color=self.bone_color,
+ transform=sp.Transforms.scale(
+ np.array([1, joint_radius, joint_radius])
+ ),
+ )
+ self.bone_meshes.append(bone_mesh)
+ self.bone_pairs.append((j, pa, bone_mesh))
+
+ def add_mesh_to_frames(self, sp_frame, jpos):
+ sp_frame.add_mesh(self.floor_mesh)
+ # joint
+ for j, pos in enumerate(jpos):
+ sp_frame.add_mesh(
+ self.joint_meshes[j], transform=sp.Transforms.translate(pos)
+ )
+
+ # bone
+ vec = []
+ for j, pa, _ in self.bone_pairs:
+ vec.append((jpos[j] - jpos[pa]))
+ vec = np.stack(vec)
+ dist = np.linalg.norm(vec, axis=-1)
+ vec = torch.tensor(vec / dist[..., None])
+ aa = quaternion_to_angle_axis(
+ quat_between_two_vec(torch.tensor([-1.0, 0.0, 0.0]).expand_as(vec), vec)
+ ).numpy()
+ angle = np.linalg.norm(aa, axis=-1, keepdims=True)
+ axis = aa / (angle + 1e-6)
+
+ for (j, pa, bone_mesh), angle_i, axis_i, dist_i in zip(
+ self.bone_pairs, angle, axis, dist
+ ):
+ transform = sp.Transforms.translate((jpos[pa] + jpos[j]) * 0.5)
+ transform = transform @ sp.Transforms.RotationMatrixFromAxisAngle(
+ axis_i, angle_i
+ )
+ transform = transform @ sp.Transforms.Scale(np.array([dist_i, 1, 1]))
+ sp_frame.add_mesh(bone_mesh, transform=transform)
+
+ def add_joint_constr_meshes_to_frames(self, sp_frame, jpos, joint_mask):
+ # joint constraints
+ for j, pos in enumerate(jpos):
+ if joint_mask[j].any():
+ sp_frame.add_mesh(
+ self.joint_constr_meshes[j], transform=sp.Transforms.translate(pos)
+ )
+
+
+class ScenepicVisualizer:
+ def __init__(
+ self,
+ smpl_model_dir=None,
+ device=torch.device("cpu"),
+ show_skeleton_jpos=True,
+ show_ik_smpl_pose=False,
+ **kwargs,
+ ):
+ super().__init__(**kwargs)
+ self.smpl_dict = {
+ "neutral": SMPL(smpl_model_dir, create_transl=False, gender="neutral").to(
+ device
+ ),
+ "male": SMPL(smpl_model_dir, create_transl=False, gender="male").to(device),
+ "female": SMPL(smpl_model_dir, create_transl=False, gender="female").to(
+ device
+ ),
+ }
+ smpl = self.smpl_dict["male"]
+ faces = smpl.faces.copy()
+ self.smpl_faces = faces = np.hstack([np.ones_like(faces[:, [0]]) * 3, faces])
+ self.smpl_joint_parents = smpl.parents.cpu().numpy()
+ self.device = device
+ self.color_sequences = [
+ ["Yellow", "Green", "Teal"],
+ ["Yellow", "Red", "Teal"],
+ ["Yellow", "Blue", "Teal"],
+ ["Yellow", "Purple", "Teal"],
+ ["Yellow", "Orange", "Teal"],
+ ]
+ self.show_skeleton_jpos = show_skeleton_jpos
+ self.show_ik_smpl_pose = show_ik_smpl_pose
+
+ def load_default_camera(self):
+ return sp.Camera(
+ center=(5, 0, 1.5),
+ look_at=(0, 0, 0.8),
+ up_dir=(0, 0, 1),
+ fov_y_degrees=45.0,
+ aspect_ratio=1.0,
+ far_crop_distance=1000.0,
+ )
+
+ def vis_smpl_scene(self, smpl_seq=None, html_path=None, window_size=(400, 400)):
+ scene = self.generate_smpl_scene(smpl_seq, window_size=window_size)
+ scene.save_as_html(html_path)
+
+ def generate_smpl_scene(self, smpl_seq=None, window_size=None):
+ scene = sp.Scene()
+ main = scene.create_canvas_3d(width=window_size[0], height=window_size[1])
+
+ if "pose" in smpl_seq or "joints_pos" in smpl_seq: # single person
+ smpl_seq = {"skel0": smpl_seq}
+ smpl_seq = {
+ k: v.copy() for k, v in smpl_seq.items()
+ } # copy to avoid inplace modification
+
+ num_fr = -1
+ for i, (skel_name, pose_dict) in enumerate(smpl_seq.items()):
+ colors = self.color_sequences[i % len(self.color_sequences)]
+ normal_shape_len = {"pose": 2, "trans": 2, "shape": 2, "joints_pos": 3}
+ for key in ["pose", "trans", "shape", "joints_pos"]:
+ if (
+ key in pose_dict
+ and len(pose_dict[key].shape) > normal_shape_len[key]
+ ):
+ pose_dict[key] = pose_dict[key][0]
+
+ if self.show_ik_smpl_pose and "pose" in pose_dict:
+ pose_dict["skeleton_fk"] = SkeletonActor(
+ scene,
+ skel_name,
+ self.smpl_joint_parents,
+ joint_color=colors[0],
+ bone_color=colors[1],
+ )
+
+ pose = pose_dict["pose"].to(self.device)
+ trans = pose_dict["trans"].to(self.device)
+ shape = pose_dict["shape"].to(self.device)
+ num_fr = max(num_fr, pose.shape[0])
+
+ # print(pose[..., :3].view(-1, 3))
+ gender = pose_dict.get("gender", "neutral")
+ smpl_motion = self.smpl_dict[gender](
+ global_orient=pose[..., :3],
+ body_pose=pose[..., 3:],
+ betas=shape,
+ root_trans=trans,
+ return_full_pose=True,
+ orig_joints=True,
+ )
+ smpl_joints = smpl_motion.joints
+ pose_dict["joints_fk"] = smpl_joints
+ if "offset" in pose_dict:
+ pose_dict["joints_fk"] = pose_dict["joints_fk"] + pose_dict[
+ "offset"
+ ].to(self.device)
+
+ if "joints_pos" in pose_dict:
+ num_fr = max(num_fr, pose_dict["joints_pos"].shape[0])
+ if "offset" in pose_dict:
+ pose_dict["joints_pos"] = pose_dict["joints_pos"] + pose_dict[
+ "offset"
+ ].to(self.device)
+ pose_dict["skeleton_jpos"] = SkeletonActor(
+ scene,
+ f"{skel_name}_jpos",
+ self.smpl_joint_parents,
+ joint_color=colors[0],
+ bone_color=colors[2] if "skeleton_fk" in pose_dict else colors[1],
+ joint_constr_color="Brown" if skel_name == "gt" else "Cyan",
+ )
+ if "keyframe_idx" in pose_dict:
+ pose_dict["skeleton_keyframe_jpos"] = SkeletonActor(
+ scene,
+ f"{skel_name}_keyframe_jpos",
+ self.smpl_joint_parents,
+ joint_color="Orange",
+ bone_color="Yellow",
+ )
+ if "target_jpos" in pose_dict:
+ pose_dict["skeleton_keyframe_jpos_unknownt"] = SkeletonActor(
+ scene,
+ f"{skel_name}_keyframe_jpos_unknownt",
+ self.smpl_joint_parents,
+ joint_color="Orange",
+ bone_color="Pink",
+ joint_constr_color="Purple",
+ )
+
+ if not self.show_skeleton_jpos:
+ main.set_layer_settings(
+ {
+ f"{skel_name}_jpos": {"filled": False}
+ for skel_name, pose_dict in smpl_seq.items()
+ if "skeleton_jpos" in pose_dict
+ }
+ )
+
+ frame_list = []
+ cam_list = {}
+ for fr in range(num_fr):
+ main_frame = main.create_frame()
+ frame_list.append(main_frame)
+ main_frame.camera = self.load_default_camera()
+
+ coord_ax = scene.create_mesh(layer_id="coord")
+ coord_ax.add_coordinate_axes()
+ main_frame.add_mesh(coord_ax)
+
+ for i, (skel_name, pose_dict) in enumerate(smpl_seq.items()):
+ colors = self.color_sequences[i % len(self.color_sequences)]
+ color_i = getattr(sp.Colors, colors[1], sp.Colors.Green)
+
+ if "skeleton_fk" in pose_dict:
+ ind = min(fr, pose_dict["joints_fk"].shape[0] - 1)
+ pose_dict["skeleton_fk"].add_mesh_to_frames(
+ main_frame, pose_dict["joints_fk"][ind].cpu().numpy()
+ )
+ if "skeleton_jpos" in pose_dict:
+ ind = min(fr, pose_dict["joints_pos"].shape[0] - 1)
+
+ if pose_dict.get("keyframe_idx", None) is not None and any(
+ [ind + i in pose_dict["keyframe_idx"] for i in [-1, 0, 1]]
+ ):
+ pose_dict["skeleton_keyframe_jpos"].add_mesh_to_frames(
+ main_frame, pose_dict["joints_pos"][ind].cpu().numpy()
+ )
+ else:
+ pose_dict["skeleton_jpos"].add_mesh_to_frames(
+ main_frame, pose_dict["joints_pos"][ind].cpu().numpy()
+ )
+
+ if "local_joints_mask" in pose_dict:
+ joint_pos = pose_dict["joints_pos"][ind, :22]
+ if pose_dict["mask_type"] == "random_feat_mask":
+ joint_mask = pose_dict["local_joints_mask"][ind] > 0.0
+ else:
+ joint_mask = (
+ sum(
+ [
+ pose_dict["local_joints_mask"][
+ max(
+ 0,
+ min(
+ ind + i,
+ pose_dict[
+ "local_joints_mask"
+ ].shape[0]
+ - 1,
+ ),
+ )
+ ]
+ for i in [-1, 0, 1]
+ ]
+ )
+ > 0.0
+ )
+ joint_mask = joint_mask.view(-1, 3)
+ if joint_mask.any():
+ pose_dict[
+ "skeleton_jpos"
+ ].add_joint_constr_meshes_to_frames(
+ main_frame,
+ joint_pos.cpu().numpy(),
+ joint_mask.cpu().numpy(),
+ )
+
+ if "T_w2c" in pose_dict:
+ if skel_name not in cam_list:
+ cam_list[skel_name] = []
+ color = cm.jet(int((fr / num_fr) * 255))[:3]
+
+ ind = min(fr, pose_dict["T_w2c"].shape[0] - 1)
+ T_w2c = pose_dict["T_w2c"][ind].detach().cpu().numpy()
+ cam_intrinsics = None
+ cam_R = T_w2c[:3, :3]
+ cam_t = T_w2c[:3, 3]
+ cam_pos = -(cam_R.T @ cam_t[..., None])[..., 0]
+ cam_mesh = scene.create_mesh(
+ shared_color=sp.Color(1.0, 0.0, 0.0),
+ layer_id=f"cam_pos_{skel_name}",
+ )
+ cam_mesh.add_sphere(
+ transform=sp.Transforms.Scale(0.1), color=color
+ )
+ cam_mesh.enable_instancing(positions=cam_pos, colors=color)
+ main_frame.add_mesh(cam_mesh)
+
+ cam_ext = T_w2c.copy()
+ cam_ext[1:3, :] *= -1
+ cam = self.load_camera_from_ext_int(cam_ext, cam_intrinsics)
+ main_frame.camera = cam
+ cam_frustum = scene.create_mesh(layer_id=f"camera_{skel_name}")
+ cam_frustum_fr = scene.create_mesh(
+ layer_id=f"camera_{skel_name}_fr"
+ )
+ cam_frustum.add_camera_frustum(cam, color=color)
+ cam_frustum_fr.add_camera_frustum(cam, color=color_i)
+ # cam_list[skel_name].append(cam_mesh)
+ if fr % 10 == 0:
+ if "vis_all_cam" in pose_dict and pose_dict["vis_all_cam"]:
+ cam_list[skel_name].append(cam_frustum)
+
+ main_frame.add_mesh(cam_frustum_fr)
+ # if 'text' in smpl_seq:
+ # label = scene.create_label(text=smpl_seq['text'], color=sp.Colors.White, size_in_pixels=60, offset_distance=0.0, horizontal_align='center', camera_space=True)
+ # main_frame.add_label(label=label, position=[0.0, 1.5, -5.0])
+ if len(cam_list) > 0:
+ for frame in frame_list:
+ for skel_name in cam_list:
+ for cam in cam_list[skel_name]:
+ frame.add_mesh(cam)
+
+ return scene
+
+ def load_camera_from_ext_int(self, cam_ext, cam_intrinsics=None):
+ if cam_intrinsics is None:
+ vfovy = 50
+ aspect_ratio = 1
+
+ img_width = 1000
+ img_height = 1000
+ f_fullframe = 43.3
+ diag_fullframe = (24**2 + 36**2) ** 0.5
+ diag_img = (img_width**2 + img_height**2) ** 0.5
+ fy = diag_img / diag_fullframe * f_fullframe
+ vfovy = 2.0 * np.arctan(img_height / (2.0 * fy))
+ aspect_ratio = img_width / img_height
+ else:
+ img_width = cam_intrinsics[0, 2] * 2
+ img_height = cam_intrinsics[1, 2] * 2
+ fy = cam_intrinsics[1, 1]
+ # vfov = 2. * np.arctan(np.sqrt(img_height ** 2 + img_width ** 2) / (2. * fy))
+ vfovy = 2.0 * np.arctan(img_height / (2.0 * fy))
+ # vfov = 55.0 * math.pi / 180.0
+ aspect_ratio = img_width / img_height
+
+ # c2w = np.eye(4)
+ # R_c2w[:3, :3] = cam_ext[:3, :3]
+ # t_c2w = cam_ext[:3, 3]
+
+ # cam = sp.Camera(
+ # center=translation[None],
+ # rotation=rotation,
+ # fov_y_degrees=float(np.degrees(vfovy)),
+ # aspect_ratio=aspect_ratio,
+ # far_crop_distance=200.0
+ # )
+ # world_to_camera = np.eye(4)
+ # world_to_camera[:3, :3] = R_c2w[:3, :3].T
+ # world_to_camera[:3, 3] = -np.dot(R_c2w[:3, :3].T, t_c2w)
+
+ cam = sp.Camera(
+ world_to_camera=cam_ext,
+ fov_y_degrees=float(np.degrees(vfovy)),
+ aspect_ratio=aspect_ratio,
+ far_crop_distance=200.0,
+ )
+
+ return cam
+
+
+if __name__ == "__main__":
+ sp_visualizer = ScenepicVisualizer("data/smpl_data", device="cuda")
+
+ smpl_pose, smpl_trans = torch.load(f"out/smpl_seq.pt")
+ ind = 0
+ smpl_seq = {
+ "pose": smpl_pose[ind],
+ "trans": smpl_trans[ind],
+ "shape": torch.zeros_like(smpl_pose[ind, :, :10]),
+ "gender": "male",
+ }
+ smpl_seq2 = smpl_seq.copy()
+ # smpl_seq2['trans'] = smpl_seq2['trans'].clone()
+ # smpl_seq2['trans'][..., 1] += 0.5
+ smpl_seq2["offset"] = torch.tensor([0.0, 0.8, 0.0])
+
+ smpl_seq_all = {"skel0": smpl_seq, "skel1": smpl_seq2}
+
+ # smpl_seq1 = {
+ # 'pose': torch.zeros((80, 72)),
+ # 'trans': torch.zeros((80, 3)),
+ # 'shape': torch.zeros((80, 10)),
+ # 'gender': 'male'
+ # }
+
+ # smpl_seq2 = {
+ # 'pose': torch.rand((1, 72)).expand_as(smpl_seq1['pose']),
+ # 'trans': torch.zeros((80, 3)),
+ # 'shape': torch.zeros((80, 10)),
+ # 'gender': 'male'
+ # }
+
+ # smpl_seq_interp = {}
+ # for key in smpl_seq1:
+ # if key in {'gender'}:
+ # continue
+ # smpl_seq_interp[key] = torch.zeros_like(smpl_seq1[key])
+ # for fr in range(smpl_seq1['pose'].shape[0]):
+ # s = fr / (smpl_seq1['pose'].shape[0] - 1)
+ # smpl_seq_interp[key][[fr]] = smpl_seq1[key][[fr]] * (1 - s) + smpl_seq2[key][[fr]] * s
+
+ sp_visualizer.vis_smpl_scene(smpl_seq_all, "out/test.html")
diff --git a/genmo/utils/vis_utils.py b/genmo/utils/vis_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..152524dd15e654ceafcc668c4500e145edcd2411
--- /dev/null
+++ b/genmo/utils/vis_utils.py
@@ -0,0 +1,827 @@
+import os
+
+import cv2
+import numpy as np
+import torch
+from einops import einsum
+from moviepy.editor import VideoFileClip
+
+from genmo.utils.geo_transform import apply_T_on_points, compute_T_ayfz2ay
+from genmo.utils.rotation_conversions import axis_angle_to_matrix
+from genmo.utils.video_io_utils import get_writer
+from genmo.utils.vis.renderer import (
+ Renderer,
+ get_global_cameras_static,
+ get_global_cameras_static_v2,
+ get_ground_params_from_points,
+)
+from genmo.utils.vis.vis_scenepic import ScenepicVisualizer
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import create_camera_sensor
+
+sp_visualizer = ScenepicVisualizer("./third_party/GVHMR/inputs/checkpoints/body_models/smpl", device="cuda")
+
+CRF = 23 # 17 is lossless, every +6 halves the mp4 size
+color_sequences = [
+ "Yellow",
+ "Green",
+ "Teal",
+ "Red",
+ "Blue",
+ "Purple",
+ "Orange",
+ "Pink",
+ "Brown",
+ "Gray",
+ "Black",
+ "White",
+]
+color_rgb = (
+ np.array(
+ [
+ [255, 255, 0],
+ [0, 255, 0],
+ [0, 255, 255],
+ [255, 0, 0],
+ [0, 0, 255],
+ [255, 0, 255],
+ [255, 165, 0],
+ [255, 20, 147],
+ [165, 42, 42],
+ [169, 169, 169],
+ [0, 0, 0],
+ [255, 255, 255],
+ ]
+ )
+ / 255.0
+)
+
+
+def move_to_start_point_face_z(verts, J_regressor):
+ "XZ to origin, Start from the ground, Face-Z"
+ # position
+ verts = verts.clone() # (L, V, 3)
+ offset = einsum(J_regressor, verts[0], "j v, v i -> j i")[0] # (3)
+ offset[1] = verts[:, :, [1]].min()
+ verts = verts - offset
+ # face direction
+ T_ay2ayfz = compute_T_ayfz2ay(
+ einsum(J_regressor, verts[[0]], "j v, l v i -> l j i"), inverse=True
+ )
+ verts = apply_T_on_points(verts, T_ay2ayfz)
+ return verts
+
+
+def visualize_smpl_scene(
+ vis_type, index, vid, j3d, gt_j3d, transform_mode=None, keyframes=None
+):
+ if transform_mode == "global":
+ global_rot = axis_angle_to_matrix(torch.tensor([np.pi / 2, 0, 0])).cuda()
+ j3d = (global_rot @ j3d.transpose(1, 2)).transpose(1, 2)
+ if gt_j3d is not None:
+ gt_j3d = (global_rot @ gt_j3d.transpose(1, 2)).transpose(1, 2)
+ elif transform_mode == "local":
+ global_rot = axis_angle_to_matrix(torch.tensor([-np.pi / 2, 0, 0])).cuda()
+ j3d = (global_rot @ j3d.transpose(1, 2)).transpose(1, 2)
+ j3d[..., 2] += 0.8
+ if gt_j3d is not None:
+ gt_j3d = (global_rot @ gt_j3d.transpose(1, 2)).transpose(1, 2)
+ gt_j3d[..., 2] += 0.8
+ smpl_seq = {
+ "pred": {
+ "joints_pos": j3d,
+ },
+ }
+ if keyframes is not None:
+ smpl_seq["pred"]["keyframe_idx"] = keyframes.tolist()
+ if gt_j3d is not None:
+ smpl_seq["gt"] = {
+ "joints_pos": gt_j3d,
+ }
+ vid_ = vid.replace("/", "_")
+ fname = f"{index:03d}-{vid_}"
+ if len(fname) > 100:
+ fname = fname[:100]
+ html_file = f"out/{vis_type}/{fname}.html"
+ os.makedirs(os.path.dirname(html_file), exist_ok=True)
+ sp_visualizer.vis_smpl_scene(smpl_seq, html_file)
+ return {}
+
+
+def visualize_intermediate_smpl_scene(
+ vis_type, index, vid, j3d_list, gt_j3d, transform_mode=None
+):
+ if transform_mode == "global":
+ global_rot = axis_angle_to_matrix(torch.tensor([np.pi / 2, 0, 0])).cuda()
+ j3d_list = [
+ (global_rot @ j3d.transpose(1, 2)).transpose(1, 2) for j3d in j3d_list
+ ]
+ gt_j3d = (global_rot @ gt_j3d.transpose(1, 2)).transpose(1, 2)
+ elif transform_mode == "local":
+ global_rot = axis_angle_to_matrix(torch.tensor([-np.pi / 2, 0, 0])).cuda()
+ j3d_list = [
+ (global_rot @ j3d.transpose(1, 2)).transpose(1, 2) for j3d in j3d_list
+ ]
+ gt_j3d = (global_rot @ gt_j3d.transpose(1, 2)).transpose(1, 2)
+ for i in range(len(j3d_list)):
+ j3d_list[i][..., 2] = j3d_list[i][..., 2] + 0.8
+ gt_j3d[..., 2] += 0.8
+ smpl_seq = {}
+ for i, j3d in enumerate(j3d_list):
+ smpl_seq[f"pred_{i}"] = {
+ "joints_pos": j3d,
+ }
+
+ vid_ = vid.replace("/", "_")
+ fname = f"{index:03d}-{vid_}"
+ html_file = f"out/{vis_type}/{fname}.html"
+ os.makedirs(os.path.dirname(html_file), exist_ok=True)
+ sp_visualizer.vis_smpl_scene(smpl_seq, html_file)
+ return {}
+
+
+def visualize_smplmesh_scene(
+ vis_type, index, vid, pred_ay_verts, gt_ay_verts, J_regressor, faces_smpl
+):
+ verts_glob = move_to_start_point_face_z(pred_ay_verts, J_regressor)
+ joints_glob = einsum(J_regressor, verts_glob, "j v, l v i -> l j i") # (L, J, 3)
+ length = gt_ay_verts.shape[0]
+ render_length = min(length, 200)
+ verts_glob = verts_glob[:render_length]
+ joints_glob = joints_glob[:render_length]
+ global_R, global_T, global_lights = get_global_cameras_static(
+ verts_glob.cpu(),
+ beta=2.0,
+ cam_height_degree=20,
+ target_center_height=1.0,
+ )
+
+ vid_ = vid.replace("/", "_")
+ fname = f"{index:03d}-{vid_}"
+ global_video_path = f"out/{vis_type}_video/{fname}.mp4"
+ os.makedirs(os.path.dirname(global_video_path), exist_ok=True)
+ # length, width, height = get_video_lwh(global_video_path)
+ width, height = 512, 512
+ _, _, K = create_camera_sensor(width, height, 24) # render as 24mm lens
+
+ # renderer
+ renderer = Renderer(width, height, device="cuda", faces=faces_smpl, K=K, bin_size=0)
+
+ # -- render mesh -- #
+ scale, cx, cz = get_ground_params_from_points(joints_glob[:, 0], verts_glob)
+ renderer.set_ground(scale * 1.5, cx, cz)
+ color = torch.ones(3).float().cuda() * 0.8
+
+ writer = get_writer(global_video_path, fps=30, crf=CRF)
+ for i in range(render_length):
+ cameras = renderer.create_camera(global_R[i], global_T[i])
+ img = renderer.render_with_ground(
+ verts_glob[[i]], color[None], cameras, global_lights
+ )
+ writer.write_frame(img)
+ writer.close()
+
+
+def visualize_intermediate_smplmesh_scene(
+ vis_type, index, vid, pred_ay_verts_list, gt_ay_verts, J_regressor, faces_smpl
+):
+ verts_glob_list = [
+ move_to_start_point_face_z(pred_ay_verts, J_regressor)
+ for pred_ay_verts in pred_ay_verts_list
+ ]
+ joints_glob_list = [
+ einsum(J_regressor, verts_glob, "j v, l v i -> l j i")
+ for verts_glob in verts_glob_list
+ ]
+ length = gt_ay_verts.shape[0]
+ render_length = min(length, 200)
+ verts_glob_list = [verts_glob[:render_length] for verts_glob in verts_glob_list]
+ joints_glob_list = [joints_glob[:render_length] for joints_glob in joints_glob_list]
+ global_R, global_T, global_lights = get_global_cameras_static(
+ verts_glob_list[0].cpu(),
+ beta=2.0,
+ cam_height_degree=20,
+ target_center_height=1.0,
+ )
+
+ vid_ = vid.replace("/", "_")
+ fname = f"{index:03d}-{vid_}"
+ global_video_path = f"out/{vis_type}_video/{fname}.mp4"
+ os.makedirs(os.path.dirname(global_video_path), exist_ok=True)
+ # length, width, height = get_video_lwh(global_video_path)
+ width, height = 512, 512
+ _, _, K = create_camera_sensor(width, height, 24) # render as 24mm lens
+
+ # renderer
+ renderer = Renderer(width, height, device="cuda", faces=faces_smpl, K=K, bin_size=0)
+
+ # -- render mesh -- #
+ scale, cx, cz = get_ground_params_from_points(
+ joints_glob_list[0][:, 0], verts_glob_list[0]
+ )
+ renderer.set_ground(scale * 1.5, cx, cz)
+ # color = torch.ones(3).float().cuda() * 0.8
+ colors = torch.stack(
+ [
+ torch.from_numpy(color_rgb[i % len(color_rgb)]).float().cuda()
+ for i in range(len(verts_glob_list))
+ ],
+ dim=0,
+ )
+
+ writer = get_writer(global_video_path, fps=30, crf=CRF)
+ for i in range(render_length):
+ cameras = renderer.create_camera(global_R[i], global_T[i])
+ verts = torch.cat([verts_glob[[i]] for verts_glob in verts_glob_list], dim=0)
+ img = renderer.render_with_ground(
+ verts, colors, cameras, global_lights, opacity=float(i / render_length)
+ )
+ writer.write_frame(img)
+ writer.close()
+
+
+def visualize_intermediate_smplmesh_scene_img(
+ vis_type, index, vid, pred_ay_verts_list, gt_ay_verts, J_regressor, faces_smpl
+):
+ import open3d as o3d
+
+ from genmo.utils.vis.o3d_render import Settings, create_meshes, get_ground
+
+ verts_glob_list = [
+ move_to_start_point_face_z(pred_ay_verts, J_regressor)
+ for pred_ay_verts in pred_ay_verts_list
+ ]
+ joints_glob_list = [
+ einsum(J_regressor, verts_glob, "j v, l v i -> l j i")
+ for verts_glob in verts_glob_list
+ ]
+ length = verts_glob_list[0].shape[0]
+ render_length = min(length, 200)
+ verts_glob_list = [
+ verts_glob[:render_length] for verts_glob in verts_glob_list
+ ] # (N, T, V, 3)
+ joints_glob_list = [
+ joints_glob[:render_length] for joints_glob in joints_glob_list
+ ] # (N, T, J, 3)
+
+ device = verts_glob_list[0].device
+
+ vid_ = vid.replace("/", "_")
+ fname = f"{index:03d}-{vid_}"
+ global_video_path = f"out/{vis_type}_video/{fname}.mp4"
+ os.makedirs(os.path.dirname(global_video_path), exist_ok=True)
+ # length, width, height = get_video_lwh(global_video_path)
+ width, height = 640 * 4, 480 * 4
+ # _, _, K = create_camera_sensor(width, height, 24) # render as 24mm lens
+
+ renderer = o3d.visualization.rendering.OffscreenRenderer(width, height)
+ # mat_settings = Settings()
+
+ color_purple = torch.tensor([0.69019608, 0.39215686, 0.95686275]).to(device)
+ color_green = torch.tensor([0.46666667, 0.90196078, 0.74901961]).to(device)
+ color_light_purple = torch.tensor([1.0, 0.65490196, 0.95294118]).to(device)
+
+ # -- render mesh -- #
+ scale, cx, cz = get_ground_params_from_points(
+ joints_glob_list[0][:, 0], verts_glob_list[0]
+ )
+ ground_geometry = get_ground(scale * 1.5, cx, cz)
+ # color = torch.ones(3).float().cuda() * 0.8
+ colors = torch.stack(
+ [
+ torch.from_numpy(color_rgb[i % len(color_rgb)]).float().cuda()
+ for i in range(len(verts_glob_list))
+ ],
+ dim=0,
+ )
+
+ verts_glob_list = torch.stack(verts_glob_list, dim=0).transpose(
+ 1, 0
+ ) # (T, N, V, 3)
+ T, N, V, _ = verts_glob_list.shape
+
+ position, target, up = get_global_cameras_static_v2(
+ # verts_list[0].cpu(),
+ verts_glob_list.reshape(-1, V, 3).cpu().clone(),
+ beta=1.5,
+ cam_height_degree=20,
+ target_center_height=1.0,
+ )
+ camera = renderer.scene.camera
+ camera.look_at(target[:, None], position[:, None], up[:, None])
+
+ gv, gf, gc = ground_geometry
+ ground_mesh = create_meshes(gv, gf, gc[..., :3])
+ renderer.scene.add_geometry(
+ "mesh_ground", ground_mesh, o3d.visualization.rendering.MaterialRecord()
+ )
+
+ # trans_mat_box = mat_settings._materials[Settings.Transparency]
+ # lit_mat_box = mat_settings._materials[Settings.LIT]
+
+ colors = color_purple[None, :].repeat(T, 1)
+ colors[:, 0] = torch.linspace(color_green[0], color_purple[0], T)
+ colors[:, 1] = torch.linspace(color_green[1], color_purple[1], T)
+ colors[:, 2] = torch.linspace(color_green[2], color_purple[2], T)
+
+ colors_trans = torch.zeros_like(colors)
+ colors_trans[:, 0] = torch.linspace(color_green[0], color_light_purple[0], T)
+ colors_trans[:, 1] = torch.linspace(color_green[1], color_light_purple[1], T)
+ colors_trans[:, 2] = torch.linspace(color_green[2], color_light_purple[2], T)
+ faces = (
+ torch.from_numpy(faces_smpl.astype("int"))
+ .unsqueeze(0)
+ .to(device)
+ .expand(N, -1, -1)
+ )
+
+ # colors = torch.stack([torch.from_numpy(color_rgb[i % len(color_rgb)]).float().cuda() for i in range(len(verts_glob_list))], dim=0)
+
+ for i in range(N):
+ faces_list = list(torch.unbind(faces, dim=0)) # + [gf]
+ for t, verts in enumerate(verts_glob_list):
+ if t % 20 != 0:
+ continue
+ verts = list(torch.unbind(verts_glob_list[t], dim=0)) # + [gv]
+ mat = o3d.visualization.rendering.MaterialRecord()
+ mat.base_color = [0.9, 0.9, 0.9, (i + 1) / N]
+ mat.shader = Settings.Transparency
+ # mat.opacity = (i + 1) / N
+ mat.thickness = 1.0
+ mat.transmission = 1.0
+ mat.absorption_distance = 10
+ mat.absorption_color = [0.5, 0.5, 0.5]
+
+ # mesh = create_meshes(verts[i], faces_list[i], colors[t] if i == 0 else colors_trans[t])
+ mesh = create_meshes(verts[i], faces_list[i], colors[t])
+ renderer.scene.add_geometry(f"mesh_{i}_{t}", mesh, mat)
+ # renderer.scene.add_geometry(
+ # f"mesh_{i}_{t}", mesh, lit_mat_box if i == 0 else trans_mat_box
+ # )
+
+ img = renderer.render_to_image()
+ os.makedirs(os.path.dirname(f"out/{vis_type}_time/{fname}.png"), exist_ok=True)
+ # cv2.imwrite(f"out/{vis_type}_time/{fname}.png", img)
+ o3d.io.write_image(f"out/{vis_type}_time/{fname}.png", img)
+
+
+def visualize_smpl_scene_mp4(
+ vis_type,
+ index,
+ vid,
+ v3d,
+ gt_j3d,
+ faces_smpl,
+ J_regressor,
+ transform_mode=None,
+ keyframes=None,
+ audio_clip=None,
+):
+ verts_glob = move_to_start_point_face_z(v3d, J_regressor)
+
+ global_R, global_T, global_lights = get_global_cameras_static(
+ verts_glob.cpu(),
+ beta=2.0,
+ )
+ vid_ = vid.replace("/", "_")
+ fname = f"{index:03d}-{vid_}"
+ if len(fname) > 100:
+ fname = fname[:100]
+ video_path = f"out/{vis_type}_video/{fname}.mp4"
+ os.makedirs(os.path.dirname(video_path), exist_ok=True)
+
+ length = verts_glob.shape[0]
+ # length, width, height = get_video_lwh(video_path)
+ width, height = 512, 512
+ _, _, K = create_camera_sensor(width, height, 24) # render as 24mm lens
+
+ renderer = Renderer(width, height, device="cuda", faces=faces_smpl, K=K, bin_size=0)
+
+ # -- render mesh -- #
+ scale, cx, cz = get_ground_params_from_points(verts_glob[:, 0], verts_glob)
+ renderer.set_ground(scale * 1.5, cx, cz)
+ color = torch.ones(3).float().cuda() * 0.8
+
+ writer = get_writer(video_path, fps=30, crf=CRF)
+ for i in range(length):
+ cameras = renderer.create_camera(global_R[i], global_T[i])
+ img = renderer.render_with_ground(
+ verts_glob[[i]], color[None], cameras, global_lights
+ )
+ writer.write_frame(img)
+ writer.close()
+
+ if audio_clip is not None:
+ video_clip = VideoFileClip(video_path)
+ video_with_audio = video_clip.set_audio(audio_clip)
+ # video_clip.write_videofile(video_path, audio=True)
+ video_with_audio.write_videofile(
+ video_path.replace(".mp4", "_audio.mp4"), codec="libx264", audio_codec="aac"
+ )
+
+ # os.makedirs(os.path.dirname(f"out/{vis_type}_audio/{fname}.mp3"), exist_ok=True)
+ # audio_clip.write_audiofile(f"out/{vis_type}_audio/{fname}.mp3")
+
+ return {}
+
+
+def visualize_smplmesh_scene_img(
+ vis_type,
+ index,
+ vid,
+ pred_ay_verts,
+ gt_ay_verts,
+ J_regressor,
+ faces_smpl,
+ vid2=None,
+ start_ind=300,
+ end_ind=550,
+):
+ import open3d as o3d
+
+ from genmo.utils.vis.o3d_render import Settings, create_meshes, get_ground
+
+ verts_glob_list = move_to_start_point_face_z(pred_ay_verts, J_regressor)
+ joints_glob_list = einsum(J_regressor, verts_glob_list, "j v, l v i -> l j i")
+ length = verts_glob_list.shape[0]
+
+ render_length = min(length, 1600)
+ render_length = length
+ verts_glob_list = verts_glob_list[:render_length] # (T, V, 3)
+ joints_glob_list = joints_glob_list[:render_length] # (T, J, 3)
+
+ device = verts_glob_list.device
+ if vid2 is None:
+ vid2 = vid
+ vid_ = vid.replace("/", "_")
+ fname = f"{index:03d}-{vid_}"
+ global_video_path = f"out/{vis_type}_video/{fname}.mp4"
+ os.makedirs(os.path.dirname(global_video_path), exist_ok=True)
+ # length, width, height = get_video_lwh(global_video_path)
+ width, height = 640 * 4, 480 * 4
+ # _, _, K = create_camera_sensor(width, height, 24) # render as 24mm lens
+ # start_ind = 300
+ # end_ind =550
+
+ renderer = o3d.visualization.rendering.OffscreenRenderer(width, height)
+ mat_settings = Settings()
+
+ color_purple = torch.tensor([0.69019608, 0.39215686, 0.95686275]).to(device)
+ color_green = torch.tensor([0.46666667, 0.90196078, 0.74901961]).to(device)
+ color_light_purple = torch.tensor([1.0, 0.65490196, 0.95294118]).to(device)
+
+ all_colors = torch.load("colors.pth")
+ color_ids = [0, 1, 2, 3, 4, 11]
+ import ipdb
+
+ ipdb.set_trace()
+ print(all_colors[color_ids], color_purple)
+
+ # -- render mesh -- #
+ scale, cx, cz = get_ground_params_from_points(
+ joints_glob_list[:, 0], verts_glob_list
+ )
+ ground_geometry = get_ground(scale * 1.5, cx, cz)
+ # color = torch.ones(3).float().cuda() * 0.8
+
+ T, V, _ = verts_glob_list.shape
+
+ position, target, up = get_global_cameras_static_v2(
+ # verts_list[0].cpu(),
+ verts_glob_list.cpu().clone(),
+ beta=1.2,
+ cam_height_degree=15,
+ target_center_height=1.0,
+ vec_rot=-90,
+ )
+ camera = renderer.scene.camera
+ camera.look_at(target[:, None], position[:, None], up[:, None])
+
+ gv, gf, gc = ground_geometry
+ ground_mesh = create_meshes(gv, gf, gc[..., :3])
+ ground_mat = o3d.visualization.rendering.MaterialRecord()
+ ground_mat.base_color = [0.9, 0.9, 0.9, 1]
+ # ground_mat.shader = "defaultUnlit"
+ # ground_mesh.paint_uniform_color([1.0, 1.0, 1.0])
+ renderer.scene.add_geometry("mesh_ground", ground_mesh, ground_mat)
+
+ white_color = np.array([1.0, 1.0, 1.0], dtype=np.float32)
+ # light_position = np.array([10.0, 1.0, 0.0], dtype=np.float32)
+
+ renderer.scene.scene.add_directional_light(
+ "light", white_color, np.array([0, -0.5, -1]), 1e5, True
+ )
+ # renderer.scene.scene.add_point_light("light", white_color, light_position, 1e5, 1e2, True)
+
+ # trans_mat_box = mat_settings._materials[Settings.Transparency]
+ lit_mat_box = mat_settings._materials[Settings.LIT]
+
+ colors = color_purple[None, :].repeat(T, 1)
+ colors[:, 0] = torch.linspace(color_green[0], color_purple[0], T)
+ colors[:, 1] = torch.linspace(color_green[1], color_purple[1], T)
+ colors[:, 2] = torch.linspace(color_green[2], color_purple[2], T)
+
+ colors_trans = torch.zeros_like(colors)
+ colors_trans[:, 0] = torch.linspace(color_green[0], color_light_purple[0], T)
+ colors_trans[:, 1] = torch.linspace(color_green[1], color_light_purple[1], T)
+ colors_trans[:, 2] = torch.linspace(color_green[2], color_light_purple[2], T)
+ faces = torch.from_numpy(faces_smpl.astype("int")).to(device)
+
+ # colors = torch.stack([torch.from_numpy(color_rgb[i % len(color_rgb)]).float().cuda() for i in range(len(verts_glob_list))], dim=0)
+
+ # faces_list = list(torch.unbind(faces, dim=0)) # + [gf]
+ T_c2w = camera.get_view_matrix()
+ R_w2c = T_c2w[:3, :3].T
+ for t, verts in enumerate(verts_glob_list):
+ # if start_ind <= t <= end_ind and t % 60 != 0:
+ # continue
+ # if (t % 30 == 0 and (t < start_ind or t > end_ind)) or (t in [290, 330, 450, 480, 510, 540, 600, 660]):
+ if (t % 30 == 0 and (t <= start_ind or t >= end_ind)) or (t % 30 == 0):
+ if t >= length - 600 and not (t % 30 == 0):
+ continue
+ t_text = t + 600 - length
+ if 0 <= t_text < 30 or 120 < t_text <= 240:
+ continue
+
+ verts = verts_glob_list[t] # + [gv]
+ mat = o3d.visualization.rendering.MaterialRecord()
+ mat.base_color = [0.9, 0.9, 0.9, 0.3 + t * 0.7 / T]
+ mat.shader = Settings.Transparency
+ # mat.opacity = (i + 1) / N
+ mat.thickness = 1.0
+ mat.transmission = 1.0
+ mat.absorption_distance = 10
+ mat.absorption_color = [0.5, 0.5, 0.5]
+
+ # mesh = create_meshes(verts, faces, colors[t])
+ if t < 240:
+ set_color = all_colors[color_ids[0]]
+ elif t < 500:
+ set_color = all_colors[color_ids[1]]
+ elif t < 800:
+ set_color = all_colors[color_ids[2]]
+ elif t < 1050:
+ set_color = all_colors[color_ids[3]]
+ elif t < 1200:
+ set_color = all_colors[color_ids[4]]
+ else:
+ set_color = all_colors[color_ids[5]]
+ set_color = color_purple.cpu().numpy()
+
+ set_color = torch.from_numpy(set_color).float().to(device)
+ mesh = create_meshes(verts, faces, set_color)
+ renderer.scene.add_geometry(f"mesh_{t}", mesh, lit_mat_box)
+
+ # if (t % 120 == 0 and (t < start_ind or t > end_ind)) or (t in [210, 630]):
+ if (t % 120 == 0 and (t < start_ind or t > end_ind)) and False:
+ if t >= length - 600:
+ continue
+ if t > end_ind:
+ pid = vid2[:2]
+ sid = vid2[3:]
+ else:
+ pid = vid[:2]
+ sid = vid[3:]
+ img_path = f"/mnt/dhd/body-pose-dataset/EMDB/{pid}/{sid}/images/{t:05d}.jpg"
+ img = cv2.imread(img_path)
+ offset = verts.mean(dim=0).cpu().numpy()
+ offset[1] += 2
+
+ img_h, img_w = img.shape[:2]
+ # img = cv2.resize(img, (img_w // 8, img_h // 8))
+ img_mesh = convert_image_to_mesh(img[:, :, ::-1], offset, R_w2c)
+ # rotate the image mesh to face the camera
+
+ # img_mesh = o3d.geometry.TriangleMesh.create_box(
+ # width=1, height=0.1, depth=1
+ # )
+ # img_mesh.compute_vertex_normals()
+ img_mat = o3d.visualization.rendering.MaterialRecord()
+ # img_mat.base_color = [0.9, 0.9, 0.9, 1]
+ img_mat.shader = "defaultLit"
+ # img_mat.albedo_img = o3d.geometry.Image(img)
+ # img_mesh.paint_uniform_color([1, 1, 1])
+ # img_mat.texture = o3d.geometry.Image(img)
+ renderer.scene.add_geometry(f"img_mesh_{t}", img_mesh, img_mat)
+ # mesh = create_meshes(verts[i], faces_list[i], colors[t])
+ # renderer.scene.add_geometry(f"mesh_{i}_{t}", mesh, mat)
+
+ img = renderer.render_to_image()
+ os.makedirs(
+ os.path.dirname(f"out/{vis_type}_scene_time/{fname}.png"), exist_ok=True
+ )
+ # cv2.imwrite(f"out/{vis_type}_scene_time/{fname}.png", img)
+ o3d.io.write_image(f"out/{vis_type}_scene_time/{fname}.png", img)
+
+
+def visualize_smplmesh_scene_video(
+ vis_type,
+ index,
+ vid,
+ pred_ay_verts,
+ gt_ay_verts,
+ J_regressor,
+ faces_smpl,
+ vid2=None,
+ start_ind=300,
+ end_ind=550,
+ caption=None,
+):
+ import open3d as o3d
+ from tqdm import tqdm
+
+ from genmo.utils.vis.o3d_render import Settings, create_meshes, get_ground
+
+ verts_glob_list = move_to_start_point_face_z(pred_ay_verts, J_regressor)
+ joints_glob_list = einsum(J_regressor, verts_glob_list, "j v, l v i -> l j i")
+ length = verts_glob_list.shape[0]
+
+ render_length = min(length, 800)
+ verts_glob_list = verts_glob_list[:render_length] # (T, V, 3)
+ joints_glob_list = joints_glob_list[:render_length] # (T, J, 3)
+
+ if vid2 is None:
+ vid2 = vid
+ device = verts_glob_list.device
+ vid_ = vid.replace("/", "_")
+ vid2_ = vid2.replace("/", "_")
+ fname = f"{index:03d}-{vid_}-{vid2_}"
+ if caption is not None:
+ fname = (
+ f"{fname}-{caption.replace(' ', '_').replace('/', '_').replace(',', '_')}"
+ )
+ fname = fname[:100]
+ # global_video_path = f"out/{vis_type}_video/{fname}.mp4"
+ global_video_path = f"out/v1-t-v2/{fname}.mp4"
+ if os.path.exists(global_video_path):
+ rand_idx = torch.randint(0, 1000000, (1,)).item()
+ global_video_path = global_video_path.replace(".mp4", f"_{rand_idx}.mp4")
+ os.makedirs(os.path.dirname(global_video_path), exist_ok=True)
+ writer = get_writer(global_video_path, fps=30, crf=CRF)
+
+ # length, width, height = get_video_lwh(global_video_path)
+ width, height = 640 * 4, 480 * 4
+ # _, _, K = create_camera_sensor(width, height, 24) # render as 24mm lens
+
+ mat_settings = Settings()
+
+ color_purple = torch.tensor([0.69019608, 0.39215686, 0.95686275]).to(device)
+ color_green = torch.tensor([0.46666667, 0.90196078, 0.74901961]).to(device)
+ color_light_purple = torch.tensor([1.0, 0.65490196, 0.95294118]).to(device)
+
+ # -- render mesh -- #
+ scale, cx, cz = get_ground_params_from_points(
+ joints_glob_list[:, 0], verts_glob_list
+ )
+ ground_geometry = get_ground(scale * 1.5, cx, cz)
+ # color = torch.ones(3).float().cuda() * 0.8
+
+ T, V, _ = verts_glob_list.shape
+
+ position, target, up = get_global_cameras_static_v2(
+ # verts_list[0].cpu(),
+ verts_glob_list.cpu().clone(),
+ beta=1.3,
+ cam_height_degree=20,
+ target_center_height=1.0,
+ )
+
+ # trans_mat_box = mat_settings._materials[Settings.Transparency]
+ lit_mat_box = mat_settings._materials[Settings.LIT]
+
+ colors = color_purple[None, :].repeat(T, 1)
+ colors[:, 0] = torch.linspace(color_green[0], color_purple[0], T)
+ colors[:, 1] = torch.linspace(color_green[1], color_purple[1], T)
+ colors[:, 2] = torch.linspace(color_green[2], color_purple[2], T)
+
+ colors_trans = torch.zeros_like(colors)
+ colors_trans[:, 0] = torch.linspace(color_green[0], color_light_purple[0], T)
+ colors_trans[:, 1] = torch.linspace(color_green[1], color_light_purple[1], T)
+ colors_trans[:, 2] = torch.linspace(color_green[2], color_light_purple[2], T)
+ faces = torch.from_numpy(faces_smpl.astype("int")).to(device)
+
+ # colors = torch.stack([torch.from_numpy(color_rgb[i % len(color_rgb)]).float().cuda() for i in range(len(verts_glob_list))], dim=0)
+ renderer = o3d.visualization.rendering.OffscreenRenderer(width, height)
+
+ camera = renderer.scene.camera
+ camera.look_at(target[:, None], position[:, None], up[:, None])
+
+ gv, gf, gc = ground_geometry
+ ground_mesh = create_meshes(gv, gf, gc[..., :3])
+ renderer.scene.add_geometry(
+ "mesh_ground", ground_mesh, o3d.visualization.rendering.MaterialRecord()
+ )
+ # faces_list = list(torch.unbind(faces, dim=0)) # + [gf]
+
+ T_c2w = camera.get_view_matrix()
+ R_w2c = T_c2w[:3, :3].T
+ for t, verts in tqdm(enumerate(verts_glob_list)):
+ verts = verts_glob_list[t] # + [gv]
+ mat = o3d.visualization.rendering.MaterialRecord()
+ mat.base_color = [0.9, 0.9, 0.9, 0.3 + t * 0.7 / T]
+ mat.shader = Settings.Transparency
+ # mat.opacity = (i + 1) / N
+ mat.thickness = 1.0
+ mat.transmission = 1.0
+ mat.absorption_distance = 10
+ mat.absorption_color = [0.5, 0.5, 0.5]
+
+ mesh = create_meshes(verts, faces, colors[t])
+ if t > 0:
+ renderer.scene.remove_geometry(f"mesh_{t - 1}")
+ if t - 1 < start_ind or t - 1 > end_ind:
+ renderer.scene.remove_geometry(f"img_mesh_{t - 1}")
+ renderer.scene.add_geometry(f"mesh_{t}", mesh, lit_mat_box)
+ # mesh = create_meshes(verts[i], faces_list[i], colors[t])
+ # renderer.scene.add_geometry(f"mesh_{i}_{t}", mesh, mat)
+
+ if t < start_ind or t > end_ind:
+ if t > end_ind:
+ pid = vid2[:2]
+ sid = vid2[3:]
+ else:
+ pid = vid[:2]
+ sid = vid[3:]
+ img_path = f"/mnt/dhd/body-pose-dataset/EMDB/{pid}/{sid}/images/{t:05d}.jpg"
+ input_img = cv2.imread(img_path)
+ offset = verts.mean(dim=0).cpu().numpy()
+ offset[1] += 1.5
+
+ img_h, img_w = input_img.shape[:2]
+ input_img = cv2.resize(input_img, (img_w // 4, img_h // 4))
+ img_mesh = convert_image_to_mesh(input_img[:, :, ::-1], offset, R_w2c)
+ renderer.scene.add_geometry(
+ f"img_mesh_{t}", img_mesh, mat_settings._materials[Settings.LIT]
+ )
+ img = renderer.render_to_image()
+ # import ipdb; ipdb.set_trace()
+ # o3d.io.write_image("out/tmp.png", img)
+ # img = cv2.imread("out/tmp.png")
+ writer.write_frame(np.array(img))
+ writer.close()
+ return global_video_path
+ # os.makedirs(
+ # os.path.dirname(f"out/{vis_type}_scene_time/{fname}.png"), exist_ok=True
+ # )
+ # # cv2.imwrite(f"out/{vis_type}_scene_time/{fname}.png", img)
+
+
+def convert_image_to_mesh(img, offset, R_c2w):
+ import open3d as o3d
+
+ img = np.asarray(img)
+
+ # Instead of backprojecting, just convert img to an actual 3D plane with Z=0
+
+ # Create 3D vertex for each pixel location
+ xvalues = np.arange(img.shape[1])
+ yvalues = np.arange(img.shape[0])[::-1].copy()
+ x_loc, y_loc = np.meshgrid(xvalues, yvalues)
+ z_loc = np.zeros_like(x_loc)
+
+ # Scale down before making 3D vertices
+ x_loc = x_loc / xvalues.shape[0] * 1.5
+ y_loc = (
+ y_loc / xvalues.shape[0] * 1.5
+ ) # Keep aspect ratio same by dividing with same denominator. Now image width is 1 meter in 3d.
+
+ vertices = np.stack((x_loc, y_loc, z_loc), axis=2).reshape(-1, 3)
+ vertices = np.matmul(R_c2w, vertices.T).T
+ vertices = vertices + offset[None]
+
+ vertex_colors = img.reshape(-1, 3) / 255.0
+
+ # Create triangles between each pair of neighboring vertices
+ # Connect positions (i,j), (i+1,j) and (i,j+1) to make one triangle and (i, j+1), (i+1,j) and (i+1,j+1) to make
+ # another triangle.
+ # Pixel (i,j) is in vertices array at location i + j*xvalues.shape[0]
+
+ vertex_positions = np.arange(xvalues.size * yvalues.size)
+ # Reshape into 2D grid and discard last row and column
+ vertex_positions = vertex_positions.reshape(yvalues.size, xvalues.size)[
+ :-1, :-1
+ ].flatten()
+
+ # Now create triangles (keep vertices in anticlockwise order when making triangles)
+ top_triangles = np.vstack(
+ ((vertex_positions + 1, vertex_positions, vertex_positions + xvalues.shape[0]))
+ ).transpose(1, 0)
+ vertex_positions = np.arange(xvalues.size * yvalues.size)
+ vertex_positions = vertex_positions.reshape(yvalues.size, xvalues.size)[
+ 1:, 1:
+ ].flatten()
+ bottom_triangles = np.vstack(
+ ((vertex_positions - 1, vertex_positions, vertex_positions - xvalues.shape[0]))
+ ).transpose(1, 0)
+ triangles = np.vstack((top_triangles, bottom_triangles))
+
+ mesh: o3d.geometry.TriangleMesh = o3d.geometry.TriangleMesh(
+ o3d.utility.Vector3dVector(vertices), o3d.utility.Vector3iVector(triangles)
+ )
+ mesh.compute_vertex_normals()
+ mesh.vertex_colors = o3d.utility.Vector3dVector(vertex_colors)
+ """
+ Flip the y and z axis according to opencv to opengl transformation.
+ See - https://stackoverflow.com/questions/44375149/opencv-to-opengl-coordinate-system-transform
+ """
+ # mesh.transform([[1, 0, 0, 0], [0, -1, 0, 0], [0, 0, -1, 0], [0, 0, 0, 1]])
+ return mesh
diff --git a/inference.sh b/inference.sh
new file mode 100644
index 0000000000000000000000000000000000000000..542180f72280ad9b6d1633a9ecc1cdb9cf9598bc
--- /dev/null
+++ b/inference.sh
@@ -0,0 +1 @@
+python scripts/demo/infer_video.py video_path=videos/test_10.mp4 ckpt_path=s050000.ckpt exp=genmo_lg static_cam=false use_sam_masking=true run_hamer=true
\ No newline at end of file
diff --git a/install.sh b/install.sh
new file mode 100644
index 0000000000000000000000000000000000000000..42185180be22e90ec085120dcbb93aa74e1e72f7
--- /dev/null
+++ b/install.sh
@@ -0,0 +1,25 @@
+# 1. Add the conda environment's library path to the linker path
+export LIBRARY_PATH=/root/miniconda3/envs/gvhmr/lib:$LIBRARY_PATH
+
+# 2. Specifically tell the linker where to find CUDA libraries
+export LD_LIBRARY_PATH=/root/miniconda3/envs/gvhmr/lib:$LD_LIBRARY_PATH
+
+# 3. If you have a system CUDA, add that too just in case
+export LIBRARY_PATH=/usr/local/cuda-12.1/lib64:$LIBRARY_PATH
+
+ conda install -c nvidia cccl
+
+ conda install -c nvidia cuda-toolkit gxx_linux-64
+
+
+ # 1. Force the linker to look in your conda environment's lib folder
+export LDFLAGS="-L/root/miniconda3/envs/gvhmr/lib -Wl,-rpath,/root/miniconda3/envs/gvhmr/lib"
+
+# 2. Add it to the standard search paths
+export LIBRARY_PATH="/root/miniconda3/envs/gvhmr/lib:$LIBRARY_PATH"
+export LD_LIBRARY_PATH="/root/miniconda3/envs/gvhmr/lib:$LD_LIBRARY_PATH"
+
+# 3. If you have a system-wide CUDA, add that as a fallback
+export LIBRARY_PATH="/usr/local/cuda-12.1/lib64:$LIBRARY_PATH"
+
+conda install -c nvidia cuda-cudart-static_linux-64=12.1 cuda-nvcc-static_linux-64=12.1 -y
\ No newline at end of file
diff --git a/label.sh b/label.sh
new file mode 100644
index 0000000000000000000000000000000000000000..ae8e256084c49a2a98a2eb60478268592060206b
--- /dev/null
+++ b/label.sh
@@ -0,0 +1,9 @@
+python scripts/label_videos.py \
+ --video ./videos/VRM_JG6Z7WA.mp4 \
+ --output ./labels.jsonl \
+ --debug-dir ./debug_dets \
+ --vitpose-filter \
+ --vitpose-min-joints 2 \
+ --vitpose-conf-threshold 0.3 \
+ --vitpose-filter \
+ --vitpose-filter-all
\ No newline at end of file
diff --git a/labels.json b/labels.json
new file mode 100644
index 0000000000000000000000000000000000000000..7a86ac42b2c57b0f12e633f09a6b16e3714c5da0
--- /dev/null
+++ b/labels.json
@@ -0,0 +1,101 @@
+{
+ "videos": [
+ {
+ "video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA_clip_0.mp4",
+ "total_duration_sec": 9.0,
+ "usable_duration_sec": 6.0,
+ "num_segments": 8,
+ "num_usable_segments": 4,
+ "segments": [
+ {
+ "start_sec": 0.0,
+ "end_sec": 2.0,
+ "dynamic_persons": 1,
+ "static_detections": 0,
+ "avg_confidence": 0.9042157828807831,
+ "avg_bbox_area_pct": 0.15445075894579474,
+ "bbox_variance": 0.0,
+ "usable": true,
+ "reason": null
+ },
+ {
+ "start_sec": 2.0,
+ "end_sec": 3.0,
+ "dynamic_persons": 0,
+ "static_detections": 0,
+ "avg_confidence": 0.0,
+ "avg_bbox_area_pct": 0.0,
+ "bbox_variance": 0.0,
+ "usable": false,
+ "reason": "no_person"
+ },
+ {
+ "start_sec": 3.0,
+ "end_sec": 5.0,
+ "dynamic_persons": 1,
+ "static_detections": 0,
+ "avg_confidence": 0.6771576702594757,
+ "avg_bbox_area_pct": 0.17748631606867282,
+ "bbox_variance": 0.0,
+ "usable": true,
+ "reason": null
+ },
+ {
+ "start_sec": 5.0,
+ "end_sec": 6.0,
+ "dynamic_persons": 1,
+ "static_detections": 0,
+ "avg_confidence": 0.4357699453830719,
+ "avg_bbox_area_pct": 0.07207973150559413,
+ "bbox_variance": 0.0,
+ "usable": false,
+ "reason": "low_confidence"
+ },
+ {
+ "start_sec": 6.0,
+ "end_sec": 7.0,
+ "dynamic_persons": 1,
+ "static_detections": 0,
+ "avg_confidence": 0.5237488150596619,
+ "avg_bbox_area_pct": 0.061591827015817904,
+ "bbox_variance": 0.0,
+ "usable": true,
+ "reason": null
+ },
+ {
+ "start_sec": 7.0,
+ "end_sec": 8.0,
+ "dynamic_persons": 1,
+ "static_detections": 0,
+ "avg_confidence": 0.4154590368270874,
+ "avg_bbox_area_pct": 0.09554578239535108,
+ "bbox_variance": 0.0,
+ "usable": false,
+ "reason": "low_confidence"
+ },
+ {
+ "start_sec": 8.0,
+ "end_sec": 9.0,
+ "dynamic_persons": 1,
+ "static_detections": 0,
+ "avg_confidence": 0.5311647057533264,
+ "avg_bbox_area_pct": 0.2209286838107639,
+ "bbox_variance": 0.0,
+ "usable": true,
+ "reason": null
+ },
+ {
+ "start_sec": 9.0,
+ "end_sec": 10.0,
+ "dynamic_persons": 1,
+ "static_detections": 0,
+ "avg_confidence": 0.44936051964759827,
+ "avg_bbox_area_pct": 0.16834459092881945,
+ "bbox_variance": 0.0,
+ "usable": false,
+ "reason": "low_confidence"
+ }
+ ]
+ }
+ ]
+}
\ No newline at end of file
diff --git a/labels.jsonl b/labels.jsonl
new file mode 100644
index 0000000000000000000000000000000000000000..39028ec60ea05ff4d8eb378820f49ed1429a9a79
--- /dev/null
+++ b/labels.jsonl
@@ -0,0 +1,716 @@
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 0.0, "end_sec": 8.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 8.0, "end_sec": 9.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5298700332641602, "avg_bbox_area_pct": 0.010859977816358024, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 9.0, "end_sec": 11.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 11.0, "end_sec": 108.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7156517487211326, "avg_bbox_area_pct": 0.3087665050497367, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 108.0, "end_sec": 109.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 109.0, "end_sec": 117.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.745468869805336, "avg_bbox_area_pct": 0.5287311544536073, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 117.0, "end_sec": 118.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 118.0, "end_sec": 119.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.4829714894294739, "avg_bbox_area_pct": 0.012777313420801987, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 119.0, "end_sec": 121.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 121.0, "end_sec": 124.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6826531787713369, "avg_bbox_area_pct": 0.5765211467978395, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 124.0, "end_sec": 125.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 125.0, "end_sec": 147.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7526712864637375, "avg_bbox_area_pct": 0.5442567644215593, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 147.0, "end_sec": 150.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 150.0, "end_sec": 157.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7153174238545554, "avg_bbox_area_pct": 0.31709705171130953, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 157.0, "end_sec": 158.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 158.0, "end_sec": 166.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7394165135920048, "avg_bbox_area_pct": 0.6309571405104648, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 166.0, "end_sec": 167.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 167.0, "end_sec": 170.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7859500050544739, "avg_bbox_area_pct": 0.6488420299639918, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 170.0, "end_sec": 171.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 171.0, "end_sec": 173.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.643202543258667, "avg_bbox_area_pct": 0.37185983916859566, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 173.0, "end_sec": 174.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 174.0, "end_sec": 176.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.819955050945282, "avg_bbox_area_pct": 0.4644908613040123, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 176.0, "end_sec": 177.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 0.0, "end_sec": 8.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 8.0, "end_sec": 9.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5298700332641602, "avg_bbox_area_pct": 0.010859977816358024, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 9.0, "end_sec": 11.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 11.0, "end_sec": 108.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7156517487211326, "avg_bbox_area_pct": 0.3087665050497367, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 108.0, "end_sec": 109.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 109.0, "end_sec": 117.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.745468869805336, "avg_bbox_area_pct": 0.5287311544536073, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 117.0, "end_sec": 118.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 118.0, "end_sec": 119.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.4829714894294739, "avg_bbox_area_pct": 0.012777313420801987, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 119.0, "end_sec": 121.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 121.0, "end_sec": 124.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6826531787713369, "avg_bbox_area_pct": 0.5765211467978395, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 124.0, "end_sec": 125.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 125.0, "end_sec": 147.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7526712864637375, "avg_bbox_area_pct": 0.5442567644215593, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 147.0, "end_sec": 150.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 150.0, "end_sec": 157.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7153174238545554, "avg_bbox_area_pct": 0.31709705171130953, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 157.0, "end_sec": 158.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 158.0, "end_sec": 166.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7394165135920048, "avg_bbox_area_pct": 0.6309571405104648, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 166.0, "end_sec": 167.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 167.0, "end_sec": 170.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7859500050544739, "avg_bbox_area_pct": 0.6488420299639918, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 170.0, "end_sec": 171.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 171.0, "end_sec": 173.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.643202543258667, "avg_bbox_area_pct": 0.37185983916859566, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 173.0, "end_sec": 174.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 174.0, "end_sec": 176.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.819955050945282, "avg_bbox_area_pct": 0.4644908613040123, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 176.0, "end_sec": 177.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 177.0, "end_sec": 185.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7422709614038467, "avg_bbox_area_pct": 0.3925352007077064, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 185.0, "end_sec": 186.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 186.0, "end_sec": 196.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.674255883693695, "avg_bbox_area_pct": 0.5368624222366898, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 196.0, "end_sec": 197.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 197.0, "end_sec": 199.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6580421179533005, "avg_bbox_area_pct": 0.733458568431713, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 199.0, "end_sec": 200.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 200.0, "end_sec": 203.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.809789796670278, "avg_bbox_area_pct": 0.4993893369823817, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 203.0, "end_sec": 204.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 204.0, "end_sec": 208.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7048537135124207, "avg_bbox_area_pct": 0.352830968786169, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 208.0, "end_sec": 209.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 209.0, "end_sec": 212.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8253923654556274, "avg_bbox_area_pct": 0.5007177031089248, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 212.0, "end_sec": 213.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 213.0, "end_sec": 244.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7340064529449709, "avg_bbox_area_pct": 0.43187158728158603, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 244.0, "end_sec": 245.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 245.0, "end_sec": 258.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7724014749893775, "avg_bbox_area_pct": 0.4080624073749481, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 258.0, "end_sec": 259.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 259.0, "end_sec": 265.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7738728721936544, "avg_bbox_area_pct": 0.47720878946437756, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 265.0, "end_sec": 267.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 267.0, "end_sec": 306.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7829218736061683, "avg_bbox_area_pct": 0.3606301512783108, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 306.0, "end_sec": 308.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 308.0, "end_sec": 377.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7605451030143793, "avg_bbox_area_pct": 0.3401769022891252, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 377.0, "end_sec": 378.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 378.0, "end_sec": 391.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5167739666425265, "avg_bbox_area_pct": 0.15364861498304694, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 391.0, "end_sec": 392.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 392.0, "end_sec": 403.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8312831033359874, "avg_bbox_area_pct": 0.36516340981428874, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 403.0, "end_sec": 405.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 405.0, "end_sec": 459.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.812526156504949, "avg_bbox_area_pct": 0.41680115778695837, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 459.0, "end_sec": 461.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 461.0, "end_sec": 502.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7453105486020809, "avg_bbox_area_pct": 0.363366205938535, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 502.0, "end_sec": 503.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 503.0, "end_sec": 524.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7220587545917148, "avg_bbox_area_pct": 0.38108089063533396, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 524.0, "end_sec": 525.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 525.0, "end_sec": 630.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7518549399716513, "avg_bbox_area_pct": 0.34388790414530973, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 630.0, "end_sec": 633.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 633.0, "end_sec": 638.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7845604419708252, "avg_bbox_area_pct": 0.2630388485001929, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 638.0, "end_sec": 641.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 641.0, "end_sec": 672.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7679022963969938, "avg_bbox_area_pct": 0.38120946157220736, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 672.0, "end_sec": 673.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 673.0, "end_sec": 724.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6953776297616023, "avg_bbox_area_pct": 0.3606798837699142, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 724.0, "end_sec": 726.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 726.0, "end_sec": 731.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.836204218864441, "avg_bbox_area_pct": 0.31444334430459103, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 731.0, "end_sec": 732.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 732.0, "end_sec": 734.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6886181235313416, "avg_bbox_area_pct": 0.463685905550733, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 734.0, "end_sec": 735.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 735.0, "end_sec": 746.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7357966628941622, "avg_bbox_area_pct": 0.4472833943821899, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 746.0, "end_sec": 747.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 747.0, "end_sec": 753.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7724084754784902, "avg_bbox_area_pct": 0.4824383067692259, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 753.0, "end_sec": 755.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 755.0, "end_sec": 759.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6599370613694191, "avg_bbox_area_pct": 0.5527715536988812, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 759.0, "end_sec": 761.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 761.0, "end_sec": 816.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7335107310251756, "avg_bbox_area_pct": 0.28521201462322876, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 816.0, "end_sec": 817.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.599391296505928, "avg_bbox_area_pct": 0.23393153061101468, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 817.0, "end_sec": 823.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6143738130728403, "avg_bbox_area_pct": 0.2606831416377315, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 823.0, "end_sec": 824.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.4402656555175781, "avg_bbox_area_pct": 0.22027199827594524, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 824.0, "end_sec": 826.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6622984111309052, "avg_bbox_area_pct": 0.3461745123215664, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 826.0, "end_sec": 828.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 828.0, "end_sec": 832.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5257512107491493, "avg_bbox_area_pct": 0.2189247300889757, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 832.0, "end_sec": 833.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6551225483417511, "avg_bbox_area_pct": 0.16928132987316744, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 833.0, "end_sec": 846.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6246378375933721, "avg_bbox_area_pct": 0.22031637309510027, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 846.0, "end_sec": 848.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.3613896518945694, "avg_bbox_area_pct": 0.214651481722608, "bbox_variance": 0.0, "usable": false, "reason": "low_confidence"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 848.0, "end_sec": 875.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5415235360463461, "avg_bbox_area_pct": 0.2959643242116055, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 875.0, "end_sec": 876.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5606665909290314, "avg_bbox_area_pct": 0.20714330602575232, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 876.0, "end_sec": 877.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6235802173614502, "avg_bbox_area_pct": 0.17004438235435956, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 877.0, "end_sec": 878.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6342715322971344, "avg_bbox_area_pct": 0.17574348355516975, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 878.0, "end_sec": 902.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5816046297550201, "avg_bbox_area_pct": 0.25488123324672873, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 902.0, "end_sec": 903.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 903.0, "end_sec": 905.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5601617097854614, "avg_bbox_area_pct": 0.2619167827088156, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 905.0, "end_sec": 906.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5112910270690918, "avg_bbox_area_pct": 0.19706223476080248, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 906.0, "end_sec": 971.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7399500351685744, "avg_bbox_area_pct": 0.24672823728613927, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 971.0, "end_sec": 972.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 972.0, "end_sec": 1014.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7955081846032824, "avg_bbox_area_pct": 0.21155740548413612, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1014.0, "end_sec": 1022.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1022.0, "end_sec": 1023.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.739115297794342, "avg_bbox_area_pct": 0.20216157889660494, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1023.0, "end_sec": 1027.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1027.0, "end_sec": 1028.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.4184567332267761, "avg_bbox_area_pct": 0.01462652700918692, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1028.0, "end_sec": 1032.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1032.0, "end_sec": 1079.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7862131018587883, "avg_bbox_area_pct": 0.2708274808843085, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1079.0, "end_sec": 1080.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7562277615070343, "avg_bbox_area_pct": 0.17557421272183643, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1080.0, "end_sec": 1141.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7718810160629085, "avg_bbox_area_pct": 0.26482654675759904, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1141.0, "end_sec": 1142.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1142.0, "end_sec": 1175.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7974752910209425, "avg_bbox_area_pct": 0.2844464401319327, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1175.0, "end_sec": 1178.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1178.0, "end_sec": 1185.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7883438553128924, "avg_bbox_area_pct": 0.32001956139081794, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1185.0, "end_sec": 1187.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1187.0, "end_sec": 1204.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7868308530134314, "avg_bbox_area_pct": 0.2738207564565178, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1204.0, "end_sec": 1205.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1205.0, "end_sec": 1220.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7866829713185628, "avg_bbox_area_pct": 0.22932244345582564, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1220.0, "end_sec": 1221.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7601432502269745, "avg_bbox_area_pct": 0.2750991256148727, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1221.0, "end_sec": 1228.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7508945124489921, "avg_bbox_area_pct": 0.34081434461805554, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1228.0, "end_sec": 1229.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7428703606128693, "avg_bbox_area_pct": 0.28981654791184414, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1229.0, "end_sec": 1238.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7396152549319797, "avg_bbox_area_pct": 0.31265516493055556, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1238.0, "end_sec": 1239.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5054189413785934, "avg_bbox_area_pct": 0.2227036464361497, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1239.0, "end_sec": 1242.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7737449606259664, "avg_bbox_area_pct": 0.14024784543386704, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1242.0, "end_sec": 1243.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7802169322967529, "avg_bbox_area_pct": 0.19483216839072143, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1243.0, "end_sec": 1249.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7257813711961111, "avg_bbox_area_pct": 0.3167046315385481, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1249.0, "end_sec": 1250.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7634005844593048, "avg_bbox_area_pct": 0.22211608133198302, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1250.0, "end_sec": 1254.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7895286828279495, "avg_bbox_area_pct": 0.23898494767554013, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1254.0, "end_sec": 1255.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6148070991039276, "avg_bbox_area_pct": 0.20645992326147763, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1255.0, "end_sec": 1259.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7233529984951019, "avg_bbox_area_pct": 0.19137785734953705, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1259.0, "end_sec": 1260.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6266395747661591, "avg_bbox_area_pct": 0.22776197645399304, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1260.0, "end_sec": 1261.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7922869920730591, "avg_bbox_area_pct": 0.14455839180652005, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1261.0, "end_sec": 1263.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6861878708004951, "avg_bbox_area_pct": 0.22374238944347993, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1263.0, "end_sec": 1276.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7462709889962122, "avg_bbox_area_pct": 0.253212226367744, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1276.0, "end_sec": 1277.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1277.0, "end_sec": 1308.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7440180951549161, "avg_bbox_area_pct": 0.31915135756188395, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1308.0, "end_sec": 1310.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6758342832326889, "avg_bbox_area_pct": 0.24904344723548422, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1310.0, "end_sec": 1316.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.67212542394797, "avg_bbox_area_pct": 0.42151464642811226, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1316.0, "end_sec": 1317.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6719513535499573, "avg_bbox_area_pct": 0.2884955512152778, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1317.0, "end_sec": 1354.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7771796165285884, "avg_bbox_area_pct": 0.2579746380495078, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1354.0, "end_sec": 1355.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6006399989128113, "avg_bbox_area_pct": 0.2074861201533565, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1355.0, "end_sec": 1356.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1356.0, "end_sec": 1359.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5622204939524332, "avg_bbox_area_pct": 0.24715800720968364, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1359.0, "end_sec": 1361.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6092827692627907, "avg_bbox_area_pct": 0.2050760226779514, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1361.0, "end_sec": 1381.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7980061888694763, "avg_bbox_area_pct": 0.21201629902404032, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1381.0, "end_sec": 1382.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6251963973045349, "avg_bbox_area_pct": 0.17539534203800156, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1382.0, "end_sec": 1383.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1383.0, "end_sec": 1386.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6254582206408182, "avg_bbox_area_pct": 0.21503366729359566, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1386.0, "end_sec": 1388.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1388.0, "end_sec": 1389.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.42509159445762634, "avg_bbox_area_pct": 0.21587616343557098, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1389.0, "end_sec": 1390.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1390.0, "end_sec": 1392.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.42837636172771454, "avg_bbox_area_pct": 0.06916879536193093, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1392.0, "end_sec": 1393.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1393.0, "end_sec": 1402.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7667882508701749, "avg_bbox_area_pct": 0.1397600573211377, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1402.0, "end_sec": 1404.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5187130272388458, "avg_bbox_area_pct": 0.19069976053120177, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1404.0, "end_sec": 1406.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8039670586585999, "avg_bbox_area_pct": 0.21609847457320602, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1406.0, "end_sec": 1410.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.601176805794239, "avg_bbox_area_pct": 0.17933336611147283, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1410.0, "end_sec": 1415.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.816373634338379, "avg_bbox_area_pct": 0.2380628677179784, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1415.0, "end_sec": 1416.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6090268343687057, "avg_bbox_area_pct": 0.22306981216242283, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1416.0, "end_sec": 1419.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6606877247492472, "avg_bbox_area_pct": 0.3207581922743055, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1419.0, "end_sec": 1420.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6138791739940643, "avg_bbox_area_pct": 0.2370355526017554, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1420.0, "end_sec": 1421.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6903096437454224, "avg_bbox_area_pct": 0.2784399715470679, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1421.0, "end_sec": 1432.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7612546546892687, "avg_bbox_area_pct": 0.241347922722231, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1432.0, "end_sec": 1434.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8383523225784302, "avg_bbox_area_pct": 0.20008930724344137, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1434.0, "end_sec": 1440.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8284551997979482, "avg_bbox_area_pct": 0.20115699171529386, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1440.0, "end_sec": 1445.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8003306984901428, "avg_bbox_area_pct": 0.22565840506847992, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1445.0, "end_sec": 1446.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.762101024389267, "avg_bbox_area_pct": 0.1372653348946277, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1446.0, "end_sec": 1450.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8056871294975281, "avg_bbox_area_pct": 0.19554460690345293, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1450.0, "end_sec": 1451.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7497442960739136, "avg_bbox_area_pct": 0.17205612747757523, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1451.0, "end_sec": 1452.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.9006332755088806, "avg_bbox_area_pct": 0.14629490981867285, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1452.0, "end_sec": 1453.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7034122943878174, "avg_bbox_area_pct": 0.13161618833188657, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1453.0, "end_sec": 1471.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.783574769894282, "avg_bbox_area_pct": 0.30096547486523706, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1471.0, "end_sec": 1472.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1472.0, "end_sec": 1488.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7721960972994566, "avg_bbox_area_pct": 0.24980660615143951, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1488.0, "end_sec": 1498.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.745188656449318, "avg_bbox_area_pct": 0.18803494451075425, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1498.0, "end_sec": 1499.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7806546688079834, "avg_bbox_area_pct": 0.20168071228780865, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1499.0, "end_sec": 1510.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7396828776056116, "avg_bbox_area_pct": 0.19924564314477242, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1510.0, "end_sec": 1519.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.713588168223699, "avg_bbox_area_pct": 0.2046629050925926, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1519.0, "end_sec": 1521.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.742208942770958, "avg_bbox_area_pct": 0.19075336974344134, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1521.0, "end_sec": 1598.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7751257752443289, "avg_bbox_area_pct": 0.19457194010416667, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1598.0, "end_sec": 1599.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8016725778579712, "avg_bbox_area_pct": 0.19228936842930172, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1599.0, "end_sec": 1606.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7350378291947501, "avg_bbox_area_pct": 0.23194292233314043, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1606.0, "end_sec": 1608.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1608.0, "end_sec": 1615.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7495998697621482, "avg_bbox_area_pct": 0.2582455561400721, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1615.0, "end_sec": 1616.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1616.0, "end_sec": 1618.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6816407144069672, "avg_bbox_area_pct": 0.22594385217737267, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1618.0, "end_sec": 1619.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6646848320960999, "avg_bbox_area_pct": 0.18721257716049383, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1619.0, "end_sec": 1621.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7130466401576996, "avg_bbox_area_pct": 0.1846813362027392, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1621.0, "end_sec": 1623.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6125450879335403, "avg_bbox_area_pct": 0.19637836597583913, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1623.0, "end_sec": 1624.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6528957486152649, "avg_bbox_area_pct": 0.19380934727044752, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1624.0, "end_sec": 1625.0, "dynamic_persons": 3, "static_detections": 0, "avg_confidence": 0.5392289857069651, "avg_bbox_area_pct": 0.19652921449009772, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1625.0, "end_sec": 1627.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5555556118488312, "avg_bbox_area_pct": 0.24881273057725695, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1627.0, "end_sec": 1636.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5663854132095972, "avg_bbox_area_pct": 0.21343463411056454, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1636.0, "end_sec": 1637.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6535671949386597, "avg_bbox_area_pct": 0.21440789870273919, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1637.0, "end_sec": 1638.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1638.0, "end_sec": 1644.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6436004638671875, "avg_bbox_area_pct": 0.24347052680121525, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1644.0, "end_sec": 1645.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1645.0, "end_sec": 1660.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6286231517791748, "avg_bbox_area_pct": 0.19863765964586552, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1660.0, "end_sec": 1661.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1661.0, "end_sec": 1675.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6138805938618523, "avg_bbox_area_pct": 0.22032950971492385, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1675.0, "end_sec": 1676.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1676.0, "end_sec": 1687.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6210675022818826, "avg_bbox_area_pct": 0.2396089835118766, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1687.0, "end_sec": 1688.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1688.0, "end_sec": 1705.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6267911581432118, "avg_bbox_area_pct": 0.31655144845911276, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1705.0, "end_sec": 1706.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1706.0, "end_sec": 1707.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6201421022415161, "avg_bbox_area_pct": 0.08453666781201774, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1707.0, "end_sec": 1708.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1708.0, "end_sec": 1709.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6937569975852966, "avg_bbox_area_pct": 0.13253556616512346, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1709.0, "end_sec": 1711.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1711.0, "end_sec": 1717.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6169558962186178, "avg_bbox_area_pct": 0.10653595221876609, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1717.0, "end_sec": 1718.0, "dynamic_persons": 3, "static_detections": 0, "avg_confidence": 0.6189529895782471, "avg_bbox_area_pct": 0.14425924167711549, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1718.0, "end_sec": 1737.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6489040192804838, "avg_bbox_area_pct": 0.11425529970760233, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1737.0, "end_sec": 1738.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1738.0, "end_sec": 1743.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6712007045745849, "avg_bbox_area_pct": 0.09635420735677083, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1743.0, "end_sec": 1744.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7226238250732422, "avg_bbox_area_pct": 0.14903437861689814, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1744.0, "end_sec": 1747.0, "dynamic_persons": 4, "static_detections": 0, "avg_confidence": 0.6524128367503484, "avg_bbox_area_pct": 0.1616956232015979, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1747.0, "end_sec": 1750.0, "dynamic_persons": 3, "static_detections": 0, "avg_confidence": 0.6840562687979804, "avg_bbox_area_pct": 0.1624436366705247, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1750.0, "end_sec": 1751.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6365471184253693, "avg_bbox_area_pct": 0.17139860930266204, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1751.0, "end_sec": 1753.0, "dynamic_persons": 3, "static_detections": 0, "avg_confidence": 0.6395860115687052, "avg_bbox_area_pct": 0.13835094310619214, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1753.0, "end_sec": 1754.0, "dynamic_persons": 4, "static_detections": 0, "avg_confidence": 0.5402426719665527, "avg_bbox_area_pct": 0.15998630476586612, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1754.0, "end_sec": 1756.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7048329710960388, "avg_bbox_area_pct": 0.13873212272738233, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1756.0, "end_sec": 1757.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6749147176742554, "avg_bbox_area_pct": 0.17095787519290123, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1757.0, "end_sec": 1758.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8217679858207703, "avg_bbox_area_pct": 0.23220772448881172, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1758.0, "end_sec": 1782.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6823552660644054, "avg_bbox_area_pct": 0.26286283234019336, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1782.0, "end_sec": 1783.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6678254902362823, "avg_bbox_area_pct": 0.20079896526572144, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1783.0, "end_sec": 1786.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7554793159166971, "avg_bbox_area_pct": 0.2889349139178241, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1786.0, "end_sec": 1787.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5356227457523346, "avg_bbox_area_pct": 0.1788705105251736, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1787.0, "end_sec": 1788.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6580737233161926, "avg_bbox_area_pct": 0.14640691309799383, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1788.0, "end_sec": 1790.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5411960482597351, "avg_bbox_area_pct": 0.1252823045518663, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1790.0, "end_sec": 1797.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.645488053560257, "avg_bbox_area_pct": 0.1553959782245508, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1797.0, "end_sec": 1798.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5767699480056763, "avg_bbox_area_pct": 0.13603706265673227, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1798.0, "end_sec": 1801.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5982308189074198, "avg_bbox_area_pct": 0.12760581938818158, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1801.0, "end_sec": 1803.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1803.0, "end_sec": 1808.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5293199896812439, "avg_bbox_area_pct": 0.13671997673128858, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1808.0, "end_sec": 1809.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1809.0, "end_sec": 1814.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5237399101257324, "avg_bbox_area_pct": 0.1152494800708912, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1814.0, "end_sec": 1815.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1815.0, "end_sec": 1816.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.4939725399017334, "avg_bbox_area_pct": 0.12985209900655864, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1816.0, "end_sec": 1817.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1817.0, "end_sec": 1837.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6377182990312577, "avg_bbox_area_pct": 0.12147232427714783, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1837.0, "end_sec": 1838.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6453697383403778, "avg_bbox_area_pct": 0.25755415551456406, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1838.0, "end_sec": 1841.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6022402246793112, "avg_bbox_area_pct": 0.2743624237437307, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1841.0, "end_sec": 1842.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5835753381252289, "avg_bbox_area_pct": 0.0985670150945216, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1842.0, "end_sec": 1852.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7569757878780365, "avg_bbox_area_pct": 0.15019916977117093, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1852.0, "end_sec": 1853.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1853.0, "end_sec": 1863.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6054648607969284, "avg_bbox_area_pct": 0.1937640426070602, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1863.0, "end_sec": 1864.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1864.0, "end_sec": 1867.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.82319708665212, "avg_bbox_area_pct": 0.3034964453928755, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1867.0, "end_sec": 1871.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5957357920706272, "avg_bbox_area_pct": 0.21047253173074604, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1871.0, "end_sec": 1887.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6082744542509317, "avg_bbox_area_pct": 0.14216453599341122, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1887.0, "end_sec": 1888.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7702455222606659, "avg_bbox_area_pct": 0.16847875147690006, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1888.0, "end_sec": 1902.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7195978249822345, "avg_bbox_area_pct": 0.12107946109939924, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1902.0, "end_sec": 1903.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6313386857509613, "avg_bbox_area_pct": 0.11011846094955632, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1903.0, "end_sec": 1904.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.778401255607605, "avg_bbox_area_pct": 0.141565242814429, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1904.0, "end_sec": 1910.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7175593425830206, "avg_bbox_area_pct": 0.2567801543812693, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1910.0, "end_sec": 1911.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8126366138458252, "avg_bbox_area_pct": 0.31704993730709874, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1911.0, "end_sec": 1913.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6398265287280083, "avg_bbox_area_pct": 0.16940595085238233, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1913.0, "end_sec": 1917.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7321412861347198, "avg_bbox_area_pct": 0.17041392196843652, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1917.0, "end_sec": 1918.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8025122582912445, "avg_bbox_area_pct": 0.11890941101827739, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1918.0, "end_sec": 1920.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7932980954647064, "avg_bbox_area_pct": 0.11582281418788581, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1920.0, "end_sec": 1923.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7936307390530905, "avg_bbox_area_pct": 0.1755236540115419, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1923.0, "end_sec": 1924.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8150386810302734, "avg_bbox_area_pct": 0.1610815731095679, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1924.0, "end_sec": 1925.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7413496375083923, "avg_bbox_area_pct": 0.2752515251253858, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1925.0, "end_sec": 1930.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7787216305732727, "avg_bbox_area_pct": 0.24364107259114584, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1930.0, "end_sec": 1933.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6045666684707006, "avg_bbox_area_pct": 0.14362231972776812, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1933.0, "end_sec": 1943.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7408762216567993, "avg_bbox_area_pct": 0.13810746934678822, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1943.0, "end_sec": 1945.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6818010360002518, "avg_bbox_area_pct": 0.17199880717713156, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1945.0, "end_sec": 1981.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.81609422299597, "avg_bbox_area_pct": 0.13579661365399145, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1981.0, "end_sec": 1984.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6154637485742569, "avg_bbox_area_pct": 0.15672272356449332, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1984.0, "end_sec": 1988.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8120100796222687, "avg_bbox_area_pct": 0.4407490407684703, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1988.0, "end_sec": 1989.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5380788296461105, "avg_bbox_area_pct": 0.1475595733265818, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1989.0, "end_sec": 1990.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7701777815818787, "avg_bbox_area_pct": 0.11741811493296682, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1990.0, "end_sec": 1991.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7168194353580475, "avg_bbox_area_pct": 0.11025697307822144, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 1991.0, "end_sec": 2018.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7429257852059824, "avg_bbox_area_pct": 0.15262513937337463, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2018.0, "end_sec": 2019.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7050831615924835, "avg_bbox_area_pct": 0.10669846522955248, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2019.0, "end_sec": 2023.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7758935689926147, "avg_bbox_area_pct": 0.12051141432773921, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2023.0, "end_sec": 2027.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6450166925787926, "avg_bbox_area_pct": 0.1286398051697531, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2027.0, "end_sec": 2029.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7760427296161652, "avg_bbox_area_pct": 0.12567823621961807, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2029.0, "end_sec": 2032.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6972978015740713, "avg_bbox_area_pct": 0.11252857098363556, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2032.0, "end_sec": 2034.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7528762519359589, "avg_bbox_area_pct": 0.14256978352864585, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2034.0, "end_sec": 2035.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8257640600204468, "avg_bbox_area_pct": 0.1347049176251447, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2035.0, "end_sec": 2036.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8348667025566101, "avg_bbox_area_pct": 0.2954687198591821, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2036.0, "end_sec": 2039.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6929511775573095, "avg_bbox_area_pct": 0.2439011461548354, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2039.0, "end_sec": 2041.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6249272525310516, "avg_bbox_area_pct": 0.25652585630063657, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2041.0, "end_sec": 2042.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8324554264545441, "avg_bbox_area_pct": 0.12970072051625192, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2042.0, "end_sec": 2046.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5628747865557671, "avg_bbox_area_pct": 0.32030345210322625, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2046.0, "end_sec": 2064.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7151267164283328, "avg_bbox_area_pct": 0.26197575698664155, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2064.0, "end_sec": 2066.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6846987903118134, "avg_bbox_area_pct": 0.3919024130738812, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2066.0, "end_sec": 2067.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7199741899967194, "avg_bbox_area_pct": 0.2722477062837577, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2067.0, "end_sec": 2068.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7222606539726257, "avg_bbox_area_pct": 0.3919551745756173, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2068.0, "end_sec": 2072.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.720182791352272, "avg_bbox_area_pct": 0.2723324453094859, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2072.0, "end_sec": 2073.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6644085049629211, "avg_bbox_area_pct": 0.39165702160493826, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2073.0, "end_sec": 2075.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7262144386768341, "avg_bbox_area_pct": 0.278432093490789, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2075.0, "end_sec": 2077.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7917138040065765, "avg_bbox_area_pct": 0.136484533239294, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2077.0, "end_sec": 2080.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7724708716074625, "avg_bbox_area_pct": 0.11131476257073046, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2080.0, "end_sec": 2085.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7429888844490051, "avg_bbox_area_pct": 0.13303180911217205, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2085.0, "end_sec": 2086.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5708269327878952, "avg_bbox_area_pct": 0.15798044463734567, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2086.0, "end_sec": 2088.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7312033176422119, "avg_bbox_area_pct": 0.1414301441333912, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2088.0, "end_sec": 2089.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8049972653388977, "avg_bbox_area_pct": 0.23171747655044367, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2089.0, "end_sec": 2096.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8560921975544521, "avg_bbox_area_pct": 0.18642137961412866, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2096.0, "end_sec": 2098.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6919247955083847, "avg_bbox_area_pct": 0.16061190758222416, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2098.0, "end_sec": 2100.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5929425954818726, "avg_bbox_area_pct": 0.2642091200086806, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2100.0, "end_sec": 2101.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8080624938011169, "avg_bbox_area_pct": 0.22559199203679592, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2101.0, "end_sec": 2102.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.61412113904953, "avg_bbox_area_pct": 0.14994366681134258, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2102.0, "end_sec": 2111.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7993894086943732, "avg_bbox_area_pct": 0.27859538932559585, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2111.0, "end_sec": 2112.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.845553457736969, "avg_bbox_area_pct": 0.3879736328125, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2112.0, "end_sec": 2113.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8309268951416016, "avg_bbox_area_pct": 0.262938692069348, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2113.0, "end_sec": 2114.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8646736145019531, "avg_bbox_area_pct": 0.38849440586419753, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2114.0, "end_sec": 2117.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8235489626725515, "avg_bbox_area_pct": 0.2762072151089892, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2117.0, "end_sec": 2118.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8454161286354065, "avg_bbox_area_pct": 0.38556327160493825, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2118.0, "end_sec": 2119.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.823845624923706, "avg_bbox_area_pct": 0.2820033546730324, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2119.0, "end_sec": 2125.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8209722439448038, "avg_bbox_area_pct": 0.43656219859182094, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2125.0, "end_sec": 2126.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5381138771772385, "avg_bbox_area_pct": 0.19977842731240356, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2126.0, "end_sec": 2127.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7417857050895691, "avg_bbox_area_pct": 0.14959422923900462, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2127.0, "end_sec": 2128.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5595627874135971, "avg_bbox_area_pct": 0.16840245376398533, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2128.0, "end_sec": 2129.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7060161232948303, "avg_bbox_area_pct": 0.14651156201774693, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2129.0, "end_sec": 2134.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7746960699558259, "avg_bbox_area_pct": 0.1438978655779803, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2134.0, "end_sec": 2136.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7595397233963013, "avg_bbox_area_pct": 0.13324118155020254, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2136.0, "end_sec": 2137.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8035041391849518, "avg_bbox_area_pct": 0.12652875358675733, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2137.0, "end_sec": 2153.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8040983434766531, "avg_bbox_area_pct": 0.13250963328797138, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2153.0, "end_sec": 2158.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6657271087169647, "avg_bbox_area_pct": 0.15816672393422065, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2158.0, "end_sec": 2162.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8032331466674805, "avg_bbox_area_pct": 0.10713161138840663, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2162.0, "end_sec": 2163.0, "dynamic_persons": 3, "static_detections": 0, "avg_confidence": 0.5484908719857534, "avg_bbox_area_pct": 0.173960119176794, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2163.0, "end_sec": 2164.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.727159172296524, "avg_bbox_area_pct": 0.18189253442081404, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2164.0, "end_sec": 2165.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6591126918792725, "avg_bbox_area_pct": 0.10896112889419367, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2165.0, "end_sec": 2166.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6518218219280243, "avg_bbox_area_pct": 0.21982892071759258, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2166.0, "end_sec": 2167.0, "dynamic_persons": 3, "static_detections": 0, "avg_confidence": 0.583522786696752, "avg_bbox_area_pct": 0.19288007571373456, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2167.0, "end_sec": 2170.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6731642087300619, "avg_bbox_area_pct": 0.22960276913740998, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2170.0, "end_sec": 2184.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7487444047416959, "avg_bbox_area_pct": 0.24601446942257083, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2184.0, "end_sec": 2185.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7720727324485779, "avg_bbox_area_pct": 0.31344673816068674, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2185.0, "end_sec": 2187.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7273826599121094, "avg_bbox_area_pct": 0.1665695755570023, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2187.0, "end_sec": 2188.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7990745306015015, "avg_bbox_area_pct": 0.2026188979914159, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2188.0, "end_sec": 2193.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6230405330657959, "avg_bbox_area_pct": 0.18951686077353397, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2193.0, "end_sec": 2195.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.4996917024254799, "avg_bbox_area_pct": 0.18558248260874807, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2195.0, "end_sec": 2196.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5283420085906982, "avg_bbox_area_pct": 0.1236546947337963, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2196.0, "end_sec": 2197.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.4806523025035858, "avg_bbox_area_pct": 0.20264493965808256, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2197.0, "end_sec": 2198.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5393003225326538, "avg_bbox_area_pct": 0.12515902295524692, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2198.0, "end_sec": 2202.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.4679768681526184, "avg_bbox_area_pct": 0.22035413577232832, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2202.0, "end_sec": 2204.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.47429464757442474, "avg_bbox_area_pct": 0.5395041684751157, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2204.0, "end_sec": 2206.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5921004563570023, "avg_bbox_area_pct": 0.18433002048068575, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2206.0, "end_sec": 2212.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.723713884751002, "avg_bbox_area_pct": 0.1732408587922775, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2212.0, "end_sec": 2215.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5165703396002451, "avg_bbox_area_pct": 0.11422894387578768, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2215.0, "end_sec": 2223.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7403253987431526, "avg_bbox_area_pct": 0.1481952252800082, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2223.0, "end_sec": 2229.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.686931607623895, "avg_bbox_area_pct": 0.1474435883196293, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2229.0, "end_sec": 2234.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6124090075492858, "avg_bbox_area_pct": 0.21044930314429014, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2234.0, "end_sec": 2245.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6817934485999021, "avg_bbox_area_pct": 0.13184931557052734, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2245.0, "end_sec": 2247.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7198481261730194, "avg_bbox_area_pct": 0.13035996802059222, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2247.0, "end_sec": 2248.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7104324400424957, "avg_bbox_area_pct": 0.14819490409191743, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2248.0, "end_sec": 2260.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7289336572090784, "avg_bbox_area_pct": 0.15654243940188559, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2260.0, "end_sec": 2261.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6021186858415604, "avg_bbox_area_pct": 0.27513560353973765, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2261.0, "end_sec": 2265.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7449441850185394, "avg_bbox_area_pct": 0.1328580352406443, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2265.0, "end_sec": 2266.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.632187008857727, "avg_bbox_area_pct": 0.11482833673924575, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2266.0, "end_sec": 2268.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6349661648273468, "avg_bbox_area_pct": 0.13102808069299768, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2268.0, "end_sec": 2270.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7001106142997742, "avg_bbox_area_pct": 0.15874671465084877, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2270.0, "end_sec": 2271.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.36427003145217896, "avg_bbox_area_pct": 0.33587646484375, "bbox_variance": 0.0, "usable": false, "reason": "low_confidence"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2271.0, "end_sec": 2272.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6996414065361023, "avg_bbox_area_pct": 0.10734177577642746, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2272.0, "end_sec": 2273.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6072337031364441, "avg_bbox_area_pct": 0.15196183946397568, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2273.0, "end_sec": 2279.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6914444516102473, "avg_bbox_area_pct": 0.28649345366552537, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2279.0, "end_sec": 2285.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6438059359788895, "avg_bbox_area_pct": 0.16082354070718397, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2285.0, "end_sec": 2286.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8064531683921814, "avg_bbox_area_pct": 0.16192634488329474, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2286.0, "end_sec": 2288.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6817826926708221, "avg_bbox_area_pct": 0.13694679354443962, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2288.0, "end_sec": 2291.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.751239279905955, "avg_bbox_area_pct": 0.13460116665059155, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2291.0, "end_sec": 2292.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6337898075580597, "avg_bbox_area_pct": 0.13140096782166283, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2292.0, "end_sec": 2293.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7834749221801758, "avg_bbox_area_pct": 0.13021743586033951, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2293.0, "end_sec": 2294.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.814103364944458, "avg_bbox_area_pct": 0.14698236912856866, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2294.0, "end_sec": 2332.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.716258314879317, "avg_bbox_area_pct": 0.28793212890625, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2332.0, "end_sec": 2333.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5817688405513763, "avg_bbox_area_pct": 0.144086612654321, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2333.0, "end_sec": 2336.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7489748001098633, "avg_bbox_area_pct": 0.1204321891878858, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2336.0, "end_sec": 2340.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7979545667767525, "avg_bbox_area_pct": 0.1244518741560571, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2340.0, "end_sec": 2359.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7529091427200719, "avg_bbox_area_pct": 0.1195711610036829, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2359.0, "end_sec": 2361.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5892810374498367, "avg_bbox_area_pct": 0.19072831518856098, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2361.0, "end_sec": 2362.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7072789669036865, "avg_bbox_area_pct": 0.5151051311728395, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2362.0, "end_sec": 2363.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.580165833234787, "avg_bbox_area_pct": 0.09639867335190008, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2363.0, "end_sec": 2367.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6463920772075653, "avg_bbox_area_pct": 0.10541298383547935, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2367.0, "end_sec": 2368.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7845600545406342, "avg_bbox_area_pct": 0.10266179967809606, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2368.0, "end_sec": 2408.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8000337928533554, "avg_bbox_area_pct": 0.13539951672966097, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2408.0, "end_sec": 2409.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5797278583049774, "avg_bbox_area_pct": 0.15096547067901234, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2409.0, "end_sec": 2413.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6582633703947067, "avg_bbox_area_pct": 0.11862513224283855, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2413.0, "end_sec": 2415.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8140659481287003, "avg_bbox_area_pct": 0.17413028293185764, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2415.0, "end_sec": 2441.0, "dynamic_persons": 1, "static_detections": 1, "avg_confidence": 0.7343810636263627, "avg_bbox_area_pct": 0.13808794041531264, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2441.0, "end_sec": 2443.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6923333331942558, "avg_bbox_area_pct": 0.17982213338216146, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2443.0, "end_sec": 2444.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6782483458518982, "avg_bbox_area_pct": 0.14562557267554013, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2444.0, "end_sec": 2447.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5509426395098368, "avg_bbox_area_pct": 0.18486043796617802, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2447.0, "end_sec": 2448.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7339099049568176, "avg_bbox_area_pct": 0.12655122733410493, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2448.0, "end_sec": 2457.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7480019595887926, "avg_bbox_area_pct": 0.17057565948109568, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2457.0, "end_sec": 2468.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8580320369113575, "avg_bbox_area_pct": 0.13546329999210857, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2468.0, "end_sec": 2470.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7285255938768387, "avg_bbox_area_pct": 0.1666618384843991, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2470.0, "end_sec": 2471.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6849440932273865, "avg_bbox_area_pct": 0.19865523726851853, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2471.0, "end_sec": 2473.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6013473123311996, "avg_bbox_area_pct": 0.1704237648292824, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2473.0, "end_sec": 2474.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6789517402648926, "avg_bbox_area_pct": 0.23151187849633487, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2474.0, "end_sec": 2475.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5519505590200424, "avg_bbox_area_pct": 0.2255025378568673, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2475.0, "end_sec": 2476.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.45085227489471436, "avg_bbox_area_pct": 0.21125703788097994, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2476.0, "end_sec": 2496.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5218527553937374, "avg_bbox_area_pct": 0.16472458769679635, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2496.0, "end_sec": 2497.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.570229709148407, "avg_bbox_area_pct": 0.10651792173032408, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2497.0, "end_sec": 2504.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5747247764042446, "avg_bbox_area_pct": 0.11712627969301147, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2504.0, "end_sec": 2531.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8294267168751469, "avg_bbox_area_pct": 0.13692485793627832, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2531.0, "end_sec": 2534.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7136362791061401, "avg_bbox_area_pct": 0.2042177616323463, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2534.0, "end_sec": 2538.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6365591064095497, "avg_bbox_area_pct": 0.287689224054784, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2538.0, "end_sec": 2539.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2539.0, "end_sec": 2543.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7179993018507957, "avg_bbox_area_pct": 0.3364644519193673, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2543.0, "end_sec": 2549.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.676969937980175, "avg_bbox_area_pct": 0.2166916421019001, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2549.0, "end_sec": 2560.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7369937571612272, "avg_bbox_area_pct": 0.17779845917398815, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2560.0, "end_sec": 2561.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2561.0, "end_sec": 2569.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8344806432723999, "avg_bbox_area_pct": 0.12367674556779273, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2569.0, "end_sec": 2573.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5480034686625004, "avg_bbox_area_pct": 0.14934466420868298, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2573.0, "end_sec": 2578.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7436557531356811, "avg_bbox_area_pct": 0.11862291124131945, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2578.0, "end_sec": 2615.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8082097406322891, "avg_bbox_area_pct": 0.11628595982864375, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2615.0, "end_sec": 2625.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8116965413093566, "avg_bbox_area_pct": 0.14179081669560184, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2625.0, "end_sec": 2626.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.774794340133667, "avg_bbox_area_pct": 0.14217433599778162, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2626.0, "end_sec": 2627.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8223035335540771, "avg_bbox_area_pct": 0.2554833682966821, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2627.0, "end_sec": 2629.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7260388731956482, "avg_bbox_area_pct": 0.09679820873119213, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2629.0, "end_sec": 2650.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7656755078406561, "avg_bbox_area_pct": 0.127170368892035, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2650.0, "end_sec": 2651.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5866517722606659, "avg_bbox_area_pct": 0.1757744381751543, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2651.0, "end_sec": 2658.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6731513908931187, "avg_bbox_area_pct": 0.11786759741512345, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2658.0, "end_sec": 2659.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7474893927574158, "avg_bbox_area_pct": 0.15883622157720872, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2659.0, "end_sec": 2663.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7862201035022736, "avg_bbox_area_pct": 0.1510168758439429, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2663.0, "end_sec": 2665.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8034710586071014, "avg_bbox_area_pct": 0.15804770764009451, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2665.0, "end_sec": 2673.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7568929977715015, "avg_bbox_area_pct": 0.22231704523533952, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2673.0, "end_sec": 2674.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5659268647432327, "avg_bbox_area_pct": 0.21018302258150076, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2674.0, "end_sec": 2728.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7754077679581113, "avg_bbox_area_pct": 0.1378739666164766, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2728.0, "end_sec": 2729.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.67946457862854, "avg_bbox_area_pct": 0.16424465980058836, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2729.0, "end_sec": 2731.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5705210119485855, "avg_bbox_area_pct": 0.11299850275487075, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2731.0, "end_sec": 2735.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6506017632782459, "avg_bbox_area_pct": 0.12317674424913194, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2735.0, "end_sec": 2736.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8610231280326843, "avg_bbox_area_pct": 0.34474910783179014, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2736.0, "end_sec": 2738.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5960644781589508, "avg_bbox_area_pct": 0.19208123289508583, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2738.0, "end_sec": 2742.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8539163768291473, "avg_bbox_area_pct": 0.25628800757137343, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2742.0, "end_sec": 2745.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6981244484583536, "avg_bbox_area_pct": 0.17630248521090533, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2745.0, "end_sec": 2746.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6722701191902161, "avg_bbox_area_pct": 0.482942195939429, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2746.0, "end_sec": 2747.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2747.0, "end_sec": 2751.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7590148001909256, "avg_bbox_area_pct": 0.18189397741247107, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2751.0, "end_sec": 2754.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5448752492666245, "avg_bbox_area_pct": 0.2344398065260899, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2754.0, "end_sec": 2755.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5032328367233276, "avg_bbox_area_pct": 0.2693160144193673, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2755.0, "end_sec": 2756.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.4506537616252899, "avg_bbox_area_pct": 0.1854554729697145, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2756.0, "end_sec": 2759.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5196335017681122, "avg_bbox_area_pct": 0.23539281523276748, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2759.0, "end_sec": 2760.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.615963339805603, "avg_bbox_area_pct": 0.16075576593846452, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2760.0, "end_sec": 2767.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7140486666134426, "avg_bbox_area_pct": 0.12445740714905752, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2767.0, "end_sec": 2768.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6977725923061371, "avg_bbox_area_pct": 0.13208356692467207, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2768.0, "end_sec": 2775.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6897477720464978, "avg_bbox_area_pct": 0.12203893806148038, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2775.0, "end_sec": 2781.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7606171319882075, "avg_bbox_area_pct": 0.1755934695648068, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2781.0, "end_sec": 2782.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8110703825950623, "avg_bbox_area_pct": 0.1310126711998457, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2782.0, "end_sec": 2786.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7892902940511703, "avg_bbox_area_pct": 0.16468830720877942, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2786.0, "end_sec": 2788.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8391539454460144, "avg_bbox_area_pct": 0.34985093858506944, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2788.0, "end_sec": 2791.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7340527971585592, "avg_bbox_area_pct": 0.21075643405992803, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2791.0, "end_sec": 2807.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7462655082345009, "avg_bbox_area_pct": 0.24266266010425708, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2807.0, "end_sec": 2809.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6641606241464615, "avg_bbox_area_pct": 0.23552203519844714, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2809.0, "end_sec": 2812.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7428411245346069, "avg_bbox_area_pct": 0.12642752439396862, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2812.0, "end_sec": 2813.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6055275797843933, "avg_bbox_area_pct": 0.10107797504943095, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2813.0, "end_sec": 2816.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7412596742312113, "avg_bbox_area_pct": 0.11408791122122557, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2816.0, "end_sec": 2817.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5479167997837067, "avg_bbox_area_pct": 0.12317721143180942, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2817.0, "end_sec": 2818.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7240546941757202, "avg_bbox_area_pct": 0.11381513430748456, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2818.0, "end_sec": 2819.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.644506424665451, "avg_bbox_area_pct": 0.09961469108675733, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2819.0, "end_sec": 2820.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8075674772262573, "avg_bbox_area_pct": 0.11405494972511573, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2820.0, "end_sec": 2824.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7507704570889473, "avg_bbox_area_pct": 0.15841981770079813, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2824.0, "end_sec": 2825.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7965513467788696, "avg_bbox_area_pct": 0.12369930314429012, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2825.0, "end_sec": 2828.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6730872243642807, "avg_bbox_area_pct": 0.12636268898292824, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2828.0, "end_sec": 2830.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8620454370975494, "avg_bbox_area_pct": 0.11163926866319444, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2830.0, "end_sec": 2831.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5818672478199005, "avg_bbox_area_pct": 0.2903345781491127, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2831.0, "end_sec": 2843.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7036513239145279, "avg_bbox_area_pct": 0.2043021402241271, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2843.0, "end_sec": 2844.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5917801707983017, "avg_bbox_area_pct": 0.19438580171561537, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2844.0, "end_sec": 2847.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7679793039957682, "avg_bbox_area_pct": 0.30547834985050154, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2847.0, "end_sec": 2848.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.609959065914154, "avg_bbox_area_pct": 0.13122555956428433, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2848.0, "end_sec": 2850.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7267879247665405, "avg_bbox_area_pct": 0.11377362663363233, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2850.0, "end_sec": 2853.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6120584706465403, "avg_bbox_area_pct": 0.1401129351329411, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2853.0, "end_sec": 2884.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7746131900818117, "avg_bbox_area_pct": 0.1357825971001033, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2884.0, "end_sec": 2886.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7487377822399139, "avg_bbox_area_pct": 0.12258680743935668, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2886.0, "end_sec": 2887.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7990342974662781, "avg_bbox_area_pct": 0.13629471390335648, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2887.0, "end_sec": 2888.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6165625303983688, "avg_bbox_area_pct": 0.13236018428096064, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2888.0, "end_sec": 2889.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6171883344650269, "avg_bbox_area_pct": 0.15018803349247686, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2889.0, "end_sec": 2891.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7719439566135406, "avg_bbox_area_pct": 0.21004357985508293, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2891.0, "end_sec": 2892.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7711315155029297, "avg_bbox_area_pct": 0.11591142065731096, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2892.0, "end_sec": 2893.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.77656090259552, "avg_bbox_area_pct": 0.10418853006245177, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2893.0, "end_sec": 2894.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7735442519187927, "avg_bbox_area_pct": 0.12132088366849923, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2894.0, "end_sec": 2896.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6238531246781349, "avg_bbox_area_pct": 0.13649289543246046, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2896.0, "end_sec": 2897.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7611863017082214, "avg_bbox_area_pct": 0.11925743573977624, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2897.0, "end_sec": 2899.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6871510520577431, "avg_bbox_area_pct": 0.17110059196566357, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2899.0, "end_sec": 2900.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6502107381820679, "avg_bbox_area_pct": 0.12642924744405865, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2900.0, "end_sec": 2901.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7658264935016632, "avg_bbox_area_pct": 0.1397537871937693, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2901.0, "end_sec": 2920.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7661363388362684, "avg_bbox_area_pct": 0.13535626248464405, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2920.0, "end_sec": 2921.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8094512522220612, "avg_bbox_area_pct": 0.13741761007426698, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2921.0, "end_sec": 2974.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8117577056839781, "avg_bbox_area_pct": 0.14968349667381856, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2974.0, "end_sec": 2975.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.715180367231369, "avg_bbox_area_pct": 0.15322733561197915, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2975.0, "end_sec": 2977.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8025554716587067, "avg_bbox_area_pct": 0.13103874300733026, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2977.0, "end_sec": 2978.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6117697805166245, "avg_bbox_area_pct": 0.1524597469376929, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2978.0, "end_sec": 2979.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7018663883209229, "avg_bbox_area_pct": 0.13105349693769291, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2979.0, "end_sec": 2980.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7918370962142944, "avg_bbox_area_pct": 0.145916145230517, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2980.0, "end_sec": 2996.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7588553987443447, "avg_bbox_area_pct": 0.13638296763102215, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 2996.0, "end_sec": 3002.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7660335898399353, "avg_bbox_area_pct": 0.12144686914765786, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3002.0, "end_sec": 3003.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8057200312614441, "avg_bbox_area_pct": 0.23603850188078704, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3003.0, "end_sec": 3004.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6394309103488922, "avg_bbox_area_pct": 0.28412584846402394, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3004.0, "end_sec": 3005.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.47266387939453125, "avg_bbox_area_pct": 0.3869817286361883, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3005.0, "end_sec": 3006.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5537759363651276, "avg_bbox_area_pct": 0.26496164580922066, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3006.0, "end_sec": 3010.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5670686736702919, "avg_bbox_area_pct": 0.38160317503375774, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3010.0, "end_sec": 3012.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.539761520922184, "avg_bbox_area_pct": 0.2797864183967496, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3012.0, "end_sec": 3013.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7533257603645325, "avg_bbox_area_pct": 0.30325288749035495, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3013.0, "end_sec": 3015.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6087014898657799, "avg_bbox_area_pct": 0.1606845733265818, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3015.0, "end_sec": 3019.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.786356970667839, "avg_bbox_area_pct": 0.34574699496045525, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3019.0, "end_sec": 3022.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6339927464723587, "avg_bbox_area_pct": 0.12937638647762345, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3022.0, "end_sec": 3025.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8201561570167542, "avg_bbox_area_pct": 0.16413004808465148, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3025.0, "end_sec": 3027.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7532098293304443, "avg_bbox_area_pct": 0.13708631539050445, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3027.0, "end_sec": 3029.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7922140955924988, "avg_bbox_area_pct": 0.1299984402126736, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3029.0, "end_sec": 3030.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8233552277088165, "avg_bbox_area_pct": 0.15492314091435186, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3030.0, "end_sec": 3031.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8134741187095642, "avg_bbox_area_pct": 0.14383142541956018, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3031.0, "end_sec": 3034.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6725228677193323, "avg_bbox_area_pct": 0.1571900707605935, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3034.0, "end_sec": 3037.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7898854811986288, "avg_bbox_area_pct": 0.13681725270463607, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3037.0, "end_sec": 3038.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.3943508267402649, "avg_bbox_area_pct": 0.26805739414544755, "bbox_variance": 0.0, "usable": false, "reason": "low_confidence"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3038.0, "end_sec": 3041.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5641066879034042, "avg_bbox_area_pct": 0.11341365610130531, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3041.0, "end_sec": 3042.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.4373866021633148, "avg_bbox_area_pct": 0.26891221788194447, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3042.0, "end_sec": 3043.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5154906213283539, "avg_bbox_area_pct": 0.14271289589964314, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3043.0, "end_sec": 3046.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5502240757147471, "avg_bbox_area_pct": 0.2293945915316358, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3046.0, "end_sec": 3047.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5087253153324127, "avg_bbox_area_pct": 0.09236485610773534, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3047.0, "end_sec": 3049.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.49839210510253906, "avg_bbox_area_pct": 0.26775664906442903, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3049.0, "end_sec": 3050.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.44832228124141693, "avg_bbox_area_pct": 0.10421010335286458, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3050.0, "end_sec": 3057.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5701604230063302, "avg_bbox_area_pct": 0.16995785237199623, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3057.0, "end_sec": 3058.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.45037874579429626, "avg_bbox_area_pct": 0.12664091133777006, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3058.0, "end_sec": 3074.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8067289963364601, "avg_bbox_area_pct": 0.17279548880494672, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3074.0, "end_sec": 3076.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3076.0, "end_sec": 3079.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5975933074951172, "avg_bbox_area_pct": 0.0989323670187114, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3079.0, "end_sec": 3081.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5603909641504288, "avg_bbox_area_pct": 0.13518611578293788, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3081.0, "end_sec": 3085.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7114621847867966, "avg_bbox_area_pct": 0.1174097188313802, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3085.0, "end_sec": 3086.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7064354717731476, "avg_bbox_area_pct": 0.1384469265407986, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3086.0, "end_sec": 3092.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7547211945056915, "avg_bbox_area_pct": 0.21363419897762348, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3092.0, "end_sec": 3094.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7753565013408661, "avg_bbox_area_pct": 0.13739603301625192, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3094.0, "end_sec": 3098.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6725919619202614, "avg_bbox_area_pct": 0.31763893786771796, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3098.0, "end_sec": 3099.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3099.0, "end_sec": 3100.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6354572474956512, "avg_bbox_area_pct": 0.09911867494936343, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3100.0, "end_sec": 3105.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7422118306159973, "avg_bbox_area_pct": 0.12688471890673225, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3105.0, "end_sec": 3106.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7619966566562653, "avg_bbox_area_pct": 0.14756798638237847, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3106.0, "end_sec": 3112.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7553934653600057, "avg_bbox_area_pct": 0.11342680000964506, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3112.0, "end_sec": 3113.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5878225117921829, "avg_bbox_area_pct": 0.13903921998577354, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3113.0, "end_sec": 3118.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7149950385093689, "avg_bbox_area_pct": 0.11386803747106482, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3118.0, "end_sec": 3119.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6365536451339722, "avg_bbox_area_pct": 0.09838720627772955, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3119.0, "end_sec": 3120.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8546769618988037, "avg_bbox_area_pct": 0.19187379436728394, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3120.0, "end_sec": 3125.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5968344449996948, "avg_bbox_area_pct": 0.11299963152850115, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3125.0, "end_sec": 3129.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6636395305395126, "avg_bbox_area_pct": 0.17470363287278162, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3129.0, "end_sec": 3133.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7429031208157539, "avg_bbox_area_pct": 0.1622885442663122, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3133.0, "end_sec": 3134.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7005603909492493, "avg_bbox_area_pct": 0.2363084731867284, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3134.0, "end_sec": 3137.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7406185269355774, "avg_bbox_area_pct": 0.16670830243899495, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3137.0, "end_sec": 3139.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6945524215698242, "avg_bbox_area_pct": 0.21075822995032795, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3139.0, "end_sec": 3140.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6149324178695679, "avg_bbox_area_pct": 0.31311200930748456, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3140.0, "end_sec": 3141.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7210100889205933, "avg_bbox_area_pct": 0.4450099163290895, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3141.0, "end_sec": 3142.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.4613118916749954, "avg_bbox_area_pct": 0.27435449670862266, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3142.0, "end_sec": 3145.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6106597781181335, "avg_bbox_area_pct": 0.18658787856867284, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3145.0, "end_sec": 3147.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.4251779317855835, "avg_bbox_area_pct": 0.21747185224368248, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3147.0, "end_sec": 3151.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6194803416728973, "avg_bbox_area_pct": 0.19353569124951775, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3151.0, "end_sec": 3152.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7389253973960876, "avg_bbox_area_pct": 0.09986722592954284, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3152.0, "end_sec": 3185.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7678821755178047, "avg_bbox_area_pct": 0.1376009032381103, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3185.0, "end_sec": 3188.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6929852366447449, "avg_bbox_area_pct": 0.2506858580789448, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3188.0, "end_sec": 3189.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5507166385650635, "avg_bbox_area_pct": 0.33777015215084877, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3189.0, "end_sec": 3191.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.804424837231636, "avg_bbox_area_pct": 0.336654433262201, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3191.0, "end_sec": 3197.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6255359699328741, "avg_bbox_area_pct": 0.28265624497653036, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3197.0, "end_sec": 3198.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7475451827049255, "avg_bbox_area_pct": 0.3356871654369213, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3198.0, "end_sec": 3199.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8100076913833618, "avg_bbox_area_pct": 0.45995771243248457, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3199.0, "end_sec": 3201.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8213127702474594, "avg_bbox_area_pct": 0.33441794689790705, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3201.0, "end_sec": 3210.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7612390319506327, "avg_bbox_area_pct": 0.37806373962486073, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3210.0, "end_sec": 3211.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3211.0, "end_sec": 3212.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7063502073287964, "avg_bbox_area_pct": 0.10713845335407021, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3212.0, "end_sec": 3215.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7191004753112793, "avg_bbox_area_pct": 0.21659433498304073, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3215.0, "end_sec": 3217.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7795258164405823, "avg_bbox_area_pct": 0.30521131727430556, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3217.0, "end_sec": 3218.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5801822692155838, "avg_bbox_area_pct": 0.17379389256606867, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3218.0, "end_sec": 3220.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7368974983692169, "avg_bbox_area_pct": 0.266357278706115, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3220.0, "end_sec": 3223.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7958376705646515, "avg_bbox_area_pct": 0.172386115431295, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3223.0, "end_sec": 3225.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.79487344622612, "avg_bbox_area_pct": 0.13278615692515433, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3225.0, "end_sec": 3226.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6024904698133469, "avg_bbox_area_pct": 0.1535770293812693, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3226.0, "end_sec": 3228.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8040550351142883, "avg_bbox_area_pct": 0.13054655098620757, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3228.0, "end_sec": 3230.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6951560080051422, "avg_bbox_area_pct": 0.14693663420500577, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3230.0, "end_sec": 3235.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8545502305030823, "avg_bbox_area_pct": 0.19986354347511576, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3235.0, "end_sec": 3236.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.703798234462738, "avg_bbox_area_pct": 0.17888358033733603, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3236.0, "end_sec": 3241.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5854869246482849, "avg_bbox_area_pct": 0.18127942648051698, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3241.0, "end_sec": 3243.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7917978316545486, "avg_bbox_area_pct": 0.10923166910807292, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3243.0, "end_sec": 3244.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8228266835212708, "avg_bbox_area_pct": 0.30333375530478396, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3244.0, "end_sec": 3245.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.747818261384964, "avg_bbox_area_pct": 0.0976025390625, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3245.0, "end_sec": 3253.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7271819859743118, "avg_bbox_area_pct": 0.12035345006872106, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3253.0, "end_sec": 3255.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7341779917478561, "avg_bbox_area_pct": 0.14188766479492188, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3255.0, "end_sec": 3263.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7652891352772713, "avg_bbox_area_pct": 0.16928037666980134, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3263.0, "end_sec": 3264.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3264.0, "end_sec": 3265.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5724528431892395, "avg_bbox_area_pct": 0.14215747974537038, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3265.0, "end_sec": 3266.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6965325176715851, "avg_bbox_area_pct": 0.13647350546754436, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3266.0, "end_sec": 3280.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7849622879709516, "avg_bbox_area_pct": 0.17841726313192377, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3280.0, "end_sec": 3282.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6243032962083817, "avg_bbox_area_pct": 0.1334356670615114, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3282.0, "end_sec": 3283.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5969443321228027, "avg_bbox_area_pct": 0.09725579155815972, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3283.0, "end_sec": 3284.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5847269296646118, "avg_bbox_area_pct": 0.10197680438006365, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3284.0, "end_sec": 3289.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6669210314750671, "avg_bbox_area_pct": 0.11962709026572145, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3289.0, "end_sec": 3292.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6296045730511347, "avg_bbox_area_pct": 0.20321135344328703, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3292.0, "end_sec": 3299.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6204473376274109, "avg_bbox_area_pct": 0.2693601373474014, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3299.0, "end_sec": 3313.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8000597506761551, "avg_bbox_area_pct": 0.19875421917627728, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3313.0, "end_sec": 3314.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6668544411659241, "avg_bbox_area_pct": 0.11047032485773534, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3314.0, "end_sec": 3315.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5601334422826767, "avg_bbox_area_pct": 0.16251208270037615, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3315.0, "end_sec": 3316.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.38182708621025085, "avg_bbox_area_pct": 0.11505213607976467, "bbox_variance": 0.0, "usable": false, "reason": "low_confidence"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3316.0, "end_sec": 3324.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7439838536083698, "avg_bbox_area_pct": 0.11784134664653259, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3324.0, "end_sec": 3325.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.666069895029068, "avg_bbox_area_pct": 0.12181734061535493, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3325.0, "end_sec": 3330.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7440213084220886, "avg_bbox_area_pct": 0.11858188355999229, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3330.0, "end_sec": 3332.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6528646349906921, "avg_bbox_area_pct": 0.10682675962094908, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3332.0, "end_sec": 3334.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.755603164434433, "avg_bbox_area_pct": 0.16252836627724732, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3334.0, "end_sec": 3340.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6839457377791405, "avg_bbox_area_pct": 0.24420519291128154, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3340.0, "end_sec": 3342.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8002564609050751, "avg_bbox_area_pct": 0.2544958872853974, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3342.0, "end_sec": 3348.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7897903074820837, "avg_bbox_area_pct": 0.23721618526757005, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3348.0, "end_sec": 3350.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6600692570209503, "avg_bbox_area_pct": 0.13306660216531635, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3350.0, "end_sec": 3351.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7608968019485474, "avg_bbox_area_pct": 0.2401011676552855, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3351.0, "end_sec": 3354.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6982832749684652, "avg_bbox_area_pct": 0.29163713620032794, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3354.0, "end_sec": 3359.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7181960612535476, "avg_bbox_area_pct": 0.1947175436137635, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3359.0, "end_sec": 3360.0, "dynamic_persons": 3, "static_detections": 0, "avg_confidence": 0.6479767064253489, "avg_bbox_area_pct": 0.13796138810522762, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3360.0, "end_sec": 3361.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7806126475334167, "avg_bbox_area_pct": 0.20582615228346834, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3361.0, "end_sec": 3364.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7936736941337585, "avg_bbox_area_pct": 0.1340851005899563, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3364.0, "end_sec": 3365.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7348816990852356, "avg_bbox_area_pct": 0.10171366373697917, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3365.0, "end_sec": 3370.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6204959690570832, "avg_bbox_area_pct": 0.16136217056086033, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3370.0, "end_sec": 3372.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6621250212192535, "avg_bbox_area_pct": 0.11536052374192227, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3372.0, "end_sec": 3373.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6241212487220764, "avg_bbox_area_pct": 0.10778172622492284, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3373.0, "end_sec": 3375.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7672347277402878, "avg_bbox_area_pct": 0.11016346872588734, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3375.0, "end_sec": 3376.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6980814337730408, "avg_bbox_area_pct": 0.10757927035108025, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3376.0, "end_sec": 3378.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5902935862541199, "avg_bbox_area_pct": 0.1560057896743586, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3378.0, "end_sec": 3390.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7528605510791143, "avg_bbox_area_pct": 0.16849423632209684, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3390.0, "end_sec": 3392.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3392.0, "end_sec": 3396.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5653383359313011, "avg_bbox_area_pct": 0.32470271357783564, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3396.0, "end_sec": 3397.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3397.0, "end_sec": 3404.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6836119379316058, "avg_bbox_area_pct": 0.28562479977885247, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3404.0, "end_sec": 3406.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8337470442056656, "avg_bbox_area_pct": 0.22263242368344904, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3406.0, "end_sec": 3409.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7898996074994405, "avg_bbox_area_pct": 0.28639796127507716, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3409.0, "end_sec": 3411.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.3795744329690933, "avg_bbox_area_pct": 0.5167618965808256, "bbox_variance": 0.0, "usable": false, "reason": "low_confidence"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3411.0, "end_sec": 3475.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7296664179302752, "avg_bbox_area_pct": 0.3581118684933509, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3475.0, "end_sec": 3476.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3476.0, "end_sec": 3477.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.46788284182548523, "avg_bbox_area_pct": 0.048259092731240356, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3477.0, "end_sec": 3562.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3562.0, "end_sec": 3563.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.39237847924232483, "avg_bbox_area_pct": 0.227367862654321, "bbox_variance": 0.0, "usable": false, "reason": "low_confidence"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3563.0, "end_sec": 3627.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7584704030305147, "avg_bbox_area_pct": 0.3221403277361834, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3627.0, "end_sec": 3629.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.6731403842568398, "avg_bbox_area_pct": 0.23115258864414542, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3629.0, "end_sec": 3633.0, "dynamic_persons": 1, "static_detections": 1, "avg_confidence": 0.747947096824646, "avg_bbox_area_pct": 0.17500512393904322, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3633.0, "end_sec": 3634.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3634.0, "end_sec": 3635.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8206733465194702, "avg_bbox_area_pct": 0.23713812934027778, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3635.0, "end_sec": 3637.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3637.0, "end_sec": 3640.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.646055539449056, "avg_bbox_area_pct": 0.20657518426086677, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3640.0, "end_sec": 3647.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3647.0, "end_sec": 3653.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7763245950142542, "avg_bbox_area_pct": 0.17360571590470678, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3653.0, "end_sec": 3654.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.574862465262413, "avg_bbox_area_pct": 0.18334058973524306, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3654.0, "end_sec": 3674.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7112200349569321, "avg_bbox_area_pct": 0.2533670383029514, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3674.0, "end_sec": 3676.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.3743056654930115, "avg_bbox_area_pct": 0.20973448199990355, "bbox_variance": 0.0, "usable": false, "reason": "low_confidence"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3676.0, "end_sec": 3689.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5159432314909421, "avg_bbox_area_pct": 0.3561501469480799, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3689.0, "end_sec": 3690.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3690.0, "end_sec": 3691.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.4979454278945923, "avg_bbox_area_pct": 0.19200670030381944, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3691.0, "end_sec": 3692.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3692.0, "end_sec": 3694.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6390490531921387, "avg_bbox_area_pct": 0.2415107934268904, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3694.0, "end_sec": 3695.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3695.0, "end_sec": 3696.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5887824296951294, "avg_bbox_area_pct": 0.10533474392361111, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3696.0, "end_sec": 3698.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3698.0, "end_sec": 3699.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5481196045875549, "avg_bbox_area_pct": 0.06936277036313658, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3699.0, "end_sec": 3700.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.36514297127723694, "avg_bbox_area_pct": 0.05559968924816744, "bbox_variance": 0.0, "usable": false, "reason": "low_confidence"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3700.0, "end_sec": 3701.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3701.0, "end_sec": 3711.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5624190896749497, "avg_bbox_area_pct": 0.07174994461624709, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3711.0, "end_sec": 3717.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.3655831515789032, "avg_bbox_area_pct": 0.09308678897810571, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3717.0, "end_sec": 3718.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.4114591181278229, "avg_bbox_area_pct": 0.09327025613667052, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3718.0, "end_sec": 3719.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.3522477447986603, "avg_bbox_area_pct": 0.07169608410493827, "bbox_variance": 0.0, "usable": false, "reason": "low_confidence"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3719.0, "end_sec": 3724.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3724.0, "end_sec": 3749.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7250602686405182, "avg_bbox_area_pct": 0.14610946481963735, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3749.0, "end_sec": 3750.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.3950442671775818, "avg_bbox_area_pct": 0.1349759966061439, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3750.0, "end_sec": 3760.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7190123915672302, "avg_bbox_area_pct": 0.27927556393470293, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3760.0, "end_sec": 3762.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.73302361369133, "avg_bbox_area_pct": 0.17405356324749227, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3762.0, "end_sec": 3764.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5888531804084778, "avg_bbox_area_pct": 0.15267366385754244, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3764.0, "end_sec": 3768.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3768.0, "end_sec": 3774.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6844950318336487, "avg_bbox_area_pct": 0.32237559126237786, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3774.0, "end_sec": 3776.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3776.0, "end_sec": 3777.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6535285115242004, "avg_bbox_area_pct": 0.2564088722511574, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3777.0, "end_sec": 3780.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3780.0, "end_sec": 3804.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6647906539340814, "avg_bbox_area_pct": 0.36270411236295974, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3804.0, "end_sec": 3805.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3805.0, "end_sec": 3811.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.5265195220708847, "avg_bbox_area_pct": 0.46336532781153555, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3811.0, "end_sec": 3812.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.814565122127533, "avg_bbox_area_pct": 0.2354951759620949, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3812.0, "end_sec": 3814.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.4742688536643982, "avg_bbox_area_pct": 0.24117349506896218, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3814.0, "end_sec": 3815.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5460800528526306, "avg_bbox_area_pct": 0.2000439679181134, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3815.0, "end_sec": 3817.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6691089570522308, "avg_bbox_area_pct": 0.3953349699797454, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3817.0, "end_sec": 3820.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7786709864934286, "avg_bbox_area_pct": 0.2759216208124357, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3820.0, "end_sec": 3821.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8435958027839661, "avg_bbox_area_pct": 0.3691775173611111, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3821.0, "end_sec": 3823.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7465824633836746, "avg_bbox_area_pct": 0.30161327974295915, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3823.0, "end_sec": 3833.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7641504347324372, "avg_bbox_area_pct": 0.30082584183304395, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3833.0, "end_sec": 3834.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3834.0, "end_sec": 3835.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8420389890670776, "avg_bbox_area_pct": 0.4514604130497685, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3835.0, "end_sec": 3836.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3836.0, "end_sec": 3837.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7977710366249084, "avg_bbox_area_pct": 0.3886456524884259, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3837.0, "end_sec": 3839.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3839.0, "end_sec": 3840.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7931423187255859, "avg_bbox_area_pct": 0.4045894820601852, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3840.0, "end_sec": 3843.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3843.0, "end_sec": 3856.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.7569769850144019, "avg_bbox_area_pct": 0.6444568253650286, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3856.0, "end_sec": 3859.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3859.0, "end_sec": 3869.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8264600753784179, "avg_bbox_area_pct": 0.4879455807532794, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3869.0, "end_sec": 3870.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7660022079944611, "avg_bbox_area_pct": 0.34893364800347226, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3870.0, "end_sec": 3871.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8252179622650146, "avg_bbox_area_pct": 0.4465821216724537, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3871.0, "end_sec": 3875.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.5973461717367172, "avg_bbox_area_pct": 0.28518442224573204, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3875.0, "end_sec": 3877.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6264539808034897, "avg_bbox_area_pct": 0.4427834593219522, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3877.0, "end_sec": 3880.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7724592983722687, "avg_bbox_area_pct": 0.31106943389515823, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3880.0, "end_sec": 3881.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.737215518951416, "avg_bbox_area_pct": 0.5384294825424383, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3881.0, "end_sec": 3883.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.7316130846738815, "avg_bbox_area_pct": 0.3037563145013503, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3883.0, "end_sec": 3895.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6222153132160505, "avg_bbox_area_pct": 0.4085576023682646, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3895.0, "end_sec": 3896.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.649704098701477, "avg_bbox_area_pct": 0.38154355649594907, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3896.0, "end_sec": 3904.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.8568584322929382, "avg_bbox_area_pct": 0.4458044320565683, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3904.0, "end_sec": 3910.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.804749940832456, "avg_bbox_area_pct": 0.29621574621632263, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3910.0, "end_sec": 3911.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.825668454170227, "avg_bbox_area_pct": 0.5198509837962964, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3911.0, "end_sec": 3915.0, "dynamic_persons": 2, "static_detections": 0, "avg_confidence": 0.8171084672212601, "avg_bbox_area_pct": 0.24742471012068384, "bbox_variance": 0.0, "usable": false, "reason": "multiple_persons"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3915.0, "end_sec": 3919.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3919.0, "end_sec": 3920.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.6009199023246765, "avg_bbox_area_pct": 0.16394074616608798, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3920.0, "end_sec": 3925.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3925.0, "end_sec": 3927.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.4733169376850128, "avg_bbox_area_pct": 0.23629698953510803, "bbox_variance": 0.0, "usable": true, "reason": null}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3927.0, "end_sec": 3933.0, "dynamic_persons": 0, "static_detections": 0, "avg_confidence": 0.0, "avg_bbox_area_pct": 0.0, "bbox_variance": 0.0, "usable": false, "reason": "no_person"}
+{"video": "/root/miko/puni/train/PromptHMR/GENMO/videos/VRM_JG6Z7WA.mp4", "start_sec": 3933.0, "end_sec": 3935.0, "dynamic_persons": 1, "static_detections": 0, "avg_confidence": 0.525958240032196, "avg_bbox_area_pct": 0.19891710069444446, "bbox_variance": 0.0, "usable": true, "reason": null}
diff --git a/labels_test.json b/labels_test.json
new file mode 100644
index 0000000000000000000000000000000000000000..7b1686b992c014ac6989390dc192a339f7a2ec09
--- /dev/null
+++ b/labels_test.json
@@ -0,0 +1,9 @@
+{
+ "videos": [
+ {
+ "video": "videos/VRM_JG6Z7WA.mp4",
+ "error": "No frames extracted",
+ "segments": []
+ }
+ ]
+}
\ No newline at end of file
diff --git a/npz_out/out_global_smplx.npz b/npz_out/out_global_smplx.npz
new file mode 100644
index 0000000000000000000000000000000000000000..90ef26b7a3b753f9230d3d6fa61d0015e444753e
--- /dev/null
+++ b/npz_out/out_global_smplx.npz
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b5c6ac2dbc579a31ddb7b8748fdde6691c33996acc4a4f771c8b8e70e7bd8357
+size 212600
diff --git a/prepare_unity_data.py b/prepare_unity_data.py
new file mode 100644
index 0000000000000000000000000000000000000000..8b750cd8fc5abbf300922ead6df94eea2c740659
--- /dev/null
+++ b/prepare_unity_data.py
@@ -0,0 +1,186 @@
+import numpy as np
+import json
+import os
+import re
+from difflib import SequenceMatcher
+from tqdm import tqdm
+
+# --- CONFIGURATION ---
+INPUT_TRAIN_JSON = "./train.json"
+NPZ_FOLDER = "./npz_data"
+OUTPUT_DIR = "./unity_ready_json"
+FPS = 30.0
+
+# Styles
+STYLE_FILLIAN = 0
+STYLE_BIBOO = 1
+STYLE_ANNY = 2 # Default for unknown characters
+STYLE_LAPWING = 3 # Vlog Override
+
+def fuzzy_match_name(text, target, threshold=0.75):
+ tokens = re.split(r'[^a-z]+', text.lower())
+ for token in tokens:
+ if len(token) < 3: continue
+ if SequenceMatcher(None, token, target).ratio() >= threshold:
+ return True
+ return False
+
+def get_base_style(filename):
+ """
+ Determines global style based on filename.
+ Hierarchy: Biboo -> Fillian -> Anny (Default).
+ """
+ clean_name = filename.lower()
+
+ # 1. Check for Biboo
+ if fuzzy_match_name(clean_name, "biboo", threshold=0.8):
+ return STYLE_BIBOO
+
+ # 2. Check for Fillian
+ if fuzzy_match_name(clean_name, "fillian", threshold=0.8) or "filian" in clean_name:
+ return STYLE_FILLIAN
+
+ # 3. Default fallback for Miltina, Anny, and others
+ return STYLE_ANNY
+
+def is_vlog_label(label_entry):
+ """
+ Checks if label indicates vlogging/handheld camera.
+ CRITICAL FIX: Explicitly excludes 'end of vlog' or 'place camera back'.
+ """
+ proc_label = label_entry.get("proc_label", "").lower()
+
+ # 1. EXCLUSION RULES (If these exist, it is NOT vlogging)
+ if "place camera back" in proc_label or "end of vlog" in proc_label:
+ return False
+
+ # 2. INCLUSION RULES
+ if "vlog" in proc_label:
+ return True
+
+ if "act_cat" in label_entry:
+ for cat in label_entry["act_cat"]:
+ if "vlog" in cat.lower():
+ return True
+
+ return False
+
+def is_transition_label(label_entry):
+ """Checks if this is a generic transition label."""
+ proc = label_entry.get("proc_label", "").lower()
+ return "transition" in proc
+
+def process_single_entry(entry_id, entry_data):
+ npz_filename = entry_data.get("feat_p")
+ npz_path = os.path.join(NPZ_FOLDER, npz_filename)
+
+ if not os.path.exists(npz_path):
+ return
+
+ # 1. Load Data
+ try:
+ data = np.load(npz_path)
+ poses = data['poses']
+ trans = data['trans']
+ betas = data['betas']
+
+ if poses.ndim == 3: poses = poses[0]
+ if trans.ndim == 3: trans = trans[0]
+
+ num_frames = poses.shape[0]
+ except Exception as e:
+ print(f"❌ Error loading {npz_filename}: {e}")
+ return
+
+ # 2. Determine Base Style
+ base_style = get_base_style(npz_filename)
+
+ # Initialize all frames with the Base Style
+ frame_styles = np.full(num_frames, base_style, dtype=int)
+
+ # 3. Apply Vlog Logic (State Machine Override)
+ if "frame_ann" in entry_data and "labels" in entry_data["frame_ann"]:
+ # Sort labels by time
+ labels = sorted(entry_data["frame_ann"]["labels"], key=lambda x: x.get("start_t", 0))
+
+ previous_was_vlog = False
+
+ for label in labels:
+ start_t = label.get("start_t", 0.0)
+ end_t = label.get("end_t", 0.0)
+
+ s_f = max(0, int(start_t * FPS))
+ e_f = min(num_frames, int(end_t * FPS))
+
+ if e_f <= s_f: continue
+
+ if is_vlog_label(label):
+ # Vlog Label -> Set to Lapwing Style
+ frame_styles[s_f:e_f] = STYLE_LAPWING
+ previous_was_vlog = True
+
+ elif is_transition_label(label) and previous_was_vlog:
+ # Transition immediately after Vlog -> Collapse Gap (Keep as Vlog)
+ frame_styles[s_f:e_f] = STYLE_LAPWING
+ # Keep state as true
+
+ else:
+ # Regular action OR "Place camera back" -> Reset to Base Style
+ previous_was_vlog = False
+
+ # 4. Construct Frame Data
+ frames_data = []
+ poses_list = np.round(poses, 4).tolist()
+ trans_list = np.round(trans, 4).tolist()
+ betas_list = np.round(betas, 4).tolist()
+ styles_list = frame_styles.tolist()
+
+ for i in range(num_frames):
+ frame_entry = {
+ "i": i,
+ "p": poses_list[i],
+ "t": trans_list[i],
+ "b": betas_list,
+ "s": styles_list[i]
+ }
+ frames_data.append(frame_entry)
+
+ # 5. Save JSON
+ clean_name = os.path.splitext(npz_filename)[0]
+ output_filename = f"{entry_id}_{clean_name}.json"
+ output_path = os.path.join(OUTPUT_DIR, output_filename)
+
+ style_debug = "Anny"
+ if base_style == STYLE_FILLIAN: style_debug = "Fillian"
+ elif base_style == STYLE_BIBOO: style_debug = "Biboo"
+
+ wrapper = {
+ "fps": FPS,
+ "video_ref": entry_data.get("video_ref_path", ""),
+ "base_style_debug": style_debug,
+ "frames": frames_data
+ }
+
+ with open(output_path, 'w') as f:
+ json.dump(wrapper, f, separators=(',', ':'))
+
+def main():
+ if not os.path.exists(OUTPUT_DIR):
+ os.makedirs(OUTPUT_DIR)
+
+ print(f"📂 Loading Train JSON: {INPUT_TRAIN_JSON}")
+ if not os.path.exists(INPUT_TRAIN_JSON):
+ print("❌ Train JSON not found.")
+ return
+
+ with open(INPUT_TRAIN_JSON, 'r') as f:
+ train_index = json.load(f)
+
+ print(f"🚀 Processing {len(train_index)} sequences...")
+ for key, val in tqdm(train_index.items()):
+ process_single_entry(key, val)
+
+ print("✅ Conversion Complete.")
+
+if __name__ == "__main__":
+ main()
\ No newline at end of file
diff --git a/requirements.txt b/requirements.txt
new file mode 100644
index 0000000000000000000000000000000000000000..3c57768004a749a71d633e2290c791a1de0ac689
--- /dev/null
+++ b/requirements.txt
@@ -0,0 +1,52 @@
+timm==0.6.7
+numpy==1.23.5
+# timm==0.9.12 # For HMR2.0a feature extraction
+
+# Lightning + Hydra
+lightning==2.3.0
+hydra-core==1.3
+hydra-zen
+hydra_colorlog
+rich
+
+# Common utilities
+
+jupyter
+matplotlib
+ipdb
+setuptools>=68.0
+black
+tensorboardX
+opencv-python
+ffmpeg-python
+scikit-image
+termcolor
+einops
+imageio==2.34.1
+av # imageio[pyav], improved performance over imageio[ffmpeg]
+joblib
+
+# Diffusion
+# diffusers[torch]==0.19.3
+# transformers==4.31.0
+
+# 3D-Vision
+trimesh
+chumpy
+smplx
+# open3d==0.17.0
+wis3d
+
+# 2D-Pose
+ultralytics==8.2.42 # YOLO
+cython_bbox
+lapx
+
+# motiondiff
+lmdb==1.4.1
+transformers==4.30.2
+sentencepiece==0.2.0
+pre-commit
+
+moviepy==1.0.3
+scenepic
diff --git a/s050000.ckpt b/s050000.ckpt
new file mode 100644
index 0000000000000000000000000000000000000000..42bd013fdd07fa5b98ae4564c23e0e80106925bc
--- /dev/null
+++ b/s050000.ckpt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3c78abf4f708a35038a93b2379b812576763aea3f3b8c3299abf2510975c4b0c
+size 5544443162
diff --git a/scripts/analyze_vis_video.py b/scripts/analyze_vis_video.py
new file mode 100644
index 0000000000000000000000000000000000000000..9ff969d4132c4e051e3e100557067cbb0bee0316
--- /dev/null
+++ b/scripts/analyze_vis_video.py
@@ -0,0 +1,190 @@
+#!/usr/bin/env python3
+from __future__ import annotations
+
+import argparse
+from dataclasses import dataclass
+from pathlib import Path
+
+import cv2
+import numpy as np
+
+
+@dataclass(frozen=True)
+class ColorSpec:
+ name: str
+ rgb: tuple[int, int, int]
+
+
+def _parse_rgb(s: str) -> tuple[int, int, int]:
+ parts = [p.strip() for p in s.split(",")]
+ if len(parts) != 3:
+ raise ValueError(f"Expected 'R,G,B', got: {s!r}")
+ rgb = tuple(int(p) for p in parts)
+ if any(c < 0 or c > 255 for c in rgb):
+ raise ValueError(f"RGB out of range: {rgb}")
+ return rgb # type: ignore[return-value]
+
+
+def _mask_by_color_dist(frame_rgb: np.ndarray, rgb: tuple[int, int, int], thr: float) -> np.ndarray:
+ target = np.array(rgb, dtype=np.float32).reshape(1, 1, 3)
+ diff = frame_rgb.astype(np.float32) - target
+ dist = np.sqrt(np.sum(diff * diff, axis=2))
+ return (dist <= float(thr)).astype(np.uint8)
+
+
+def _bbox_from_mask(mask: np.ndarray) -> tuple[int, int, int, int] | None:
+ ys, xs = np.where(mask > 0)
+ if ys.size == 0:
+ return None
+ x0 = int(xs.min())
+ x1 = int(xs.max())
+ y0 = int(ys.min())
+ y1 = int(ys.max())
+ return x0, y0, x1, y1
+
+
+def _centroid_from_mask(mask: np.ndarray) -> tuple[float, float] | None:
+ ys, xs = np.where(mask > 0)
+ if ys.size == 0:
+ return None
+ return float(xs.mean()), float(ys.mean())
+
+
+def _largest_cc(mask: np.ndarray, min_pixels: int) -> np.ndarray:
+ # Keep only the largest connected component to suppress single-pixel compression noise.
+ num_labels, labels, stats, _ = cv2.connectedComponentsWithStats(mask, connectivity=8)
+ if num_labels <= 1:
+ return mask
+ # label 0 is background
+ areas = stats[1:, cv2.CC_STAT_AREA]
+ best_i = int(np.argmax(areas)) + 1
+ if int(stats[best_i, cv2.CC_STAT_AREA]) < int(min_pixels):
+ return np.zeros_like(mask)
+ return (labels == best_i).astype(np.uint8)
+
+
+def _pick_rgb_order(frame: np.ndarray, colors: list[ColorSpec], thr: float) -> str:
+ # Frames are likely RGB (Renderer outputs RGB), but some pipelines write BGR.
+ # Pick the order that yields more colored pixels for the provided colors.
+ frame_rgb = frame
+ frame_bgr_as_rgb = frame[..., ::-1]
+ score_rgb = 0
+ score_bgr = 0
+ for c in colors:
+ score_rgb += int(_mask_by_color_dist(frame_rgb, c.rgb, thr).sum())
+ score_bgr += int(_mask_by_color_dist(frame_bgr_as_rgb, c.rgb, thr).sum())
+ return "rgb" if score_rgb >= score_bgr else "bgr"
+
+
+def _touches_border(bbox: tuple[int, int, int, int], w: int, h: int, margin: int) -> bool:
+ x0, y0, x1, y1 = bbox
+ return (
+ x0 <= margin
+ or y0 <= margin
+ or (w - 1 - x1) <= margin
+ or (h - 1 - y1) <= margin
+ )
+
+
+def main() -> int:
+ ap = argparse.ArgumentParser()
+ ap.add_argument("video", type=str)
+ ap.add_argument("--gt_rgb", type=str, default="0,255,0", help="GT mesh RGB, e.g. 0,255,0")
+ ap.add_argument("--pred_rgb", type=str, default="176,100,244", help="Pred mesh RGB, e.g. 176,100,244")
+ ap.add_argument("--thr", type=float, default=60.0, help="Color distance threshold")
+ ap.add_argument("--margin", type=int, default=2, help="Border margin in pixels to count as 'out of frame'")
+ ap.add_argument("--min_pixels", type=int, default=200, help="Min pixels for largest CC to be considered present")
+ ap.add_argument("--max_frames", type=int, default=0, help="If >0, analyze only first N frames")
+ args = ap.parse_args()
+
+ video_path = Path(args.video)
+ if not video_path.exists():
+ raise FileNotFoundError(str(video_path))
+
+ gt = ColorSpec("gt", _parse_rgb(args.gt_rgb))
+ pred = ColorSpec("pred", _parse_rgb(args.pred_rgb))
+
+ cap = cv2.VideoCapture(str(video_path))
+ if not cap.isOpened():
+ raise RuntimeError(f"Failed to open video: {video_path}")
+
+ ok, frame0 = cap.read()
+ if not ok:
+ raise RuntimeError(f"Empty video: {video_path}")
+ h, w = frame0.shape[:2]
+
+ order = _pick_rgb_order(frame0, [gt, pred], float(args.thr))
+ def as_rgb(frame: np.ndarray) -> np.ndarray:
+ return frame if order == "rgb" else frame[..., ::-1]
+
+ # rewind to frame 0
+ cap.set(cv2.CAP_PROP_POS_FRAMES, 0)
+
+ gt_centroids: list[tuple[float, float] | None] = []
+ pred_centroids: list[tuple[float, float] | None] = []
+ gt_borders: list[bool] = []
+ pred_borders: list[bool] = []
+ dp_centroid: list[float] = []
+
+ frame_idx = 0
+ while True:
+ ok, frame = cap.read()
+ if not ok:
+ break
+ if args.max_frames and frame_idx >= int(args.max_frames):
+ break
+
+ frame_rgb = as_rgb(frame)
+ gt_mask = _mask_by_color_dist(frame_rgb, gt.rgb, float(args.thr))
+ pred_mask = _mask_by_color_dist(frame_rgb, pred.rgb, float(args.thr))
+ gt_mask = _largest_cc(gt_mask, int(args.min_pixels))
+ pred_mask = _largest_cc(pred_mask, int(args.min_pixels))
+
+ gt_bbox = _bbox_from_mask(gt_mask)
+ pred_bbox = _bbox_from_mask(pred_mask)
+ gt_ctr = _centroid_from_mask(gt_mask)
+ pred_ctr = _centroid_from_mask(pred_mask)
+
+ gt_centroids.append(gt_ctr)
+ pred_centroids.append(pred_ctr)
+ gt_borders.append(bool(gt_bbox is not None and _touches_border(gt_bbox, w, h, int(args.margin))))
+ pred_borders.append(bool(pred_bbox is not None and _touches_border(pred_bbox, w, h, int(args.margin))))
+
+ if gt_ctr is not None and pred_ctr is not None:
+ dx = pred_ctr[0] - gt_ctr[0]
+ dy = pred_ctr[1] - gt_ctr[1]
+ dp_centroid.append(float(np.sqrt(dx * dx + dy * dy)))
+ else:
+ dp_centroid.append(float("nan"))
+
+ frame_idx += 1
+
+ cap.release()
+
+ n = frame_idx
+ gt_border_frac = float(np.mean(gt_borders)) if n > 0 else float("nan")
+ pred_border_frac = float(np.mean(pred_borders)) if n > 0 else float("nan")
+
+ def _first_last(xs: list[tuple[float, float] | None]):
+ first = next((v for v in xs if v is not None), None)
+ last = next((v for v in reversed(xs) if v is not None), None)
+ return first, last
+
+ gt_first, gt_last = _first_last(gt_centroids)
+ pred_first, pred_last = _first_last(pred_centroids)
+
+ dp_arr = np.array(dp_centroid, dtype=np.float32)
+ dp_mean = float(np.nanmean(dp_arr)) if np.isfinite(dp_arr).any() else float("nan")
+ dp_max = float(np.nanmax(dp_arr)) if np.isfinite(dp_arr).any() else float("nan")
+
+ print(f"video={video_path}")
+ print(f"frames={n} size={w}x{h} order={order} thr={float(args.thr):.1f} margin={int(args.margin)}")
+ print(f"gt_border_frac={gt_border_frac:.3f} pred_border_frac={pred_border_frac:.3f}")
+ print(f"centroid_dist_px mean={dp_mean:.2f} max={dp_max:.2f}")
+ print(f"gt_centroid first/last={gt_first}/{gt_last}")
+ print(f"pred_centroid first/last={pred_first}/{pred_last}")
+ return 0
+
+
+if __name__ == '__main__':
+ raise SystemExit(main())
diff --git a/scripts/blender_load_camera.py b/scripts/blender_load_camera.py
new file mode 100644
index 0000000000000000000000000000000000000000..af6efd74b474a96608c51fe58e2125729a84bb62
--- /dev/null
+++ b/scripts/blender_load_camera.py
@@ -0,0 +1,97 @@
+from pathlib import Path
+
+import bpy
+import numpy as np
+import math
+from mathutils import Matrix, Vector
+
+# Hardcode your NPZ path here. Use raw string or forward slashes to avoid backslash escapes.
+NPZ_PATH = Path(r"D:\Users\Adam\Downloads\VRM_JG6Z7WA_clip_0_global_smplx.npz")
+CAMERA_NAME = "GenmoCamera"
+SPHERE_NAME = "TestSphere"
+FPS = 30
+ROTATE_LOCAL_Z_DEG = 180.0
+ROTATE_LOCAL_Y_DEG = 180.0
+
+
+def _ensure_camera(name: str) -> bpy.types.Object:
+ cam_obj = bpy.data.objects.get(name)
+ if cam_obj is None:
+ cam_data = bpy.data.cameras.new(name)
+ cam_obj = bpy.data.objects.new(name, cam_data)
+ bpy.context.collection.objects.link(cam_obj)
+ return cam_obj
+
+
+def _apply_fov(cam_obj: bpy.types.Object, npz_data: np.lib.npyio.NpzFile) -> None:
+ cam_data = cam_obj.data
+ fov_y_deg = float(npz_data["fov_y_deg"]) if "fov_y_deg" in npz_data else None
+ fov_x_deg = float(npz_data["fov_x_deg"]) if "fov_x_deg" in npz_data else None
+
+ if (fov_x_deg is None or fov_y_deg is None) and "K_fullimg" in npz_data:
+ K_fullimg = npz_data["K_fullimg"]
+ if K_fullimg.ndim == 3:
+ K0 = K_fullimg[0]
+ else:
+ K0 = K_fullimg
+ fx = float(K0[0, 0])
+ fy = float(K0[1, 1])
+ cx = float(K0[0, 2])
+ cy = float(K0[1, 2])
+ width = 2.0 * cx
+ height = 2.0 * cy
+ if fov_x_deg is None and fx > 0 and width > 0:
+ fov_x_deg = math.degrees(2.0 * math.atan(width / (2.0 * fx)))
+ if fov_y_deg is None and fy > 0 and height > 0:
+ fov_y_deg = math.degrees(2.0 * math.atan(height / (2.0 * fy)))
+
+ if fov_y_deg is not None:
+ cam_data.angle_y = math.radians(float(fov_y_deg))
+ elif fov_x_deg is not None:
+ cam_data.angle = math.radians(float(fov_x_deg))
+
+
+def _ensure_sphere(name: str) -> bpy.types.Object:
+ obj = bpy.data.objects.get(name)
+ if obj is None:
+ bpy.ops.mesh.primitive_uv_sphere_add(radius=0.2, location=(0.0, 0.0, 0.0))
+ obj = bpy.context.active_object
+ obj.name = name
+ return obj
+
+
+def main():
+ npz_path = NPZ_PATH
+ if not npz_path.is_absolute():
+ npz_path = Path(bpy.path.abspath("//")) / npz_path
+ data = np.load(str(npz_path))
+ if "camera_transform" not in data:
+ raise ValueError("NPZ is missing 'camera_transform'")
+ camera_transform = data["camera_transform"] # (F, 4, 4)
+
+ scene = bpy.context.scene
+ scene.render.fps = FPS
+ scene.frame_start = 1
+ scene.frame_end = int(camera_transform.shape[0])
+
+ cam_obj = _ensure_camera(CAMERA_NAME)
+ scene.camera = cam_obj
+ _ensure_sphere(SPHERE_NAME)
+ _apply_fov(cam_obj, data)
+
+ for i, T in enumerate(camera_transform):
+ frame = i + 1
+ mat = Matrix(T.tolist())
+ rot_z = Matrix.Rotation(math.radians(ROTATE_LOCAL_Z_DEG), 3, Vector((0.0, 0.0, 1.0)))
+ rot_y = Matrix.Rotation(math.radians(ROTATE_LOCAL_Y_DEG), 3, Vector((0.0, 1.0, 0.0)))
+ rot_local = rot_y @ rot_z # first local Z, then local Y
+ new_rot = mat.to_3x3() @ rot_local
+ new_mat = new_rot.to_4x4()
+ new_mat.translation = mat.to_translation()
+ cam_obj.matrix_world = new_mat
+ cam_obj.keyframe_insert(data_path="location", frame=frame)
+ cam_obj.keyframe_insert(data_path="rotation_euler", frame=frame)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/check_unity_consistency.py b/scripts/check_unity_consistency.py
new file mode 100644
index 0000000000000000000000000000000000000000..484e5bc64816406cf9737f02951306edb885ebfc
--- /dev/null
+++ b/scripts/check_unity_consistency.py
@@ -0,0 +1,164 @@
+#!/usr/bin/env python3
+"""
+Check internal consistency of Unity processed samples.
+
+This script compares:
+ - saved `smpl_params_w` vs `smpl_params_c` + `T_w2c`-derived world params
+ - saved `cam_angvel` (if present) vs recomputed from `T_w2c`
+
+If these disagree by non-trivial amounts, fine-tuning can "break global" because the
+network is asked to fit mutually inconsistent targets/condition signals.
+"""
+
+from __future__ import annotations
+
+import argparse
+from pathlib import Path
+from typing import Dict, Any, Iterable
+
+import torch
+
+from genmo.utils.geo_transform import compute_cam_angvel, normalize_T_w2c
+from genmo.utils.rotation_conversions import axis_angle_to_matrix, matrix_to_axis_angle
+
+
+def _load_pt(path: Path) -> Dict[str, Any]:
+ return torch.load(path, map_location="cpu", weights_only=False)
+
+
+def _axis_angle_angle_deg(a: torch.Tensor, b: torch.Tensor) -> torch.Tensor:
+ """Geodesic angle between rotations in axis-angle form. Shapes (..., 3). Returns (...,) degrees."""
+ Ra = axis_angle_to_matrix(a.float())
+ Rb = axis_angle_to_matrix(b.float())
+ Rrel = Ra.transpose(-1, -2) @ Rb
+ aa = matrix_to_axis_angle(Rrel)
+ return aa.norm(dim=-1) * (180.0 / torch.pi)
+
+
+def _derive_world_from_c_and_T(
+ smpl_params_c: Dict[str, torch.Tensor], T_w2c: torch.Tensor
+) -> Dict[str, torch.Tensor]:
+ """
+ Derive world translation/orientation from camera params and T_w2c.
+
+ Assumes:
+ p_w = R_c2w * p_c + t_c2w
+ R_w = R_c2w * R_c
+ """
+ R_w2c = T_w2c[:, :3, :3].float() # (L,3,3)
+ t_w2c = T_w2c[:, :3, 3].float() # (L,3)
+ R_c2w = R_w2c.transpose(-1, -2)
+ t_c2w = -torch.einsum("fij,fj->fi", R_c2w, t_w2c)
+
+ transl_c = smpl_params_c["transl"].float()
+ transl_w = torch.einsum("fij,fj->fi", R_c2w, transl_c) + t_c2w
+
+ R_c = axis_angle_to_matrix(smpl_params_c["global_orient"].float())
+ R_w = torch.einsum("fij,fjk->fik", R_c2w, R_c)
+ go_w = matrix_to_axis_angle(R_w)
+
+ return {"transl": transl_w, "global_orient": go_w}
+
+
+def _maybe_slice(x: Any, start: int, end: int) -> Any:
+ if isinstance(x, dict):
+ return {k: _maybe_slice(v, start, end) for k, v in x.items()}
+ if isinstance(x, torch.Tensor):
+ return x[start:end]
+ return x
+
+
+def _iter_paths(root: Path, glob_pat: str) -> Iterable[Path]:
+ if root.is_file():
+ yield root
+ return
+ yield from sorted(root.glob(glob_pat))
+
+
+def main() -> None:
+ ap = argparse.ArgumentParser()
+ ap.add_argument(
+ "--root",
+ type=Path,
+ default=Path("processed_dataset/genmo_features"),
+ help="A single .pt file or a directory containing .pt files.",
+ )
+ ap.add_argument("--glob", type=str, default="*.pt")
+ ap.add_argument("--max", type=int, default=10, help="Max files to check")
+ ap.add_argument("--frame0_only", action="store_true", help="Check only frame 0")
+ args = ap.parse_args()
+
+ paths = list(_iter_paths(args.root, args.glob))
+ if not paths:
+ raise SystemExit(f"No files found under {args.root} with glob {args.glob!r}")
+ paths = paths[: max(1, int(args.max))]
+
+ all_stats = []
+ for p in paths:
+ d = _load_pt(p)
+ smpl_c = d["smpl_params_c"]
+ smpl_w = d["smpl_params_w"]
+ T_w2c = d["T_w2c"]
+
+ if args.frame0_only:
+ smpl_c = _maybe_slice(smpl_c, 0, 1)
+ smpl_w = _maybe_slice(smpl_w, 0, 1)
+ T_w2c = _maybe_slice(T_w2c, 0, 1)
+
+ derived = _derive_world_from_c_and_T(smpl_c, T_w2c)
+
+ diff_t = smpl_w["transl"].float() - derived["transl"] # (L, 3)
+ dt = diff_t.norm(dim=-1) # (L,)
+ diff_t_abs_mean = diff_t.abs().mean(dim=0)
+ diff_t_abs_max = diff_t.abs().max(dim=0)[0]
+ dgo = _axis_angle_angle_deg(smpl_w["global_orient"].float(), derived["global_orient"]) # (L,)
+
+ # cam_angvel consistency
+ cam_av_saved = d.get("cam_angvel", None)
+ cam_av_deg = None
+ if cam_av_saved is not None:
+ cam_av_saved = cam_av_saved[: T_w2c.shape[0]]
+ Tw = T_w2c.float()
+ if Tw.ndim == 3 and Tw.shape[-2:] == (4, 4) and Tw.shape[0] >= 2:
+ Tw = normalize_T_w2c(Tw)
+ cam_av_re = compute_cam_angvel(Tw[:, :3, :3]) # (L,6)
+ # matrix_to_rotation_6d convention: compare directly in 6D space
+ cam_av_deg = (cam_av_saved.float() - cam_av_re.float()).abs().mean().item()
+
+ stat = {
+ "file": p.name,
+ "max_transl_err_m": float(dt.max().item()),
+ "mean_transl_err_m": float(dt.mean().item()),
+ "mean_abs_dx": float(diff_t_abs_mean[0].item()),
+ "mean_abs_dy": float(diff_t_abs_mean[1].item()),
+ "mean_abs_dz": float(diff_t_abs_mean[2].item()),
+ "max_abs_dx": float(diff_t_abs_max[0].item()),
+ "max_abs_dy": float(diff_t_abs_max[1].item()),
+ "max_abs_dz": float(diff_t_abs_max[2].item()),
+ "max_go_err_deg": float(dgo.max().item()),
+ "mean_go_err_deg": float(dgo.mean().item()),
+ "cam_angvel_absmean_diff": None if cam_av_deg is None else float(cam_av_deg),
+ }
+ all_stats.append(stat)
+
+ print(
+ f"{p.name}: "
+ f"transl_err mean={stat['mean_transl_err_m']:.4f}m max={stat['max_transl_err_m']:.4f}m | "
+ f"|Δt| mean_abs(x,y,z)=({stat['mean_abs_dx']:.4f},{stat['mean_abs_dy']:.4f},{stat['mean_abs_dz']:.4f}) "
+ f"max_abs(x,y,z)=({stat['max_abs_dx']:.4f},{stat['max_abs_dy']:.4f},{stat['max_abs_dz']:.4f}) | "
+ f"go_err mean={stat['mean_go_err_deg']:.2f}° max={stat['max_go_err_deg']:.2f}°"
+ + (
+ ""
+ if stat["cam_angvel_absmean_diff"] is None
+ else f" | cam_angvel_absmean_diff={stat['cam_angvel_absmean_diff']:.6f}"
+ )
+ )
+
+ # Aggregate
+ max_transl = max(s["max_transl_err_m"] for s in all_stats)
+ max_go = max(s["max_go_err_deg"] for s in all_stats)
+ print(f"\nAggregate over {len(all_stats)} files: max_transl_err_m={max_transl:.4f}, max_go_err_deg={max_go:.2f}")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/clip_videos.py b/scripts/clip_videos.py
new file mode 100644
index 0000000000000000000000000000000000000000..b79d3bbdaa7c61cb4890340dc7e8b37b43510640
--- /dev/null
+++ b/scripts/clip_videos.py
@@ -0,0 +1,136 @@
+#!/usr/bin/env python3
+"""
+Video Clipping Script
+
+Reads a labels.jsonl file (produced by label_videos.py) and extracts the usable
+segments into separate video files using ffmpeg.
+
+Usage:
+ python clip_videos.py --labels labels.jsonl --output-dir clips/
+"""
+
+import os
+import sys
+import json
+import argparse
+import subprocess
+from dataclasses import dataclass
+from typing import List, Dict
+from collections import defaultdict
+
+@dataclass
+class Clip:
+ video_path: str
+ start_sec: float
+ end_sec: float
+ output_filename: str
+
+def parse_args():
+ parser = argparse.ArgumentParser(description="Clip videos based on labels.jsonl")
+ parser.add_argument("--labels", required=True, help="Path to labels.jsonl file")
+ parser.add_argument("--output-dir", required=True, help="Directory to save clips")
+ parser.add_argument("--min-duration", type=float, default=4.0, help="Minimum duration in seconds (default: 4.0)")
+ parser.add_argument("--dry-run", action="store_true", help="Print commands without executing")
+ return parser.parse_args()
+
+def load_clips(labels_path: str, min_duration: float = 0.0) -> List[Clip]:
+ clips = []
+ if not os.path.exists(labels_path):
+ print(f"Error: Labels file not found: {labels_path}")
+ return []
+
+ with open(labels_path, 'r') as f:
+ for i, line in enumerate(f):
+ try:
+ data = json.loads(line)
+ except json.JSONDecodeError:
+ print(f"Warning: Skipping invalid JSON on line {i+1}")
+ continue
+
+ if not data.get('usable'):
+ continue
+
+ video_path = data['video']
+ start = float(data['start_sec'])
+ end = float(data['end_sec'])
+ duration = end - start
+
+ if duration < min_duration:
+ continue
+
+ # Create a safe filename
+ video_basename = os.path.splitext(os.path.basename(video_path))[0]
+ # Format: VideoName_Start_End.mp4 (e.g. MyVideo_005.50_010.00.mp4)
+ filename = f"{video_basename}_{start:06.2f}_{end:06.2f}.mp4"
+
+ clips.append(Clip(
+ video_path=video_path,
+ start_sec=start,
+ end_sec=end,
+ output_filename=filename
+ ))
+
+ return clips
+
+def process_clips(clips: List[Clip], output_dir: str, dry_run: bool = False):
+ os.makedirs(output_dir, exist_ok=True)
+
+ # Group by video to potentially optimize (though currently we treat each clip independently)
+ # If we wanted to batch, we could, but ffmpeg seeking is fast enough with -ss
+
+ for i, clip in enumerate(clips):
+ output_path = os.path.join(output_dir, clip.output_filename)
+
+ if os.path.exists(output_path):
+ print(f"[{i+1}/{len(clips)}] Skipping existing: {output_path}")
+ continue
+
+ print(f"[{i+1}/{len(clips)}] Clipping: {clip.video_path} -> {output_path}")
+ print(f" Range: {clip.start_sec}s to {clip.end_sec}s")
+
+ duration = clip.end_sec - clip.start_sec
+
+ # Use system ffmpeg if available (likely has AV1 decoder), otherwise fallback to env ffmpeg
+ ffmpeg_bin = "/usr/bin/ffmpeg"
+ if not os.path.exists(ffmpeg_bin):
+ ffmpeg_bin = "ffmpeg"
+
+ cmd = [
+ ffmpeg_bin,
+ '-y', # Overwrite output
+ '-hide_banner', '-loglevel', 'error',
+ '-ss', str(clip.start_sec),
+ '-i', clip.video_path,
+ '-t', str(duration),
+ '-c:v', 'libx264',
+ '-crf', '18', # High quality
+ '-preset', 'slow', # Better compression/quality tradeoff
+ '-c:a', 'copy', # Copy audio stream without re-encoding
+ output_path
+ ]
+
+ if dry_run:
+ print("Running:", " ".join(cmd))
+ else:
+ try:
+ subprocess.run(cmd, check=True)
+ except subprocess.CalledProcessError as e:
+ print(f"Error clipping {clip.video_path}: {e}")
+
+def main():
+ args = parse_args()
+
+ print(f"Reading labels from: {args.labels}")
+ print(f"Minimum clip duration: {args.min_duration}s")
+ clips = load_clips(args.labels, min_duration=args.min_duration)
+
+ if not clips:
+ print("No usable clips found in labels file.")
+ return
+
+ print(f"Found {len(clips)} usable clips.")
+ process_clips(clips, args.output_dir, args.dry_run)
+ print("Done!")
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/combine_prompthmr_genmo.py b/scripts/combine_prompthmr_genmo.py
new file mode 100644
index 0000000000000000000000000000000000000000..f2c63299bfa1b1eec2de77ff27f688deea2f37c1
--- /dev/null
+++ b/scripts/combine_prompthmr_genmo.py
@@ -0,0 +1,194 @@
+#!/usr/bin/env python3
+"""
+Combine PromptHMR camera with GENMO incam pose to produce proper global SMPLX.
+
+Usage:
+ python combine_prompthmr_genmo.py \
+ --prompthmr-results results/VRM_JG6Z7WA_clip_0/results.pkl \
+ --genmo-results outputs/infer_video/VRM_JG6Z7WA_clip_0/hmr4d_results.pt \
+ --output outputs/infer_video/VRM_JG6Z7WA_clip_0/combined_global_smplx.npz
+"""
+
+import argparse
+import sys
+from pathlib import Path
+
+import joblib
+import numpy as np
+import torch
+
+# Add GENMO to path
+SCRIPT_DIR = Path(__file__).resolve().parent
+GENMO_ROOT = SCRIPT_DIR.parent
+sys.path.insert(0, str(GENMO_ROOT))
+
+from genmo.utils.rotation_conversions import matrix_to_axis_angle
+
+
+def transform_smpl_params(root_orient, transl, R, t, smpl_t_pose_pelvis):
+ """
+ Transform SMPL params from camera space to world space.
+
+ Args:
+ root_orient: [B, 3, 3] rotation matrices
+ transl: [B, 3] translations
+ R: [B, 3, 3] camera rotation (world-to-camera or Rwc)
+ t: [B, 3] camera translation
+ smpl_t_pose_pelvis: [3] T-pose pelvis offset
+
+ Returns:
+ root_orient: [B, 3, 3] transformed rotation
+ transl: [B, 3] transformed translation
+ """
+ smpl_t_pose_pelvis = smpl_t_pose_pelvis[None, :, None]
+ transl = transl.unsqueeze(-1).float()
+ t = t.unsqueeze(-1)
+
+ transl = R @ (smpl_t_pose_pelvis + transl) + t - smpl_t_pose_pelvis
+ root_orient = R @ root_orient
+ transl = transl.squeeze(-1)
+ return root_orient, transl
+
+
+def main():
+ parser = argparse.ArgumentParser(description="Combine PromptHMR camera with GENMO incam")
+ parser.add_argument("--prompthmr-results", type=Path, required=True,
+ help="Path to PromptHMR results.pkl")
+ parser.add_argument("--genmo-results", type=Path, required=True,
+ help="Path to GENMO hmr4d_results.pt")
+ parser.add_argument("--output", type=Path, required=True,
+ help="Output NPZ path")
+ parser.add_argument("--track-id", type=int, default=None,
+ help="PromptHMR track ID to use (default: longest track)")
+ args = parser.parse_args()
+
+ # Load PromptHMR results
+ print(f"Loading PromptHMR results from {args.prompthmr_results}")
+ phmr = joblib.load(args.prompthmr_results)
+
+ # Load GENMO results
+ print(f"Loading GENMO results from {args.genmo_results}")
+ genmo = torch.load(args.genmo_results, map_location="cpu", weights_only=False)
+
+ # Get PromptHMR camera (world coordinates)
+ camera_world = phmr["camera_world"]
+ Rwc_full = torch.from_numpy(camera_world["Rwc"]).float() # [N, 3, 3]
+ Twc_full = torch.from_numpy(camera_world["Twc"]).float() # [N, 3]
+
+ print(f"PromptHMR camera frames: {Rwc_full.shape[0]}")
+
+ # Get GENMO incam params
+ incam = genmo["smpl_params_incam"]
+
+ # Convert global_orient to rotation matrix if needed
+ global_orient = incam["global_orient"]
+ if isinstance(global_orient, np.ndarray):
+ global_orient = torch.from_numpy(global_orient)
+
+ # Handle different shapes - could be axis-angle [B, 3] or rotation matrix [B, 3, 3]
+ if global_orient.ndim == 4:
+ global_orient = global_orient.squeeze(1) # [B, 1, 3, 3] -> [B, 3, 3]
+
+ if global_orient.ndim == 2 and global_orient.shape[-1] == 3:
+ # Axis-angle [B, 3] -> need to convert to rotation matrix
+ from genmo.utils.rotation_conversions import axis_angle_to_matrix
+ root_orient_mat = axis_angle_to_matrix(global_orient.float())
+ print("Converted global_orient from axis-angle to rotation matrix")
+ elif global_orient.ndim == 3 and global_orient.shape[-1] == 3 and global_orient.shape[-2] == 3:
+ root_orient_mat = global_orient.float()
+ else:
+ raise ValueError(f"Unexpected global_orient shape: {global_orient.shape}")
+
+ transl = incam["transl"]
+ if isinstance(transl, np.ndarray):
+ transl = torch.from_numpy(transl)
+ transl = transl.float()
+
+ num_genmo_frames = root_orient_mat.shape[0]
+ num_phmr_frames = Rwc_full.shape[0]
+ print(f"GENMO incam frames: {num_genmo_frames}")
+
+ # Handle FPS mismatch - PromptHMR might be 60fps, GENMO 30fps
+ if num_phmr_frames != num_genmo_frames:
+ ratio = num_phmr_frames / num_genmo_frames
+ print(f"FPS mismatch detected (ratio={ratio:.2f}), resampling PromptHMR camera...")
+
+ # Resample camera to match GENMO frame count
+ indices = (torch.arange(num_genmo_frames) * ratio).long().clamp(0, num_phmr_frames - 1)
+ Rwc = Rwc_full[indices]
+ Twc = Twc_full[indices]
+ else:
+ Rwc = Rwc_full
+ Twc = Twc_full
+
+ num_frames = num_genmo_frames
+ print(f"Using {num_frames} frames")
+
+ # Estimate T-pose pelvis offset (approximate)
+ # In SMPL-X, the pelvis is roughly at origin in T-pose
+ smpl_t_pose_pelvis = torch.zeros(3)
+
+ # Transform from camera to world
+ print("Transforming GENMO incam to world coordinates using PromptHMR camera...")
+ root_orient_world, transl_world = transform_smpl_params(
+ root_orient_mat, transl, Rwc, Twc, smpl_t_pose_pelvis
+ )
+
+ # Convert rotation matrices to axis-angle
+ root_orient_aa = matrix_to_axis_angle(root_orient_world)
+
+ # Get body pose from incam
+ body_pose = incam["body_pose"]
+ if isinstance(body_pose, np.ndarray):
+ body_pose = torch.from_numpy(body_pose)
+ body_pose = body_pose[:num_frames]
+
+ if body_pose.ndim == 4:
+ body_pose = matrix_to_axis_angle(body_pose)
+ body_pose = body_pose.reshape(num_frames, -1)
+
+ # Build full pose
+ full_pose = torch.cat([root_orient_aa, body_pose], dim=-1)
+
+ # Pad to 165 (55 joints * 3)
+ current_size = full_pose.shape[-1]
+ if current_size < 165:
+ padding = torch.zeros(num_frames, 165 - current_size)
+ full_pose = torch.cat([full_pose, padding], dim=-1)
+
+ # Get betas
+ betas = incam["betas"]
+ if isinstance(betas, np.ndarray):
+ betas = torch.from_numpy(betas)
+ betas = betas[0, :10].detach().numpy()
+
+ # Build camera transform (Rwc, Twc as 4x4)
+ camera_transform = np.eye(4)[None].repeat(num_frames, axis=0)
+ camera_transform[:, :3, :3] = Rwc.numpy()
+ camera_transform[:, :3, 3] = Twc.numpy()
+
+ # Get intrinsics from PromptHMR
+ K = np.eye(3)
+ K[0, 0] = camera_world["img_focal"]
+ K[1, 1] = camera_world["img_focal"]
+ K[0, 2] = camera_world["img_center"][0]
+ K[1, 2] = camera_world["img_center"][1]
+
+ # Save NPZ
+ out_dict = {
+ "mocap_framerate": 30,
+ "gender": "neutral",
+ "betas": betas,
+ "trans": transl_world.detach().numpy(),
+ "poses": full_pose.detach().numpy(),
+ "camera_transform": camera_transform,
+ "K_fullimg": K,
+ }
+
+ args.output.parent.mkdir(parents=True, exist_ok=True)
+ np.savez(args.output, **out_dict)
+ print(f"Saved combined global SMPLX to {args.output}")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/compare_debug_with_dataset.py b/scripts/compare_debug_with_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..34a350b55d35b3db5af1e8ac151712bfc264baa4
--- /dev/null
+++ b/scripts/compare_debug_with_dataset.py
@@ -0,0 +1,160 @@
+#!/usr/bin/env python3
+from __future__ import annotations
+
+import argparse
+import json
+import sys
+from pathlib import Path
+
+import torch
+
+REPO_ROOT = Path(__file__).resolve().parents[1]
+if str(REPO_ROOT) not in sys.path:
+ sys.path.insert(0, str(REPO_ROOT))
+GVHMR_ROOT = REPO_ROOT / "third_party" / "GVHMR"
+if GVHMR_ROOT.is_dir() and str(GVHMR_ROOT) not in sys.path:
+ sys.path.insert(0, str(GVHMR_ROOT))
+
+from genmo.utils.eval_utils import compute_camcoord_metrics
+from genmo.utils.geo_transform import compute_cam_angvel, compute_cam_tvel, normalize_T_w2c
+from genmo.utils.rotation_conversions import axis_angle_to_matrix, matrix_to_axis_angle
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+def _to_tensor(x):
+ if isinstance(x, torch.Tensor):
+ return x
+ return torch.as_tensor(x)
+
+
+def _slice(x: torch.Tensor, n: int) -> torch.Tensor:
+ return x[:n].clone()
+
+
+def _compare_arrays(a: torch.Tensor, b: torch.Tensor):
+ if a.shape != b.shape:
+ return {"shape_a": list(a.shape), "shape_b": list(b.shape)}
+ diff = a - b
+ return {
+ "mae": float(diff.abs().mean().item()),
+ "rmse": float((diff.pow(2).mean().sqrt()).item()),
+ }
+
+
+def _load_smplx_tools(device: torch.device):
+ smplx = make_smplx("supermotion").to(device).eval()
+ smplx2smpl_path = (
+ REPO_ROOT / "third_party" / "GVHMR" / "inputs" / "checkpoints" / "body_models" / "smplx2smpl_sparse.pt"
+ )
+ j_reg_path = (
+ REPO_ROOT / "third_party" / "GVHMR" / "inputs" / "checkpoints" / "body_models" / "smpl_neutral_J_regressor.pt"
+ )
+ smplx2smpl = torch.load(smplx2smpl_path, map_location=device)
+ j_reg = torch.load(j_reg_path, map_location=device)
+ return smplx, smplx2smpl, j_reg
+
+
+def _smplx_to_j3d(params: dict, smplx, smplx2smpl, j_reg):
+ out = smplx(**params)
+ verts = out.vertices if hasattr(out, "vertices") else out[0].vertices
+ verts = torch.stack([torch.matmul(smplx2smpl, v) for v in verts])
+ j3d = torch.einsum("jv,fvi->fji", j_reg, verts)
+ return verts, j3d
+
+
+def _prepare_params(params: dict, n: int, device: torch.device):
+ out = {}
+ for k, v in params.items():
+ if not isinstance(v, torch.Tensor):
+ v = torch.as_tensor(v)
+ out[k] = _slice(v, n).to(device)
+ return out
+
+
+def _min_len(*lens: int) -> int:
+ lens = [int(x) for x in lens if x is not None]
+ return min(lens) if lens else 0
+
+
+def main() -> None:
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--debug-io", required=True, help="Path to debug_io.pt")
+ parser.add_argument("--dataset-pt", required=True, help="Path to dataset .pt file")
+ parser.add_argument("--num-frames", type=int, default=60)
+ parser.add_argument("--out", default="outputs/debug_compare.json")
+ args = parser.parse_args()
+
+ debug = torch.load(args.debug_io, map_location="cpu", weights_only=False)
+ data = torch.load(args.dataset_pt, map_location="cpu", weights_only=False)
+
+ inputs = debug.get("inputs", {})
+ outputs = debug.get("outputs", {})
+
+ n = int(args.num_frames)
+ results = {"num_frames": n, "inputs": {}, "metrics": {}, "formatted_dataset": True}
+
+ dataset_inputs = dict(data)
+ if "T_w2c" in data:
+ T_w2c = _to_tensor(data["T_w2c"]).float()
+ if T_w2c.ndim == 3:
+ normed_T_w2c = normalize_T_w2c(T_w2c)
+ dataset_inputs["cam_angvel"] = compute_cam_angvel(normed_T_w2c[:, :3, :3])
+ dataset_inputs["cam_tvel"] = compute_cam_tvel(normed_T_w2c[:, :3, 3])
+
+ # Compare input tensors (if present in both).
+ input_keys = [
+ "bbx_xys",
+ "kp2d",
+ "K_fullimg",
+ "cam_angvel",
+ "cam_tvel",
+ "T_w2c",
+ ]
+ for k in input_keys:
+ if k in inputs and k in dataset_inputs:
+ a_t = _to_tensor(inputs[k])
+ b_t = _to_tensor(dataset_inputs[k])
+ n_in = _min_len(n, a_t.shape[0], b_t.shape[0])
+ a = _slice(a_t, n_in)
+ b = _slice(b_t, n_in)
+ results["inputs"][k] = _compare_arrays(a, b)
+
+ # Compare incam SMPLX (pred vs GT from dataset).
+ pred_key = "smpl_params_incam"
+ if pred_key not in outputs:
+ pred_key = "pred_smpl_params_incam"
+
+ if pred_key in outputs and "smpl_params_c" in data:
+ device = torch.device("cpu")
+ smplx, smplx2smpl, j_reg = _load_smplx_tools(device)
+ pred_raw = outputs[pred_key]
+ gt_raw = data["smpl_params_c"]
+ pred_len = int(_to_tensor(pred_raw["global_orient"]).shape[0])
+ gt_len = int(_to_tensor(gt_raw["global_orient"]).shape[0])
+ n_eval = _min_len(n, pred_len, gt_len)
+ pred_params = _prepare_params(pred_raw, n_eval, device)
+ gt_params = _prepare_params(gt_raw, n_eval, device)
+
+ pred_verts, pred_j3d = _smplx_to_j3d(pred_params, smplx, smplx2smpl, j_reg)
+ gt_verts, gt_j3d = _smplx_to_j3d(gt_params, smplx, smplx2smpl, j_reg)
+
+ metrics = compute_camcoord_metrics(
+ {
+ "pred_j3d": pred_j3d,
+ "target_j3d": gt_j3d,
+ "pred_verts": pred_verts,
+ "target_verts": gt_verts,
+ }
+ )
+ results["metrics"] = {k: float(v.mean()) for k, v in metrics.items()}
+
+ out_path = Path(args.out)
+ out_path.parent.mkdir(parents=True, exist_ok=True)
+ with out_path.open("w") as f:
+ json.dump(results, f, indent=2)
+
+ print(f"Wrote {out_path}")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/compare_unity_sample.py b/scripts/compare_unity_sample.py
new file mode 100644
index 0000000000000000000000000000000000000000..5a8a1f5624c7c222268b11697879fff72561d0ef
--- /dev/null
+++ b/scripts/compare_unity_sample.py
@@ -0,0 +1,120 @@
+#!/usr/bin/env python3
+from __future__ import annotations
+
+import argparse
+import json
+import os
+import sys
+from pathlib import Path
+
+import torch
+
+# Ensure repo + GVHMR roots are importable.
+REPO_ROOT = Path(__file__).resolve().parents[1]
+if str(REPO_ROOT) not in sys.path:
+ sys.path.insert(0, str(REPO_ROOT))
+GVHMR_ROOT = REPO_ROOT / "third_party" / "GVHMR"
+if GVHMR_ROOT.is_dir() and str(GVHMR_ROOT) not in sys.path:
+ sys.path.insert(0, str(GVHMR_ROOT))
+
+from hydra import compose, initialize_config_dir
+from hydra.utils import instantiate
+
+from genmo.datamodule.mocap_trainX_testY import collate_fn
+from genmo.utils.eval_utils import compute_camcoord_metrics
+from genmo.utils.net_utils import load_pretrained_model, to_cuda
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+def _load_cfg(config_name: str, overrides: list[str]):
+ with initialize_config_dir(config_dir=str(REPO_ROOT / "configs"), version_base=None):
+ return compose(config_name=config_name, overrides=overrides)
+
+
+def _select_sample(dataset, idx: int | None):
+ if idx is None:
+ idx = 0
+ return dataset[int(idx)]
+
+
+def main() -> None:
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--ckpt", required=True, help="Path to checkpoint")
+ parser.add_argument("--config-name", default="finetune_unity")
+ parser.add_argument("--root", default=None, help="Dataset root (processed_dataset)")
+ parser.add_argument("--sample-idx", type=int, default=0)
+ parser.add_argument("--out-dir", default="outputs/debug_unity_sample")
+ args = parser.parse_args()
+
+ overrides = []
+ if args.root:
+ overrides.append(f"train_datasets.unity.root={args.root}")
+ overrides.append(f"test_datasets.unity_val.root={args.root}")
+
+ cfg = _load_cfg(args.config_name, overrides)
+
+ dataset = instantiate(cfg.test_datasets.unity_val, _recursive_=False)
+ sample = _select_sample(dataset, args.sample_idx)
+ batch = collate_fn([sample], mode="val", collate_cfg=cfg.data.collate_cfg)
+
+ device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
+ model = instantiate(cfg.model, _recursive_=False).to(device)
+ load_pretrained_model(model, args.ckpt)
+ model.eval()
+
+ if device.type == "cuda":
+ batch = to_cuda(batch)
+
+ with torch.inference_mode():
+ outputs = model.validation(batch, test_mode="default", batch_idx=0, dataloader_idx=0)
+
+ smplx = make_smplx("supermotion").to(device)
+ j_reg_path = REPO_ROOT / "third_party" / "GVHMR" / "inputs" / "checkpoints" / "body_models" / "smpl_neutral_J_regressor.pt"
+ smplx2smpl_path = REPO_ROOT / "third_party" / "GVHMR" / "inputs" / "checkpoints" / "body_models" / "smplx2smpl_sparse.pt"
+ J_regressor = torch.load(j_reg_path, map_location=device)
+ smplx2smpl = torch.load(smplx2smpl_path, map_location=device)
+
+ gt_params_c = {k: v[0] for k, v in batch["smpl_params_c"].items()}
+ pred_params_c = outputs["pred_smpl_params_incam"]
+
+ gt_out = smplx(**gt_params_c)
+ pred_out = smplx(**pred_params_c)
+
+ gt_verts = torch.stack([torch.matmul(smplx2smpl, v) for v in gt_out.vertices])
+ pred_verts = torch.stack([torch.matmul(smplx2smpl, v) for v in pred_out.vertices])
+
+ gt_j3d = torch.einsum("jv,fvi->fji", J_regressor, gt_verts)
+ pred_j3d = torch.einsum("jv,fvi->fji", J_regressor, pred_verts)
+
+ mask = batch["mask"]["valid"][0] if "mask" in batch and "valid" in batch["mask"] else None
+ metrics = compute_camcoord_metrics(
+ {
+ "pred_j3d": pred_j3d,
+ "target_j3d": gt_j3d,
+ "pred_verts": pred_verts,
+ "target_verts": gt_verts,
+ },
+ mask=mask,
+ )
+
+ out_dir = Path(args.out_dir)
+ out_dir.mkdir(parents=True, exist_ok=True)
+ with open(out_dir / "metrics.json", "w") as f:
+ json.dump({k: float(v.mean()) for k, v in metrics.items()}, f, indent=2)
+
+ torch.save(
+ {
+ "pred_smpl_params_incam": pred_params_c,
+ "gt_smpl_params_c": gt_params_c,
+ "pred_j3d": pred_j3d.detach().cpu(),
+ "gt_j3d": gt_j3d.detach().cpu(),
+ },
+ out_dir / "pred_gt.pt",
+ )
+
+ print("Wrote:", out_dir / "metrics.json")
+ print("Wrote:", out_dir / "pred_gt.pt")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/convert_phmr_cam_to_genmo.py b/scripts/convert_phmr_cam_to_genmo.py
new file mode 100644
index 0000000000000000000000000000000000000000..bb335885d3c06fb50c5fbab0b9bebf2b1f14337f
--- /dev/null
+++ b/scripts/convert_phmr_cam_to_genmo.py
@@ -0,0 +1,107 @@
+#!/usr/bin/env python3
+"""
+Convert PromptHMR camera to GENMO slam.pt format.
+
+This allows GENMO to use PromptHMR's DROID-SLAM camera estimation
+instead of its own DPVO, ensuring consistent camera understanding.
+
+Usage:
+ python convert_phmr_cam_to_genmo.py \
+ --prompthmr-results ../PromptHMR/results/VRM_JG6Z7WA_clip_0/results.pkl \
+ --genmo-output-dir outputs/infer_video/VRM_JG6Z7WA_clip_0 \
+ --target-fps 30
+"""
+
+import argparse
+from pathlib import Path
+
+import joblib
+import numpy as np
+import torch
+
+
+def main():
+ parser = argparse.ArgumentParser(description="Convert PromptHMR camera to GENMO slam.pt")
+ parser.add_argument("--prompthmr-results", type=Path, required=True,
+ help="Path to PromptHMR results.pkl")
+ parser.add_argument("--genmo-output-dir", type=Path, required=True,
+ help="GENMO output directory (will write to preprocess/slam.pt)")
+ parser.add_argument("--target-fps", type=int, default=30,
+ help="Target FPS for GENMO (default: 30)")
+ parser.add_argument("--source-fps", type=int, default=None,
+ help="Source FPS from PromptHMR (auto-detect if not set)")
+ args = parser.parse_args()
+
+ # Load PromptHMR results
+ print(f"Loading PromptHMR results from {args.prompthmr_results}")
+ phmr = joblib.load(args.prompthmr_results)
+
+ # Get camera in world-to-camera format (Rcw, Tcw)
+ camera_world = phmr["camera_world"]
+ Rcw = torch.from_numpy(camera_world["Rcw"]).float() # [N, 3, 3] world-to-camera rotation
+ Tcw = torch.from_numpy(camera_world["Tcw"]).float() # [N, 3] world-to-camera translation
+
+ num_phmr_frames = Rcw.shape[0]
+ print(f"PromptHMR camera frames: {num_phmr_frames}")
+
+ # Build T_w2c (4x4 world-to-camera transform)
+ T_w2c = torch.eye(4).unsqueeze(0).repeat(num_phmr_frames, 1, 1)
+ T_w2c[:, :3, :3] = Rcw
+ T_w2c[:, :3, 3] = Tcw
+
+ # Handle FPS conversion if needed
+ if args.source_fps is not None:
+ source_fps = args.source_fps
+ else:
+ # Try to detect from PromptHMR
+ # PromptHMR typically runs at video's native FPS (often 60)
+ # GENMO typically runs at 30fps
+ # Default assumption: PromptHMR is 2x GENMO
+ source_fps = 60 # assume 60fps if not specified
+
+ if source_fps != args.target_fps:
+ ratio = source_fps / args.target_fps
+ target_frames = int(num_phmr_frames / ratio)
+ print(f"Resampling from {source_fps}fps to {args.target_fps}fps ({num_phmr_frames} -> {target_frames} frames)")
+
+ # Resample by selecting every Nth frame
+ indices = (torch.arange(target_frames) * ratio).long().clamp(0, num_phmr_frames - 1)
+ T_w2c = T_w2c[indices]
+ print(f"Resampled to {T_w2c.shape[0]} frames")
+
+ # Save to GENMO's expected location
+ preprocess_dir = args.genmo_output_dir / "preprocess"
+ preprocess_dir.mkdir(parents=True, exist_ok=True)
+ slam_path = preprocess_dir / "slam.pt"
+
+ # Backup existing slam.pt if present
+ if slam_path.exists():
+ backup_path = preprocess_dir / "slam_dpvo_backup.pt"
+ if not backup_path.exists():
+ import shutil
+ shutil.copy(slam_path, backup_path)
+ print(f"Backed up original DPVO slam.pt to {backup_path}")
+
+ # Save in GENMO's expected format (N, 4, 4)
+ torch.save(T_w2c, slam_path)
+ print(f"Saved PromptHMR camera to {slam_path}")
+ print(f"Camera shape: {T_w2c.shape}")
+
+ # Also save K if available
+ K = np.eye(3)
+ K[0, 0] = camera_world["img_focal"]
+ K[1, 1] = camera_world["img_focal"]
+ K[0, 2] = camera_world["img_center"][0]
+ K[1, 2] = camera_world["img_center"][1]
+
+ k_path = preprocess_dir / "K_phmr.pt"
+ torch.save(torch.from_numpy(K).float(), k_path)
+ print(f"Saved intrinsics K to {k_path}")
+
+ print("\nNow you can run GENMO inference - it will use PromptHMR's camera!")
+ print("Make sure to delete hmr4d_results.pt first to force re-inference:")
+ print(f" rm {args.genmo_output_dir}/hmr4d_results.pt")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/demo/demo_text.py b/scripts/demo/demo_text.py
new file mode 100644
index 0000000000000000000000000000000000000000..02e2fd944e0a6593357dd2e8b2ae3bc70789f7a7
--- /dev/null
+++ b/scripts/demo/demo_text.py
@@ -0,0 +1,732 @@
+import argparse
+import os
+import subprocess
+from glob import glob
+from pathlib import Path
+
+import cv2
+import ffmpeg
+import hydra
+import imageio.v3 as iio
+import numpy as np
+import open3d as o3d
+import torch
+from einops import einsum
+from hydra import compose, initialize_config_module
+from PIL import Image, ImageDraw, ImageFont
+from tqdm import tqdm
+from genmo.utils.tools import rsync_file_from_remote, find_last_version
+import sys
+print(sys.path)
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import (
+ convert_K_to_K4,
+ create_camera_sensor,
+ estimate_K,
+ get_bbx_xys_from_xyxy,
+)
+from genmo.utils.geo_transform import (
+ apply_T_on_points,
+ compute_cam_angvel,
+ compute_cam_tvel,
+ compute_T_ayfz2ay,
+ normalize_T_w2c,
+)
+from genmo.utils.net_utils import detach_to_cpu, to_cuda
+from third_party.GVHMR.hmr4d.utils.preproc import (
+ Extractor,
+ Tracker,
+ VitPoseExtractor,
+)
+from genmo.utils.pylogger import Log
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+from genmo.utils.video_io_utils import (
+ concat_videos,
+ get_video_lwh,
+ get_video_reader,
+ get_writer,
+ merge_videos_horizontal,
+ read_video_np,
+ save_video,
+)
+from genmo.utils.vis.cv2_utils import (
+ draw_bbx_xyxy_on_image_batch,
+ draw_coco17_skeleton_batch,
+)
+from genmo.utils.vis.o3d_render import Settings, create_meshes, get_ground
+from genmo.utils.vis.renderer import (
+ Renderer,
+ get_global_cameras,
+ get_global_cameras_static,
+ get_global_cameras_static_v2,
+ get_ground_params_from_points,
+)
+from genmo.utils.rotation_conversions import quaternion_to_matrix
+
+CRF = 23 # 17 is lossless, every +6 halves the mp4 size
+
+
+def create_text_video(
+ output_path,
+ text,
+ fps=30,
+ num_frames=90,
+ width=1280,
+ height=720,
+ font_path="arial.ttf",
+ font_size=60,
+ text_color=(255, 255, 255),
+):
+ """
+ Create a video with text centered on a black background.
+ Text will automatically wrap if it's too long for the screen width.
+
+ Args:
+ output_path: Path to save the output video
+ text: Text to display
+ fps: Frames per second
+ num_frames: Total number of frames
+ width: Video width
+ height: Video height
+ font_path: Path to font file
+ font_size: Font size
+ text_color: RGB tuple for text color
+ """
+ # Create VideoWriter object
+ fourcc = cv2.VideoWriter_fourcc(*"mp4v") # or 'XVID'
+ video = cv2.VideoWriter(output_path, fourcc, fps, (width, height))
+
+ # Create a black frame with text
+ try:
+ font = ImageFont.truetype(font_path, font_size)
+ except IOError:
+ # Fallback to default font if specified font not found
+ font = ImageFont.load_default(size=font_size)
+
+ # Create a black image with text
+ img = Image.new("RGB", (width, height), color=(0, 0, 0))
+ draw = ImageDraw.Draw(img)
+
+ # Calculate max width for text (with some margin)
+ max_text_width = width * 0.9
+
+ # Wrap text
+ lines = []
+ words = text.split()
+ current_line = words[0]
+
+ for word in words[1:]:
+ # Check if adding this word exceeds the max width
+ test_line = current_line + " " + word
+ test_width = draw.textbbox((0, 0), test_line, font=font)[2]
+
+ if test_width <= max_text_width:
+ current_line = test_line
+ else:
+ lines.append(current_line)
+ current_line = word
+
+ lines.append(current_line) # Add the last line
+
+ # Calculate total text height
+ line_height = font_size * 1.2 # Add some line spacing
+ total_text_height = len(lines) * line_height
+
+ # Calculate starting y position to center all text vertically
+ y_position = (height - total_text_height) // 2
+
+ # Draw each line of text centered horizontally
+ for line in lines:
+ line_width = draw.textbbox((0, 0), line, font=font)[2]
+ x_position = (width - line_width) // 2
+ draw.text((x_position, y_position), line, font=font, fill=text_color)
+ y_position += line_height
+
+ # Convert PIL Image to OpenCV format
+ frame = np.array(img)
+ # Convert RGB to BGR (OpenCV uses BGR)
+ frame = cv2.cvtColor(frame, cv2.COLOR_RGB2BGR)
+
+ # Write the frame to video multiple times
+ for _ in range(num_frames):
+ video.write(frame)
+
+ # Release the video writer
+ video.release()
+ print(f"Video saved to {output_path}")
+
+
+@torch.no_grad()
+def run_preprocess_text(cfg):
+ Log.info("[Preprocess] Start text!")
+ tic = Log.time()
+
+ text_1 = cfg.text1
+
+ text_length = cfg.text_length
+ bbx_xys = torch.zeros(text_length, 3)
+
+ return_data = {
+ "meta": {
+ "vid": "text",
+ "caption": [text_1],
+ },
+ "length": torch.tensor(text_length),
+ "bbx_xys": bbx_xys,
+ "K_fullimg": torch.eye(3).repeat(text_length, 3, 3),
+ "f_imgseq": torch.zeros(text_length, 1024),
+ "kp2d": torch.zeros(text_length, 17, 3),
+ # "cam_angvel": torch.zeros_like(
+ # compute_cam_angvel(torch.eye(3)[None].repeat(text_length, 1, 1))
+ # ),
+ "cam_angvel": compute_cam_angvel(torch.eye(3)[None].repeat(text_length, 1, 1)),
+ "cam_tvel": torch.zeros(text_length, 3),
+ "R_w2c": torch.eye(3).reshape(1, 3, 3).repeat(text_length, 1, 1),
+ "T_w2c": torch.eye(4).reshape(1, 4, 4).repeat(text_length, 1, 1),
+ "gt_T_w2c": torch.eye(4).reshape(1, 4, 4).repeat(text_length, 1, 1),
+ "gender": "neutral",
+ "caption": text_1,
+ "has_text": torch.tensor([True]),
+ "mask": {
+ "valid": torch.ones(text_length),
+ # "vitpose": False,
+ # "bbx_xys": False,
+ # "f_imgseq": False,
+ # "spv_incam_only": False,
+ "has_img_mask": torch.zeros(text_length).bool(),
+ "has_2d_mask": torch.zeros(text_length).bool(),
+ "has_cam_mask": torch.ones(text_length).bool(),
+ "has_audio_mask": torch.zeros(text_length).bool(),
+ "has_music_mask": torch.zeros(text_length).bool(),
+ },
+ }
+ return return_data
+
+
+@torch.no_grad()
+def run_preprocess(cfg, vid=1):
+ Log.info(f"[Preprocess] Start {vid}!")
+ tic = Log.time()
+ paths = cfg.paths
+ # video_path = cfg.video_path
+ if vid == 1:
+ video_path = cfg.video1_path
+ else:
+ video_path = cfg.video2_path
+ bbx_path = paths.bbx1 if vid == 1 else paths.bbx2
+ bbx_xyxy_video_overlay_path = (
+ paths.bbx_xyxy_video_overlay1 if vid == 1 else paths.bbx_xyxy_video_overlay2
+ )
+ vitpose_path = paths.vitpose1 if vid == 1 else paths.vitpose2
+ vitpose_video_overlay_path = (
+ paths.vitpose_video_overlay1 if vid == 1 else paths.vitpose_video_overlay2
+ )
+ slam_path = paths.slam1 if vid == 1 else paths.slam2
+ static_cam = cfg.static_cam1 if vid == 1 else cfg.static_cam2
+ vimo_pred_path = paths.vimo_pred1 if vid == 1 else paths.vimo_pred2
+ vit_features_path = paths.vit_features1 if vid == 1 else paths.vit_features2
+ verbose = cfg.verbose
+
+ # Get bbx tracking result
+ if not Path(bbx_path).exists():
+ tracker = Tracker()
+ bbx_xyxy = tracker.get_one_track(video_path).float() # (L, 4)
+ bbx_xys = get_bbx_xys_from_xyxy(
+ bbx_xyxy, base_enlarge=1.2
+ ).float() # (L, 3) apply aspect ratio and enlarge
+ torch.save({"bbx_xyxy": bbx_xyxy, "bbx_xys": bbx_xys}, bbx_path)
+ del tracker
+ else:
+ bbx_xys = torch.load(bbx_path)["bbx_xys"]
+ Log.info(f"[Preprocess] bbx (xyxy, xys) from {bbx_path}")
+ if verbose:
+ video = read_video_np(video_path)
+ bbx_xyxy = torch.load(bbx_path)["bbx_xyxy"]
+ video_overlay = draw_bbx_xyxy_on_image_batch(bbx_xyxy, video)
+ save_video(video_overlay, bbx_xyxy_video_overlay_path)
+
+ # Get VitPose
+ if not Path(vitpose_path).exists():
+ vitpose_extractor = VitPoseExtractor()
+ vitpose = vitpose_extractor.extract(video_path, bbx_xys)
+ torch.save(vitpose, vitpose_path)
+ del vitpose_extractor
+ else:
+ vitpose = torch.load(vitpose_path)
+ Log.info(f"[Preprocess] vitpose from {vitpose_path}")
+ if verbose:
+ video = read_video_np(video_path)
+ video_overlay = draw_coco17_skeleton_batch(video, vitpose, 0.5)
+ save_video(video_overlay, vitpose_video_overlay_path)
+
+ if isinstance(vitpose, tuple):
+ vitpose = vitpose[0]
+
+ # Get DROID-SLAM results
+ if not static_cam: # use slam to get cam rotation
+ if not Path(slam_path).exists():
+ length, width, height = get_video_lwh(video_path)
+ K_fullimg = estimate_K(width, height)
+ intrinsics = convert_K_to_K4(K_fullimg)
+ cam_int = [
+ K_fullimg[0, 0],
+ K_fullimg[1, 1],
+ K_fullimg[0, 2],
+ K_fullimg[1, 2],
+ ]
+ out_dir = os.path.dirname(slam_path)
+ np.save(f"{out_dir}/cam_int.npy", cam_int)
+
+ # parse video to frames
+ video = read_video_np(video_path)
+ img_dir = os.path.join(os.path.dirname(video_path), f"imgs_{vid}")
+ os.makedirs(img_dir, exist_ok=True)
+ for i, frame in enumerate(video):
+ cv2.imwrite(f"{img_dir}/{i:06d}.jpg", frame[..., ::-1])
+ i += 1
+
+ cmd = f"python tools/estimate_camera_dir.py --img_dir {img_dir} --out_dir {out_dir}"
+ Log.info(f"[DROID-SLAM] {cmd}")
+ subprocess.run(cmd, shell=True)
+
+ else:
+ Log.info(f"[Preprocess] slam results from {slam_path}")
+ else:
+ length, width, height = get_video_lwh(video_path)
+ K_fullimg = estimate_K(width, height)
+ intrinsics = convert_K_to_K4(K_fullimg)
+ cam_int = [
+ K_fullimg[0, 0],
+ K_fullimg[1, 1],
+ K_fullimg[0, 2],
+ K_fullimg[1, 2],
+ ]
+ out_dir = os.path.dirname(slam_path)
+ np.save(f"{out_dir}/cam_int.npy", cam_int)
+
+ # Get vit features
+ if not Path(vit_features_path).exists():
+ extractor = Extractor()
+ vit_features = extractor.extract_video_features(video_path, bbx_xys)
+ torch.save(vit_features, vit_features_path)
+ del extractor
+ else:
+ Log.info(f"[Preprocess] vit_features from {vit_features_path}")
+
+ Log.info(f"[Preprocess] End. Time elapsed: {Log.time() - tic:.2f}s")
+
+
+def render_incam(cfg, vid_slice, vid=1):
+ incam_video_path = (
+ Path(cfg.paths.incam_video1) if vid == 1 else Path(cfg.paths.incam_video2)
+ )
+ if incam_video_path.exists():
+ Log.info(f"[Render Incam] Video already exists at {incam_video_path}")
+ return
+
+ pred_full = torch.load(cfg.paths.hmr4d_results)
+ pred = {"smpl_params_incam": pred_full["smpl_params_incam"]}
+ start_idx, end_idx = vid_slice[vid]
+ for k in pred_full.keys():
+ if k not in ["smpl_params_incam", "K_fullimg"]:
+ continue
+ if isinstance(pred_full[k], dict):
+ pred[k] = {}
+ for kk in pred_full[k].keys():
+ pred[k][kk] = pred_full[k][kk][start_idx:end_idx]
+ else:
+ pred[k] = pred_full[k][start_idx:end_idx]
+
+ smplx = make_smplx("supermotion").cuda()
+ smplx2smpl = torch.load("inputs/checkpoints/body_models/smplx2smpl_sparse.pt").cuda()
+ faces_smpl = make_smplx("smpl").faces
+
+ # smpl
+ smplx_out = smplx(**to_cuda(pred["smpl_params_incam"]))
+ pred_c_verts = torch.stack(
+ [torch.matmul(smplx2smpl, v_) for v_ in smplx_out.vertices]
+ )
+
+ # -- rendering code -- #
+ video_path = cfg.video1_path if vid == 1 else cfg.video2_path
+ video_30fps_path = str(video_path).replace(".mp4", "_30fps.mp4")
+
+ # convert to 30 fps
+ fps = cfg.orig_fps1 if vid == 1 else cfg.orig_fps2
+ if fps != 30:
+ stream = ffmpeg.input(video_path).filter("setpts", f"{30.0 / fps}*PTS")
+ output = ffmpeg.output(stream, video_30fps_path)
+ ffmpeg.run(output, overwrite_output=True, quiet=True)
+ else:
+ video_30fps_path = video_path
+
+ length, width, height = get_video_lwh(video_30fps_path)
+ K = pred["K_fullimg"][0]
+
+ # renderer
+ renderer = Renderer(width, height, device="cuda", faces=faces_smpl, K=K)
+ reader = get_video_reader(video_30fps_path) # (F, H, W, 3), uint8, numpy
+ bbx_path = cfg.paths.bbx1 if vid == 1 else cfg.paths.bbx2
+ bbx_xys_render = torch.load(bbx_path)["bbx_xys"]
+ color = torch.ones(3).float().cuda() * 0.8
+ color_purple = torch.tensor([0.69019608, 0.39215686, 0.95686275]).cuda()
+ color[0] = color_purple[0]
+ color[1] = color_purple[1]
+ color[2] = color_purple[2]
+
+ # -- render mesh -- #
+ verts_incam = pred_c_verts
+ writer = get_writer(incam_video_path, fps=30, crf=CRF)
+ assert abs(get_video_lwh(video_30fps_path)[0] - len(verts_incam)) < 10, (
+ f"Video length mismatch: {get_video_lwh(video_30fps_path)[0]} != {len(verts_incam)}"
+ )
+ for i, img_raw in tqdm(
+ enumerate(reader),
+ total=get_video_lwh(video_30fps_path)[0],
+ desc=f"Rendering Incam",
+ ):
+ if i >= verts_incam.shape[0]:
+ break
+ img = renderer.render_mesh(
+ verts_incam[i].cuda(), img_raw, [color[0], color[1], color[2]]
+ )
+
+ # # bbx
+ # bbx_xys_ = bbx_xys_render[i].cpu().numpy()
+ # lu_point = (bbx_xys_[:2] - bbx_xys_[2:] / 2).astype(int)
+ # rd_point = (bbx_xys_[:2] + bbx_xys_[2:] / 2).astype(int)
+ # img = cv2.rectangle(img, lu_point, rd_point, (255, 178, 102), 2)
+
+ writer.write_frame(img)
+ writer.close()
+ reader.close()
+
+
+def render_global_o3d(cfg, orig_fps):
+ global_video_path = Path(cfg.paths.global_video)
+ # if global_video_path.exists():
+ # Log.info(f"[Render Global] Video already exists at {global_video_path}")
+ # return
+
+ debug_cam = False
+ pred = torch.load(cfg.paths.hmr4d_results)
+ smplx = make_smplx("supermotion").cuda()
+ smplx2smpl = torch.load("inputs/checkpoints/body_models/smplx2smpl_sparse.pt").cuda()
+ faces_smpl = make_smplx("smpl").faces
+ J_regressor = torch.load(
+ "inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt"
+ ).cuda()
+
+ # smpl
+ smplx_out = smplx(**to_cuda(pred["smpl_params_global"]))
+ pred_ay_verts = torch.stack(
+ [torch.matmul(smplx2smpl, v_) for v_ in smplx_out.vertices]
+ )
+
+ def move_to_start_point_face_z(verts, J_regressor):
+ "XZ to origin, Start from the ground, Face-Z"
+ # position
+ verts = verts.clone() # (L, V, 3)
+ offset = einsum(J_regressor, verts[0], "j v, v i -> j i")[0] # (3)
+ offset[1] = verts[:, :, [1]].min()
+ verts = verts - offset
+ # face direction
+ T_ay2ayfz = compute_T_ayfz2ay(
+ einsum(J_regressor, verts[[0]], "j v, l v i -> l j i"), inverse=True
+ )
+ verts = apply_T_on_points(verts, T_ay2ayfz)
+ return verts
+
+ verts_glob_list = move_to_start_point_face_z(pred_ay_verts, J_regressor)
+ joints_glob_list = einsum(J_regressor, verts_glob_list, "j v, l v i -> l j i")
+ length = verts_glob_list.shape[0]
+
+ # -- rendering code -- #
+ video_path = cfg.text1_video_path
+ # orig_fps = cv2.VideoCapture(video_path).get(cv2.CAP_PROP_FPS)
+ length, width, height = get_video_lwh(video_path)
+ _, _, K = create_camera_sensor(width, height, 24) # render as 24mm lens
+ device = verts_glob_list.device
+ # global_video_path = f"out/{vis_type}_video/{fname}.mp4"
+ os.makedirs(os.path.dirname(global_video_path), exist_ok=True)
+ writer = get_writer(global_video_path, fps=orig_fps, crf=CRF)
+
+ mat_settings = Settings()
+
+ color_purple = torch.tensor([0.69019608, 0.39215686, 0.95686275]).to(device)
+ color_green = torch.tensor([0.46666667, 0.90196078, 0.74901961]).to(device)
+ color_light_purple = torch.tensor([1.0, 0.65490196, 0.95294118]).to(device)
+
+ # -- render mesh -- #
+ scale, cx, cz = get_ground_params_from_points(
+ joints_glob_list[:, 0], verts_glob_list
+ )
+ scale = max(scale, 3)
+ ground_geometry = get_ground(scale * 1.5, cx, cz)
+ # color = torch.ones(3).float().cuda() * 0.8
+
+ T, V, _ = verts_glob_list.shape
+
+ position, target, up = get_global_cameras_static_v2(
+ # verts_list[0].cpu(),
+ verts_glob_list.cpu().clone(),
+ beta=3.0,
+ # beta=4.0,
+ cam_height_degree=20,
+ target_center_height=1.0,
+ )
+
+ trans_mat_box = mat_settings._materials[Settings.Transparency]
+ lit_mat_box = mat_settings._materials[Settings.LIT]
+
+ colors = color_purple[None, :].repeat(T, 1)
+ # colors[:, 0] = torch.linspace(color_green[0], color_purple[0], T)
+ # colors[:, 1] = torch.linspace(color_green[1], color_purple[1], T)
+ # colors[:, 2] = torch.linspace(color_green[2], color_purple[2], T)
+
+ colors_trans = torch.zeros_like(colors)
+ colors_trans[:, 0] = torch.linspace(color_green[0], color_light_purple[0], T)
+ colors_trans[:, 1] = torch.linspace(color_green[1], color_light_purple[1], T)
+ colors_trans[:, 2] = torch.linspace(color_green[2], color_light_purple[2], T)
+ faces = torch.from_numpy(faces_smpl.astype("int")).to(device)
+
+ # colors = torch.stack([torch.from_numpy(color_rgb[i % len(color_rgb)]).float().cuda() for i in range(len(verts_glob_list))], dim=0)
+ renderer = o3d.visualization.rendering.OffscreenRenderer(width, height)
+ # renderer.scene.camera.set_projection(K[0, 0], K[1, 1], K[0, 2], K[1, 2], width, height, 0.1, 100.0)
+ renderer.scene.camera.set_projection(
+ K.cpu().double().numpy(), 0.1, 100.0, float(width), float(height)
+ )
+
+ camera = renderer.scene.camera
+ camera.look_at(target[:, None], position[:, None], up[:, None])
+
+ gv, gf, gc = ground_geometry
+ ground_mesh = create_meshes(gv, gf, gc[..., :3])
+ renderer.scene.add_geometry(
+ "mesh_ground", ground_mesh, o3d.visualization.rendering.MaterialRecord()
+ )
+ # faces_list = list(torch.unbind(faces, dim=0)) # + [gf]
+
+ T_c2w = camera.get_view_matrix()
+ R_w2c = T_c2w[:3, :3].T
+ for t, verts in tqdm(enumerate(verts_glob_list)):
+ verts = verts_glob_list[t] # + [gv]
+ mat = o3d.visualization.rendering.MaterialRecord()
+ mat.base_color = [0.9, 0.9, 0.9, 0.3 + t * 0.7 / T]
+ mat.shader = Settings.Transparency
+ # mat.opacity = (i + 1) / N
+ mat.thickness = 1.0
+ mat.transmission = 1.0
+ mat.absorption_distance = 10
+ mat.absorption_color = [0.5, 0.5, 0.5]
+
+ mesh = create_meshes(verts, faces, colors[t])
+ if t > 0:
+ renderer.scene.remove_geometry(f"mesh_{t - 1}")
+ # img = renderer.render_to_image()
+
+ renderer.scene.add_geometry(f"mesh_{t}", mesh, lit_mat_box)
+ # mesh = create_meshes(verts[i], faces_list[i], colors[t])
+ # renderer.scene.add_geometry(f"mesh_{i}_{t}", mesh, mat)
+
+ img = renderer.render_to_image()
+ # import ipdb; ipdb.set_trace()
+ # o3d.io.write_image("out/tmp.png", img)
+ # img = cv2.imread("out/tmp.png")
+ writer.write_frame(np.array(img))
+ writer.close()
+ print(f"Saved to {global_video_path}")
+
+
+def render_global(cfg):
+ global_video_path = Path(cfg.paths.global_video)
+ if global_video_path.exists():
+ Log.info(f"[Render Global] Video already exists at {global_video_path}")
+ return
+
+ debug_cam = False
+ pred = torch.load(cfg.paths.hmr4d_results)
+ smplx = make_smplx("supermotion").cuda()
+ smplx2smpl = torch.load("inputs/checkpoints/body_models/smplx2smpl_sparse.pt").cuda()
+ faces_smpl = make_smplx("smpl").faces
+ J_regressor = torch.load(
+ "inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt"
+ ).cuda()
+
+ # smpl
+ smplx_out = smplx(**to_cuda(pred["smpl_params_global"]))
+ pred_ay_verts = torch.stack(
+ [torch.matmul(smplx2smpl, v_) for v_ in smplx_out.vertices]
+ )
+
+ def move_to_start_point_face_z(verts):
+ "XZ to origin, Start from the ground, Face-Z"
+ # position
+ verts = verts.clone() # (L, V, 3)
+ offset = einsum(J_regressor, verts[0], "j v, v i -> j i")[0] # (3)
+ offset[1] = verts[:, :, [1]].min()
+ verts = verts - offset
+ # face direction
+ T_ay2ayfz = compute_T_ayfz2ay(
+ einsum(J_regressor, verts[[0]], "j v, l v i -> l j i"), inverse=True
+ )
+ verts = apply_T_on_points(verts, T_ay2ayfz)
+ return verts, T_ay2ayfz
+
+ verts_glob, T_ay2ayfz = move_to_start_point_face_z(pred_ay_verts)
+ joints_glob = einsum(J_regressor, verts_glob, "j v, l v i -> l j i") # (L, J, 3)
+ # global_R, global_T, global_lights = get_global_cameras_static(
+ # verts_glob.cpu(),
+ # beta=2.0,
+ # cam_height_degree=20,
+ # target_center_height=1.0,
+ # )
+ global_R, global_T, global_lights = get_global_cameras(
+ verts_glob.cpu(),
+ )
+ # pred_T_w2c = pred["net_outputs"]["pred_T_w2c"].to(T_ay2ayfz)
+ # T_w2c = (T_ay2ayfz @ pred_T_w2c.inverse()).inverse()
+
+ # # pred_T_c2w = pred_T_w2c.inverse()
+ # global_R = T_w2c[:, :3, :3]
+ # global_T = T_w2c[:, :3, 3]
+
+ # -- rendering code -- #
+ video_path = cfg.video_path
+ length, width, height = get_video_lwh(video_path)
+ _, _, K = create_camera_sensor(width, height, 24) # render as 24mm lens
+
+ # renderer
+ renderer = Renderer(width, height, device="cuda", faces=faces_smpl, K=K, bin_size=0)
+ # renderer = Renderer(width, height, device="cuda", faces=faces_smpl, K=K, bin_size=0)
+
+ # -- render mesh -- #
+ scale, cx, cz = get_ground_params_from_points(joints_glob[:, 0], verts_glob)
+ renderer.set_ground(scale * 1.5, cx, cz)
+ color = torch.ones(3).float().cuda() * 0.8
+
+ render_length = length if not debug_cam else 8
+ writer = get_writer(global_video_path, fps=30, crf=CRF)
+ for i in tqdm(range(render_length), desc=f"Rendering Global"):
+ cameras = renderer.create_camera(global_R[i], global_T[i])
+ img = renderer.render_with_ground(
+ verts_glob[[i]], color[None], cameras, global_lights
+ )
+ writer.write_frame(img)
+ writer.close()
+
+
+@hydra.main(version_base="1.3", config_path="../../configs", config_name="demo")
+def main(cfg):
+ # Parse args with proper Hydra override support
+ if cfg.text1_file is not None:
+ text_file = open(cfg.text1_file, "r")
+ cfg.text1 = text_file.read().strip()
+
+ cfg.text1_video_name = cfg.text1.replace(" ", "_").replace(".", "")
+ cfg.text1_video_path = os.path.join(cfg.output_dir, cfg.text1.replace(" ", "_").replace(".", "") + ".mp4")
+
+ # Output
+ Log.info(f"[Output Dir]: {cfg.output_dir}")
+ Path(cfg.output_dir).mkdir(parents=True, exist_ok=True)
+
+ paths = cfg.paths
+ Log.info(f"[GPU]: {torch.cuda.get_device_name()}")
+ Log.info(f"[GPU]: {torch.cuda.get_device_properties('cuda')}")
+
+ # ===== Preprocess and save to disk ===== #
+ data_text = run_preprocess_text(cfg)
+ length = cfg.text_length
+ width, height = 1280, 720
+
+ # generate text video
+ text_video_path = Path(cfg.text1_video_path)
+ if not text_video_path.exists() or True:
+ Log.info("[Generate Text Video]")
+ create_text_video(
+ text_video_path,
+ cfg.text1,
+ fps=30,
+ num_frames=cfg.text_length,
+ width=width,
+ height=height,
+ font_size=int(min(width, height) * 0.1),
+ )
+
+ # merge data
+ data = dict()
+ tot_length = data_text["length"]
+ # multi_text_data = {
+ # "vid": ["text1"],
+ # "caption": [cfg.text1],
+ # "text_ind": [0],
+ # "window_start": [0],
+ # "window_end": [1],
+ # }
+ # multi_text_data["window_start"] = torch.tensor(multi_text_data["window_start"])
+ # multi_text_data["window_end"] = torch.tensor(multi_text_data["window_end"])
+ data_text["meta"] = [
+ {
+ "vid": "text1",
+ "caption": cfg.text1,
+ # "multi_text_data": multi_text_data,
+ }
+ ]
+ data = data_text
+
+ debug = False
+ if debug:
+ data = data_text
+ data["meta"] = [
+ {
+ "vid1": cfg.video1_name,
+ "caption": cfg.text1,
+ "eval_gen_only": True,
+ # "multi_text_data": multi_text_data,
+ }
+ ]
+ # ===== HMR4D ===== #
+ if not Path(paths.hmr4d_results).exists():
+ Log.info("[GENMO] Predicting")
+ model = hydra.utils.instantiate(cfg.model, _recursive_=False)
+
+ test_cp = cfg.get("test_checkpoint", "last")
+ if cfg.version is None:
+ version = find_last_version(cfg.ckpt_dir)
+ if version is None or cfg.get("rsync_ckpt", False):
+ remote_ckpt_dir = os.path.join(cfg.remote_results_path, cfg.data_name, cfg.exp_name)
+ version = find_last_version(remote_ckpt_dir, cp=test_cp)
+ ckpt_path = os.path.join("outputs", cfg.data_name, cfg.exp_name, f"version_{version}", "checkpoints", "last.ckpt")
+ print(f"rsyncing from remote {remote_ckpt_dir} to {ckpt_path}")
+ os.makedirs(os.path.dirname(ckpt_path), exist_ok=True)
+ rsync_file_from_remote(
+ ckpt_path,
+ remote_ckpt_dir,
+ # "outputs",
+ cfg.ckpt_dir,
+ hostname="cs-oci-ord-dc-03",
+ )
+ else:
+ ckpt_path = os.path.join(cfg.ckpt_dir, f"version_{version}", "checkpoints", "last.ckpt")
+
+ model.load_pretrained_model(ckpt_path)
+ model = model.eval().cuda()
+ tic = Log.sync_time()
+ pred = model.predict(data, static_cam=False, postproc=True)
+ pred = detach_to_cpu(pred)
+ data_time = data["length"] / 30
+ Log.info(
+ f"[GENMO] Elapsed: {Log.sync_time() - tic:.2f}s for data-length={data_time:.1f}s"
+ )
+ torch.save(pred, paths.hmr4d_results)
+
+ # ===== Render ===== #
+ render_global_o3d(cfg, 30)
+ if not Path(paths.incam_global_horiz_video).exists():
+ Log.info("[Merge Videos]")
+ merge_videos_horizontal(
+ [cfg.text1_video_path, paths.global_video], paths.incam_global_horiz_video
+ )
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/demo/hamer_inference.py b/scripts/demo/hamer_inference.py
new file mode 100644
index 0000000000000000000000000000000000000000..58f3d05090d01f5e65c9feaa6428bbdd8c48ada5
--- /dev/null
+++ b/scripts/demo/hamer_inference.py
@@ -0,0 +1,377 @@
+"""
+HaMeR (Hand Mesh Recovery) wrapper for integration with GENMO.
+Runs HaMeR on detected hand bounding boxes and returns MANO parameters.
+"""
+import sys
+import os
+from pathlib import Path
+
+# Add HaMeR to path BEFORE any other imports
+# This needs to be at the absolute path level
+_SCRIPT_DIR = Path(__file__).resolve().parent
+_GENMO_ROOT = _SCRIPT_DIR.parent.parent # GENMO/scripts/demo -> GENMO
+HAMER_ROOT = _GENMO_ROOT / "third_party" / "hamer"
+
+if str(HAMER_ROOT) not in sys.path:
+ sys.path.insert(0, str(HAMER_ROOT))
+
+import torch
+import numpy as np
+import cv2
+import mmcv
+from tqdm import tqdm
+
+
+class HaMeRInference:
+ """
+ HaMeR wrapper for hand mesh recovery.
+
+ Input: Hand bounding boxes + video frames
+ Output: MANO parameters (hand_pose, global_orient, betas)
+ """
+
+ def __init__(self, device='cuda:0'):
+ self.device = torch.device(device)
+ self._model = None
+ self._model_cfg = None
+
+ def _load_model(self):
+ """Lazy load HaMeR model."""
+ if self._model is not None:
+ return
+
+ # Override CACHE_DIR_HAMER to point to the correct location BEFORE importing
+ import hamer.configs
+ hamer.configs.CACHE_DIR_HAMER = str(HAMER_ROOT / "_DATA")
+
+ from hamer.configs import CACHE_DIR_HAMER
+ from hamer.models import load_hamer, DEFAULT_CHECKPOINT
+
+ # The checkpoint path also needs to be updated
+ checkpoint_path = HAMER_ROOT / "_DATA" / "hamer_ckpts" / "checkpoints" / "hamer.ckpt"
+
+ if not checkpoint_path.exists():
+ raise FileNotFoundError(f"HaMeR checkpoint not found at {checkpoint_path}. Run fetch_demo_data.sh in third_party/hamer/")
+
+ # Load HaMeR
+ self._model, self._model_cfg = load_hamer(str(checkpoint_path))
+ self._model = self._model.to(self.device)
+ self._model.eval()
+
+ print(f"[HaMeR] Loaded model from {checkpoint_path}")
+
+ def _prepare_input(self, frame, bbox, is_right):
+ """
+ Prepare input for HaMeR model.
+
+ Args:
+ frame: (H, W, 3) BGR image
+ bbox: [x1, y1, x2, y2] hand bounding box
+ is_right: bool, True for right hand
+
+ Returns:
+ batch dict for HaMeR model
+ """
+ from hamer.datasets.vitdet_dataset import ViTDetDataset, DEFAULT_MEAN, DEFAULT_STD
+
+ # Validate frame shape - must be (H, W, 3)
+ if frame is None:
+ raise ValueError("Frame is None")
+ if not isinstance(frame, np.ndarray):
+ raise ValueError(f"Frame must be numpy array, got {type(frame)}")
+ if frame.ndim != 3:
+ raise ValueError(f"Frame must be 3D (H, W, C), got {frame.ndim}D with shape {frame.shape}")
+ if frame.shape[2] != 3:
+ raise ValueError(f"Frame must have 3 channels, got {frame.shape[2]} with shape {frame.shape}")
+
+ # Ensure frame is contiguous and correct dtype
+ if not frame.flags['C_CONTIGUOUS']:
+ frame = np.ascontiguousarray(frame)
+ if frame.dtype != np.uint8:
+ frame = frame.astype(np.uint8)
+
+ # Create dataset for single hand
+ boxes = np.array([bbox])
+ right = np.array([1 if is_right else 0])
+
+ dataset = ViTDetDataset(
+ self._model_cfg,
+ frame, # BGR image
+ boxes,
+ right,
+ rescale_factor=2.0
+ )
+
+ return dataset[0]
+
+ @torch.no_grad()
+ def predict_single(self, frame, bbox, is_right):
+ """
+ Predict MANO parameters for a single hand.
+
+ Args:
+ frame: (H, W, 3) BGR image
+ bbox: [x1, y1, x2, y2] hand bounding box
+ is_right: bool, True for right hand
+
+ Returns:
+ dict with:
+ - hand_pose: (15, 3) axis-angle for finger joints
+ - global_orient: (3,) axis-angle for wrist
+ - betas: (10,) shape parameters
+ - vertices: (778, 3) mesh vertices
+ - keypoints_3d: (21, 3) 3D hand joints
+ """
+ self._load_model()
+
+ from hamer.utils import recursive_to
+
+ batch = self._prepare_input(frame, bbox, is_right)
+ # Add batch dimension to all array-like values (both numpy and torch)
+ # ViTDetDataset returns numpy arrays, not torch tensors, so we need to handle both
+ processed_batch = {}
+ for k, v in batch.items():
+ if isinstance(v, torch.Tensor):
+ processed_batch[k] = v.unsqueeze(0)
+ elif isinstance(v, np.ndarray):
+ # Add batch dimension to numpy array and convert to tensor
+ processed_batch[k] = torch.from_numpy(v).unsqueeze(0)
+ else:
+ processed_batch[k] = v
+ batch = recursive_to(processed_batch, self.device)
+
+ out = self._model(batch)
+
+ # Extract MANO parameters
+ pred_mano = out['pred_mano_params']
+
+ # Convert rotation matrices to axis-angle
+ global_orient_rotmat = pred_mano['global_orient'][0] # (1, 3, 3)
+ hand_pose_rotmat = pred_mano['hand_pose'][0] # (15, 3, 3)
+ betas = pred_mano['betas'][0] # (10,)
+
+ # Mirror for left hand (HaMeR predicts right-hand rotations)
+ if not is_right:
+ mirror = torch.diag(torch.tensor([-1.0, 1.0, 1.0], device=global_orient_rotmat.device))
+ global_orient_rotmat = mirror @ global_orient_rotmat @ mirror
+ hand_pose_rotmat = mirror @ hand_pose_rotmat @ mirror
+
+ global_orient_aa = self._rotmat_to_axis_angle(global_orient_rotmat.reshape(-1, 3, 3)) # (1, 3)
+ hand_pose_aa = self._rotmat_to_axis_angle(hand_pose_rotmat.reshape(-1, 3, 3)) # (15, 3)
+
+ # Compute full-frame camera translation for rendering
+ from hamer.utils.renderer import cam_crop_to_full
+ right_val = 1 if is_right else 0
+ multiplier = (2 * right_val - 1)
+ pred_cam = out['pred_cam'][0].detach().float().clone()
+ pred_cam[1] = multiplier * pred_cam[1]
+ box_center = torch.as_tensor(batch["box_center"], device=pred_cam.device, dtype=pred_cam.dtype)
+ box_size = torch.as_tensor(batch["box_size"], device=pred_cam.device, dtype=pred_cam.dtype)
+ img_size = torch.as_tensor(batch["img_size"], device=pred_cam.device, dtype=pred_cam.dtype)
+ scaled_focal_length = (
+ self._model_cfg.EXTRA.FOCAL_LENGTH / self._model_cfg.MODEL.IMAGE_SIZE * img_size.max()
+ )
+ cam_t_full = cam_crop_to_full(
+ pred_cam.unsqueeze(0), box_center, box_size, img_size, scaled_focal_length
+ )[0].detach().cpu().numpy()
+
+ verts = out['pred_vertices'][0].detach().cpu().numpy()
+ if not is_right:
+ verts[:, 0] *= -1.0
+
+ return {
+ 'hand_pose': hand_pose_aa.cpu().numpy(), # (15, 3)
+ 'global_orient': global_orient_aa.cpu().numpy().squeeze(0), # (3,)
+ 'betas': betas.cpu().numpy(), # (10,)
+ 'vertices': verts, # (778, 3)
+ 'keypoints_3d': out['pred_keypoints_3d'][0].cpu().numpy(), # (21, 3)
+ 'cam_t': cam_t_full, # (3,)
+ 'focal_length': float(scaled_focal_length),
+ 'is_right': right_val,
+ }
+
+ def _rotmat_to_axis_angle(self, rotmat):
+ """Convert rotation matrices to axis-angle representation."""
+ from pytorch3d.transforms import matrix_to_axis_angle
+ return matrix_to_axis_angle(rotmat)
+
+ @torch.no_grad()
+ def predict_video(self, video_path, left_bboxes, right_bboxes, masks=None):
+ """
+ Predict MANO parameters for all frames in video.
+
+ Args:
+ video_path: Path to video file
+ left_bboxes: List of left hand bboxes (None if not visible)
+ right_bboxes: List of right hand bboxes (None if not visible)
+ masks: Optional list of SAM masks
+
+ Returns:
+ left_hand_params: List of dicts with MANO params (or None)
+ right_hand_params: List of dicts with MANO params (or None)
+ """
+ self._load_model()
+
+ if isinstance(video_path, str):
+ video = mmcv.VideoReader(video_path)
+ else:
+ video = video_path
+
+ L = len(left_bboxes)
+ left_results = []
+ right_results = []
+
+ for i in tqdm(range(L), desc="HaMeR Hands"):
+ frame = video[i]
+
+ # Validate frame read from video
+ if frame is None:
+ print(f"[HaMeR] Warning: frame {i} is None, skipping")
+ left_results.append(None)
+ right_results.append(None)
+ continue
+ if not isinstance(frame, np.ndarray):
+ print(f"[HaMeR] Warning: frame {i} has unexpected type {type(frame)}, skipping")
+ left_results.append(None)
+ right_results.append(None)
+ continue
+ if frame.ndim != 3 or frame.shape[2] != 3:
+ print(f"[HaMeR] Warning: frame {i} has unexpected shape {frame.shape}, expected (H, W, 3), skipping")
+ left_results.append(None)
+ right_results.append(None)
+ continue
+
+ # Apply mask if available
+ if masks is not None and i < len(masks) and masks[i] is not None:
+ mask = masks[i]
+ if isinstance(mask, torch.Tensor):
+ mask = mask.numpy()
+ frame_h, frame_w = frame.shape[:2]
+ if mask.shape[0] != frame_h or mask.shape[1] != frame_w:
+ mask = cv2.resize(mask.astype(np.uint8), (frame_w, frame_h), interpolation=cv2.INTER_NEAREST)
+ gray_bg = np.full_like(frame, 128)
+ mask_3ch = mask[:, :, None].astype(bool)
+ frame = np.where(mask_3ch, frame, gray_bg)
+
+ # Validate post-mask frame shape
+ if frame.ndim != 3 or frame.shape[2] != 3:
+ print(f"[HaMeR] Warning: frame {i} after masking has unexpected shape {frame.shape}, skipping")
+ left_results.append(None)
+ right_results.append(None)
+ continue
+
+ # Left hand
+ if left_bboxes[i] is not None:
+ try:
+ left_result = self.predict_single(frame, left_bboxes[i], is_right=False)
+ except Exception as e:
+ print(f"[HaMeR] Left hand frame {i} failed: {e}")
+ left_result = None
+ else:
+ left_result = None
+ left_results.append(left_result)
+
+ # Right hand
+ if right_bboxes[i] is not None:
+ try:
+ right_result = self.predict_single(frame, right_bboxes[i], is_right=True)
+ except Exception as e:
+ print(f"[HaMeR] Right hand frame {i} failed: {e}")
+ right_result = None
+ else:
+ right_result = None
+ right_results.append(right_result)
+
+ return left_results, right_results
+
+
+def mano_to_smplx_hands(
+ left_results,
+ right_results,
+ num_frames,
+ smooth_alpha=None,
+ median_window=7,
+ mean_window=5,
+ max_delta=0.1,
+):
+ """
+ Convert HaMeR MANO results to SMPL-X hand pose format.
+
+ SMPL-X expects:
+ - left_hand_pose: (L, 15, 3) axis-angle
+ - right_hand_pose: (L, 15, 3) axis-angle
+
+ Args:
+ left_results: List of dicts from HaMeR (or None)
+ right_results: List of dicts from HaMeR (or None)
+ num_frames: Total number of frames
+
+ Returns:
+ left_hand_pose: (L, 15, 3) numpy array
+ right_hand_pose: (L, 15, 3) numpy array
+ """
+ left_hand_pose = np.zeros((num_frames, 15, 3), dtype=np.float32)
+ right_hand_pose = np.zeros((num_frames, 15, 3), dtype=np.float32)
+ left_valid = np.zeros(num_frames, dtype=bool)
+ right_valid = np.zeros(num_frames, dtype=bool)
+
+ for i in range(num_frames):
+ if left_results[i] is not None:
+ left_hand_pose[i] = left_results[i]['hand_pose']
+ left_valid[i] = True
+ if right_results[i] is not None:
+ right_hand_pose[i] = right_results[i]['hand_pose']
+ right_valid[i] = True
+
+ # Forward-fill missing frames to avoid jitter on occlusions
+ for i in range(1, num_frames):
+ if not left_valid[i]:
+ left_hand_pose[i] = left_hand_pose[i - 1]
+ if not right_valid[i]:
+ right_hand_pose[i] = right_hand_pose[i - 1]
+
+ # Median filter to suppress outliers
+ if median_window is not None and median_window > 1:
+ half = median_window // 2
+ left_filtered = left_hand_pose.copy()
+ right_filtered = right_hand_pose.copy()
+ for i in range(num_frames):
+ s = max(0, i - half)
+ e = min(num_frames, i + half + 1)
+ left_filtered[i] = np.median(left_hand_pose[s:e], axis=0)
+ right_filtered[i] = np.median(right_hand_pose[s:e], axis=0)
+ left_hand_pose = left_filtered
+ right_hand_pose = right_filtered
+
+ # Centered moving average to remove high-frequency jitter without lag
+ if mean_window is not None and mean_window > 1:
+ half = mean_window // 2
+ left_smoothed = left_hand_pose.copy()
+ right_smoothed = right_hand_pose.copy()
+ for i in range(num_frames):
+ s = max(0, i - half)
+ e = min(num_frames, i + half + 1)
+ left_smoothed[i] = left_hand_pose[s:e].mean(axis=0)
+ right_smoothed[i] = right_hand_pose[s:e].mean(axis=0)
+ left_hand_pose = left_smoothed
+ right_hand_pose = right_smoothed
+
+ # Optional EMA smoothing (disabled by default)
+ if smooth_alpha is not None and smooth_alpha > 0.0:
+ for i in range(1, num_frames):
+ left_hand_pose[i] = (1.0 - smooth_alpha) * left_hand_pose[i - 1] + smooth_alpha * left_hand_pose[i]
+ right_hand_pose[i] = (1.0 - smooth_alpha) * right_hand_pose[i - 1] + smooth_alpha * right_hand_pose[i]
+
+ # Clamp per-frame rotation change to suppress jitter spikes
+ if max_delta is not None and max_delta > 0.0:
+ for i in range(1, num_frames):
+ left_diff = left_hand_pose[i] - left_hand_pose[i - 1]
+ right_diff = right_hand_pose[i] - right_hand_pose[i - 1]
+ left_norm = np.linalg.norm(left_diff, axis=-1, keepdims=True)
+ right_norm = np.linalg.norm(right_diff, axis=-1, keepdims=True)
+ left_scale = np.minimum(1.0, max_delta / (left_norm + 1e-8))
+ right_scale = np.minimum(1.0, max_delta / (right_norm + 1e-8))
+ left_hand_pose[i] = left_hand_pose[i - 1] + left_diff * left_scale
+ right_hand_pose[i] = right_hand_pose[i - 1] + right_diff * right_scale
+
+ return left_hand_pose, right_hand_pose
diff --git a/scripts/demo/infer_video.py b/scripts/demo/infer_video.py
new file mode 100644
index 0000000000000000000000000000000000000000..8c74eee9dbad2c80bdf73256230bc90a152c1a91
--- /dev/null
+++ b/scripts/demo/infer_video.py
@@ -0,0 +1,1624 @@
+import logging
+import sys
+import gc
+import json
+from pathlib import Path
+from tqdm import tqdm # <--- Added for progress bars
+from types import SimpleNamespace
+
+# Ensure repo root is on sys.path when running as a script.
+REPO_ROOT = Path(__file__).resolve().parents[2]
+if str(REPO_ROOT) not in sys.path:
+ sys.path.insert(0, str(REPO_ROOT))
+
+import ffmpeg
+import hydra
+import numpy as np
+import cv2
+import torch
+from omegaconf import DictConfig
+
+from genmo.utils.geo_transform import (
+ apply_T_on_points,
+ compute_cam_angvel,
+ compute_T_ayfz2ay,
+)
+from genmo.utils.net_utils import detach_to_cpu, to_cuda
+from genmo.utils.pylogger import Log
+from genmo.utils.video_io_utils import (
+ copy_file,
+ get_video_lwh,
+ get_video_reader,
+ get_writer,
+ merge_videos_horizontal,
+)
+from genmo.utils.rotation_conversions import matrix_to_axis_angle
+from genmo.utils.vis.renderer import (
+ Renderer,
+ get_global_cameras_static,
+ get_ground_params_from_points,
+ perspective_projection,
+)
+
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import (
+ create_camera_sensor,
+ estimate_K,
+ get_bbx_xys_from_xyxy,
+ )
+from third_party.GVHMR.hmr4d.utils.preproc.tracker import Tracker
+from third_party.GVHMR.hmr4d.utils.preproc.vitpose import VitPoseExtractor
+from third_party.GVHMR.hmr4d.utils.preproc.vitfeat_extractor import Extractor
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+def _is_nonempty_file(path: str | Path) -> bool:
+ p = Path(path)
+ try:
+ return p.exists() and p.is_file() and p.stat().st_size > 0
+ except OSError:
+ return False
+
+
+def _is_valid_video(path: str | Path) -> bool:
+ if not _is_nonempty_file(path):
+ return False
+ try:
+ probe = ffmpeg.probe(str(path))
+ streams = probe.get("streams", [])
+ return any(s.get("codec_type") == "video" for s in streams)
+ except Exception:
+ return False
+
+
+def _get_smplx_batch_len(params: dict) -> int | None:
+ for key in ("body_pose", "global_orient"):
+ value = params.get(key)
+ if isinstance(value, (torch.Tensor, np.ndarray)) and value.ndim >= 2:
+ return value.shape[0]
+ betas = params.get("betas")
+ if isinstance(betas, (torch.Tensor, np.ndarray)) and betas.ndim == 2:
+ return betas.shape[0]
+ return None
+
+
+def _normalize_smplx_hand_pose(hand_pose, target_len: int | None):
+ if hand_pose is None:
+ return None
+ if isinstance(hand_pose, np.ndarray):
+ hand_pose = torch.from_numpy(hand_pose)
+ if not isinstance(hand_pose, torch.Tensor):
+ return hand_pose
+ if hand_pose.ndim == 1:
+ hand_pose = hand_pose.unsqueeze(0)
+ if hand_pose.ndim == 2 and hand_pose.shape[-1] == 45:
+ hand_pose = hand_pose.reshape(hand_pose.shape[0], 15, 3)
+ if hand_pose.ndim == 3 and hand_pose.shape[1:] == (15, 3):
+ if target_len is not None and hand_pose.shape[0] != target_len:
+ if hand_pose.shape[0] == 1:
+ hand_pose = hand_pose.repeat(target_len, 1, 1)
+ else:
+ Log.warning(
+ f"[SMPL-X] Hand pose batch ({hand_pose.shape[0]}) "
+ f"does not match body batch ({target_len}); rendering may fail."
+ )
+ return hand_pose
+
+
+def _vitpose_to_hand_bboxes(
+ kp2d,
+ bbx_xys,
+ conf_thresh=0.5,
+ wrist_scale=1.0,
+ wrist_shift=0.0,
+):
+ left_bboxes = []
+ right_bboxes = []
+ for i in range(len(kp2d)):
+ kp = kp2d[i]
+ left_box = None
+ right_box = None
+ if kp.shape[0] >= 11:
+ l_w = kp[9]
+ l_e = kp[7]
+ r_w = kp[10]
+ r_e = kp[8]
+ if l_w[2] > conf_thresh:
+ forearm_vec = l_w[:2] - l_e[:2] if l_e[2] > conf_thresh else np.zeros(2)
+ forearm = np.linalg.norm(forearm_vec)
+ # Fallback to body-relative size if forearm is foreshortened
+ # 0.15 * body_size is approx head size, reasonable for hand
+ size = max(forearm * wrist_scale, float(bbx_xys[i][2]) * 0.15)
+
+ # Center on wrist (shift=0.0) covers hand regardless of angle
+ center = l_w[:2] + (forearm_vec * wrist_shift)
+ cx, cy = center[0], center[1]
+ left_box = np.array([cx - size / 2, cy - size / 2, cx + size / 2, cy + size / 2])
+ if r_w[2] > conf_thresh:
+ forearm_vec = r_w[:2] - r_e[:2] if r_e[2] > conf_thresh else np.zeros(2)
+ forearm = np.linalg.norm(forearm_vec)
+ size = max(forearm * wrist_scale, float(bbx_xys[i][2]) * 0.15)
+
+ center = r_w[:2] + (forearm_vec * wrist_shift)
+ cx, cy = center[0], center[1]
+ right_box = np.array([cx - size / 2, cy - size / 2, cx + size / 2, cy + size / 2])
+ left_bboxes.append(left_box)
+ right_bboxes.append(right_box)
+ return left_bboxes, right_bboxes
+
+
+def _hand_pose_to_flat(hand_pose, target_len: int | None):
+ pose = _normalize_smplx_hand_pose(hand_pose, target_len)
+ if isinstance(pose, np.ndarray):
+ pose = torch.from_numpy(pose)
+ if isinstance(pose, torch.Tensor):
+ if pose.ndim == 3:
+ pose = pose.reshape(pose.shape[0], -1)
+ return pose
+ return pose
+
+
+def _normalize_smplx_params(params: dict) -> dict:
+ target_len = _get_smplx_batch_len(params)
+ if target_len is None:
+ return params
+ normalized = {}
+ for k, v in params.items():
+ if k in ("left_hand_pose", "right_hand_pose"):
+ normalized[k] = _normalize_smplx_hand_pose(v, target_len)
+ continue
+ if isinstance(v, np.ndarray):
+ v = torch.from_numpy(v)
+ if isinstance(v, torch.Tensor) and v.ndim >= 1:
+ if v.shape[0] == 1 and target_len > 1:
+ v = v.repeat(target_len, *([1] * (v.ndim - 1)))
+ normalized[k] = v
+ return normalized
+
+
+def _ensure_smplx_pose_defaults(params: dict) -> dict:
+ target_len = _get_smplx_batch_len(params)
+ if target_len is None:
+ return params
+ ref = params.get("body_pose")
+ if ref is None:
+ ref = params.get("global_orient")
+ if ref is None:
+ ref = params.get("betas")
+ if isinstance(ref, np.ndarray):
+ ref = torch.from_numpy(ref)
+ device = ref.device if isinstance(ref, torch.Tensor) else None
+ dtype = ref.dtype if isinstance(ref, torch.Tensor) else None
+ def _zeros_pose():
+ return torch.zeros((target_len, 3), device=device, dtype=dtype)
+ for key in ("jaw_pose", "leye_pose", "reye_pose"):
+ if key not in params or params[key] is None:
+ params[key] = _zeros_pose()
+ if params.get("expression") is None:
+ expr_dim = None
+ betas = params.get("betas")
+ if isinstance(betas, np.ndarray):
+ betas = torch.from_numpy(betas)
+ if isinstance(betas, torch.Tensor) and betas.ndim == 2:
+ expr_dim = betas.shape[1]
+ if expr_dim is None:
+ expr_dim = 10
+ params["expression"] = torch.zeros((target_len, expr_dim), device=device, dtype=dtype)
+ return params
+
+
+def _clamp_int(x: float, lo: int, hi: int) -> int:
+ return int(max(lo, min(hi, int(round(float(x)))))) if hi >= lo else int(round(float(x)))
+
+
+def _face_bbox_from_coco17_kp(
+ kp17_xyc: np.ndarray,
+ frame_w: int,
+ frame_h: int,
+ min_conf: float = 0.3,
+ min_points: int = 2,
+ scale: float = 2.2,
+ min_size_px: int = 32,
+) -> tuple[int, int, int, int] | None:
+ """
+ Estimate a face bbox from COCO17 keypoints (nose/eyes/ears only).
+
+ Returns xyxy in pixel ints or None if face not confidently visible.
+
+ Visibility criteria (to prevent hallucinated keypoints when back is turned):
+ - Nose + at least one eye must be visible, OR
+ - Both eyes must be visible
+ - If ears are much more confident than nose/eyes, face is likely turned away
+ """
+ if kp17_xyc is None:
+ return None
+ kp = np.asarray(kp17_xyc)
+ if kp.shape[0] < 5 or kp.shape[1] < 3:
+ return None
+
+ # COCO17: 0 nose, 1 l_eye, 2 r_eye, 3 l_ear, 4 r_ear
+ # Extract confidences for visibility check
+ nose_conf = float(kp[0, 2])
+ leye_conf = float(kp[1, 2])
+ reye_conf = float(kp[2, 2])
+ lear_conf = float(kp[3, 2])
+ rear_conf = float(kp[4, 2])
+
+ # Check if keypoint is valid (in bounds and confident enough)
+ def is_valid(j: int) -> bool:
+ x, y, c = float(kp[j, 0]), float(kp[j, 1]), float(kp[j, 2])
+ return c >= float(min_conf) and 0 <= x < frame_w and 0 <= y < frame_h
+
+ nose_visible = is_valid(0)
+ leye_visible = is_valid(1)
+ reye_visible = is_valid(2)
+ lear_visible = is_valid(3)
+ rear_visible = is_valid(4)
+
+ # Strong visibility check: require frontal face evidence
+ # Case 1: Nose + at least one eye visible (front or side profile)
+ # Case 2: Both eyes visible (face is front-facing)
+ has_frontal_face = (nose_visible and (leye_visible or reye_visible)) or (leye_visible and reye_visible)
+
+ if not has_frontal_face:
+ return None
+
+ # Back-facing check: if ears have much higher confidence than frontal keypoints,
+ # the pose estimator is likely hallucinating based on body structure, not actual face
+ frontal_max_conf = max(nose_conf, leye_conf, reye_conf)
+ ear_max_conf = max(lear_conf, rear_conf)
+
+ # If ears are confident but frontal keypoints are barely above threshold,
+ # and ears are significantly more confident (relative ratio check)
+ if ear_max_conf > 0.5 and frontal_max_conf < 0.5:
+ # Ears are clearly visible but frontal face is marginal = back/side facing
+ return None
+
+ if ear_max_conf > frontal_max_conf * 1.5 and frontal_max_conf < 0.6:
+ # Ears much more confident than frontal = likely back facing
+ return None
+
+ # Collect valid face points for bbox
+ face_idx = [0, 1, 2, 3, 4]
+ pts = []
+ for j in face_idx:
+ if is_valid(j):
+ x, y = float(kp[j, 0]), float(kp[j, 1])
+ pts.append((x, y))
+
+ if len(pts) < int(min_points):
+ return None
+
+ xs = [p[0] for p in pts]
+ ys = [p[1] for p in pts]
+ x1, y1, x2, y2 = min(xs), min(ys), max(xs), max(ys)
+
+ cx, cy = (x1 + x2) * 0.5, (y1 + y2) * 0.5
+ # Make it more square - use max of w/h as base, then scale
+ base_size = max(x2 - x1, y2 - y1, 1.0)
+ w = base_size * float(scale)
+ h = base_size * float(scale) * 1.15 # Taller to include forehead and chin
+ w = max(w, float(min_size_px))
+ h = max(h, float(min_size_px))
+
+ # Shift center down slightly to include more chin
+ cy = cy + h * 0.05
+ x1 = _clamp_int(cx - w * 0.5, 0, frame_w - 1)
+ y1 = _clamp_int(cy - h * 0.5, 0, frame_h - 1)
+ x2 = _clamp_int(cx + w * 0.5, 0, frame_w - 1)
+ y2 = _clamp_int(cy + h * 0.5, 0, frame_h - 1)
+ if x2 <= x1 or y2 <= y1:
+ return None
+ return x1, y1, x2, y2
+
+
+class _EmotionClassifier:
+ """
+ Optional HF Transformers facial expression classifier.
+ Uses local cache by default to avoid network fetches.
+ """
+
+ def __init__(
+ self,
+ model_id_or_path: str,
+ cache_dir: str | None = None,
+ local_files_only: bool = True,
+ device: str | None = None,
+ ) -> None:
+ self.available = False
+ self.error = None
+ self.device = device
+ self.model_id_or_path = model_id_or_path
+ self.is_clip = model_id_or_path.lower().startswith("clip:")
+
+ if self.device in (None, "auto"):
+ self.device = "cuda" if torch.cuda.is_available() else "cpu"
+
+ if self.is_clip:
+ self._init_clip(model_id_or_path[5:], cache_dir, local_files_only)
+ else:
+ self._init_classifier(model_id_or_path, cache_dir, local_files_only)
+
+ def _init_classifier(self, model_id: str, cache_dir: str | None, local_files_only: bool):
+ try:
+ from transformers import AutoImageProcessor, AutoModelForImageClassification
+ except Exception as e:
+ self.error = f"transformers import failed: {e}"
+ return
+
+ try:
+ self.processor = AutoImageProcessor.from_pretrained(
+ model_id,
+ cache_dir=cache_dir,
+ local_files_only=bool(local_files_only),
+ )
+ self.model = AutoModelForImageClassification.from_pretrained(
+ model_id,
+ cache_dir=cache_dir,
+ local_files_only=bool(local_files_only),
+ ).to(self.device)
+ self.model.eval()
+ self.available = True
+ except Exception as e:
+ self.error = str(e)
+
+ def _init_clip(self, model_id: str, cache_dir: str | None, local_files_only: bool):
+ try:
+ from transformers import CLIPProcessor, CLIPModel
+ except Exception as e:
+ self.error = f"CLIP import failed: {e}"
+ return
+
+ try:
+ self.processor = CLIPProcessor.from_pretrained(
+ model_id,
+ cache_dir=cache_dir,
+ local_files_only=bool(local_files_only),
+ )
+ self.model = CLIPModel.from_pretrained(
+ model_id,
+ cache_dir=cache_dir,
+ local_files_only=bool(local_files_only),
+ use_safetensors=True,
+ ).to(self.device)
+ self.model.eval()
+ # Emotion prompts for zero-shot
+ self.emotions = ["happy", "sad", "angry", "fear", "surprise", "disgust", "neutral"]
+ self.prompts = [f"a face showing {e} emotion" for e in self.emotions]
+ self.available = True
+ except Exception as e:
+ self.error = str(e)
+
+ @torch.no_grad()
+ def predict_top1(self, face_bgr: np.ndarray) -> tuple[str, float] | None:
+ if not self.available:
+ return None
+ if face_bgr is None or face_bgr.size == 0:
+ return None
+ rgb = cv2.cvtColor(face_bgr, cv2.COLOR_BGR2RGB)
+
+ if self.is_clip:
+ inputs = self.processor(text=self.prompts, images=rgb, return_tensors="pt", padding=True)
+ inputs = {k: v.to(self.device) for k, v in inputs.items()}
+ outputs = self.model(**inputs)
+ logits = outputs.logits_per_image[0]
+ probs = torch.softmax(logits, dim=0)
+ score, idx = torch.max(probs, dim=0)
+ label = self.emotions[int(idx)]
+ return str(label), float(score.detach().cpu().item())
+ else:
+ inputs = self.processor(images=rgb, return_tensors="pt")
+ inputs = {k: v.to(self.device) for k, v in inputs.items()}
+ outputs = self.model(**inputs)
+ logits = outputs.logits[0]
+ probs = torch.softmax(logits, dim=0)
+ score, idx = torch.max(probs, dim=0)
+ label = self.model.config.id2label.get(int(idx), str(int(idx)))
+ return str(label), float(score.detach().cpu().item())
+
+
+def _load_emotion_cache(path: str | Path) -> list[dict]:
+ path = Path(path)
+ if not path.exists():
+ return []
+ if path.suffix == ".pt":
+ return torch.load(path, map_location="cpu")
+
+ # Fallback to jsonl
+ out: list[dict] = []
+ with path.open("r", encoding="utf-8") as f:
+ for line in f:
+ line = line.strip()
+ if not line:
+ continue
+ try:
+ out.append(json.loads(line))
+ except Exception:
+ continue
+ return out
+
+
+def _compute_or_load_emotions(
+ video_path: str,
+ kp2d: np.ndarray,
+ out_path: str | Path,
+ *,
+ enabled: bool,
+ model_id: str,
+ cache_dir: str | None,
+ local_files_only: bool,
+ min_kpt_conf: float,
+ min_visible_kpts: int,
+ face_bbox_scale: float,
+) -> list[dict] | None:
+ out_path = Path(out_path)
+ if not enabled:
+ return None
+ if out_path.exists() and out_path.stat().st_size > 0:
+ return _load_emotion_cache(out_path)
+
+ reader = get_video_reader(video_path)
+ length, width, height = get_video_lwh(video_path)
+ length = min(int(length), int(len(kp2d)))
+
+ clf = _EmotionClassifier(
+ model_id_or_path=model_id,
+ cache_dir=cache_dir,
+ local_files_only=local_files_only,
+ device="auto",
+ )
+ if not clf.available:
+ Log.warning(
+ "[Emotion] Classifier unavailable; writing visibility-only labels. "
+ f"Reason: {clf.error}"
+ )
+ else:
+ Log.info(f"[Emotion] Loaded model: {model_id} (device={clf.device})")
+
+ out_path.parent.mkdir(parents=True, exist_ok=True)
+ results: list[dict] = []
+
+ # Run inference
+ for i, frame in tqdm(enumerate(reader), total=length, desc="[Emotion] Face crop + classify", unit="frame"):
+ if i >= length:
+ break
+ bbox = _face_bbox_from_coco17_kp(
+ kp17_xyc=np.asarray(kp2d[i]),
+ frame_w=int(width),
+ frame_h=int(height),
+ min_conf=float(min_kpt_conf),
+ min_points=int(min_visible_kpts),
+ scale=float(face_bbox_scale),
+ )
+ entry: dict = {
+ "frame": int(i),
+ "visible": bool(bbox is not None),
+ "bbox_xyxy": None,
+ "label": None,
+ "score": None,
+ }
+ if bbox is not None:
+ x1, y1, x2, y2 = bbox
+ face = frame[y1:y2, x1:x2]
+ pred = clf.predict_top1(face) if clf.available else None
+ entry["bbox_xyxy"] = [int(x1), int(y1), int(x2), int(y2)]
+ if pred is not None:
+ entry["label"], entry["score"] = pred[0], float(pred[1])
+ else:
+ entry["label"] = "unknown"
+ entry["score"] = None
+ results.append(entry)
+
+ # Save as .pt
+ torch.save(results, out_path)
+ Log.info(f"[Emotion] Saved cache to {out_path}")
+
+ return results
+
+
+def _dedupe_stream_logging_handlers():
+ root = logging.getLogger()
+ stream_handlers = [h for h in root.handlers if isinstance(h, logging.StreamHandler)]
+ if len(stream_handlers) <= 1:
+ return
+
+ seen_streams = set()
+ for h in list(root.handlers):
+ if not isinstance(h, logging.StreamHandler):
+ continue
+ stream = getattr(h, "stream", None)
+ if stream in seen_streams:
+ root.removeHandler(h)
+ else:
+ seen_streams.add(stream)
+
+
+
+
+
+
+
+
+def _convert_dpvo_to_matrix(traj_7d):
+ """
+ Converts DPVO (N, 7) output [x, y, z, qx, qy, qz, qw] (Camera-to-World)
+ into T_w2c (N, 4, 4) (World-to-Camera Matrix).
+ """
+ if isinstance(traj_7d, np.ndarray):
+ traj_7d = torch.from_numpy(traj_7d).float()
+
+ # Extract Translation (C2W)
+ t_c2w = traj_7d[:, :3] # (N, 3)
+
+ # Extract Quaternion (C2W) -> DPVO usually [qx, qy, qz, qw]
+ qw, qx, qy, qz = traj_7d[:, 6], traj_7d[:, 3], traj_7d[:, 4], traj_7d[:, 5]
+
+ # Build Rotation Matrix R_c2w from Quaternions
+ N = traj_7d.shape[0]
+ R_c2w = torch.zeros((N, 3, 3), dtype=traj_7d.dtype, device=traj_7d.device)
+
+ # Diagonal
+ R_c2w[:, 0, 0] = 1 - 2*(qy**2 + qz**2)
+ R_c2w[:, 1, 1] = 1 - 2*(qx**2 + qz**2)
+ R_c2w[:, 2, 2] = 1 - 2*(qx**2 + qy**2)
+
+ # Off-diagonal
+ R_c2w[:, 0, 1] = 2*(qx*qy - qz*qw)
+ R_c2w[:, 0, 2] = 2*(qx*qz + qy*qw)
+ R_c2w[:, 1, 0] = 2*(qx*qy + qz*qw)
+ R_c2w[:, 1, 2] = 2*(qy*qz - qx*qw)
+ R_c2w[:, 2, 0] = 2*(qx*qz - qy*qw)
+ R_c2w[:, 2, 1] = 2*(qy*qz + qx*qw)
+
+ # Convert C2W -> W2C
+ # R_w2c = R_c2w^T
+ # t_w2c = -R_c2w^T * t_c2w
+ R_w2c = R_c2w.transpose(1, 2)
+ t_w2c = -torch.bmm(R_w2c, t_c2w.unsqueeze(-1)).squeeze(-1)
+
+ # Build 4x4
+ T_w2c = torch.eye(4).unsqueeze(0).repeat(N, 1, 1).to(traj_7d.device)
+ T_w2c[:, :3, :3] = R_w2c
+ T_w2c[:, :3, 3] = t_w2c
+
+ return T_w2c
+
+
+def _load_or_compute_camera_motion(
+ cfg: DictConfig,
+ video_path: str,
+ length: int,
+ width: int,
+ height: int,
+ slam_path: str | None = None,
+ masks: list = None,
+):
+ if cfg.static_cam:
+ R_w2c = torch.eye(3)[None].repeat(length, 1, 1)
+ cam_angvel = compute_cam_angvel(R_w2c)
+ cam_tvel = torch.zeros(length, 3)
+ T_w2c = torch.eye(4)[None].repeat(length, 1, 1)
+ gt_T_w2c = torch.eye(4)[None].repeat(length, 1, 1)
+ return R_w2c, cam_angvel, cam_tvel, T_w2c, gt_T_w2c
+
+ slam_path = Path(slam_path or "")
+ if not str(slam_path).strip() or str(slam_path) in {".", "./"}:
+ raise RuntimeError("Non-static camera requested but `paths.slam` is not set.")
+
+ # Clean memory before SLAM
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ if not slam_path.exists():
+ Log.info(f"[Preprocess] Estimating camera motion (saving to {slam_path})")
+ if masks is not None:
+ Log.info(f"[Preprocess] Using {sum(1 for m in masks if m is not None)} SAM masks for masked SLAM")
+ slam_path.parent.mkdir(parents=True, exist_ok=True)
+ from third_party.GVHMR.hmr4d.utils.preproc.slam import SLAMModel
+
+
+ f_mm = getattr(cfg, "f_mm", None)
+ if f_mm is not None:
+ K_fullimg = create_camera_sensor(width, height, f_mm)[2]
+ else:
+ K_fullimg = estimate_K(width, height)
+
+ if isinstance(K_fullimg, torch.Tensor):
+ K_fullimg = K_fullimg.cpu().numpy()
+
+ # Respect SLAM resize: scale intrinsics if frames are resized before SLAM.
+ slam_resize = float(getattr(cfg, "slam_resize", 1.0))
+ fx, fy, cx, cy = K_fullimg[0, 0], K_fullimg[1, 1], K_fullimg[0, 2], K_fullimg[1, 2]
+ fx *= slam_resize
+ fy *= slam_resize
+ cx *= slam_resize
+ cy *= slam_resize
+ intrinsics = torch.tensor([fx, fy, cx, cy])
+
+ # Run SLAM in inference mode + Progress Bar
+ with torch.inference_mode():
+ # Pass masks to SLAMModel for masked camera tracking
+ slam = SLAMModel(video_path, width, height, intrinsics, buffer=4000, resize=slam_resize, masks=masks)
+
+ # Progress Bar for SLAM
+ pbar = tqdm(total=length, desc="[DPVO] Tracking Camera (masked)" if masks else "[DPVO] Tracking Camera", unit="frame")
+
+ try:
+ while True:
+ if not slam.track():
+ break
+ pbar.update(1)
+
+ traj = slam.process()
+ torch.save(traj, slam_path)
+ finally:
+ pbar.close()
+ # Explicit cleanup
+ del slam
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ # Load to CPU
+ traj = torch.load(slam_path, map_location='cpu')
+ if isinstance(traj, np.ndarray):
+ traj = torch.from_numpy(traj).float()
+
+ # Convert DPVO (N, 7) to Matrix (N, 4, 4) if needed
+ if traj.dim() == 2 and traj.shape[1] == 7:
+ Log.info("[Preprocess] Converting DPVO trajectory (N,7) to T_w2c Matrix (N,4,4)")
+ traj = _convert_dpvo_to_matrix(traj)
+
+ traj = traj[:length]
+ if len(traj) < length:
+ last = traj[-1:]
+ padding = last.repeat(length - len(traj), 1, 1)
+ traj = torch.cat([traj, padding], dim=0)
+
+ T_w2c = traj
+ R_w2c = T_w2c[:, :3, :3].clone()
+ gt_T_w2c = T_w2c.clone()
+ t = T_w2c[:, :3, 3].clone()
+
+ cam_tvel = torch.zeros_like(t)
+ cam_tvel[1:] = t[1:] - t[:-1]
+ cam_angvel = compute_cam_angvel(R_w2c)
+
+ return R_w2c, cam_angvel, cam_tvel, T_w2c, gt_T_w2c
+
+
+def _resample_to_30fps(in_path: str, out_path: str):
+ stream = ffmpeg.input(in_path)
+ stream = stream.filter("fps", fps=30)
+ out = ffmpeg.output(stream, out_path, vcodec="libx264", pix_fmt="yuv420p")
+ ffmpeg.run(out, overwrite_output=True, quiet=True)
+
+
+def _render_incam(
+ video_path: str,
+ pred: dict,
+ out_path: str,
+ crf: int = 23,
+ preprocess_dir: Path = None,
+):
+ """Render incam overlay with mesh, mask outline, keypoints, and bbox."""
+ _, width, height = get_video_lwh(video_path)
+
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ with torch.inference_mode():
+ # Check if hand poses are available from HaMeR
+ has_left_hand = "left_hand_pose" in pred["smpl_params_incam"] and pred["smpl_params_incam"]["left_hand_pose"] is not None
+ has_right_hand = "right_hand_pose" in pred["smpl_params_incam"] and pred["smpl_params_incam"]["right_hand_pose"] is not None
+ has_hands = has_left_hand or has_right_hand
+
+ if has_hands:
+ # Use SMPL-X with axis-angle hands (use_pca=False) for HaMeR output
+ import smplx as smplx_lib
+ smplx = smplx_lib.create(
+ model_path=REPO_ROOT / "third_party/GVHMR/inputs/checkpoints/body_models",
+ model_type="smplx",
+ gender="neutral",
+ use_pca=False, # HaMeR outputs axis-angle (45 components), not PCA
+ flat_hand_mean=True,
+ num_betas=10,
+ ).cuda()
+
+ # Build params with hand poses
+ params_incam = _normalize_smplx_params(pred["smpl_params_incam"])
+ params_incam = _ensure_smplx_pose_defaults(params_incam)
+
+ smplx_out = smplx(**to_cuda(params_incam))
+ verts_smplx = smplx_out.vertices
+ faces = smplx.faces
+ verts_incam = verts_smplx
+ Log.info("[Render] Using SMPL-X with HaMeR hands")
+ else:
+ # No hands - use supermotion model (faster, no hands)
+ smplx = make_smplx("supermotion").cuda()
+ params_incam = {k: v for k, v in pred["smpl_params_incam"].items() if k not in ["left_hand_pose", "right_hand_pose"]}
+ smplx_out = smplx(**to_cuda(params_incam))
+ if isinstance(smplx_out, tuple):
+ verts_smplx = smplx_out[0].vertices if hasattr(smplx_out[0], 'vertices') else smplx_out.vertices
+ else:
+ verts_smplx = smplx_out.vertices
+
+ smplx2smpl_path = REPO_ROOT / "third_party/GVHMR/hmr4d/utils/body_model/smplx2smpl_sparse.pt"
+ if smplx2smpl_path.exists():
+ # Convert to SMPL (smaller mesh, faster render)
+ smplx2smpl = torch.load(smplx2smpl_path).cuda()
+ faces = make_smplx("smpl").faces
+ verts_incam = torch.stack([torch.matmul(smplx2smpl, v) for v in verts_smplx])
+ else:
+ faces = smplx.faces
+ verts_incam = verts_smplx
+ Log.info("[Render] Using SMPL (no hands)")
+
+ renderer = Renderer(width, height, device="cuda", faces=faces, K=pred["K_fullimg"][0], max_faces_per_bin=100000)
+
+ # Load debug data if available
+ bbx_xyxy, bbx_xys, vitpose, masks = None, None, None, None
+ left_hand_bboxes, right_hand_bboxes = None, None
+ emotions = None
+ if preprocess_dir is not None:
+ preprocess_dir = Path(preprocess_dir)
+ if (preprocess_dir / "bbx.pt").exists():
+ bbx_data = torch.load(preprocess_dir / "bbx.pt", map_location="cpu")
+ bbx_xyxy = bbx_data.get("bbx_xyxy")
+ bbx_xys = bbx_data.get("bbx_xys")
+ if bbx_xyxy is not None:
+ bbx_xyxy = bbx_xyxy.numpy()
+ if (preprocess_dir / "vitpose.pt").exists():
+ vp = torch.load(preprocess_dir / "vitpose.pt", map_location="cpu")
+ if isinstance(vp, torch.Tensor):
+ vitpose = vp.numpy()
+ elif isinstance(vp, dict):
+ for k in ["kp2d", "keypoints", "vitpose"]:
+ if k in vp:
+ vitpose = vp[k].numpy() if isinstance(vp[k], torch.Tensor) else vp[k]
+ break
+ if (preprocess_dir / "hands.pt").exists():
+ hand_data = torch.load(preprocess_dir / "hands.pt", map_location="cpu")
+ left_hand_bboxes = hand_data.get("left_bboxes")
+ right_hand_bboxes = hand_data.get("right_bboxes")
+ if (preprocess_dir / "masks.pt").exists():
+ masks = torch.load(preprocess_dir / "masks.pt", map_location="cpu")
+ if (preprocess_dir / "emotion.jsonl").exists():
+ emotions = _load_emotion_cache(preprocess_dir / "emotion.jsonl")
+ elif (preprocess_dir / "emotion.pt").exists():
+ emotions = _load_emotion_cache(preprocess_dir / "emotion.pt")
+
+ out_path = Path(out_path)
+ tmp_path = out_path.with_name(out_path.stem + ".tmp" + out_path.suffix)
+ writer = get_writer(str(tmp_path), fps=30, crf=crf)
+ reader = get_video_reader(video_path)
+
+ total_frames = int(verts_incam.shape[0])
+
+ # COCO17 keypoint names
+ COCO17_NAMES = [
+ "nose", "l_eye", "r_eye", "l_ear", "r_ear",
+ "l_sho", "r_sho", "l_elb", "r_elb", "l_wri", "r_wri",
+ "l_hip", "r_hip", "l_knee", "r_knee", "l_ank", "r_ank"
+ ]
+ HAND_EDGES = [
+ (0, 1), (1, 2), (2, 3), (3, 4), # thumb
+ (0, 5), (5, 6), (6, 7), (7, 8), # index
+ (0, 9), (9, 10), (10, 11), (11, 12), # middle
+ (0, 13), (13, 14), (14, 15), (15, 16), # ring
+ (0, 17), (17, 18), (18, 19), (19, 20), # pinky
+ ]
+
+ try:
+ with torch.inference_mode():
+ for i, frame in tqdm(enumerate(reader), total=total_frames, desc="[Render] Incam Overlay", unit="frame"):
+ if i >= total_frames: break
+
+ # 1. Render mesh
+ img = renderer.render_mesh(verts_incam[i], frame, colors=[176, 100, 244])
+
+ # 2. Draw mask outline (cyan)
+ if masks is not None and i < len(masks) and masks[i] is not None:
+ mask = masks[i]
+ if isinstance(mask, torch.Tensor):
+ mask = mask.numpy()
+ # Resize mask to frame size if needed
+ if mask.shape[0] != height or mask.shape[1] != width:
+ mask = cv2.resize(mask.astype(np.uint8), (width, height), interpolation=cv2.INTER_NEAREST)
+ # Find contours and draw
+ contours, _ = cv2.findContours(mask.astype(np.uint8), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
+ cv2.drawContours(img, contours, -1, (255, 255, 0), 2) # Cyan outline
+
+ # 3. Draw bounding box (green)
+ if bbx_xyxy is not None and i < len(bbx_xyxy):
+ x1, y1, x2, y2 = [int(round(float(v))) for v in bbx_xyxy[i]]
+ cv2.rectangle(img, (x1, y1), (x2, y2), (0, 255, 0), 2)
+
+ # 4. Draw keypoints with names and confidence
+ if vitpose is not None and i < len(vitpose):
+ kp = vitpose[i] # (17, 3) - x, y, conf
+ for j in range(min(len(kp), 17)):
+ x, y = int(round(kp[j, 0])), int(round(kp[j, 1]))
+ conf = kp[j, 2] if kp.shape[-1] >= 3 else 1.0
+
+ if x < 0 or y < 0 or x >= width or y >= height:
+ continue
+
+ # Color: red if confident, gray if low
+ color = (0, 0, 255) if conf > 0.3 else (150, 150, 150)
+ cv2.circle(img, (x, y), 3, color, -1)
+
+ # Label with name and confidence
+ name = COCO17_NAMES[j] if j < len(COCO17_NAMES) else f"j{j}"
+ label = f"{name} {conf:.2f}"
+
+ # Background box for text
+ font = cv2.FONT_HERSHEY_SIMPLEX
+ font_scale = 0.35
+ thickness = 1
+ (tw, th), baseline = cv2.getTextSize(label, font, font_scale, thickness)
+ org = (x + 4, y - 4)
+ cv2.rectangle(img, (org[0], org[1] - th), (org[0] + tw, org[1] + baseline), (255, 255, 255), -1)
+ cv2.putText(img, label, org, font, font_scale, (0, 0, 0), thickness, cv2.LINE_AA)
+
+ # 4b. Draw hand bboxes (HaMeR / RTMPose)
+ if left_hand_bboxes is not None and i < len(left_hand_bboxes):
+ lb = left_hand_bboxes[i]
+ if lb is not None:
+ x1, y1, x2, y2 = [int(round(float(v))) for v in lb]
+ cv2.rectangle(img, (x1, y1), (x2, y2), (0, 165, 255), 2)
+ if right_hand_bboxes is not None and i < len(right_hand_bboxes):
+ rb = right_hand_bboxes[i]
+ if rb is not None:
+ x1, y1, x2, y2 = [int(round(float(v))) for v in rb]
+ cv2.rectangle(img, (x1, y1), (x2, y2), (255, 200, 0), 2)
+
+ # 4d. Emotion label (optional)
+ if emotions is not None and i < len(emotions):
+ e = emotions[i]
+ if e.get("visible") and e.get("bbox_xyxy") is not None:
+ x1, y1, x2, y2 = [int(v) for v in e["bbox_xyxy"]]
+ cv2.rectangle(img, (x1, y1), (x2, y2), (255, 0, 255), 2)
+ label = e.get("label") or "unknown"
+ score = e.get("score")
+ text = f"{label}" + (f" {float(score):.2f}" if score is not None else "")
+ cv2.putText(
+ img,
+ text,
+ (x1, max(20, y1 - 8)),
+ cv2.FONT_HERSHEY_SIMPLEX,
+ 0.7,
+ (255, 0, 255),
+ 2,
+ cv2.LINE_AA,
+ )
+
+ # 5. Frame counter
+ cv2.putText(img, f"t={i}", (10, 25), cv2.FONT_HERSHEY_SIMPLEX, 0.7, (255, 255, 255), 2, cv2.LINE_AA)
+
+ writer.write_frame(img)
+ finally:
+ writer.close()
+ reader.close()
+ tmp_path.replace(out_path)
+
+
+def _render_global(
+ video_path: str,
+ pred: dict,
+ out_path: str,
+ crf: int = 23,
+ cam_T_w2c: torch.Tensor | None = None,
+ cam_gizmo_every: int = 10,
+):
+ _, width, height = get_video_lwh(video_path)
+
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ with torch.inference_mode():
+ # Check if hand poses are available from HaMeR
+ has_left_hand = "left_hand_pose" in pred["smpl_params_global"] and pred["smpl_params_global"]["left_hand_pose"] is not None
+ has_right_hand = "right_hand_pose" in pred["smpl_params_global"] and pred["smpl_params_global"]["right_hand_pose"] is not None
+ has_hands = has_left_hand or has_right_hand
+
+ if has_hands:
+ # Use SMPL-X with axis-angle hands (use_pca=False) for HaMeR output
+ import smplx as smplx_lib
+ smplx = smplx_lib.create(
+ model_path=REPO_ROOT / "third_party/GVHMR/inputs/checkpoints/body_models",
+ model_type="smplx",
+ gender="neutral",
+ use_pca=False, # HaMeR outputs axis-angle (45 components), not PCA
+ flat_hand_mean=True,
+ num_betas=10,
+ ).cuda()
+
+ # Build params for incam (for joint estimation)
+ params_incam = _normalize_smplx_params(pred["smpl_params_incam"])
+ params_incam = _ensure_smplx_pose_defaults(params_incam)
+ smplx_incam_out = smplx(**to_cuda(params_incam))
+ joints_incam = smplx_incam_out.joints
+
+ # Build params for global
+ params_global = _normalize_smplx_params(pred["smpl_params_global"])
+ params_global = _ensure_smplx_pose_defaults(params_global)
+
+ smplx_out = smplx(**to_cuda(params_global))
+ verts, joints = smplx_out.vertices, smplx_out.joints
+ faces = smplx.faces
+ Log.info("[Render] Using SMPL-X with HaMeR hands (global)")
+ else:
+ # No hands - use supermotion model (faster)
+ smplx_incam = make_smplx("supermotion").cuda()
+ params_incam = {k: v for k, v in pred["smpl_params_incam"].items() if k not in ["left_hand_pose", "right_hand_pose"]}
+ smplx_incam_out = smplx_incam(**to_cuda(params_incam))
+ joints_incam = smplx_incam_out.joints
+
+ smplx = make_smplx("supermotion").cuda()
+ params_global = {k: v for k, v in pred["smpl_params_global"].items() if k not in ["left_hand_pose", "right_hand_pose"]}
+ smplx_out = smplx(**to_cuda(params_global))
+ verts, joints = smplx_out.vertices, smplx_out.joints
+ faces = smplx.faces
+
+ smplx2smpl_path = REPO_ROOT / "third_party/GVHMR/hmr4d/utils/body_model/smplx2smpl_sparse.pt"
+ if smplx2smpl_path.exists():
+ # Convert to SMPL (smaller mesh, faster render)
+ smplx2smpl = torch.load(smplx2smpl_path).cuda()
+ verts = torch.stack([torch.matmul(smplx2smpl, v) for v in verts])
+ faces = make_smplx("smpl").faces
+ Log.info("[Render] Using SMPL (no hands) (global)")
+
+ # Alignment
+ verts = verts.clone()
+ j_reg_path = REPO_ROOT / "third_party/GVHMR/hmr4d/utils/body_model/smpl_neutral_J_regressor.pt"
+ if j_reg_path.exists() and verts.shape[1] == 6890:
+ J_regressor = torch.load(j_reg_path).to(verts.device)
+ root0 = torch.matmul(J_regressor, verts[0])[0]
+ offset = root0.clone()
+ offset[1] = verts[..., 1].min()
+ verts = verts - offset
+ joints0 = torch.matmul(J_regressor, verts[0])
+ T_ay2ayfz = compute_T_ayfz2ay(joints0[None], inverse=True)
+ verts = apply_T_on_points(verts, T_ay2ayfz)
+ joints = torch.einsum("jv,lvi->lji", J_regressor, verts)
+ else:
+ if j_reg_path.exists() and verts.shape[1] != 6890:
+ Log.warning("[Render] Skipping SMPL J_regressor for non-SMPL verts")
+ joints = joints.clone()
+ offset = joints[0, 0].clone()
+ offset[1] = verts[..., 1].min()
+ verts = verts - offset
+ joints = joints - offset
+ T_ay2ayfz = compute_T_ayfz2ay(joints[[0]], inverse=True)
+ verts = apply_T_on_points(verts, T_ay2ayfz)
+ joints = apply_T_on_points(joints, T_ay2ayfz)
+
+ global_R, global_T, global_lights = get_global_cameras_static(
+ verts.detach().cpu(), beta=2.0, cam_height_degree=20, target_center_height=1.0
+ )
+
+ _, _, K = create_camera_sensor(width, height, 24)
+ renderer = Renderer(width, height, device="cuda", faces=faces, K=K, max_faces_per_bin=100000)
+ scale, cx, cz = get_ground_params_from_points(joints[:, 0].detach().cpu(), verts.detach().cpu())
+ renderer.set_ground(scale * 1.5, cx, cz)
+ color = torch.ones(3, device="cuda") * 0.8
+
+ # ---- Camera gizmo (estimated from incam -> global alignment) ----
+ def _estimate_T_c2w_from_joints(joints_c: torch.Tensor, joints_w: torch.Tensor) -> torch.Tensor:
+ joints_c = joints_c.to(dtype=torch.float32)
+ joints_w = joints_w.to(dtype=torch.float32)
+ mu_c = joints_c.mean(dim=1, keepdim=True)
+ mu_w = joints_w.mean(dim=1, keepdim=True)
+ Xc = joints_c - mu_c
+ Xw = joints_w - mu_w
+ H = torch.einsum("bni,bnj->bij", Xc, Xw)
+ U, _, Vt = torch.linalg.svd(H)
+ R = Vt.transpose(1, 2) @ U.transpose(1, 2)
+ det = torch.det(R)
+ Vt[det < 0, 2, :] *= -1
+ R = Vt.transpose(1, 2) @ U.transpose(1, 2)
+ t = (mu_w - torch.einsum("bij,bnj->bni", R, mu_c)).squeeze(1)
+ T = torch.eye(4, device=R.device, dtype=R.dtype).unsqueeze(0).repeat(R.shape[0], 1, 1)
+ T[:, :3, :3] = R
+ T[:, :3, 3] = t
+ return T
+
+ def _smooth_camera_pose(R_in: torch.Tensor, t_in: torch.Tensor, alpha: float = 0.2):
+ R_out = torch.zeros_like(R_in)
+ t_out = torch.zeros_like(t_in)
+ R_out[0] = R_in[0]
+ t_out[0] = t_in[0]
+ for i in range(1, R_in.shape[0]):
+ R_blend = (1.0 - alpha) * R_out[i - 1] + alpha * R_in[i]
+ U, _, Vt = torch.linalg.svd(R_blend)
+ R_i = U @ Vt
+ if torch.det(R_i) < 0:
+ U[:, -1] *= -1
+ R_i = U @ Vt
+ R_out[i] = R_i
+ t_out[i] = (1.0 - alpha) * t_out[i - 1] + alpha * t_in[i]
+ return R_out, t_out
+
+ joints_incam = joints_incam[: joints.shape[0]]
+ joints_global = joints[: joints_incam.shape[0]]
+ num_joints = min(joints_incam.shape[1], joints_global.shape[1])
+ joints_incam = joints_incam[:, :num_joints]
+ joints_global = joints_global[:, :num_joints]
+ stable_idx = [0, 1, 2, 12, 13, 14, 16, 17]
+ stable_idx = [i for i in stable_idx if i < num_joints]
+ joints_incam = joints_incam[:, stable_idx]
+ joints_global = joints_global[:, stable_idx]
+ T_c2w = _estimate_T_c2w_from_joints(joints_incam, joints_global)
+ R_c2w, t_c2w = _smooth_camera_pose(T_c2w[:, :3, :3], T_c2w[:, :3, 3], alpha=0.2)
+ R_w2c = R_c2w.transpose(1, 2)
+ t_w2c = -torch.einsum("bij,bj->bi", R_w2c, t_c2w)
+ cam_T_w2c = torch.eye(4, device=R_w2c.device, dtype=R_w2c.dtype).unsqueeze(0).repeat(R_w2c.shape[0], 1, 1)
+ cam_T_w2c[:, :3, :3] = R_w2c
+ cam_T_w2c[:, :3, 3] = t_w2c
+
+ out_path = Path(out_path)
+ tmp_path = out_path.with_name(out_path.stem + ".tmp" + out_path.suffix)
+ writer = get_writer(str(tmp_path), fps=30, crf=crf)
+
+ total_frames = int(verts.shape[0])
+
+ try:
+ with torch.inference_mode():
+ for i in tqdm(range(total_frames), desc="[Render] Global View", unit="frame"):
+ cameras = renderer.create_camera(global_R[i], global_T[i])
+ img = renderer.render_with_ground(verts[[i]], color[None], cameras, global_lights)
+ writer.write_frame(img)
+ finally:
+ writer.close()
+ tmp_path.replace(out_path)
+
+
+def _kabsch_align(src: torch.Tensor, dst: torch.Tensor):
+ """Rigid alignment (no scale). src/dst: (J, 3). Returns R,t s.t. src @ R + t = dst."""
+ src_mean = src.mean(dim=0)
+ dst_mean = dst.mean(dim=0)
+ X = src - src_mean
+ Y = dst - dst_mean
+ H = X.t() @ Y
+ U, _, Vt = torch.linalg.svd(H)
+ R = Vt.t() @ U.t()
+ if torch.det(R) < 0:
+ Vt[-1, :] *= -1
+ R = Vt.t() @ U.t()
+ t = dst_mean - src_mean @ R
+ return R, t
+
+
+def save_output_npz(
+ output_dir: Path,
+ video_name: str,
+ params: dict,
+ label: str,
+ camera_transform: np.ndarray | None = None,
+ K_fullimg: np.ndarray | None = None,
+):
+ """
+ Saves SMPL-X output in AMASS NPZ format with camera transform.
+ """
+ output_npz_path = Path(output_dir) / f"{video_name}_{label}_smplx.npz"
+ Log.info(f"[Export NPZ] Saving SMPL-X output to: {output_npz_path}")
+
+ device = params["global_orient"].device
+
+ root_orient = params["global_orient"]
+ if root_orient.ndim == 4:
+ root_orient = root_orient.squeeze(1)
+ if root_orient.ndim == 3 and root_orient.shape[-1] == 3:
+ root_orient = matrix_to_axis_angle(root_orient)
+ elif root_orient.ndim == 2 and root_orient.shape[-1] == 3:
+ pass
+
+ L = root_orient.shape[0]
+ root_orient = root_orient.reshape(L, 3)
+
+ body_pose = params["body_pose"]
+ if body_pose.ndim == 4:
+ body_pose = matrix_to_axis_angle(body_pose)
+ body_pose = body_pose.reshape(L, -1) # (L, 63) for 21 body joints
+
+ # SMPL-X full pose layout (165 total):
+ # - root_orient: 3
+ # - body_pose: 63 (21 joints * 3)
+ # - jaw_pose: 3
+ # - leye_pose: 3
+ # - reye_pose: 3
+ # - left_hand_pose: 45 (15 joints * 3)
+ # - right_hand_pose: 45 (15 joints * 3)
+
+ # Get hand poses if available
+ left_hand_pose = params.get("left_hand_pose", None)
+ right_hand_pose = params.get("right_hand_pose", None)
+
+ if left_hand_pose is not None:
+ if isinstance(left_hand_pose, np.ndarray):
+ left_hand_pose = torch.from_numpy(left_hand_pose).to(device)
+ left_hand_pose = left_hand_pose.reshape(L, -1) # (L, 45)
+ else:
+ left_hand_pose = torch.zeros(L, 45, device=device)
+
+ if right_hand_pose is not None:
+ if isinstance(right_hand_pose, np.ndarray):
+ right_hand_pose = torch.from_numpy(right_hand_pose).to(device)
+ right_hand_pose = right_hand_pose.reshape(L, -1) # (L, 45)
+ else:
+ right_hand_pose = torch.zeros(L, 45, device=device)
+
+ # Face poses (jaw, eyes) - zeros for now
+ jaw_pose = torch.zeros(L, 3, device=device)
+ leye_pose = torch.zeros(L, 3, device=device)
+ reye_pose = torch.zeros(L, 3, device=device)
+
+ # Combine into full_pose
+ full_pose = torch.cat([
+ root_orient, # 3
+ body_pose, # 63
+ jaw_pose, # 3
+ leye_pose, # 3
+ reye_pose, # 3
+ left_hand_pose, # 45
+ right_hand_pose, # 45
+ ], dim=-1) # Total: 165
+
+ out_dict = {
+ "mocap_framerate": 30,
+ "gender": "neutral",
+ "betas": params["betas"][0, :10].detach().cpu().numpy(),
+ "trans": params["transl"].detach().cpu().numpy(),
+ "poses": full_pose.detach().cpu().numpy(),
+ }
+ if camera_transform is not None:
+ out_dict["camera_transform"] = camera_transform
+ if K_fullimg is not None:
+ out_dict["K_fullimg"] = K_fullimg
+
+ with open(output_npz_path, "wb") as f:
+ np.savez(f, **out_dict)
+ Log.info("[Export NPZ] Saved successfully.")
+
+
+@hydra.main(version_base="1.3", config_path="../../configs", config_name="infer_video")
+def main(cfg: DictConfig):
+ _dedupe_stream_logging_handlers()
+
+ if cfg.video_path is None: raise ValueError("Missing `video_path`.")
+ if cfg.ckpt_path is None: raise ValueError("Missing `ckpt_path`.")
+
+ video_name = cfg.video_name or Path(cfg.video_path).stem
+ output_dir = Path(cfg.output_root) / video_name
+ preprocess_dir = output_dir / "preprocess"
+ output_dir.mkdir(parents=True, exist_ok=True)
+ preprocess_dir.mkdir(parents=True, exist_ok=True)
+
+ paths = {
+ "input_video": str(output_dir / "0_input_video.mp4"),
+ "video_30fps": str(output_dir / "0_input_video_30fps.mp4"),
+ "bbx": str(preprocess_dir / "bbx.pt"),
+ "vitpose": str(preprocess_dir / "vitpose.pt"),
+ "vit_features": str(preprocess_dir / "vit_features.pt"),
+ "hmr4d_results": str(output_dir / "hmr4d_results.pt"),
+ "incam_video": str(output_dir / "1_incam.mp4"),
+ "global_video": str(output_dir / "2_global.mp4"),
+ "incam_global_horiz_video": str(output_dir / "3_incam_global_horiz.mp4"),
+ "slam": str(preprocess_dir / "slam.pt"),
+ }
+
+ Log.info(f"[Output Dir]: {output_dir}")
+ copy_file(cfg.video_path, paths["input_video"], overwrite=False)
+ video_path = paths["input_video"]
+
+ if cfg.resample_to_30fps:
+ if not Path(paths["video_30fps"]).exists():
+ Log.info("[Video] Resampling to 30 FPS")
+ _resample_to_30fps(video_path, paths["video_30fps"])
+ video_path = paths["video_30fps"]
+
+ length, width, height = get_video_lwh(video_path)
+ Log.info(f"[Video] (L,W,H)=({length},{width},{height})")
+
+ f_mm = getattr(cfg, "f_mm", None)
+ K_est = create_camera_sensor(width, height, f_mm)[2] if f_mm else estimate_K(width, height)
+ K = K_est.clone().detach().float() if isinstance(K_est, torch.Tensor) else torch.tensor(K_est).float()
+
+ # ===== Preprocess =====
+ masks_path = str(preprocess_dir / "masks.pt")
+
+ if not Path(paths["bbx"]).exists():
+ Log.info("[Preprocess] Tracking bbox with SAM masks")
+ with torch.inference_mode():
+ tracker = Tracker()
+ bbx_xyxy, masks = tracker.get_one_track_with_masks(video_path)
+ bbx_xyxy = bbx_xyxy.float()
+ bbx_xys = get_bbx_xys_from_xyxy(bbx_xyxy, base_enlarge=1.2).float()
+ torch.save({"bbx_xyxy": bbx_xyxy, "bbx_xys": bbx_xys}, paths["bbx"])
+ torch.save(masks, masks_path) # Save masks separately
+ del tracker
+ gc.collect()
+ else:
+ # Load masks if they exist
+ masks = torch.load(masks_path) if Path(masks_path).exists() else None
+
+ # Check if SAM masking is enabled (default: True)
+ use_sam_masking = cfg.get("use_sam_masking", True)
+
+ if not Path(paths["vitpose"]).exists():
+ Log.info("[Preprocess] VitPose" + (" (with SAM masking)" if use_sam_masking else ""))
+ bbx_xys = torch.load(paths["bbx"])["bbx_xys"]
+ # Load masks for VitPose if enabled
+ masks = None
+ if use_sam_masking and Path(masks_path).exists():
+ masks = torch.load(masks_path)
+ Log.info(f"[Preprocess] Applying {sum(1 for m in masks if m is not None)} SAM masks to VitPose")
+ with torch.inference_mode():
+ vitpose_extractor = VitPoseExtractor()
+ kp2d = vitpose_extractor.extract(video_path, bbx_xys, masks=masks)
+ torch.save(kp2d, paths["vitpose"])
+ del vitpose_extractor; del bbx_xys; del masks
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ if not Path(paths["vit_features"]).exists():
+ Log.info("[Preprocess] ViT features" + (" (with SAM masking)" if use_sam_masking else ""))
+ bbx_xys = torch.load(paths["bbx"])["bbx_xys"]
+ # Load masks for ViT features if enabled
+ masks = None
+ if use_sam_masking and Path(masks_path).exists():
+ masks = torch.load(masks_path)
+ Log.info(f"[Preprocess] Applying {sum(1 for m in masks if m is not None)} SAM masks to ViT features")
+ with torch.inference_mode():
+ extractor = Extractor()
+ vit_features = extractor.extract_video_features(video_path, bbx_xys, masks=masks)
+ torch.save(vit_features, paths["vit_features"])
+ del extractor; del bbx_xys; del masks
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ # Optional: Emotion classification (face crop from COCO17 head keypoints)
+ emotion_cfg = cfg.get("emotion", None)
+ if emotion_cfg is not None and bool(getattr(emotion_cfg, "enabled", False)):
+ kp2d_for_emotion = torch.load(paths["vitpose"], map_location="cpu")
+ if isinstance(kp2d_for_emotion, torch.Tensor):
+ kp2d_for_emotion = kp2d_for_emotion.numpy()
+ elif isinstance(kp2d_for_emotion, dict):
+ for k in ["kp2d", "keypoints", "vitpose"]:
+ if k in kp2d_for_emotion:
+ kp2d_for_emotion = kp2d_for_emotion[k]
+ break
+ kp2d_for_emotion = kp2d_for_emotion.numpy() if isinstance(kp2d_for_emotion, torch.Tensor) else kp2d_for_emotion
+
+ cache_dir = getattr(emotion_cfg, "cache_dir", "./third_party/GVHMR/.cache/huggingface")
+ out_path_default = preprocess_dir / "emotion.pt"
+ out_path_cfg = getattr(emotion_cfg, "output_path", None)
+ out_path = Path(out_path_cfg) if out_path_cfg else out_path_default
+
+ _compute_or_load_emotions(
+ video_path=video_path,
+ kp2d=np.asarray(kp2d_for_emotion),
+ out_path=out_path,
+ enabled=True,
+ model_id=str(getattr(emotion_cfg, "model_id", "clip:openai/clip-vit-base-patch32")),
+ cache_dir=str(cache_dir) if cache_dir is not None else None,
+ local_files_only=bool(getattr(emotion_cfg, "local_files_only", True)),
+ min_kpt_conf=float(getattr(emotion_cfg, "min_kpt_conf", 0.3)),
+ min_visible_kpts=int(getattr(emotion_cfg, "min_visible_kpts", 2)),
+ face_bbox_scale=float(getattr(emotion_cfg, "face_bbox_scale", 2.2)),
+ )
+ if out_path != out_path_default and out_path.exists() and not out_path_default.exists():
+ try:
+ copy_file(str(out_path), str(out_path_default), overwrite=False)
+ except Exception:
+ pass
+
+ # SLAM
+ slam_path = cfg.paths.get("slam", paths["slam"]) if hasattr(cfg, "paths") else paths["slam"]
+
+ # Load masks for SLAM if enabled
+ slam_masks = None
+ if use_sam_masking and Path(masks_path).exists():
+ slam_masks = torch.load(masks_path)
+
+ # Internal function now handles inference_mode, GC, matrix conversion AND PROGRESS BAR
+ R_w2c, cam_angvel, cam_tvel, T_w2c, gt_T_w2c = _load_or_compute_camera_motion(
+ cfg, video_path, length, width, height, slam_path=slam_path, masks=slam_masks
+ )
+
+ # === LOAD DATA TO CPU FIRST ===
+ bbx_xys = torch.load(paths["bbx"], map_location="cpu")["bbx_xys"]
+ kp2d = torch.load(paths["vitpose"], map_location="cpu")
+ if isinstance(kp2d, tuple): kp2d = kp2d[0]
+ vit_features = torch.load(paths["vit_features"], map_location="cpu")
+
+ length = min(int(length), int(bbx_xys.shape[0]), int(kp2d.shape[0]), int(vit_features.shape[0]))
+
+ # Slice on CPU
+ bbx_xys = bbx_xys[:length]
+ kp2d = kp2d[:length]
+ vit_features = vit_features[:length]
+ K_fullimg = K[None].repeat(length, 1, 1).cpu()
+
+ R_w2c = R_w2c[:length].cpu()
+ cam_angvel = cam_angvel[:length].cpu()
+ cam_tvel = cam_tvel[:length].cpu()
+ T_w2c = T_w2c[:length].cpu()
+ gt_T_w2c = gt_T_w2c[:length].cpu()
+
+ data = {
+ "meta": [{"vid": video_name}],
+ "caption": "",
+ "has_text": torch.tensor([False]),
+ "length": torch.tensor(length),
+ "bbx_xys": bbx_xys,
+ "K_fullimg": K_fullimg,
+ "f_imgseq": vit_features,
+ "kp2d": kp2d,
+ "cam_angvel": cam_angvel,
+ "cam_tvel": cam_tvel,
+ "R_w2c": R_w2c,
+ "T_w2c": T_w2c,
+ "gt_T_w2c": gt_T_w2c,
+ "mask": {
+ "valid": torch.ones(length),
+ "has_img_mask": torch.ones(length).bool(),
+ "has_2d_mask": torch.ones(length).bool(),
+ "has_cam_mask": torch.ones(length).bool(),
+ "has_audio_mask": torch.zeros(length).bool(),
+ "has_music_mask": torch.zeros(length).bool(),
+ },
+ }
+
+ # ===== GENMO inference =====
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ if Path(paths["hmr4d_results"]).exists():
+ Log.info(f"[GENMO] Using cached results: {paths['hmr4d_results']}")
+ pred = torch.load(paths["hmr4d_results"], weights_only=False)
+ else:
+ Log.info("[GENMO] Loading model")
+ model = hydra.utils.instantiate(cfg.model, _recursive_=False)
+ model.load_pretrained_model(cfg.ckpt_path)
+ model = model.eval().cuda()
+
+ Log.info("[GENMO] Predicting")
+ with torch.inference_mode():
+ pred = model.predict(data, static_cam=cfg.static_cam, postproc=cfg.postproc)
+ pred = detach_to_cpu(pred)
+
+ torch.save(pred, paths["hmr4d_results"])
+ Log.info(f"[GENMO] Saved: {paths['hmr4d_results']}")
+
+ del model
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ if cfg.get("dump_io", False):
+ dump_path = Path(cfg.get("dump_io_path", output_dir / "debug_io.pt"))
+ dump_path.parent.mkdir(parents=True, exist_ok=True)
+ torch.save(detach_to_cpu({"inputs": data, "outputs": pred}), dump_path)
+ Log.info(f"[Debug] Dumped inputs/outputs to {dump_path}")
+
+ # ===== HaMeR Hand Mesh Recovery (Optional) =====
+ if cfg.get("run_hamer", False):
+ hands_cache_path = preprocess_dir / "hands.pt"
+ has_pred_hands = (
+ "left_hand_pose" in pred["smpl_params_global"]
+ and pred["smpl_params_global"]["left_hand_pose"] is not None
+ and "right_hand_pose" in pred["smpl_params_global"]
+ and pred["smpl_params_global"]["right_hand_pose"] is not None
+ )
+ if has_pred_hands:
+ Log.info("[HaMeR] Hand poses already in pred; skipping HaMeR")
+ else:
+ Log.info("[HaMeR] Running hand mesh recovery...")
+ try:
+ from scripts.demo.hamer_inference import HaMeRInference, mano_to_smplx_hands
+
+ cached = False
+ if hands_cache_path.exists():
+ hand_data = torch.load(hands_cache_path, map_location="cpu")
+ left_hand_pose = hand_data.get("left_hand_pose")
+ right_hand_pose = hand_data.get("right_hand_pose")
+ if left_hand_pose is not None and right_hand_pose is not None:
+ L = _get_smplx_batch_len(pred["smpl_params_global"])
+ left_hand_flat = _hand_pose_to_flat(left_hand_pose, L)
+ right_hand_flat = _hand_pose_to_flat(right_hand_pose, L)
+ if isinstance(left_hand_flat, torch.Tensor) and isinstance(right_hand_flat, torch.Tensor):
+ device = pred["smpl_params_global"]["body_pose"].device
+ pred["smpl_params_global"]["left_hand_pose"] = left_hand_flat.to(device)
+ pred["smpl_params_global"]["right_hand_pose"] = right_hand_flat.to(device)
+ pred["smpl_params_incam"]["left_hand_pose"] = left_hand_flat.to(device)
+ pred["smpl_params_incam"]["right_hand_pose"] = right_hand_flat.to(device)
+ cached = True
+ Log.info(f"[HaMeR] Using cached hands: {hands_cache_path}")
+
+ if not cached:
+ # Load masks for HaMeR
+ hamer_masks = None
+ if use_sam_masking and Path(masks_path).exists():
+ hamer_masks = torch.load(masks_path)
+
+ # Load person bboxes
+ bbx_xys = torch.load(paths["bbx"], map_location="cpu")["bbx_xys"]
+
+ # Step 1: Estimate hand bboxes from ViTPose wrists
+ Log.info("[HaMeR] Estimating hand bboxes from ViTPose...")
+ kp2d = torch.load(paths["vitpose"], map_location="cpu")
+ if isinstance(kp2d, tuple):
+ kp2d = kp2d[0]
+ if isinstance(kp2d, torch.Tensor):
+ kp2d = kp2d.numpy()
+ left_bboxes, right_bboxes = _vitpose_to_hand_bboxes(kp2d, bbx_xys)
+
+ # Count valid detections
+ left_valid = sum(1 for b in left_bboxes if b is not None)
+ right_valid = sum(1 for b in right_bboxes if b is not None)
+ Log.info(f"[HaMeR] Found {left_valid} left hands, {right_valid} right hands")
+
+ # Step 2: Run HaMeR on detected hands
+ if left_valid > 0 or right_valid > 0:
+ Log.info("[HaMeR] Running HaMeR inference...")
+ hamer = HaMeRInference()
+ left_results, right_results = hamer.predict_video(
+ video_path, left_bboxes, right_bboxes, masks=hamer_masks
+ )
+ mano_faces = None
+ mano_focal_length = None
+ mano_img_res = None
+ if hasattr(hamer, "_model") and hasattr(hamer._model, "mano"):
+ mano_faces = np.asarray(hamer._model.mano.faces)
+ if hasattr(hamer, "_model_cfg"):
+ mano_focal_length = float(hamer._model_cfg.EXTRA.FOCAL_LENGTH)
+ mano_img_res = int(hamer._model_cfg.MODEL.IMAGE_SIZE)
+ del hamer
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ # Step 3: Convert MANO to SMPL-X hand poses
+ num_frames = len(left_bboxes)
+ left_hand_pose, right_hand_pose = mano_to_smplx_hands(
+ left_results, right_results, num_frames
+ )
+
+ # Step 4: Inject into pred smpl_params_global
+ # SMPL-X expects hand poses as (L, 45) flattened, not (L, 15, 3)
+ Log.info("[HaMeR] Merging hand poses into SMPL-X...")
+ device = pred["smpl_params_global"]["body_pose"].device
+
+ # Flatten from (L, 15, 3) to (L, 45)
+ left_hand_flat = left_hand_pose.reshape(num_frames, -1) # (L, 45)
+ right_hand_flat = right_hand_pose.reshape(num_frames, -1) # (L, 45)
+
+ pred["smpl_params_global"]["left_hand_pose"] = torch.from_numpy(left_hand_flat).to(device)
+ pred["smpl_params_global"]["right_hand_pose"] = torch.from_numpy(right_hand_flat).to(device)
+
+ # Also update incam params
+ pred["smpl_params_incam"]["left_hand_pose"] = torch.from_numpy(left_hand_flat).to(device)
+ pred["smpl_params_incam"]["right_hand_pose"] = torch.from_numpy(right_hand_flat).to(device)
+
+ # Save hand keypoints for debugging
+ hand_data = {
+ "left_bboxes": left_bboxes,
+ "right_bboxes": right_bboxes,
+ "left_hand_pose": left_hand_pose,
+ "right_hand_pose": right_hand_pose,
+ "left_results": left_results,
+ "right_results": right_results,
+ "mano_faces": mano_faces,
+ "mano_focal_length": mano_focal_length,
+ "mano_img_res": mano_img_res,
+ }
+ torch.save(hand_data, hands_cache_path)
+ Log.info(f"[HaMeR] Saved hand data to {hands_cache_path}")
+
+ # Re-save pred with hands
+ torch.save(pred, paths["hmr4d_results"])
+ Log.info("[HaMeR] Updated hmr4d_results.pt with hand poses")
+ else:
+ Log.info("[HaMeR] No hands detected, skipping HaMeR inference")
+
+ except Exception as e:
+ Log.warning(f"[HaMeR] Failed: {e}")
+ import traceback
+ traceback.print_exc()
+
+ # Compute camera_transform via Kabsch alignment (incam -> global)
+ Log.info("[Export NPZ] Computing camera transform via Kabsch alignment...")
+ device = "cuda" if torch.cuda.is_available() else "cpu"
+
+ with torch.inference_mode():
+ smplx = make_smplx("supermotion").to(device)
+ # Note: supermotion model uses PCA hand poses (12 components), not full axis-angle (45)
+ # Remove HaMeR hand poses for Kabsch alignment (they'll be used for rendering later)
+ params_incam = {k: (v.to(device) if isinstance(v, torch.Tensor) else torch.tensor(v, device=device))
+ for k, v in pred["smpl_params_incam"].items() if k not in ["left_hand_pose", "right_hand_pose"]}
+ params_global = {k: (v.to(device) if isinstance(v, torch.Tensor) else torch.tensor(v, device=device))
+ for k, v in pred["smpl_params_global"].items() if k not in ["left_hand_pose", "right_hand_pose"]}
+ joints_incam = smplx(**params_incam).joints
+ joints_global = smplx(**params_global).joints
+
+ # Use all 22 joints for more robust alignment
+ joint_count = min(22, joints_incam.shape[1], joints_global.shape[1])
+ num_frames = joints_incam.shape[0]
+
+ if cfg.static_cam:
+ # STATIC CAMERA: Compute from frame 0 only, freeze for all frames
+ Log.info("[Export NPZ] Static camera mode: using frame-0 alignment for all frames")
+ incam_f0 = joints_incam[0, :joint_count, :]
+ glob_f0 = joints_global[0, :joint_count, :]
+ R0, t0 = _kabsch_align(incam_f0, glob_f0)
+ T0 = torch.eye(4, device=R0.device, dtype=R0.dtype)
+ T0[:3, :3] = R0
+ T0[:3, 3] = t0
+ camera_transform = T0.unsqueeze(0).repeat(num_frames, 1, 1).detach().cpu().numpy()
+ else:
+ # DYNAMIC CAMERA: Compute per-frame + temporal smoothing (EMA)
+ Log.info("[Export NPZ] Dynamic camera mode: per-frame alignment with temporal smoothing")
+ camera_transforms = []
+ for f in range(num_frames):
+ incam_f = joints_incam[f, :joint_count, :]
+ glob_f = joints_global[f, :joint_count, :]
+ Rf, tf = _kabsch_align(incam_f, glob_f)
+ T = torch.eye(4, device=Rf.device, dtype=Rf.dtype)
+ T[:3, :3] = Rf
+ T[:3, 3] = tf
+ camera_transforms.append(T)
+ camera_transforms = torch.stack(camera_transforms)
+
+ # Apply exponential moving average smoothing (alpha=0.1 for heavy smoothing)
+ alpha = 0.1
+ smoothed = torch.zeros_like(camera_transforms)
+ smoothed[0] = camera_transforms[0]
+ for f in range(1, num_frames):
+ # Smooth rotation via SVD re-orthogonalization
+ R_blend = (1.0 - alpha) * smoothed[f-1, :3, :3] + alpha * camera_transforms[f, :3, :3]
+ U, _, Vt = torch.linalg.svd(R_blend)
+ R_smooth = U @ Vt
+ if torch.det(R_smooth) < 0:
+ U[:, -1] *= -1
+ R_smooth = U @ Vt
+ smoothed[f, :3, :3] = R_smooth
+ # Smooth translation
+ smoothed[f, :3, 3] = (1.0 - alpha) * smoothed[f-1, :3, 3] + alpha * camera_transforms[f, :3, 3]
+ smoothed[f, 3, 3] = 1.0
+ camera_transform = smoothed.detach().cpu().numpy()
+
+ # Get K_fullimg
+ K_fullimg_np = K.detach().cpu().numpy() if isinstance(K, torch.Tensor) else K
+
+ save_output_npz(
+ output_dir,
+ video_name,
+ pred["smpl_params_global"],
+ "global",
+ camera_transform=camera_transform,
+ K_fullimg=K_fullimg_np,
+ )
+
+ # ===== Render =====
+ if cfg.render_incam and not _is_valid_video(paths["incam_video"]):
+ _render_incam(video_path, pred, paths["incam_video"], crf=int(cfg.render_crf), preprocess_dir=preprocess_dir)
+
+ if cfg.render_global and not _is_valid_video(paths["global_video"]):
+ _render_global(
+ video_path,
+ pred,
+ paths["global_video"],
+ crf=int(cfg.render_crf),
+ cam_T_w2c=data.get("T_w2c", None),
+ )
+
+ if cfg.render_side_by_side and _is_valid_video(paths["incam_video"]) and _is_valid_video(paths["global_video"]):
+ if not _is_valid_video(paths["incam_global_horiz_video"]):
+ Log.info("[Render] Side-by-side")
+ merge_videos_horizontal([paths["incam_video"], paths["global_video"]], paths["incam_global_horiz_video"])
+
+
+if __name__ == "__main__":
+ from hydra.core.global_hydra import GlobalHydra
+ if GlobalHydra.instance().is_initialized():
+ GlobalHydra.instance().clear()
+ main()
diff --git a/scripts/demo/render_npz_global.py b/scripts/demo/render_npz_global.py
new file mode 100644
index 0000000000000000000000000000000000000000..9ca5b1e7d0e112b061985e317bbd64cf4414462f
--- /dev/null
+++ b/scripts/demo/render_npz_global.py
@@ -0,0 +1,170 @@
+import argparse
+from pathlib import Path
+from typing import Dict, Tuple
+
+import numpy as np
+import torch
+
+from genmo.utils.geo_transform import apply_T_on_points, compute_T_ayfz2ay
+from genmo.utils.video_io_utils import get_writer
+from genmo.utils.vis.renderer import (
+ Renderer,
+ get_global_cameras_static,
+ get_ground_params_from_points,
+)
+from third_party.GVHMR.hmr4d.utils.geo.hmr_cam import create_camera_sensor
+from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+
+
+def _load_motion_npz(npz_path: Path) -> Tuple[np.ndarray, np.ndarray, np.ndarray, float, str]:
+ with np.load(npz_path, allow_pickle=True) as d:
+ poses = np.asarray(d["poses"], dtype=np.float32)
+ trans = np.asarray(d["trans"], dtype=np.float32)
+ betas = np.asarray(d["betas"], dtype=np.float32).reshape(-1)
+ fps = float(np.asarray(d.get("mocap_framerate", 30.0)))
+ gender = str(np.asarray(d.get("gender", "neutral")))
+ if poses.ndim != 2 or poses.shape[1] < 66:
+ raise ValueError(f"Expected poses (F,165) or (F,>=66); got {poses.shape}")
+ if trans.ndim != 2 or trans.shape[1] != 3:
+ raise ValueError(f"Expected trans (F,3); got {trans.shape}")
+ if betas.shape[0] < 10:
+ betas = np.pad(betas, (0, 10 - betas.shape[0]))
+ betas = betas[:10]
+ if trans.shape[0] != poses.shape[0]:
+ raise ValueError(f"poses and trans length mismatch: {poses.shape[0]} vs {trans.shape[0]}")
+ return poses, trans, betas, fps, gender
+
+
+def _split_smplx_poses(poses165: torch.Tensor) -> Dict[str, torch.Tensor]:
+ # SMPL-X pose layout: [global(3), body(63), jaw(3), leye(3), reye(3), lhand(45), rhand(45)] = 165
+ global_orient = poses165[:, 0:3]
+ body_pose = poses165[:, 3:66]
+ extra = poses165[:, 66:]
+ params = {
+ "global_orient": global_orient,
+ "body_pose": body_pose,
+ }
+ if extra.shape[1] >= 99:
+ params.update(
+ {
+ "jaw_pose": extra[:, 0:3],
+ "leye_pose": extra[:, 3:6],
+ "reye_pose": extra[:, 6:9],
+ "left_hand_pose": extra[:, 9:54],
+ "right_hand_pose": extra[:, 54:99],
+ }
+ )
+ return params
+
+
+def _try_smplx_forward(smplx, params: Dict[str, torch.Tensor]) -> torch.Tensor:
+ try:
+ out = smplx(**params)
+ verts = out.vertices if hasattr(out, "vertices") else out[0].vertices
+ return verts
+ except (TypeError, RuntimeError):
+ # Fallback: model variant doesn't take hand/face params (or expects PCA hand pose dims).
+ keep = {k: v for k, v in params.items() if k in {"global_orient", "body_pose", "betas", "transl"}}
+ out = smplx(**keep)
+ verts = out.vertices if hasattr(out, "vertices") else out[0].vertices
+ return verts
+
+
+def main():
+ ap = argparse.ArgumentParser()
+ ap.add_argument("--npz", required=True, type=str)
+ ap.add_argument("--out", required=True, type=str)
+ ap.add_argument("--max_frames", type=int, default=300, help="Max rendered frames (uniformly sampled). Use -1 for all.")
+ ap.add_argument("--size", type=int, default=512)
+ ap.add_argument("--f_mm", type=float, default=24.0)
+ ap.add_argument("--crf", type=int, default=23)
+ args = ap.parse_args()
+
+ npz_path = Path(args.npz)
+ out_path = Path(args.out)
+ out_path.parent.mkdir(parents=True, exist_ok=True)
+
+ poses, trans, betas, fps, gender = _load_motion_npz(npz_path)
+ total = poses.shape[0]
+ if args.max_frames is None or args.max_frames == 0:
+ raise ValueError("--max_frames must be -1 or a positive integer")
+ if args.max_frames < 0 or args.max_frames >= total:
+ idxs = np.arange(total, dtype=np.int64)
+ else:
+ idxs = np.linspace(0, total - 1, int(args.max_frames), dtype=np.int64)
+ # Keep the original FPS even when subsampling frames (avoids tiny fps which can
+ # overflow pyav's rational conversion on some builds).
+ fps_out = fps
+
+ device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
+ smplx = make_smplx("supermotion").to(device).eval()
+
+ poses_t = torch.from_numpy(poses[idxs]).to(device)
+ trans_t = torch.from_numpy(trans[idxs]).to(device)
+ betas_t = torch.from_numpy(betas[None]).to(device).repeat(len(idxs), 1)
+
+ params = _split_smplx_poses(poses_t)
+ params["betas"] = betas_t
+ params["transl"] = trans_t
+
+ with torch.inference_mode():
+ verts_smplx = _try_smplx_forward(smplx, params)
+
+ # Convert to SMPL topology if possible (better matches the regressor + faster render).
+ smplx2smpl_path = Path("third_party/GVHMR/hmr4d/utils/body_model/smplx2smpl_sparse.pt")
+ if smplx2smpl_path.exists():
+ smplx2smpl = torch.load(smplx2smpl_path, map_location=device)
+ verts = torch.stack([torch.matmul(smplx2smpl, v) for v in verts_smplx])
+ faces = make_smplx("smpl", gender="male").faces
+ else:
+ verts = verts_smplx
+ faces = smplx.faces
+
+ # Align like infer_video.py (ground + face-Z)
+ j_reg_path = Path("third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt")
+ J_reg = torch.load(j_reg_path, map_location=device) if j_reg_path.exists() else None
+ if J_reg is not None and verts.shape[1] == J_reg.shape[-1]:
+ root0 = torch.matmul(J_reg, verts[0])[0]
+ offset = root0.clone()
+ else:
+ J_reg = None
+ offset = verts[0].mean(0)
+ offset[1] = verts[..., 1].min()
+ verts = verts - offset
+
+ if J_reg is not None:
+ joints0 = torch.matmul(J_reg, verts[0])[None]
+ T_ay2ayfz = compute_T_ayfz2ay(joints0, inverse=True)
+ verts = apply_T_on_points(verts, T_ay2ayfz)
+
+ size = int(args.size)
+ _, _, K = create_camera_sensor(size, size, float(args.f_mm))
+ renderer = Renderer(size, size, device=device, faces=faces, K=K.to(device), bin_size=0)
+
+ global_R, global_T, global_lights = get_global_cameras_static(
+ verts.detach().cpu(),
+ beta=2.0,
+ cam_height_degree=20,
+ target_center_height=1.0,
+ device=str(device),
+ )
+ if J_reg is not None:
+ roots = torch.einsum("jv,fvi->fji", J_reg, verts)[..., 0, :]
+ else:
+ roots = verts.mean(1)
+ scale, cx, cz = get_ground_params_from_points(roots.detach().cpu(), verts.detach().cpu())
+ renderer.set_ground(scale * 1.5, cx, cz)
+
+ writer = get_writer(str(out_path), fps=float(fps_out), crf=int(args.crf))
+ try:
+ color = torch.tensor([[0.8, 0.2, 0.8]], device=device) # purple-ish
+ for i in range(verts.shape[0]):
+ cameras = renderer.create_camera(global_R[i], global_T[i])
+ img = renderer.render_with_ground(verts[[i]], color, cameras, global_lights)
+ writer.write_frame(img.astype(np.uint8))
+ finally:
+ writer.close()
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/export_npz_from_results.py b/scripts/export_npz_from_results.py
new file mode 100644
index 0000000000000000000000000000000000000000..be8870d1407d4d11db564fa6c49697e112a9c250
--- /dev/null
+++ b/scripts/export_npz_from_results.py
@@ -0,0 +1,168 @@
+import argparse
+import sys
+from pathlib import Path
+from typing import Dict
+
+import cv2
+import numpy as np
+import torch
+
+# Reuse the logging helper from Genmo if available, otherwise fall back.
+try:
+ from genmo.utils.pylogger import Log # type: ignore
+except Exception: # pragma: no cover - optional dependency
+ class _FallbackLog:
+ @staticmethod
+ def info(msg):
+ print(msg)
+ Log = _FallbackLog()
+
+try:
+ from genmo.utils.rotation_conversions import matrix_to_axis_angle # type: ignore
+except Exception:
+ def matrix_to_axis_angle(mat: torch.Tensor) -> torch.Tensor:
+ """Minimal reimplementation for fallback."""
+ return torch.from_numpy(np.stack([cv2.Rodrigues(r.cpu().numpy())[0][:, 0] for r in mat], axis=0))
+
+
+def save_output_npz(
+ output_dir: Path,
+ video_name: str,
+ params: Dict[str, torch.Tensor],
+ label: str,
+ camera_transform: np.ndarray | None = None,
+ K_fullimg: np.ndarray | None = None,
+):
+ """
+ Saves SMPL-X output in the requested NPZ format (matching Genmo demo exporter).
+ """
+ output_npz_path = Path(output_dir) / f"{video_name}_{label}_smplx.npz"
+ Log.info(f"[Export NPZ] Saving SMPL-X output to: {output_npz_path}")
+
+ device = params["global_orient"].device
+
+ root_orient = params["global_orient"]
+ if root_orient.ndim == 4:
+ root_orient = root_orient.squeeze(1)
+ if root_orient.ndim == 3 and root_orient.shape[-1] == 3:
+ root_orient = matrix_to_axis_angle(root_orient)
+ elif root_orient.ndim == 2 and root_orient.shape[-1] == 3:
+ pass
+
+ L = root_orient.shape[0]
+ root_orient = root_orient.reshape(L, 3)
+
+ body_pose = params["body_pose"]
+ if body_pose.ndim == 4:
+ body_pose = matrix_to_axis_angle(body_pose)
+ body_pose = body_pose.reshape(L, -1)
+
+ full_pose = torch.cat([root_orient, body_pose], dim=-1)
+ padding = torch.zeros(L, 99, device=device)
+ full_pose = torch.cat([full_pose, padding], dim=-1)
+
+ out_dict = {
+ "mocap_framerate": 30,
+ "gender": "neutral",
+ "betas": params["betas"][0, :10].detach().cpu().numpy(),
+ "trans": params["transl"].detach().cpu().numpy(),
+ "poses": full_pose.detach().cpu().numpy(),
+ }
+ if camera_transform is not None:
+ out_dict["camera_transform"] = camera_transform
+ if K_fullimg is not None:
+ out_dict["K_fullimg"] = K_fullimg
+
+ with open(output_npz_path, "wb") as f:
+ np.savez(f, **out_dict)
+ Log.info("[Export NPZ] Saved successfully.")
+
+
+def _require_gvhmr():
+ repo_root = Path(__file__).resolve().parents[2]
+ genmo_root = repo_root.parent / "GENMO"
+ gvhmr_root = genmo_root / "third_party" / "GVHMR"
+ if gvhmr_root.exists() and str(genmo_root) not in sys.path:
+ sys.path.insert(0, str(genmo_root))
+ from third_party.GVHMR.hmr4d.utils.smplx_utils import make_smplx
+ return make_smplx
+
+
+def _kabsch_align(src: torch.Tensor, dst: torch.Tensor):
+ """Rigid alignment (no scale). src/dst: (J, 3). Returns R,t s.t. src @ R + t = dst."""
+ src_mean = src.mean(dim=0)
+ dst_mean = dst.mean(dim=0)
+ X = src - src_mean
+ Y = dst - dst_mean
+ H = X.t() @ Y
+ U, _, Vt = torch.linalg.svd(H)
+ R = Vt.t() @ U.t()
+ if torch.det(R) < 0:
+ Vt[-1, :] *= -1
+ R = Vt.t() @ U.t()
+ t = dst_mean - src_mean @ R
+ return R, t
+
+
+def _apply_rt(points: torch.Tensor, R: torch.Tensor, t: torch.Tensor) -> torch.Tensor:
+ return points @ R + t
+
+
+def main():
+ parser = argparse.ArgumentParser(
+ description="Export GENMO global SMPL-X NPZ with camera transform derived from incam alignment."
+ )
+ parser.add_argument("--genmo-results", type=Path, required=True, help="Path to Genmo hmr4d_results.pt")
+ parser.add_argument("--output-dir", type=Path, required=True, help="Directory to write NPZ file")
+ parser.add_argument("--video-name", type=str, required=True, help="Base video name for outputs")
+ parser.add_argument("--joint-count", type=int, default=22, help="Number of joints to use for alignment")
+ args = parser.parse_args()
+
+ pred = torch.load(args.genmo_results, map_location="cpu", weights_only=False)
+ if "smpl_params_incam" not in pred or "smpl_params_global" not in pred:
+ raise ValueError("hmr4d_results.pt missing smpl_params_incam or smpl_params_global")
+
+ make_smplx = _require_gvhmr()
+ device = "cuda" if torch.cuda.is_available() else "cpu"
+ smplx = make_smplx("supermotion").to(device)
+
+ def _to_device(params: Dict[str, torch.Tensor]):
+ return {k: (v.to(device) if isinstance(v, torch.Tensor) else torch.tensor(v, device=device)) for k, v in params.items()}
+
+ params_incam = _to_device(pred["smpl_params_incam"])
+ params_global = _to_device(pred["smpl_params_global"])
+
+ with torch.inference_mode():
+ joints_incam = smplx(**params_incam).joints
+ joints_global = smplx(**params_global).joints
+
+ j = min(args.joint_count, joints_incam.shape[1], joints_global.shape[1])
+ camera_transforms = []
+ for f in range(joints_incam.shape[0]):
+ incam_f = joints_incam[f, :j, :]
+ glob_f = joints_global[f, :j, :]
+ Rf, tf = _kabsch_align(incam_f, glob_f)
+ T = torch.eye(4, device=Rf.device, dtype=Rf.dtype)
+ T[:3, :3] = Rf
+ T[:3, 3] = tf
+ camera_transforms.append(T)
+ camera_transform = torch.stack(camera_transforms).detach().cpu().numpy()
+
+ args.output_dir.mkdir(parents=True, exist_ok=True)
+ K_fullimg = pred.get("K_fullimg")
+
+ if isinstance(K_fullimg, torch.Tensor):
+ K_fullimg = K_fullimg.detach().cpu().numpy()
+
+ save_output_npz(
+ args.output_dir,
+ args.video_name,
+ params_global,
+ "global",
+ camera_transform=camera_transform,
+ K_fullimg=K_fullimg,
+ )
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/fix_genmo_features_world.py b/scripts/fix_genmo_features_world.py
new file mode 100644
index 0000000000000000000000000000000000000000..57e09f9957ce4b94871ba6ae507c70bb75f45317
--- /dev/null
+++ b/scripts/fix_genmo_features_world.py
@@ -0,0 +1,95 @@
+#!/usr/bin/env python3
+"""
+Flip Unity genmo_features .pt files by rotating existing smpl_params_w only.
+This keeps the world/camera untouched and avoids re-running the pipeline.
+"""
+from __future__ import annotations
+
+import argparse
+import math
+import sys
+from pathlib import Path
+
+import torch
+
+REPO_ROOT = Path(__file__).resolve().parents[1]
+if str(REPO_ROOT) not in sys.path:
+ sys.path.insert(0, str(REPO_ROOT))
+
+from genmo.utils.rotation_conversions import axis_angle_to_matrix, matrix_to_axis_angle
+
+
+def _squeeze_batch(x: torch.Tensor) -> torch.Tensor:
+ if x.ndim >= 3 and x.shape[0] == 1:
+ return x[0]
+ return x
+
+
+def _axis_angle_from_axis(axis: str, degrees: float) -> torch.Tensor:
+ axis = axis.lower().strip()
+ if axis == "x":
+ vec = torch.tensor([1.0, 0.0, 0.0], dtype=torch.float32)
+ elif axis == "y":
+ vec = torch.tensor([0.0, 1.0, 0.0], dtype=torch.float32)
+ elif axis == "z":
+ vec = torch.tensor([0.0, 0.0, 1.0], dtype=torch.float32)
+ else:
+ raise ValueError(f"Unsupported axis: {axis}")
+ return vec * float(math.radians(float(degrees)))
+
+
+def _fix_one(path: Path, out_dir: Path | None) -> None:
+ data = torch.load(path, map_location="cpu", weights_only=False)
+
+ if "smpl_params_w" not in data:
+ print(f"[skip] Missing smpl_params_w in {path.name}")
+ return
+
+ smpl_w = data["smpl_params_w"]
+ global_orient_w = _squeeze_batch(smpl_w["global_orient"]).float()
+ transl_w = _squeeze_batch(smpl_w["transl"]).float()
+
+ if global_orient_w.ndim != 2 or transl_w.ndim != 2:
+ print(f"[skip] Unexpected shapes in {path.name}")
+ return
+
+ # Hardcoded SMPLX-only flip to fix upside-down global orientation.
+ aa_fix = _axis_angle_from_axis("x", 180.0).to(global_orient_w)
+ R_fix = axis_angle_to_matrix(aa_fix[None])[0]
+ R_w = torch.matmul(R_fix[None], axis_angle_to_matrix(global_orient_w))
+ global_orient_w = matrix_to_axis_angle(R_w)
+ transl_w = torch.einsum("ij,fj->fi", R_fix, transl_w)
+
+ smpl_w["global_orient"] = global_orient_w
+ smpl_w["transl"] = transl_w
+ data["smpl_params_w"] = smpl_w
+
+ out_path = (out_dir / path.name) if out_dir else path
+ if out_dir:
+ out_dir.mkdir(parents=True, exist_ok=True)
+ torch.save(data, out_path)
+ print(f"[ok] {path.name}")
+
+
+def main() -> None:
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--features-dir", required=True, help="Path to genmo_features directory")
+ parser.add_argument("--output-dir", default=None, help="Optional output dir (defaults to in-place)")
+ args = parser.parse_args()
+
+ features_dir = Path(args.features_dir)
+ if not features_dir.is_dir():
+ raise FileNotFoundError(f"features dir not found: {features_dir}")
+
+ out_dir = Path(args.output_dir) if args.output_dir else None
+ pt_files = sorted(features_dir.glob("*.pt"))
+ if not pt_files:
+ print(f"No .pt files found in {features_dir}")
+ return
+
+ for path in pt_files:
+ _fix_one(path, out_dir)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/label_videos.py b/scripts/label_videos.py
new file mode 100644
index 0000000000000000000000000000000000000000..9460b9e24fb0da4c86e82d689c08ca35cfaa3604
--- /dev/null
+++ b/scripts/label_videos.py
@@ -0,0 +1,1308 @@
+#!/usr/bin/env python3
+"""
+Video Labeling Pipeline for GENMO Training Data
+
+Automatically labels video footage to identify clips suitable for GENMO motion capture:
+- Single person in frame (no multi-person scenes)
+- Person consistently visible
+- Filters out false positives (posters, stickers) via motion analysis
+
+Usage:
+ python label_videos.py --video path/to/video.mp4 --output labels.json
+ python label_videos.py --video-dir path/to/videos/ --output labels.json
+"""
+
+import os
+import sys
+import json
+import argparse
+import time
+import cv2
+import torch
+import numpy as np
+from PIL import Image
+from tqdm import tqdm
+from dataclasses import dataclass, asdict
+from typing import List, Dict, Optional, Tuple, Iterator, Callable
+from collections import defaultdict
+
+# Add GVHMR to path for imports
+sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "third_party", "GVHMR"))
+
+from transformers import AutoProcessor, AutoModelForZeroShotObjectDetection
+
+
+@dataclass
+class Detection:
+ """Single detection in a frame."""
+ bbox_xyxy: List[float] # [x1, y1, x2, y2]
+ confidence: float
+ area_pct: float # bbox area as percentage of frame
+
+
+@dataclass
+class Segment:
+ """A continuous segment of video with labeling info."""
+ start_sec: float
+ end_sec: float
+ dynamic_persons: int
+ static_detections: int
+ avg_confidence: float
+ avg_bbox_area_pct: float
+ bbox_variance: float
+ usable: bool
+ reason: Optional[str] = None
+
+
+@dataclass
+class TrackState:
+ """Streaming track state with running variance."""
+ last_bbox: List[float]
+ last_ts: float
+ count: int
+ mean_center: np.ndarray
+ m2_center: np.ndarray
+ mean_size: np.ndarray
+ m2_size: np.ndarray
+
+
+class VitPoseValidator:
+ """Validate that a bbox contains a complete person using ViTPose joints."""
+
+ HEAD_KP = {0, 1, 2, 3, 4} # nose, eyes, ears - any one visible = head visible
+ EXCLUDE_KP = {0, 1, 2, 3, 4} # nose, eyes, ears (excluded from body joint count)
+ UPPER_KP = {5, 6, 7, 8, 9, 10} # shoulders, elbows, wrists
+ LOWER_KP = {11, 12, 13, 14, 15, 16} # hips, knees, ankles
+
+ def __init__(
+ self,
+ config_path: str,
+ ckpt_path: str,
+ device: str,
+ min_joints: int,
+ conf_threshold: float,
+ require_upper_lower: bool,
+ min_vertical_span: float,
+ require_head: bool = True
+ ):
+ try:
+ from mmpose.apis import init_model, inference_topdown
+ except Exception as exc:
+ raise RuntimeError(f"mmpose not available: {exc}") from exc
+
+ if not os.path.exists(config_path):
+ raise RuntimeError(f"ViTPose config not found: {config_path}")
+ if not os.path.exists(ckpt_path):
+ raise RuntimeError(f"ViTPose checkpoint not found: {ckpt_path}")
+
+ self._inference_topdown = inference_topdown
+ self.pose = init_model(config_path, ckpt_path, device=device)
+ self.pose.eval()
+ self.min_joints = int(min_joints)
+ self.conf_threshold = float(conf_threshold)
+ self.require_upper_lower = bool(require_upper_lower)
+ self.min_vertical_span = float(min_vertical_span)
+ self.require_head = bool(require_head)
+
+ @torch.no_grad()
+ def is_complete(self, frame_rgb: np.ndarray, bbox_xyxy: List[float]) -> bool:
+ frame_bgr = cv2.cvtColor(frame_rgb, cv2.COLOR_RGB2BGR)
+ x1, y1, x2, y2 = bbox_xyxy
+ bbox = np.array([[x1, y1, x2, y2]], dtype=np.float32)
+
+ results = self._inference_topdown(self.pose, frame_bgr, bboxes=bbox)
+ if not results:
+ return False
+
+ pred = results[0].pred_instances
+ scores = np.asarray(pred.keypoint_scores[0]).reshape(-1)
+ keypoints = np.asarray(pred.keypoints[0]).reshape(-1, 2)
+ count = 0
+ upper_count = 0
+ lower_count = 0
+ head_visible = False
+ ys = []
+
+ for idx, score in enumerate(scores):
+ # Check head visibility (nose, eyes, or ears)
+ if idx in self.HEAD_KP and float(score) >= self.conf_threshold:
+ head_visible = True
+ if idx in self.EXCLUDE_KP:
+ continue
+ if float(score) >= self.conf_threshold:
+ count += 1
+ if idx in self.UPPER_KP:
+ upper_count += 1
+ if idx in self.LOWER_KP:
+ lower_count += 1
+ ys.append(float(keypoints[idx][1]))
+
+ if self.require_head and not head_visible:
+ return False # Head not visible
+ if count < self.min_joints:
+ return False
+ if self.require_upper_lower and (upper_count == 0 or lower_count == 0):
+ return False
+ if self.min_vertical_span > 0.0 and len(ys) >= 2:
+ span = max(ys) - min(ys)
+ bbox_h = max(1.0, float(y2) - float(y1))
+ if (span / bbox_h) < self.min_vertical_span:
+ return False
+
+ return True
+
+
+class VideoLabeler:
+ """Labels videos for GENMO training suitability."""
+
+ # Thresholds
+ STATIC_VARIANCE_THRESHOLD = 50.0 # px² - below this = static object
+ MIN_CONFIDENCE = 0.4
+ MIN_BBOX_AREA_PCT = 0.01 # 1% of frame
+ MAX_BBOX_JUMP_RATIO = 0.5 # max center movement as ratio of bbox size
+ MIN_SEGMENT_DURATION = 10.0 # seconds
+ DUPLICATE_OVERLAP_THRESHOLD = 0.1
+ DUPLICATE_IOU_THRESHOLD = 0.2
+ DUPLICATE_CENTER_RATIO = 0.75
+ DUPLICATE_CENTER_ONLY_RATIO = 0.35
+ DUPLICATE_AREA_RATIO = 3.0
+ MULTI_PERSON_MIN_AREA_PCT = 0.08
+ MULTI_PERSON_REL_AREA = 0.35
+ LOW_CONF_SMOOTH_MAX_SEC = 2.0
+
+ def __init__(
+ self,
+ sample_fps: float = 1.0,
+ debug_dir: Optional[str] = None,
+ debug_all: bool = False,
+ vitpose_filter: bool = True, # Enabled by default to filter out animals/false positives
+ vitpose_filter_all: bool = True, # Validate all detections, not just multi-person frames
+ vitpose_min_joints: int = 4,
+ vitpose_conf_threshold: float = 0.3,
+ vitpose_require_upper_lower: bool = True,
+ vitpose_min_vertical_span: float = 0.35,
+ vitpose_config: Optional[str] = None,
+ vitpose_ckpt: Optional[str] = None
+ ):
+ self.sample_fps = sample_fps
+ self.debug_dir = debug_dir
+ self.debug_all = debug_all
+ self.vitpose_filter = vitpose_filter
+ self.vitpose_filter_all = vitpose_filter_all
+ self.vitpose_min_joints = vitpose_min_joints
+ self.vitpose_conf_threshold = vitpose_conf_threshold
+ self.vitpose_require_upper_lower = vitpose_require_upper_lower
+ self.vitpose_min_vertical_span = vitpose_min_vertical_span
+ self.vitpose_config = vitpose_config
+ self.vitpose_ckpt = vitpose_ckpt
+ self.device = "cuda" if torch.cuda.is_available() else "cpu"
+ self.vitpose_validator = None
+
+ # Precision
+ if torch.cuda.is_available() and torch.cuda.get_device_properties(0).major >= 8:
+ self.dtype = torch.bfloat16
+ else:
+ self.dtype = torch.float16
+
+ print(f"[Labeler] Device: {self.device}, Precision: {self.dtype}")
+
+ # Initialize Grounding DINO
+ self._init_dino()
+ self._init_vitpose()
+
+ def _init_vitpose(self):
+ """Initialize ViTPose for validation if enabled."""
+ if not self.vitpose_filter:
+ return
+
+ config_path = self.vitpose_config or os.path.join(
+ os.path.dirname(__file__),
+ "..",
+ "third_party",
+ "GVHMR",
+ "mmpose",
+ "configs",
+ "body_2d_keypoint",
+ "topdown_heatmap",
+ "coco",
+ "vitpose_huge_finetune.py"
+ )
+ ckpt_path = self.vitpose_ckpt or os.path.join(
+ os.path.dirname(__file__),
+ "..",
+ "third_party",
+ "GVHMR",
+ "work_dirs",
+ "best_coco_AP_epoch_1.pth"
+ )
+
+ try:
+ self.vitpose_validator = VitPoseValidator(
+ config_path=config_path,
+ ckpt_path=ckpt_path,
+ device=self.device,
+ min_joints=self.vitpose_min_joints,
+ conf_threshold=self.vitpose_conf_threshold,
+ require_upper_lower=self.vitpose_require_upper_lower,
+ min_vertical_span=self.vitpose_min_vertical_span
+ )
+ print("[Labeler] ViTPose validation enabled")
+ except Exception as exc:
+ print(f"[Labeler] ViTPose validation disabled: {exc}")
+ self.vitpose_validator = None
+
+ def _init_dino(self):
+ """Initialize Grounding DINO model."""
+ model_id = "IDEA-Research/grounding-dino-tiny"
+ cache_dir = os.path.abspath(
+ os.path.join(os.path.dirname(__file__), "..", "third_party", "GVHMR", ".cache", "huggingface")
+ )
+ os.makedirs(cache_dir, exist_ok=True)
+
+ try:
+ self.processor = AutoProcessor.from_pretrained(
+ model_id, local_files_only=True, cache_dir=cache_dir
+ )
+ self.model = AutoModelForZeroShotObjectDetection.from_pretrained(
+ model_id, local_files_only=True, cache_dir=cache_dir
+ ).to(self.device)
+ print("[Labeler] Loaded Grounding DINO from cache")
+ except Exception:
+ print("[Labeler] Downloading Grounding DINO...")
+ self.processor = AutoProcessor.from_pretrained(model_id, cache_dir=cache_dir)
+ self.model = AutoModelForZeroShotObjectDetection.from_pretrained(
+ model_id, cache_dir=cache_dir
+ ).to(self.device)
+
+ self.text_prompt = "person."
+ self.box_threshold = 0.35 # Raised from 0.25 to reduce false positives (animals, etc.)
+ self.text_threshold = 0.3
+
+ def _iter_sampled_frames(
+ self,
+ video_path: str,
+ end_time: Optional[float] = None
+ ) -> Tuple[Tuple[int, int], Optional[float], Iterator[Tuple[np.ndarray, float]]]:
+ """Stream sampled frames at target FPS without loading all frames into memory."""
+ import subprocess
+
+ probe_cmd = [
+ 'ffprobe', '-v', 'error',
+ '-select_streams', 'v:0',
+ '-show_entries', 'stream=width,height,duration',
+ '-of', 'csv=p=0',
+ video_path
+ ]
+ try:
+ result = subprocess.run(probe_cmd, capture_output=True, text=True, check=True)
+ parts = result.stdout.strip().split(',')
+ width = int(parts[0])
+ height = int(parts[1])
+ duration = float(parts[2]) if len(parts) > 2 and parts[2] else None
+ except Exception as e:
+ print(f"[Labeler] ffprobe failed: {e}, falling back to OpenCV for metadata")
+ cap = cv2.VideoCapture(video_path)
+ width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
+ height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
+ fps = cap.get(cv2.CAP_PROP_FPS) or 30.0
+ duration = cap.get(cv2.CAP_PROP_FRAME_COUNT) / fps if fps > 0 else None
+ cap.release()
+
+ if duration and end_time is not None:
+ duration = min(duration, end_time)
+
+ ffmpeg_cmd = [
+ 'ffmpeg', '-v', 'warning', '-nostdin',
+ '-i', video_path,
+ '-vf', f'fps={self.sample_fps}',
+ '-f', 'rawvideo',
+ '-pix_fmt', 'rgb24',
+ ]
+ if end_time is not None:
+ ffmpeg_cmd += ['-t', str(end_time)]
+ ffmpeg_cmd.append('pipe:1')
+
+ print(f"[Labeler] Streaming frames at {self.sample_fps} fps using ffmpeg...")
+ process = subprocess.Popen(
+ ffmpeg_cmd,
+ stdout=subprocess.PIPE,
+ stderr=subprocess.DEVNULL
+ )
+
+ frame_size = width * height * 3
+
+ def iterator():
+ idx = 0
+ try:
+ while True:
+ raw = process.stdout.read(frame_size)
+ if raw is None or len(raw) < frame_size:
+ break
+ frame = np.frombuffer(raw, np.uint8).reshape((height, width, 3))
+ ts = idx / self.sample_fps
+ idx += 1
+ yield frame, ts
+ finally:
+ if process.stdout:
+ process.stdout.close()
+ process.wait()
+
+ return (width, height), duration, iterator()
+
+ def _nms(self, detections: List[Detection], iou_threshold: float = 0.5) -> List[Detection]:
+ """Apply Non-Maximum Suppression to filter overlapping/contained detections."""
+ if len(detections) <= 1:
+ return detections
+
+ # Sort by confidence (highest first)
+ sorted_dets = sorted(detections, key=lambda d: d.confidence, reverse=True)
+
+ def compute_iou(box1, box2):
+ x1 = max(box1[0], box2[0])
+ y1 = max(box1[1], box2[1])
+ x2 = min(box1[2], box2[2])
+ y2 = min(box1[3], box2[3])
+ inter_area = max(0, x2 - x1) * max(0, y2 - y1)
+ box1_area = (box1[2] - box1[0]) * (box1[3] - box1[1])
+ box2_area = (box2[2] - box2[0]) * (box2[3] - box2[1])
+ union_area = box1_area + box2_area - inter_area
+ return inter_area / union_area if union_area > 0 else 0
+
+ def compute_overlap_small(box1, box2):
+ """Intersection over smaller area; higher when one box sits on the same person."""
+ x1 = max(box1[0], box2[0])
+ y1 = max(box1[1], box2[1])
+ x2 = min(box1[2], box2[2])
+ y2 = min(box1[3], box2[3])
+ inter_area = max(0, x2 - x1) * max(0, y2 - y1)
+ box1_area = (box1[2] - box1[0]) * (box1[3] - box1[1])
+ box2_area = (box2[2] - box2[0]) * (box2[3] - box2[1])
+ small_area = min(box1_area, box2_area)
+ return inter_area / small_area if small_area > 0 else 0
+
+ def box_diag(box):
+ width = max(0, box[2] - box[0])
+ height = max(0, box[3] - box[1])
+ return np.sqrt(width * width + height * height)
+
+ def is_contained(box_small, box_large, threshold=0.7):
+ """Check if box_small is mostly contained within box_large."""
+ x1 = max(box_small[0], box_large[0])
+ y1 = max(box_small[1], box_large[1])
+ x2 = min(box_small[2], box_large[2])
+ y2 = min(box_small[3], box_large[3])
+ inter_area = max(0, x2 - x1) * max(0, y2 - y1)
+ small_area = (box_small[2] - box_small[0]) * (box_small[3] - box_small[1])
+ if small_area <= 0:
+ return False
+ return (inter_area / small_area) >= threshold
+
+ def is_near_duplicate(box1, box2, overlap_threshold=0.3, center_ratio=0.5):
+ """Suppress boxes that likely describe the same person with weak IoU."""
+ overlap_small = compute_overlap_small(box1, box2)
+ if overlap_small < overlap_threshold:
+ return False
+ c1 = ((box1[0] + box1[2]) / 2, (box1[1] + box1[3]) / 2)
+ c2 = ((box2[0] + box2[2]) / 2, (box2[1] + box2[3]) / 2)
+ dist = np.sqrt((c1[0] - c2[0])**2 + (c1[1] - c2[1])**2)
+ max_diag = max(box_diag(box1), box_diag(box2))
+ return dist <= (center_ratio * max_diag)
+
+ keep = []
+ while sorted_dets:
+ best = sorted_dets.pop(0)
+ keep.append(best)
+ # Remove detections that overlap OR are contained within the best detection
+ sorted_dets = [d for d in sorted_dets
+ if compute_iou(best.bbox_xyxy, d.bbox_xyxy) < iou_threshold
+ and not is_contained(d.bbox_xyxy, best.bbox_xyxy)
+ and not is_near_duplicate(best.bbox_xyxy, d.bbox_xyxy)]
+
+ return keep
+
+ def _dedupe_nearby(
+ self,
+ detections: List[Detection],
+ overlap_threshold: Optional[float] = None,
+ iou_threshold: Optional[float] = None,
+ center_ratio: Optional[float] = None,
+ center_only_ratio: Optional[float] = None,
+ area_ratio: Optional[float] = None
+ ) -> List[Detection]:
+ """Merge nearby detections that likely describe the same person."""
+ if len(detections) <= 1:
+ return detections
+
+ overlap_threshold = self.DUPLICATE_OVERLAP_THRESHOLD if overlap_threshold is None else overlap_threshold
+ iou_threshold = self.DUPLICATE_IOU_THRESHOLD if iou_threshold is None else iou_threshold
+ center_ratio = self.DUPLICATE_CENTER_RATIO if center_ratio is None else center_ratio
+ center_only_ratio = 0.0 if center_only_ratio is None else center_only_ratio
+ area_ratio = self.DUPLICATE_AREA_RATIO if area_ratio is None else area_ratio
+
+ def compute_iou(box1, box2):
+ x1 = max(box1[0], box2[0])
+ y1 = max(box1[1], box2[1])
+ x2 = min(box1[2], box2[2])
+ y2 = min(box1[3], box2[3])
+ inter_area = max(0, x2 - x1) * max(0, y2 - y1)
+ box1_area = (box1[2] - box1[0]) * (box1[3] - box1[1])
+ box2_area = (box2[2] - box2[0]) * (box2[3] - box2[1])
+ union_area = box1_area + box2_area - inter_area
+ return inter_area / union_area if union_area > 0 else 0
+
+ def compute_overlap_small(box1, box2):
+ x1 = max(box1[0], box2[0])
+ y1 = max(box1[1], box2[1])
+ x2 = min(box1[2], box2[2])
+ y2 = min(box1[3], box2[3])
+ inter_area = max(0, x2 - x1) * max(0, y2 - y1)
+ box1_area = (box1[2] - box1[0]) * (box1[3] - box1[1])
+ box2_area = (box2[2] - box2[0]) * (box2[3] - box2[1])
+ small_area = min(box1_area, box2_area)
+ return inter_area / small_area if small_area > 0 else 0
+
+ def box_diag(box):
+ width = max(0, box[2] - box[0])
+ height = max(0, box[3] - box[1])
+ return np.sqrt(width * width + height * height)
+
+ def center_distance(box1, box2):
+ c1 = ((box1[0] + box1[2]) / 2, (box1[1] + box1[3]) / 2)
+ c2 = ((box2[0] + box2[2]) / 2, (box2[1] + box2[3]) / 2)
+ return np.sqrt((c1[0] - c2[0])**2 + (c1[1] - c2[1])**2)
+
+ n = len(detections)
+ parent = list(range(n))
+
+ def find(x):
+ while parent[x] != x:
+ parent[x] = parent[parent[x]]
+ x = parent[x]
+ return x
+
+ def union(a, b):
+ ra, rb = find(a), find(b)
+ if ra != rb:
+ parent[rb] = ra
+
+ def should_merge(box1, box2):
+ iou = compute_iou(box1, box2)
+ overlap_small = compute_overlap_small(box1, box2)
+ max_diag = max(box_diag(box1), box_diag(box2))
+ if max_diag <= 0:
+ return False
+ dist = center_distance(box1, box2)
+ if iou >= iou_threshold or overlap_small >= overlap_threshold:
+ return dist <= (center_ratio * max_diag)
+ if center_only_ratio > 0.0:
+ area1 = max(0.0, (box1[2] - box1[0]) * (box1[3] - box1[1]))
+ area2 = max(0.0, (box2[2] - box2[0]) * (box2[3] - box2[1]))
+ if area1 <= 0 or area2 <= 0:
+ return False
+ ratio = max(area1, area2) / min(area1, area2)
+ if ratio <= area_ratio:
+ return dist <= (center_only_ratio * max_diag)
+ return False
+
+ for i in range(n):
+ box_i = detections[i].bbox_xyxy
+ for j in range(i + 1, n):
+ box_j = detections[j].bbox_xyxy
+ if should_merge(box_i, box_j):
+ union(i, j)
+
+ best_by_root = {}
+ for idx, det in enumerate(detections):
+ root = find(idx)
+ if root not in best_by_root or det.confidence > best_by_root[root].confidence:
+ best_by_root[root] = det
+
+ return list(best_by_root.values())
+
+ def _detect_frame(self, frame: np.ndarray, width: int, height: int) -> List[Detection]:
+ """Run DINO detection on a single frame."""
+ frame_area = width * height
+ img = Image.fromarray(frame)
+
+ with torch.inference_mode():
+ inputs = self.processor(
+ images=img,
+ text=self.text_prompt,
+ return_tensors="pt"
+ ).to(self.device)
+
+ outputs = self.model(**inputs)
+
+ results = self.processor.post_process_grounded_object_detection(
+ outputs,
+ inputs.input_ids,
+ threshold=self.box_threshold,
+ text_threshold=self.text_threshold,
+ target_sizes=[img.size[::-1]] # (height, width)
+ )
+
+ frame_dets = []
+ if len(results) > 0 and 'boxes' in results[0]:
+ boxes = results[0]['boxes'].cpu().numpy()
+ scores = results[0]['scores'].cpu().numpy()
+
+ for box, score in zip(boxes, scores):
+ x1, y1, x2, y2 = box
+ area = (x2 - x1) * (y2 - y1)
+ area_pct = area / frame_area
+
+ if area_pct < self.MIN_BBOX_AREA_PCT:
+ continue
+
+ frame_dets.append(Detection(
+ bbox_xyxy=[float(x1), float(y1), float(x2), float(y2)],
+ confidence=float(score),
+ area_pct=float(area_pct)
+ ))
+
+ frame_dets = self._nms(frame_dets, iou_threshold=0.5)
+ frame_dets = self._dedupe_nearby(frame_dets)
+ return frame_dets
+
+ def _detect_batch(self, frames: List[np.ndarray], width: int, height: int) -> List[List[Detection]]:
+ """Run DINO detection on a list of frames."""
+ all_detections = []
+ for frame in tqdm(frames, desc="DINO detection"):
+ all_detections.append(self._detect_frame(frame, width, height))
+ return all_detections
+
+ def _save_debug_frame(
+ self,
+ frame: np.ndarray,
+ frame_idx: int,
+ timestamp: float,
+ detections: List[Detection],
+ out_dir: str,
+ save_all: bool = False
+ ) -> Optional[Dict]:
+ """Save a single debug frame with detection boxes drawn."""
+ if not save_all and len(detections) <= 1:
+ return None
+
+ frame_dir = os.path.join(out_dir, "frames")
+ os.makedirs(frame_dir, exist_ok=True)
+
+ colors = [
+ (0, 255, 0),
+ (0, 0, 255),
+ (255, 0, 0),
+ (0, 255, 255),
+ (255, 0, 255),
+ (255, 255, 0),
+ ]
+
+ frame_bgr = cv2.cvtColor(frame, cv2.COLOR_RGB2BGR)
+ h, w = frame_bgr.shape[:2]
+
+ for det_idx, det in enumerate(detections):
+ x1, y1, x2, y2 = det.bbox_xyxy
+ x1 = int(max(0, min(w - 1, round(x1))))
+ y1 = int(max(0, min(h - 1, round(y1))))
+ x2 = int(max(0, min(w - 1, round(x2))))
+ y2 = int(max(0, min(h - 1, round(y2))))
+
+ color = colors[det_idx % len(colors)]
+ cv2.rectangle(frame_bgr, (x1, y1), (x2, y2), color, 2)
+ label = f"{det_idx} {det.confidence:.2f}"
+ cv2.putText(
+ frame_bgr,
+ label,
+ (x1 + 4, max(10, y1 - 6)),
+ cv2.FONT_HERSHEY_SIMPLEX,
+ 0.5,
+ color,
+ 1,
+ cv2.LINE_AA
+ )
+
+ filename = f"frame_{frame_idx:06d}_t{timestamp:.2f}_n{len(detections)}.jpg"
+ out_path = os.path.join(frame_dir, filename)
+ cv2.imwrite(out_path, frame_bgr)
+
+ return {
+ "frame_idx": frame_idx,
+ "timestamp": float(timestamp),
+ "num_detections": len(detections),
+ "image": filename,
+ "detections": [
+ {
+ "bbox_xyxy": det.bbox_xyxy,
+ "confidence": det.confidence,
+ "area_pct": det.area_pct
+ } for det in detections
+ ]
+ }
+
+ def _filter_frame_detections_with_vitpose(
+ self,
+ frame: np.ndarray,
+ detections: List[Detection]
+ ) -> List[Detection]:
+ """Filter detections that look like partial people (head/limbs)."""
+ if not self.vitpose_validator or not detections:
+ return detections
+ if not self.vitpose_filter_all and len(detections) <= 1:
+ return detections
+
+ keep = []
+ for det in detections:
+ if self.vitpose_validator.is_complete(frame, det.bbox_xyxy):
+ keep.append(det)
+ return keep
+
+ def _build_tracks(self, detections: List[List[Detection]], timestamps: List[float],
+ img_width: int = 1920, img_height: int = 1080) -> Dict[int, Dict]:
+ """Build detection tracks over time using center distance matching.
+
+ IoU-based tracking fails at 1fps because the person moves too much.
+ Instead, use center distance - match to the nearest previous detection.
+ """
+ tracks = {} # track_id -> {timestamps, bboxes, confidences}
+ next_track_id = 0
+ active_tracks = {} # track_id -> last_bbox
+
+ # Max distance threshold: 50% of image diagonal (for fast motion at 1fps)
+ img_diagonal = np.sqrt(img_width**2 + img_height**2)
+ MAX_DISTANCE = img_diagonal * 0.5
+
+ def bbox_center(box):
+ return ((box[0] + box[2]) / 2, (box[1] + box[3]) / 2)
+
+ def center_distance(box1, box2):
+ c1 = bbox_center(box1)
+ c2 = bbox_center(box2)
+ return np.sqrt((c1[0] - c2[0])**2 + (c1[1] - c2[1])**2)
+
+ for frame_idx, (frame_dets, ts) in enumerate(zip(detections, timestamps)):
+ matched_tracks = set()
+ unmatched_dets = list(range(len(frame_dets)))
+
+ # Match detections to existing tracks by nearest center
+ for track_id, last_bbox in list(active_tracks.items()):
+ best_dist = float('inf')
+ best_det_idx = None
+
+ for det_idx in unmatched_dets:
+ dist = center_distance(last_bbox, frame_dets[det_idx].bbox_xyxy)
+ if dist < best_dist and dist <= MAX_DISTANCE:
+ best_dist = dist
+ best_det_idx = det_idx
+
+ if best_det_idx is not None:
+ det = frame_dets[best_det_idx]
+ tracks[track_id]['timestamps'].append(ts)
+ tracks[track_id]['bboxes'].append(det.bbox_xyxy)
+ tracks[track_id]['confidences'].append(det.confidence)
+ tracks[track_id]['areas'].append(det.area_pct)
+ active_tracks[track_id] = det.bbox_xyxy
+ matched_tracks.add(track_id)
+ unmatched_dets.remove(best_det_idx)
+
+ # Create new tracks for unmatched detections
+ for det_idx in unmatched_dets:
+ det = frame_dets[det_idx]
+ tracks[next_track_id] = {
+ 'timestamps': [ts],
+ 'bboxes': [det.bbox_xyxy],
+ 'confidences': [det.confidence],
+ 'areas': [det.area_pct]
+ }
+ active_tracks[next_track_id] = det.bbox_xyxy
+ next_track_id += 1
+
+ # Remove stale tracks (not seen in 3 seconds)
+ stale_threshold = 3.0
+ for track_id in list(active_tracks.keys()):
+ if track_id not in matched_tracks:
+ last_ts = tracks[track_id]['timestamps'][-1]
+ if ts - last_ts > stale_threshold:
+ del active_tracks[track_id]
+
+ return tracks
+
+ def _is_dynamic_track(self, track: TrackState) -> bool:
+ """Decide dynamic/static using running variance."""
+ if track.count < 3:
+ return True
+ center_var = (track.m2_center / max(1, track.count - 1)).sum()
+ size_var = (track.m2_size / max(1, track.count - 1)).sum()
+ total_variance = center_var + size_var
+ return total_variance >= self.STATIC_VARIANCE_THRESHOLD
+
+ def _update_track_stats(self, track: TrackState, bbox_xyxy: List[float]) -> None:
+ """Update running mean/variance for a track."""
+ x1, y1, x2, y2 = bbox_xyxy
+ center = np.array([(x1 + x2) * 0.5, (y1 + y2) * 0.5], dtype=np.float32)
+ size = np.array([max(1.0, x2 - x1), max(1.0, y2 - y1)], dtype=np.float32)
+
+ track.count += 1
+ delta_c = center - track.mean_center
+ track.mean_center += delta_c / track.count
+ track.m2_center += delta_c * (center - track.mean_center)
+
+ delta_s = size - track.mean_size
+ track.mean_size += delta_s / track.count
+ track.m2_size += delta_s * (size - track.mean_size)
+
+ def _classify_tracks(self, tracks: Dict[int, Dict]) -> Tuple[List[int], List[int]]:
+ """Classify tracks as dynamic (real person) or static (poster/sticker)."""
+ dynamic_tracks = []
+ static_tracks = []
+
+ for track_id, track in tracks.items():
+ bboxes = np.array(track['bboxes'])
+
+ if len(bboxes) < 3:
+ # Too short to classify reliably - assume dynamic (real person)
+ dynamic_tracks.append(track_id)
+ continue
+
+ # Compute bbox center variance
+ centers = (bboxes[:, :2] + bboxes[:, 2:]) / 2 # (N, 2)
+ center_variance = np.var(centers, axis=0).sum() # px²
+
+ # Also check if bbox size changes (person moving closer/farther)
+ sizes = bboxes[:, 2:] - bboxes[:, :2] # (N, 2) widths and heights
+ size_variance = np.var(sizes, axis=0).sum()
+
+ total_variance = center_variance + size_variance
+
+ if total_variance < self.STATIC_VARIANCE_THRESHOLD:
+ static_tracks.append(track_id)
+ else:
+ dynamic_tracks.append(track_id)
+
+ return dynamic_tracks, static_tracks
+
+ def _create_segments(
+ self,
+ tracks: Dict[int, Dict],
+ dynamic_tracks: List[int],
+ static_tracks: List[int],
+ timestamps: List[float]
+ ) -> List[Segment]:
+ """Create time segments with labeling info."""
+ if not timestamps:
+ return []
+
+ video_duration = timestamps[-1]
+ segments = []
+
+ # Build per-second person count
+ time_bins = defaultdict(lambda: {'dynamic': {}, 'static': set()})
+
+ for track_id in dynamic_tracks:
+ track = tracks[track_id]
+ for ts, bbox, conf, area in zip(
+ track['timestamps'],
+ track['bboxes'],
+ track['confidences'],
+ track['areas']
+ ):
+ sec = int(ts)
+ det = Detection(
+ bbox_xyxy=list(bbox),
+ confidence=float(conf),
+ area_pct=float(area)
+ )
+ existing = time_bins[sec]['dynamic'].get(track_id)
+ if existing is None or det.confidence > existing.confidence:
+ time_bins[sec]['dynamic'][track_id] = det
+
+ for track_id in static_tracks:
+ track = tracks[track_id]
+ for ts in track['timestamps']:
+ sec = int(ts)
+ time_bins[sec]['static'].add(track_id)
+
+ # Merge consecutive seconds with same characteristics
+ # Use ceil to ensure we cover the full video duration without gaps
+ import math
+ max_sec = math.ceil(video_duration)
+ current_segment = None
+
+ for sec in range(max_sec + 1):
+ bin_data = time_bins.get(sec, {'dynamic': {}, 'static': set()})
+ detections = list(bin_data['dynamic'].values())
+ detections = self._dedupe_nearby(
+ detections,
+ center_only_ratio=self.DUPLICATE_CENTER_ONLY_RATIO,
+ area_ratio=self.DUPLICATE_AREA_RATIO
+ )
+ n_dynamic = len(detections)
+ n_static = len(bin_data['static'])
+ avg_conf = np.mean([d.confidence for d in detections]) if detections else 0.0
+ avg_area = np.mean([d.area_pct for d in detections]) if detections else 0.0
+
+ # Determine usability
+ usable = n_dynamic == 1 and avg_conf >= self.MIN_CONFIDENCE and avg_area >= self.MIN_BBOX_AREA_PCT
+ reason = None
+ if n_dynamic == 0:
+ reason = "no_person"
+ elif n_dynamic > 1:
+ reason = "multiple_persons"
+ elif avg_conf < self.MIN_CONFIDENCE:
+ reason = "low_confidence"
+ elif avg_area < self.MIN_BBOX_AREA_PCT:
+ reason = "person_too_small"
+
+ # Check if we should start a new segment
+ if current_segment is None:
+ current_segment = {
+ 'start_sec': sec,
+ 'end_sec': sec + 1,
+ 'n_dynamic': n_dynamic,
+ 'n_static': n_static,
+ 'confs': [d.confidence for d in detections],
+ 'areas': [d.area_pct for d in detections],
+ 'usable': usable,
+ 'reason': reason
+ }
+ elif (current_segment['n_dynamic'] == n_dynamic and
+ current_segment['usable'] == usable and
+ current_segment['reason'] == reason):
+ # Extend current segment
+ current_segment['end_sec'] = sec + 1
+ current_segment['confs'].extend([d.confidence for d in detections])
+ current_segment['areas'].extend([d.area_pct for d in detections])
+ else:
+ # Finish current segment and start new one
+ segments.append(self._finalize_segment(current_segment))
+ current_segment = {
+ 'start_sec': sec,
+ 'end_sec': sec + 1,
+ 'n_dynamic': n_dynamic,
+ 'n_static': n_static,
+ 'confs': [d.confidence for d in detections],
+ 'areas': [d.area_pct for d in detections],
+ 'usable': usable,
+ 'reason': reason
+ }
+
+ if current_segment:
+ segments.append(self._finalize_segment(current_segment))
+
+ # Keep all segments - the usability field indicates whether each segment is good for training
+ # Note: MIN_SEGMENT_DURATION is used to determine which segments count toward usable_duration,
+ # but all segments are included in output for complete coverage
+
+ return segments
+
+ def _finalize_segment(self, seg_data: Dict) -> Segment:
+ """Convert segment data to Segment dataclass."""
+ return Segment(
+ start_sec=float(seg_data['start_sec']),
+ end_sec=float(seg_data['end_sec']),
+ dynamic_persons=int(seg_data['n_dynamic']),
+ static_detections=int(seg_data['n_static']),
+ avg_confidence=float(np.mean(seg_data['confs'])) if seg_data['confs'] else 0.0,
+ avg_bbox_area_pct=float(np.mean(seg_data['areas'])) if seg_data['areas'] else 0.0,
+ bbox_variance=0.0, # TODO: compute if needed
+ usable=bool(seg_data['usable']), # Cast to native Python bool for JSON
+ reason=seg_data['reason']
+ )
+
+ def label_video(
+ self,
+ video_path: str,
+ end_time: Optional[float] = None,
+ segment_writer: Optional[Callable[[Segment], None]] = None
+ ) -> Dict:
+ """Label a single video and return results."""
+ print(f"\n[Labeler] Processing: {video_path}")
+
+ # Step 1: Stream sampled frames
+ (width, height), duration, frame_iter = self._iter_sampled_frames(video_path, end_time=end_time)
+
+ frame_count = 0
+ last_ts = None
+ debug_meta = []
+ if self.debug_dir:
+ video_tag = os.path.splitext(os.path.basename(video_path))[0]
+ out_dir = os.path.join(self.debug_dir, video_tag)
+ os.makedirs(out_dir, exist_ok=True)
+ else:
+ out_dir = None
+
+ total_before = 0
+ total_after = 0
+
+ active_tracks: Dict[int, TrackState] = {}
+ next_track_id = 0
+ img_diagonal = np.sqrt(width**2 + height**2)
+ max_distance = img_diagonal * 0.5
+
+ current_sec = None
+ sec_dynamic: Dict[int, Detection] = {}
+ sec_static: set = set()
+ current_segment = None
+ segments = []
+ usable_duration = 0.0
+ total_segments = 0
+ pending_segments: List[Dict] = []
+
+ def emit_segment(seg_data: Dict):
+ nonlocal usable_duration, total_segments
+ segment = self._finalize_segment(seg_data)
+ total_segments += 1
+ if segment.usable:
+ usable_duration += (segment.end_sec - segment.start_sec)
+ if segment_writer:
+ segment_writer(segment)
+ else:
+ segments.append(segment)
+
+ def should_merge_low_conf(prev_seg: Dict, mid_seg: Dict, next_seg: Dict) -> bool:
+ if mid_seg['reason'] != "low_confidence":
+ return False
+ if (mid_seg['end_sec'] - mid_seg['start_sec']) >= self.LOW_CONF_SMOOTH_MAX_SEC:
+ return False
+ return (
+ prev_seg['n_dynamic'] == next_seg['n_dynamic']
+ and prev_seg['usable'] == next_seg['usable']
+ and prev_seg['reason'] == next_seg['reason']
+ )
+
+ def merge_triplet(prev_seg: Dict, mid_seg: Dict, next_seg: Dict) -> Dict:
+ return {
+ 'start_sec': prev_seg['start_sec'],
+ 'end_sec': next_seg['end_sec'],
+ 'n_dynamic': prev_seg['n_dynamic'],
+ 'n_static': max(prev_seg['n_static'], mid_seg['n_static'], next_seg['n_static']),
+ 'confs': prev_seg['confs'] + mid_seg['confs'] + next_seg['confs'],
+ 'areas': prev_seg['areas'] + mid_seg['areas'] + next_seg['areas'],
+ 'usable': prev_seg['usable'],
+ 'reason': prev_seg['reason']
+ }
+
+ def queue_segment(seg_data: Dict):
+ pending_segments.append(seg_data)
+ while len(pending_segments) >= 3:
+ prev_seg, mid_seg, next_seg = pending_segments[0], pending_segments[1], pending_segments[2]
+ if should_merge_low_conf(prev_seg, mid_seg, next_seg):
+ merged = merge_triplet(prev_seg, mid_seg, next_seg)
+ pending_segments[:3] = [merged]
+ else:
+ emit_segment(pending_segments.pop(0))
+
+ def flush_pending_segments():
+ while pending_segments:
+ emit_segment(pending_segments.pop(0))
+
+ def finalize_sec(sec_idx: int, dynamic_map: Dict[int, Detection], static_set: set):
+ nonlocal current_segment
+ detections_list = list(dynamic_map.values())
+ detections_list = self._dedupe_nearby(
+ detections_list,
+ center_only_ratio=self.DUPLICATE_CENTER_ONLY_RATIO,
+ area_ratio=self.DUPLICATE_AREA_RATIO
+ )
+ if len(detections_list) > 1:
+ max_area = max(d.area_pct for d in detections_list)
+ detections_list = [
+ d for d in detections_list
+ if d.area_pct >= self.MULTI_PERSON_MIN_AREA_PCT
+ and d.area_pct >= (max_area * self.MULTI_PERSON_REL_AREA)
+ ]
+ n_dynamic = len(detections_list)
+ n_static = len(static_set)
+ avg_conf = np.mean([d.confidence for d in detections_list]) if detections_list else 0.0
+ avg_area = np.mean([d.area_pct for d in detections_list]) if detections_list else 0.0
+
+ usable = n_dynamic == 1 and avg_conf >= self.MIN_CONFIDENCE and avg_area >= self.MIN_BBOX_AREA_PCT
+ reason = None
+ if n_dynamic == 0:
+ reason = "no_person"
+ elif n_dynamic > 1:
+ reason = "multiple_persons"
+ elif avg_conf < self.MIN_CONFIDENCE:
+ reason = "low_confidence"
+ elif avg_area < self.MIN_BBOX_AREA_PCT:
+ reason = "person_too_small"
+
+ if current_segment is None:
+ current_segment = {
+ 'start_sec': sec_idx,
+ 'end_sec': sec_idx + 1,
+ 'n_dynamic': n_dynamic,
+ 'n_static': n_static,
+ 'confs': [d.confidence for d in detections_list],
+ 'areas': [d.area_pct for d in detections_list],
+ 'usable': usable,
+ 'reason': reason
+ }
+ elif (current_segment['n_dynamic'] == n_dynamic and
+ current_segment['usable'] == usable and
+ current_segment['reason'] == reason):
+ current_segment['end_sec'] = sec_idx + 1
+ current_segment['confs'].extend([d.confidence for d in detections_list])
+ current_segment['areas'].extend([d.area_pct for d in detections_list])
+ else:
+ queue_segment(current_segment)
+ current_segment = {
+ 'start_sec': sec_idx,
+ 'end_sec': sec_idx + 1,
+ 'n_dynamic': n_dynamic,
+ 'n_static': n_static,
+ 'confs': [d.confidence for d in detections_list],
+ 'areas': [d.area_pct for d in detections_list],
+ 'usable': usable,
+ 'reason': reason
+ }
+
+ def finalize_missing_secs(start_sec: int, end_sec: int):
+ for missing_sec in range(start_sec, end_sec + 1):
+ finalize_sec(missing_sec, {}, set())
+
+ start_time = time.time()
+ last_log = start_time
+ for idx, (frame, ts) in enumerate(frame_iter):
+ frame_count += 1
+ last_ts = ts
+ frame_dets = self._detect_frame(frame, width, height)
+ total_before += len(frame_dets)
+ frame_dets = self._filter_frame_detections_with_vitpose(frame, frame_dets)
+ total_after += len(frame_dets)
+
+ # Track matching
+ matched_tracks = set()
+ unmatched_dets = list(range(len(frame_dets)))
+ assignments: Dict[int, int] = {}
+
+ for track_id, track in list(active_tracks.items()):
+ best_dist = float('inf')
+ best_det_idx = None
+ for det_idx in unmatched_dets:
+ det = frame_dets[det_idx]
+ x1, y1, x2, y2 = det.bbox_xyxy
+ cx = (x1 + x2) * 0.5
+ cy = (y1 + y2) * 0.5
+ last = track.last_bbox
+ lx = (last[0] + last[2]) * 0.5
+ ly = (last[1] + last[3]) * 0.5
+ dist = np.sqrt((cx - lx)**2 + (cy - ly)**2)
+ if dist < best_dist and dist <= max_distance:
+ best_dist = dist
+ best_det_idx = det_idx
+
+ if best_det_idx is not None:
+ det = frame_dets[best_det_idx]
+ track.last_bbox = det.bbox_xyxy
+ track.last_ts = ts
+ self._update_track_stats(track, det.bbox_xyxy)
+ matched_tracks.add(track_id)
+ assignments[best_det_idx] = track_id
+ unmatched_dets.remove(best_det_idx)
+
+ for det_idx in unmatched_dets:
+ det = frame_dets[det_idx]
+ x1, y1, x2, y2 = det.bbox_xyxy
+ center = np.array([(x1 + x2) * 0.5, (y1 + y2) * 0.5], dtype=np.float32)
+ size = np.array([max(1.0, x2 - x1), max(1.0, y2 - y1)], dtype=np.float32)
+ active_tracks[next_track_id] = TrackState(
+ last_bbox=det.bbox_xyxy,
+ last_ts=ts,
+ count=1,
+ mean_center=center.copy(),
+ m2_center=np.zeros_like(center),
+ mean_size=size.copy(),
+ m2_size=np.zeros_like(size)
+ )
+ assignments[det_idx] = next_track_id
+ matched_tracks.add(next_track_id)
+ next_track_id += 1
+
+ # Remove stale tracks (not seen in 3 seconds)
+ stale_threshold = 3.0
+ for track_id in list(active_tracks.keys()):
+ if track_id not in matched_tracks:
+ last_ts = active_tracks[track_id].last_ts
+ if ts - last_ts > stale_threshold:
+ del active_tracks[track_id]
+
+ sec = int(ts)
+ if current_sec is None:
+ current_sec = sec
+ elif sec > current_sec:
+ finalize_sec(current_sec, sec_dynamic, sec_static)
+ if sec > current_sec + 1:
+ finalize_missing_secs(current_sec + 1, sec - 1)
+ sec_dynamic = {}
+ sec_static = set()
+ current_sec = sec
+
+ for det_idx, det in enumerate(frame_dets):
+ track_id = assignments.get(det_idx)
+ if track_id is None or track_id not in active_tracks:
+ continue
+ if self._is_dynamic_track(active_tracks[track_id]):
+ existing = sec_dynamic.get(track_id)
+ if existing is None or det.confidence > existing.confidence:
+ sec_dynamic[track_id] = det
+ else:
+ sec_static.add(track_id)
+
+ if out_dir:
+ meta = self._save_debug_frame(
+ frame,
+ idx,
+ ts,
+ frame_dets,
+ out_dir,
+ save_all=self.debug_all
+ )
+ if meta:
+ debug_meta.append(meta)
+
+ now = time.time()
+ if now - last_log >= 30.0:
+ elapsed = now - start_time
+ fps = frame_count / elapsed if elapsed > 0 else 0.0
+ if duration:
+ pct = min(100.0, (ts / duration) * 100.0) if duration > 0 else 0.0
+ print(f"[Labeler] Progress: {frame_count} frames, t={ts:.1f}s ({pct:.1f}%), {fps:.2f} fps")
+ else:
+ print(f"[Labeler] Progress: {frame_count} frames, t={ts:.1f}s, {fps:.2f} fps")
+ last_log = now
+
+ if current_sec is not None:
+ finalize_sec(current_sec, sec_dynamic, sec_static)
+
+ if current_segment:
+ queue_segment(current_segment)
+ flush_pending_segments()
+
+ if self.vitpose_validator:
+ print(f"[Labeler] ViTPose filtered detections: {total_before} -> {total_after}")
+
+ if out_dir:
+ meta_path = os.path.join(out_dir, "detections.json")
+ with open(meta_path, "w") as f:
+ json.dump(debug_meta, f, indent=2)
+
+ if frame_count == 0:
+ return {'video': video_path, 'error': 'No frames extracted', 'segments': []}
+
+ # Step 3: Build tracks
+ tracks = self._build_tracks(detections, timestamps, width, height)
+ print(f"[Labeler] Found {len(tracks)} detection tracks")
+
+ # Step 4: Classify tracks as dynamic/static
+ dynamic_tracks, static_tracks = self._classify_tracks(tracks)
+ print(f"[Labeler] Dynamic (person): {len(dynamic_tracks)}, Static (poster/sticker): {len(static_tracks)}")
+
+ # Step 5: Create segments
+ # Summary
+ total_duration = duration if duration is not None else (last_ts if last_ts is not None else 0)
+
+ print(f"[Labeler] Found {usable_duration:.0f}s usable ({total_duration:.0f}s total)")
+
+ return {
+ 'video': os.path.abspath(video_path),
+ 'total_duration_sec': total_duration,
+ 'usable_duration_sec': usable_duration,
+ 'num_segments': total_segments if segment_writer else len(segments),
+ 'segments': [asdict(s) for s in segments] if not segment_writer else []
+ }
+
+
+def main():
+ parser = argparse.ArgumentParser(description='Label videos for GENMO training suitability')
+ parser.add_argument('--video', type=str, help='Path to a single video file')
+ parser.add_argument('--video-dir', type=str, help='Path to directory containing videos')
+ parser.add_argument('--output', type=str, required=True, help='Output JSON file path')
+ parser.add_argument('--sample-fps', type=float, default=1.0, help='Frames per second to sample (default: 1.0)')
+ parser.add_argument('--end-time', type=float, default=None, help='Only process first N seconds of video')
+ parser.add_argument('--debug-dir', type=str, default=None, help='Directory to save debug frames with bboxes')
+ parser.add_argument('--debug-all', action='store_true', help='Save debug frames for all detections (default: only multi-person frames)')
+ parser.add_argument('--vitpose-filter', action='store_true', help='Filter detections using ViTPose joint visibility')
+ parser.add_argument('--vitpose-filter-all', action='store_true', help='Apply ViTPose filtering to all frames')
+ parser.add_argument('--vitpose-min-joints', type=int, default=4, help='Minimum visible joints (excluding face) to keep')
+ parser.add_argument('--vitpose-conf-threshold', type=float, default=0.3, help='Minimum joint confidence for ViTPose')
+ parser.add_argument('--vitpose-disable-upper-lower', action='store_true', help='Disable upper/lower body joint requirement')
+ parser.add_argument('--vitpose-min-vertical-span', type=float, default=0.35, help='Min joint vertical span ratio within bbox')
+ parser.add_argument('--vitpose-config', type=str, default=None, help='ViTPose config path')
+ parser.add_argument('--vitpose-ckpt', type=str, default=None, help='ViTPose checkpoint path')
+ parser.add_argument('--stream-jsonl', action='store_true', help='Stream segments as JSON Lines (append)')
+
+ args = parser.parse_args()
+
+ if not args.video and not args.video_dir:
+ parser.error("Must specify either --video or --video-dir")
+
+ # Collect video paths
+ video_paths = []
+ if args.video:
+ video_paths.append(args.video)
+ if args.video_dir:
+ for fname in os.listdir(args.video_dir):
+ if fname.endswith(('.mp4', '.avi', '.mov', '.mkv')):
+ video_paths.append(os.path.join(args.video_dir, fname))
+
+ print(f"[Labeler] Found {len(video_paths)} video(s) to process")
+
+ # Initialize labeler
+ labeler = VideoLabeler(
+ sample_fps=args.sample_fps,
+ debug_dir=args.debug_dir,
+ debug_all=args.debug_all,
+ vitpose_filter=args.vitpose_filter,
+ vitpose_filter_all=args.vitpose_filter_all,
+ vitpose_min_joints=args.vitpose_min_joints,
+ vitpose_conf_threshold=args.vitpose_conf_threshold,
+ vitpose_require_upper_lower=not args.vitpose_disable_upper_lower,
+ vitpose_min_vertical_span=args.vitpose_min_vertical_span,
+ vitpose_config=args.vitpose_config,
+ vitpose_ckpt=args.vitpose_ckpt
+ )
+
+ # Process each video
+ results = {'videos': []}
+ segment_writer = None
+ output_path = args.output
+
+ stream_jsonl = args.stream_jsonl or output_path.endswith(".jsonl")
+
+ if stream_jsonl:
+ os.makedirs(os.path.dirname(os.path.abspath(output_path)) or ".", exist_ok=True)
+ with open(output_path, "a") as f:
+ def write_segment(segment: Segment, video_path: str):
+ payload = {'video': os.path.abspath(video_path)}
+ payload.update(asdict(segment))
+ f.write(json.dumps(payload) + "\n")
+ f.flush()
+
+ segment_writer = write_segment
+ for video_path in video_paths:
+ result = labeler.label_video(
+ video_path,
+ end_time=args.end_time,
+ segment_writer=lambda seg, vp=video_path: segment_writer(seg, vp)
+ )
+ results['videos'].append(result)
+ else:
+ for video_path in video_paths:
+ result = labeler.label_video(video_path, end_time=args.end_time)
+ results['videos'].append(result)
+
+ # Save results
+ if not stream_jsonl:
+ with open(args.output, 'w') as f:
+ json.dump(results, f, indent=2)
+ print(f"\n[Labeler] Results saved to: {args.output}")
+ else:
+ print(f"\n[Labeler] Segments appended to: {args.output}")
+
+ # Print summary
+ total_usable = sum(v.get('usable_duration_sec', 0) for v in results['videos'])
+ total_duration = sum(v.get('total_duration_sec', 0) for v in results['videos'])
+ print(f"[Labeler] Total usable: {total_usable/3600:.2f} hours / {total_duration/3600:.2f} hours")
+
+
+if __name__ == '__main__':
+ main()
diff --git a/scripts/scan_unity_transl_vel_spikes.py b/scripts/scan_unity_transl_vel_spikes.py
new file mode 100644
index 0000000000000000000000000000000000000000..36fc5432e1efb66f7fcf33a761b55db9edcf9bd2
--- /dev/null
+++ b/scripts/scan_unity_transl_vel_spikes.py
@@ -0,0 +1,100 @@
+#!/usr/bin/env python3
+"""
+Scan Unity `processed_dataset/genmo_features/*.pt` for translation spikes.
+
+Why:
+ A small number of teleports/cuts (very large per-frame translation deltas) can
+ dominate the transl_vel loss (MSE in normalized space), making fine-tuning
+ look like "global breaks" even when most frames are fine.
+
+This script reports the worst sequences and how many frames exceed a threshold.
+"""
+
+from __future__ import annotations
+
+import argparse
+from pathlib import Path
+from typing import Any, Dict
+
+import torch
+
+from third_party.GVHMR.hmr4d.utils.geo.hmr_global import get_local_transl_vel
+
+
+def _load_pt(path: Path) -> Dict[str, Any]:
+ return torch.load(path, map_location="cpu", weights_only=False)
+
+
+def main() -> None:
+ ap = argparse.ArgumentParser()
+ ap.add_argument(
+ "--root",
+ type=Path,
+ default=Path("processed_dataset/genmo_features"),
+ help="Directory containing .pt files.",
+ )
+ ap.add_argument("--glob", type=str, default="*.pt")
+ ap.add_argument("--thr_m_per_frame", type=float, default=0.30)
+ ap.add_argument("--topk", type=int, default=20)
+ args = ap.parse_args()
+
+ root: Path = args.root
+ if not root.is_dir():
+ raise SystemExit(f"Not found: {root}")
+
+ paths = sorted(root.glob(args.glob))
+ if not paths:
+ raise SystemExit(f"No files under {root} matching {args.glob!r}")
+
+ thr = float(args.thr_m_per_frame)
+ rows = []
+ total_spike_frames = 0
+ total_frames = 0
+
+ for p in paths:
+ d = _load_pt(p)
+ tw = d["smpl_params_w"]["transl"].float() # (L, 3)
+ go = d["smpl_params_w"]["global_orient"].float() # (L, 3)
+
+ # world-frame deltas
+ wv = torch.zeros_like(tw)
+ wv[1:] = tw[1:] - tw[:-1]
+ wv_norm = wv.norm(dim=-1) # (L,)
+
+ # local-frame deltas (matches EnDecoder target generation)
+ lv = get_local_transl_vel(tw[None], go[None])[0] # (L, 3)
+ lv_norm = lv.norm(dim=-1) # (L,)
+
+ spike = (wv_norm > thr) | (lv_norm > thr)
+ num_spike = int(spike.sum().item())
+
+ total_spike_frames += num_spike
+ total_frames += int(spike.numel())
+
+ rows.append(
+ {
+ "file": p.name,
+ "L": int(tw.shape[0]),
+ "wv_max": float(wv_norm.max().item()),
+ "lv_max": float(lv_norm.max().item()),
+ "wv_mean": float(wv_norm.mean().item()),
+ "lv_mean": float(lv_norm.mean().item()),
+ "spike_frames": num_spike,
+ }
+ )
+
+ rows.sort(key=lambda r: (r["spike_frames"], r["wv_max"], r["lv_max"]), reverse=True)
+ print(f"Files: {len(rows)} | thr_m_per_frame={thr:.3f}")
+ print(f"Spike frames: {total_spike_frames}/{total_frames} ({(100.0*total_spike_frames/max(1,total_frames)):.3f}%)")
+ print("\nTop sequences:")
+ for r in rows[: max(1, int(args.topk))]:
+ print(
+ f"{r['file']}: L={r['L']} spike_frames={r['spike_frames']} "
+ f"wv_max={r['wv_max']:.4f} lv_max={r['lv_max']:.4f} "
+ f"wv_mean={r['wv_mean']:.4f} lv_mean={r['lv_mean']:.4f}"
+ )
+
+
+if __name__ == "__main__":
+ main()
+
diff --git a/scripts/train.py b/scripts/train.py
new file mode 100644
index 0000000000000000000000000000000000000000..cc1251c42936ed4b9648f990d329b3b6e437cc0e
--- /dev/null
+++ b/scripts/train.py
@@ -0,0 +1,262 @@
+import os
+import sys
+
+# Ensure repo root is importable when running as `python scripts/train.py`.
+# Without this, `genmo.*` may resolve from site-packages while `third_party.*`
+# (a namespace package in this repo) fails to import, which Hydra reports as
+# "Error locating target ...".
+_REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
+if _REPO_ROOT not in sys.path:
+ sys.path.insert(0, _REPO_ROOT)
+
+# GVHMR uses absolute imports like `import hmr4d...` internally, so its repo root
+# must also be importable.
+_GVHMR_ROOT = os.path.join(_REPO_ROOT, "third_party", "GVHMR")
+if os.path.isdir(_GVHMR_ROOT) and _GVHMR_ROOT not in sys.path:
+ sys.path.insert(0, _GVHMR_ROOT)
+
+import builtins
+from datetime import datetime
+
+import hydra
+import pytorch_lightning as pl
+import torch
+import torch.distributed as dist
+import yaml
+from omegaconf import DictConfig, ListConfig, OmegaConf
+from pytorch_lightning.callbacks.checkpoint import Checkpoint
+from pytorch_lightning.loggers.tensorboard import TensorBoardLogger
+
+from genmo.callbacks.autoresume_callback import AutoResume, AutoResumeCallback
+from genmo.utils.net_utils import get_resume_ckpt_path, load_pretrained_model
+from genmo.utils.pylogger import Log
+from genmo.utils.tools import (
+ find_last_version,
+ get_checkpoint_path,
+ rsync_file_from_remote,
+)
+from genmo.utils.vis.rich_logger import print_cfg
+
+OmegaConf.register_new_resolver("eval", builtins.eval)
+
+
+def _get_rank():
+ # SLURM_PROCID can be set even if SLURM is not managing the multiprocessing,
+ # therefore LOCAL_RANK needs to be checked first
+ rank_keys = ("RANK", "LOCAL_RANK", "SLURM_PROCID", "JSM_NAMESPACE_RANK")
+ for key in rank_keys:
+ rank = os.environ.get(key)
+ if rank is not None:
+ return int(rank)
+ # None to differentiate whether an environment variable was set at all
+ return 0
+
+
+global_rank = _get_rank()
+
+
+def get_callbacks(cfg: DictConfig) -> list:
+ """Parse and instantiate all the callbacks in the config.
+
+ Supports both flat and nested callback configs. Only nodes containing
+ a `_target_` are instantiated.
+ """
+ if not hasattr(cfg, "callbacks") or cfg.callbacks is None:
+ return None
+
+ def _collect_callback_nodes(node):
+ collected = []
+ if node is None:
+ return collected
+ # Dict-like node
+ if isinstance(node, (DictConfig, dict)):
+ # direct instantiable config
+ if "_target_" in node:
+ collected.append(node)
+ else:
+ for child in node.values():
+ collected.extend(_collect_callback_nodes(child))
+ # List-like node
+ elif isinstance(node, (ListConfig, list, tuple)):
+ for child in node:
+ collected.extend(_collect_callback_nodes(child))
+ # primitives are ignored
+ return collected
+
+ enable_checkpointing = cfg.pl_trainer.get("enable_checkpointing", True)
+ callbacks = []
+ for cb_conf in _collect_callback_nodes(cfg.callbacks):
+ cb = hydra.utils.instantiate(cb_conf, _recursive_=False)
+ if not enable_checkpointing and isinstance(cb, Checkpoint):
+ continue
+ callbacks.append(cb)
+ return callbacks
+
+
+def train(cfg: DictConfig) -> None:
+ """Train/Test"""
+ Log.info(f"[Exp Name]: {cfg.exp_name}")
+ # use total batch size
+ if cfg.task == "fit":
+ Log.info(
+ f"[GPU x Batch] = {cfg.pl_trainer.devices} x {cfg.data.loader_opts.train.batch_size}"
+ )
+ num_nodes = cfg.pl_trainer.get("num_nodes", 1)
+ cfg.num_test_data *= cfg.pl_trainer.devices * num_nodes
+ if (
+ "imgfeat_motionx" in cfg.test_datasets
+ and "max_num_motions" in cfg.test_datasets.imgfeat_motionx
+ ):
+ cfg.test_datasets.imgfeat_motionx.max_num_motions *= (
+ cfg.pl_trainer.devices * num_nodes
+ )
+ pl.seed_everything(cfg.seed)
+ torch.cuda.set_device(global_rank % 8) # for tinycudann default memory
+ version = None
+ tb_logger = None
+
+ if cfg.get("timing", False):
+ os.environ["DEBUG_TIMING"] = "TRUE"
+
+ if AutoResume is not None:
+ details = AutoResume.get_resume_details()
+ if details:
+ cfg.resume_mode = "last"
+ version = int(details["version"])
+ print(
+ f"[Auto Resume] Loading. checkpoint: {details['checkpoint']} version: {details['version']}"
+ )
+
+ if cfg.task == "test" and not cfg.get("no_checkpoint", False):
+ test_cp = cfg.get("test_checkpoint", "last")
+ remote_run_dir = cfg.output_dir.replace("outputs", cfg.remote_results_path)
+ version = find_last_version(remote_run_dir, cp=test_cp)
+ checkpoint_dir = f"{remote_run_dir}/version_{version}/checkpoints"
+ remote_ckpt_path = get_checkpoint_path(checkpoint_dir, test_cp)
+ if cfg.get("rsync_ckpt", False):
+ cfg.ckpt_path = remote_ckpt_path.replace(cfg.remote_results_path, "outputs")
+ if not os.path.exists(cfg.ckpt_path):
+ print(f"rsyncing from remote: {remote_ckpt_path}")
+ print(f"output_dir: {cfg.output_dir}")
+ rsync_file_from_remote(
+ cfg.ckpt_path,
+ remote_run_dir,
+ cfg.output_dir,
+ hostname="cs-oci-ord-dc-03",
+ )
+ else:
+ cfg.ckpt_path = remote_ckpt_path
+ print("ckpt path:", cfg.ckpt_path)
+ cfg.output_dir = f"{cfg.output_dir}/version_{version}"
+ cfg.logger.name = (
+ f"{cfg.exp_name}_v{version}_{datetime.now().strftime('%Y%m%d%H%M%S')}"
+ )
+ else:
+ run_root_dir = cfg.output_dir
+ if version is None and cfg.resume_mode == "last":
+ version = find_last_version(run_root_dir, cp="last")
+
+ # preparation
+ datamodule: pl.LightningDataModule = hydra.utils.instantiate(
+ cfg.data, _recursive_=False
+ )
+ model: pl.LightningModule = hydra.utils.instantiate(cfg.model, _recursive_=False)
+
+ if (
+ cfg.get("pretrain_ckpt", None) is not None
+ and cfg.ckpt_path is None
+ and cfg.resume_mode is None
+ ):
+ cfg.ckpt_path = cfg.pretrain_ckpt
+
+ if cfg.ckpt_path is not None:
+ if cfg.get("rsync_ckpt", False) and not os.path.exists(cfg.ckpt_path):
+ print(f"rsyncing from remote: {cfg.ckpt_path}")
+ cfg.ckpt_path = cfg.ckpt_path.replace(cfg.remote_results_path, "outputs")
+ local_dir = cfg.ckpt_path.split("/version_")[0]
+ os.makedirs(local_dir, exist_ok=True)
+ rsync_file_from_remote(
+ cfg.ckpt_path,
+ cfg.remote_results_path,
+ "outputs",
+ hostname="cs-oci-ord-dc-03",
+ )
+
+ ckpt = load_pretrained_model(model, cfg.ckpt_path)
+ print(f"Loaded pretrained model from {cfg.ckpt_path}")
+ if ckpt is not None:
+ print(
+ "pretrained ckpt info:",
+ {"global_step": ckpt["global_step"], "epoch": ckpt["epoch"]},
+ )
+
+ # PL callbacks and logger (TensorBoard only)
+ if cfg.task == "fit":
+ tb_logger = TensorBoardLogger(run_root_dir, version=version, name="")
+ version = tb_logger.version
+ if global_rank == 0:
+ os.makedirs(tb_logger.log_dir, exist_ok=True)
+ cfg.output_dir = tb_logger.log_dir
+
+ if cfg.pl_trainer.devices > 1 and "RANK" in os.environ:
+ dist.init_process_group("nccl")
+ dist.barrier()
+
+ if global_rank != 0:
+ if version is None:
+ version = find_last_version(run_root_dir, cp=None)
+ cfg.output_dir = f"{run_root_dir}/version_{version}"
+
+ callbacks = get_callbacks(cfg)
+ has_ckpt_cb = any([isinstance(cb, Checkpoint) for cb in callbacks])
+ if not has_ckpt_cb and cfg.pl_trainer.get("enable_checkpointing", True):
+ Log.warning("No checkpoint-callback found. Disabling PL auto checkpointing.")
+ cfg.pl_trainer = {**cfg.pl_trainer, "enable_checkpointing": False}
+ if AutoResume is not None:
+ callbacks.append(AutoResumeCallback(version))
+
+ logger = tb_logger if tb_logger is not None else False
+
+ # PL-Trainer
+ if cfg.task == "test":
+ Log.info("Test mode forces full-precision.")
+ cfg.pl_trainer = {**cfg.pl_trainer, "precision": 32}
+ trainer = pl.Trainer(
+ accelerator="gpu",
+ logger=logger if logger is not None else False,
+ callbacks=callbacks,
+ **cfg.pl_trainer,
+ )
+
+ print("=" * 20)
+ print("version:", version)
+
+ if cfg.task == "fit":
+ resume_path = None
+ if cfg.resume_mode is not None:
+ save_dir = cfg.output_dir + "/checkpoints"
+ resume_path = get_resume_ckpt_path(cfg.resume_mode, ckpt_dir=save_dir)
+ Log.info("Start Fitting...")
+ trainer.fit(
+ model,
+ datamodule.train_dataloader(),
+ datamodule.val_dataloader(),
+ ckpt_path=resume_path,
+ )
+ elif cfg.task == "test":
+ Log.info("Start Testing...")
+ trainer.test(model, datamodule.test_dataloader())
+ else:
+ raise ValueError(f"Unknown task: {cfg.task}")
+
+ Log.info("End of script.")
+
+
+@hydra.main(version_base="1.3", config_path="../configs", config_name="train")
+def main(cfg) -> None:
+ print_cfg(cfg, use_rich=True)
+ train(cfg)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/setup.py b/setup.py
new file mode 100644
index 0000000000000000000000000000000000000000..4739df3d16f444d14658c2115f63bfbf48de7e15
--- /dev/null
+++ b/setup.py
@@ -0,0 +1,10 @@
+from setuptools import find_packages, setup
+
+setup(
+ name="genmo",
+ version="1.0.0",
+ packages=find_packages(),
+ author="NVIDIA Digital Human AI Research",
+ description="GENMO: A GENeralist Model for Human MOtion",
+ url="https://github.com/NVlabs/GENMO",
+)
diff --git a/show_rotation_fix_difference.py b/show_rotation_fix_difference.py
new file mode 100644
index 0000000000000000000000000000000000000000..d2b98154fe9899c78778d11f2f1c7cbec9eacb89
--- /dev/null
+++ b/show_rotation_fix_difference.py
@@ -0,0 +1,95 @@
+#!/usr/bin/env python3
+"""
+Show the difference between old and new rotation computation methods.
+This demonstrates why order matters when combining rotations with coordinate conversions.
+"""
+import numpy as np
+from scipy.spatial.transform import Rotation as R
+
+def compare_methods():
+ """Compare old vs new rotation computation."""
+
+ # Example: Camera 45° around Y, Pelvis 90° around Y (Unity space)
+ cam_quat_unity = R.from_euler('y', 45, degrees=True).as_quat()
+ pel_quat_unity = R.from_euler('y', 90, degrees=True).as_quat()
+
+ C = np.diag([1.0, -1.0, 1.0]) # Unity -> CV conversion
+
+ print("="*70)
+ print("ROTATION COMPUTATION ORDER - COMPARISON")
+ print("="*70)
+ print()
+ print("Unity Input:")
+ print(f" Camera: 45° around Y")
+ print(f" Pelvis: 90° around Y")
+ print()
+
+ # === OLD METHOD (V1 - WRONG) ===
+ print("--- OLD METHOD (V1) ---")
+ print("Steps: 1) Compute relative in Unity, 2) Convert to CV")
+ print()
+
+ R_cam_w_unity = R.from_quat(cam_quat_unity).as_matrix()
+ R_pel_w_unity = R.from_quat(pel_quat_unity).as_matrix()
+ R_rel_unity = R_cam_w_unity.T @ R_pel_w_unity
+ R_cv_old = C @ R_rel_unity @ C
+
+ euler_old = R.from_matrix(R_cv_old).as_euler('YXZ', degrees=True)
+ print(f"Result (YXZ): yaw={euler_old[0]:7.2f}°, pitch={euler_old[1]:7.2f}°, roll={euler_old[2]:7.2f}°")
+ print()
+
+ # === NEW METHOD (V2 - CORRECT) ===
+ print("--- NEW METHOD (V2) ---")
+ print("Steps: 1) Convert to CV, 2) Compute relative in CV")
+ print()
+
+ R_cam_w_cv = C @ R_cam_w_unity @ C
+ R_pel_w_cv = C @ R_pel_w_unity @ C
+ R_w2c_cv = R_cam_w_cv.T
+ R_pel_c_cv = R_w2c_cv @ R_pel_w_cv # GENMO's formula
+
+ euler_new = R.from_matrix(R_pel_c_cv).as_euler('YXZ', degrees=True)
+ print(f"Result (YXZ): yaw={euler_new[0]:7.2f}°, pitch={euler_new[1]:7.2f}°, roll={euler_new[2]:7.2f}°")
+ print()
+
+ # === DIFFERENCE ===
+ print("--- DIFFERENCE ---")
+ diff = euler_new - euler_old
+ print(f"Δ (YXZ): yaw={diff[0]:7.2f}°, pitch={diff[1]:7.2f}°, roll={diff[2]:7.2f}°")
+ print()
+
+ # === GEOMETRIC INTERPRETATION ===
+ print("--- GEOMETRIC INTERPRETATION ---")
+ print(f"OLD: Pelvis is {45}° relative to camera in Unity,")
+ print(f" then convert whole thing to CV")
+ print(f" → Gives: {euler_old[0]:.1f}° yaw in CV camera frame")
+ print()
+ print(f"NEW: Camera is {45}° in CV, Pelvis is {90}° in CV,")
+ print(f" relative angle is {90-45}° = 45°")
+ print(f" → Gives: {euler_new[0]:.1f}° yaw in CV camera frame")
+ print()
+
+ # === WHY IT MATTERS ===
+ print("="*70)
+ print("WHY THIS MATTERS FOR GENMO")
+ print("="*70)
+ print()
+ print("GENMO was trained with rotations computed as:")
+ print(" global_orient_c = R_w2c @ R_pel_w (both in CV convention)")
+ print()
+ print("If we compute in Unity then convert:")
+ print(" global_orient_c = C @ (R_w2c_unity @ R_pel_w_unity) @ C")
+ print(" ≠ (C @ R_w2c_unity @ C) @ (C @ R_pel_w_unity @ C)")
+ print()
+ print("This mismatch caused:")
+ print(" ✗ 168° roll errors")
+ print(" ✗ 55° yaw errors")
+ print(" ✗ Training loss explosion (12 → 100+)")
+ print()
+ print("Afterfix (V2):")
+ print(" ✓ <5° rotation errors (all axes)")
+ print(" ✓ Training loss converges (~0.5-2.0)")
+ print()
+
+if __name__ == "__main__":
+ compare_methods()
diff --git a/test.py b/test.py
new file mode 100644
index 0000000000000000000000000000000000000000..cab1380b882f6a88d31edf1d8a052e94f93f7f57
--- /dev/null
+++ b/test.py
@@ -0,0 +1,41 @@
+import torch
+import os
+import sys
+
+def analyze_ckpt(path):
+ if not os.path.exists(path):
+ print(f"File not found: {path}")
+ return
+
+ size_mb = os.path.getsize(path) / (1024 * 1024)
+ print(f"\nAnalyzing {path} (Total File Size: {size_mb:.2f} MB)...")
+
+ try:
+ ckpt = torch.load(path, map_location="cpu")
+ except Exception as e:
+ print(f"Error loading checkpoint: {e}")
+ return
+
+ # 1. Measure Model Weights (state_dict)
+ state_dict_size = 0
+ if "state_dict" in ckpt:
+ for k, v in ckpt["state_dict"].items():
+ state_dict_size += v.numel() * v.element_size()
+ print(f" - Model Weights (state_dict): {state_dict_size / (1024*1024):.2f} MB")
+
+ # 2. Measure Optimizer States
+ opt_size = 0
+ if "optimizer_states" in ckpt:
+ for opt in ckpt["optimizer_states"]:
+ # Optimizer state structure can vary, this is a general traversal
+ if isinstance(opt, dict) and "state" in opt:
+ for param_id, state in opt["state"].items():
+ for k, v in state.items():
+ if torch.is_tensor(v):
+ opt_size += v.numel() * v.element_size()
+ print(f" - Optimizer States: {opt_size / (1024*1024):.2f} MB")
+
+if __name__ == "__main__":
+ # Replace with your actual paths if different
+ analyze_ckpt("s050000.ckpt")
+ analyze_ckpt("./checkpoints/last_manual.ckpt")
diff --git a/test_rotation_fix.py b/test_rotation_fix.py
new file mode 100644
index 0000000000000000000000000000000000000000..f05e64672fe757dddbee7e78230884ae5341c370
--- /dev/null
+++ b/test_rotation_fix.py
@@ -0,0 +1,47 @@
+#!/usr/bin/env python3
+"""
+Quick test to verify the Z-180° fix removal resolved the rotation mismatch.
+This should show much smaller rotation errors after reprocessing.
+"""
+import numpy as np
+from scipy.spatial.transform import Rotation as R
+
+def test_rotation_convention():
+ """Test that incam and world rotations are consistent after Z-180 fix removal."""
+
+ # Example Unity quaternions (from your data)
+ cam_quat = np.array([0.0, 0.0, 0.0, 1.0]) # Identity
+ pelvis_quat = np.array([0.0, 0.707, 0.0, 0.707]) # 90° around Y
+
+ # Unity -> CV coordinate conversion
+ C = np.diag([1.0, -1.0, 1.0])
+
+ # OLD method (with Z-180° fix)
+ R_cam_w = R.from_quat(cam_quat).as_matrix()
+ R_pel_w = R.from_quat(pelvis_quat).as_matrix()
+ R_rel_unity = R_cam_w.T @ R_pel_w
+ R_cv_old = C @ R_rel_unity @ C
+ R_final_old = R_cv_old @ R.from_euler("z", 180, degrees=True).as_matrix()
+
+ # NEW method (without Z-180° fix)
+ R_cv_new = C @ R_rel_unity @ C
+
+ # Compare
+ euler_old = R.from_matrix(R_final_old).as_euler('YXZ', degrees=True)
+ euler_new = R.from_matrix(R_cv_new).as_euler('YXZ', degrees=True)
+
+ print("=== Rotation Convention Test ===")
+ print(f"OLD (with Z-180°): yaw={euler_old[0]:7.2f}, pitch={euler_old[1]:7.2f}, roll={euler_old[2]:7.2f}")
+ print(f"NEW (no Z-180°): yaw={euler_new[0]:7.2f}, pitch={euler_new[1]:7.2f}, roll={euler_new[2]:7.2f}")
+ print(f"Difference: yaw={euler_new[0]-euler_old[0]:7.2f}, pitch={euler_new[1]-euler_old[1]:7.2f}, roll={euler_new[2]-euler_old[2]:7.2f}")
+ print()
+ print("Expected: ~180° difference in roll (confirming the fix)")
+ print()
+
+if __name__ == "__main__":
+ test_rotation_convention()
+ print("After reprocessing with the updated script, run diagnose_data.py again.")
+ print("You should see:")
+ print(" - In-camera roll error: ~0-10° (instead of 348°)")
+ print(" - World orientation errors: <5° for all axes")
+ print(" - Body pose errors: <10° mean")
diff --git a/test_single_frame.py b/test_single_frame.py
new file mode 100644
index 0000000000000000000000000000000000000000..54ceb6992657a42e92eb348e19b46c144db4d4b6
--- /dev/null
+++ b/test_single_frame.py
@@ -0,0 +1,112 @@
+#!/usr/bin/env python3
+"""
+Test rotation computation on actual Unity export data.
+This will show what's actually causing the 168° roll error.
+"""
+import json
+import numpy as np
+from scipy.spatial.transform import Rotation as R
+from pathlib import Path
+from glob import glob
+
+def test_actual_frame():
+ """Load first frame from Unity export and compare rotation computations."""
+
+ # Find Unity export
+ jsonl_files = sorted(glob("./raw_dataset/sequence_*.jsonl"))
+ if not jsonl_files:
+ print("ERROR: No Unity export found in ./raw_dataset/")
+ print("Please run this from GENMO/ directory")
+ return
+
+ with open(jsonl_files[0], "r") as f:
+ lines = f.readlines()
+
+ if len(lines) < 2:
+ print("ERROR: No data in JSONL file")
+ return
+
+ # Parse frame 0 (skip header line)
+ row = json.loads(lines[1])
+
+ print("="*70)
+ print("TESTING ACTUAL UNITY EXPORT DATA (Frame 0)")
+ print("="*70)
+ print()
+
+ # Extract Unity quaternions
+ cam_quat = np.array(row["cam_rot_world"], dtype=np.float64)
+ pel_quat = np.array(row["pelvis_rot_world"], dtype=np.float64)
+
+ print(f"Unity quaternions (XYZW):")
+ print(f" cam_rot_world: {cam_quat}")
+ print(f" pelvis_rot_world: {pel_quat}")
+ print()
+
+ # Convert to matrices
+ R_cam_w_unity = R.from_quat(cam_quat).as_matrix()
+ R_pel_w_unity = R.from_quat(pel_quat).as_matrix()
+
+ print("Unity rotations (Euler YXZ):")
+ print(f" Camera: {R.from_matrix(R_cam_w_unity).as_euler('YXZ', degrees=True)}")
+ print(f" Pelvis: {R.from_matrix(R_pel_w_unity).as_euler('YXZ', degrees=True)}")
+ print()
+
+ # Coordinate conversion
+ C = np.diag([1.0, -1.0, 1.0])
+
+ # === OLD METHOD ===
+ print("--- OLD METHOD (Compute relative, then convert) ---")
+ R_w2c_unity = R_cam_w_unity.T
+ R_rel_unity = R_w2c_unity @ R_pel_w_unity
+ R_cv_old = C @ R_rel_unity @ C
+
+ euler_old = R.from_matrix(R_cv_old).as_euler('YXZ', degrees=True)
+ print(f"global_orient_c (YXZ): yaw={euler_old[0]:7.2f}°, pitch={euler_old[1]:7.2f}°, roll={euler_old[2]:7.2f}°")
+ print()
+
+ # === NEW METHOD ===
+ print("--- NEW METHOD (Convert, then compute relative) ---")
+ R_cam_w_cv = C @ R_cam_w_unity @ C
+ R_pel_w_cv = C @ R_pel_w_unity @ C
+ R_w2c_cv = R_cam_w_cv.T
+ R_pel_c_cv = R_w2c_cv @ R_pel_w_cv
+
+ euler_new = R.from_matrix(R_pel_c_cv).as_euler('YXZ', degrees=True)
+ print(f"global_orient_c (YXZ): yaw={euler_new[0]:7.2f}°, pitch={euler_new[1]:7.2f}°, roll={euler_new[2]:7.2f}°")
+ print()
+
+ # === COMPARE WITH UNITY'S smpl_incam_rot ===
+ if "smpl_incam_rot" in row:
+ unity_incam_quat = np.array(row["smpl_incam_rot"], dtype=np.float64)
+ R_unity_incam = R.from_quat(unity_incam_quat).as_matrix()
+ # Unity gives this in Unity convention, need to convert to CV
+ R_unity_incam_cv = C @ R_unity_incam @ C
+ euler_unity = R.from_matrix(R_unity_incam_cv).as_euler('YXZ', degrees=True)
+
+ print("--- UNITY'S EXPORTED smpl_incam_rot (converted to CV) ---")
+ print(f"global_orient_c (YXZ): yaw={euler_unity[0]:7.2f}°, pitch={euler_unity[1]:7.2f}°, roll={euler_unity[2]:7.2f}°")
+ print()
+
+ print("--- COMPARISON ---")
+ diff_old = euler_old - euler_unity
+ diff_new = euler_new - euler_unity
+ print(f"OLD vs Unity: yaw={diff_old[0]:7.2f}°, pitch={diff_old[1]:7.2f}°, roll={diff_old[2]:7.2f}°")
+ print(f"NEW vs Unity: yaw={diff_new[0]:7.2f}°, pitch={diff_new[1]:7.2f}°, roll={diff_new[2]:7.2f}°")
+
+ print()
+ print("="*70)
+ print("DIAGNOSIS:")
+ diff = euler_new - euler_old
+ if np.allclose(diff, 0, atol=1.0):
+ print(" Both methods give same result!")
+ print(" The 168° roll error must come from somewhere else:")
+ print(" - Check body_pose export")
+ print(" - Check if Unity's smpl_incam_rot matches derived rotation")
+ print(" - Check SMPL model forward pass")
+ else:
+ print(f" Methods differ: yaw={diff[0]:.2f}°, pitch={diff[1]:.2f}°, roll={diff[2]:.2f}°")
+ print(" This explains the rotation error!")
+
+if __name__ == "__main__":
+ test_actual_frame()
diff --git a/third_party/GVHMR/.gitattributes b/third_party/GVHMR/.gitattributes
new file mode 100644
index 0000000000000000000000000000000000000000..ef0af28e9cebdf4d5734aa46f6170783a1e545c0
--- /dev/null
+++ b/third_party/GVHMR/.gitattributes
@@ -0,0 +1,51 @@
+*.7z filter=lfs diff=lfs merge=lfs -text
+*.arrow filter=lfs diff=lfs merge=lfs -text
+*.bin filter=lfs diff=lfs merge=lfs -text
+*.bz2 filter=lfs diff=lfs merge=lfs -text
+*.ckpt filter=lfs diff=lfs merge=lfs -text
+*.ftz filter=lfs diff=lfs merge=lfs -text
+*.gz filter=lfs diff=lfs merge=lfs -text
+*.h5 filter=lfs diff=lfs merge=lfs -text
+*.joblib filter=lfs diff=lfs merge=lfs -text
+*.lfs.* filter=lfs diff=lfs merge=lfs -text
+*.mlmodel filter=lfs diff=lfs merge=lfs -text
+*.model filter=lfs diff=lfs merge=lfs -text
+*.msgpack filter=lfs diff=lfs merge=lfs -text
+*.npy filter=lfs diff=lfs merge=lfs -text
+*.npz filter=lfs diff=lfs merge=lfs -text
+*.onnx filter=lfs diff=lfs merge=lfs -text
+*.ot filter=lfs diff=lfs merge=lfs -text
+*.parquet filter=lfs diff=lfs merge=lfs -text
+*.pb filter=lfs diff=lfs merge=lfs -text
+*.pickle filter=lfs diff=lfs merge=lfs -text
+*.pkl filter=lfs diff=lfs merge=lfs -text
+*.pt filter=lfs diff=lfs merge=lfs -text
+*.pth filter=lfs diff=lfs merge=lfs -text
+*.rar filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+saved_model/**/* filter=lfs diff=lfs merge=lfs -text
+*.tar.* filter=lfs diff=lfs merge=lfs -text
+*.tar filter=lfs diff=lfs merge=lfs -text
+*.tflite filter=lfs diff=lfs merge=lfs -text
+*.tgz filter=lfs diff=lfs merge=lfs -text
+*.wasm filter=lfs diff=lfs merge=lfs -text
+*.xz filter=lfs diff=lfs merge=lfs -text
+*.zip filter=lfs diff=lfs merge=lfs -text
+*.zst filter=lfs diff=lfs merge=lfs -text
+*tfevents* filter=lfs diff=lfs merge=lfs -text
+assets/teaser.png filter=lfs diff=lfs merge=lfs -text
+test.mp4 filter=lfs diff=lfs merge=lfs -text
+test_10.mp4 filter=lfs diff=lfs merge=lfs -text
+test_11.mp4 filter=lfs diff=lfs merge=lfs -text
+test_6.mp4 filter=lfs diff=lfs merge=lfs -text
+third_party/DroidCalib/build/temp.linux-x86_64-3.10/.ninja_deps filter=lfs diff=lfs merge=lfs -text
+third_party/DroidCalib/build/temp.linux-x86_64-3.10/src/droid.o filter=lfs diff=lfs merge=lfs -text
+third_party/DroidCalib/misc/droidcalib.png filter=lfs diff=lfs merge=lfs -text
+third_party/DroidCalib/thirdparty/lietorch/examples/registration/assets/image1.png filter=lfs diff=lfs merge=lfs -text
+third_party/DroidCalib/thirdparty/lietorch/examples/registration/assets/image2.png filter=lfs diff=lfs merge=lfs -text
+third_party/DroidCalib/thirdparty/lietorch/examples/registration/assets/image3.png filter=lfs diff=lfs merge=lfs -text
+third_party/DroidCalib/thirdparty/lietorch/examples/registration/assets/image4.png filter=lfs diff=lfs merge=lfs -text
+third_party/DroidCalib/thirdparty/lietorch/examples/registration/assets/registration.gif filter=lfs diff=lfs merge=lfs -text
+third_party/DroidCalib/thirdparty/lietorch/examples/rgbdslam/assets/floor.png filter=lfs diff=lfs merge=lfs -text
+third_party/DroidCalib/thirdparty/lietorch/examples/rgbdslam/assets/room.png filter=lfs diff=lfs merge=lfs -text
+third_party/DroidCalib/thirdparty/lietorch/lietorch.png filter=lfs diff=lfs merge=lfs -text
diff --git a/third_party/GVHMR/.gitignore b/third_party/GVHMR/.gitignore
new file mode 100644
index 0000000000000000000000000000000000000000..8da58848fe56a37dc86adab3089c4376ae9faf6d
--- /dev/null
+++ b/third_party/GVHMR/.gitignore
@@ -0,0 +1,10 @@
+outputs/
+out/
+__pycache__/
+*.pyc
+processed_dataset/
+mmpose/
+gvhmr.egg-info/
+Grounded-SAM-2/
+.cache/
+third-party/
\ No newline at end of file
diff --git a/third_party/GVHMR/.gitmodules b/third_party/GVHMR/.gitmodules
new file mode 100644
index 0000000000000000000000000000000000000000..adbb2855c89b45f7654b85e43a7f83b11b1a78ae
--- /dev/null
+++ b/third_party/GVHMR/.gitmodules
@@ -0,0 +1,6 @@
+[submodule "third-party/DROID-SLAM/thirdparty/eigen"]
+ path = third-party/DROID-SLAM/thirdparty/eigen
+ url = https://gitlab.com/libeigen/eigen.git
+[submodule "third_party/GVHMR"]
+ path = third_party/GVHMR
+ url = git@github.com:zju3dv/GVHMR.git
diff --git a/third_party/GVHMR/1e-3 b/third_party/GVHMR/1e-3
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/third_party/GVHMR/4.50 b/third_party/GVHMR/4.50
new file mode 100644
index 0000000000000000000000000000000000000000..3e12fb120be2babe77c8e72c153ba3c9d20e374a
--- /dev/null
+++ b/third_party/GVHMR/4.50
@@ -0,0 +1,17 @@
+Requirement already satisfied: transformers in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (4.29.2)
+Requirement already satisfied: filelock in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (3.14.0)
+Requirement already satisfied: huggingface-hub<1.0,>=0.14.1 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (0.36.0)
+Requirement already satisfied: numpy>=1.17 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (1.23.5)
+Requirement already satisfied: packaging>=20.0 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (25.0)
+Requirement already satisfied: pyyaml>=5.1 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (6.0.3)
+Requirement already satisfied: regex!=2019.12.17 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (2025.11.3)
+Requirement already satisfied: requests in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (2.28.2)
+Requirement already satisfied: tokenizers!=0.11.3,<0.14,>=0.11.1 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (0.13.3)
+Requirement already satisfied: tqdm>=4.27 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (4.65.2)
+Requirement already satisfied: fsspec>=2023.5.0 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from huggingface-hub<1.0,>=0.14.1->transformers) (2025.12.0)
+Requirement already satisfied: typing-extensions>=3.7.4.3 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from huggingface-hub<1.0,>=0.14.1->transformers) (4.15.0)
+Requirement already satisfied: hf-xet<2.0.0,>=1.1.3 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from huggingface-hub<1.0,>=0.14.1->transformers) (1.2.0)
+Requirement already satisfied: charset-normalizer<4,>=2 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from requests->transformers) (3.4.4)
+Requirement already satisfied: idna<4,>=2.5 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from requests->transformers) (3.11)
+Requirement already satisfied: urllib3<1.27,>=1.21.1 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from requests->transformers) (1.26.20)
+Requirement already satisfied: certifi>=2017.4.17 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from requests->transformers) (2025.11.12)
diff --git a/third_party/GVHMR/=4.50 b/third_party/GVHMR/=4.50
new file mode 100644
index 0000000000000000000000000000000000000000..3e12fb120be2babe77c8e72c153ba3c9d20e374a
--- /dev/null
+++ b/third_party/GVHMR/=4.50
@@ -0,0 +1,17 @@
+Requirement already satisfied: transformers in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (4.29.2)
+Requirement already satisfied: filelock in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (3.14.0)
+Requirement already satisfied: huggingface-hub<1.0,>=0.14.1 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (0.36.0)
+Requirement already satisfied: numpy>=1.17 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (1.23.5)
+Requirement already satisfied: packaging>=20.0 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (25.0)
+Requirement already satisfied: pyyaml>=5.1 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (6.0.3)
+Requirement already satisfied: regex!=2019.12.17 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (2025.11.3)
+Requirement already satisfied: requests in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (2.28.2)
+Requirement already satisfied: tokenizers!=0.11.3,<0.14,>=0.11.1 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (0.13.3)
+Requirement already satisfied: tqdm>=4.27 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from transformers) (4.65.2)
+Requirement already satisfied: fsspec>=2023.5.0 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from huggingface-hub<1.0,>=0.14.1->transformers) (2025.12.0)
+Requirement already satisfied: typing-extensions>=3.7.4.3 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from huggingface-hub<1.0,>=0.14.1->transformers) (4.15.0)
+Requirement already satisfied: hf-xet<2.0.0,>=1.1.3 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from huggingface-hub<1.0,>=0.14.1->transformers) (1.2.0)
+Requirement already satisfied: charset-normalizer<4,>=2 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from requests->transformers) (3.4.4)
+Requirement already satisfied: idna<4,>=2.5 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from requests->transformers) (3.11)
+Requirement already satisfied: urllib3<1.27,>=1.21.1 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from requests->transformers) (1.26.20)
+Requirement already satisfied: certifi>=2017.4.17 in /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages (from requests->transformers) (2025.11.12)
diff --git a/third_party/GVHMR/LICENSE b/third_party/GVHMR/LICENSE
new file mode 100644
index 0000000000000000000000000000000000000000..0e17249dc9ad40f52cf22cc6eebddb34e1e4da0e
--- /dev/null
+++ b/third_party/GVHMR/LICENSE
@@ -0,0 +1,36 @@
+NVIDIA License
+
+1. Definitions
+
+“Licensor” means any person or entity that distributes its Work.
+“Work” means (a) the original work of authorship made available under this license, which may include software, documentation, or other files, and (b) any additions to or derivative works thereof that are made available under this license.
+The terms “reproduce,” “reproduction,” “derivative works,” and “distribution” have the meaning as provided under U.S. copyright law; provided, however, that for the purposes of this license, derivative works shall not include works that remain separable from, or merely link (or bind by name) to the interfaces of, the Work.
+Works are “made available” under this license by including in or with the Work either (a) a copyright notice referencing the applicability of this license to the Work, or (b) a copy of this license.
+
+2. License Grant
+
+2.1 Copyright Grant. Subject to the terms and conditions of this license, each Licensor grants to you a perpetual, worldwide, non-exclusive, royalty-free, copyright license to use, reproduce, prepare derivative works of, publicly display, publicly perform, sublicense and distribute its Work and any resulting derivative works in any form.
+
+3. Limitations
+
+3.1 Redistribution. You may reproduce or distribute the Work only if (a) you do so under this license, (b) you include a complete copy of this license with your distribution, and (c) you retain without modification any copyright, patent, trademark, or attribution notices that are present in the Work.
+
+3.2 Derivative Works. You may specify that additional or different terms apply to the use, reproduction, and distribution of your derivative works of the Work (“Your Terms”) only if (a) Your Terms provide that the use limitation in Section 3.3 applies to your derivative works, and (b) you identify the specific derivative works that are subject to Your Terms. Notwithstanding Your Terms, this license (including the redistribution requirements in Section 3.1) will continue to apply to the Work itself.
+
+3.3 Use Limitation. The Work and any derivative works thereof only may be used or intended for use non-commercially. Notwithstanding the foregoing, NVIDIA Corporation and its affiliates may use the Work and any derivative works commercially. As used herein, “non-commercially” means for non-commercial academic purposes only.
+
+3.4 Patent Claims. If you bring or threaten to bring a patent claim against any Licensor (including any claim, cross-claim or counterclaim in a lawsuit) to enforce any patents that you allege are infringed by any Work, then your rights under this license from such Licensor (including the grant in Section 2.1) will terminate immediately.
+
+3.5 Trademarks. This license does not grant any rights to use any Licensor’s or its affiliates’ names, logos, or trademarks, except as necessary to reproduce the notices described in this license.
+
+3.6 Termination. If you violate any term of this license, then your rights under this license (including the grant in Section 2.1) will terminate immediately.
+
+4. Disclaimer of Warranty.
+
+THE WORK IS PROVIDED “AS IS” WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, EITHER EXPRESS OR IMPLIED, INCLUDING WARRANTIES OR CONDITIONS OF
+MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, TITLE OR NON-INFRINGEMENT. YOU BEAR THE RISK OF UNDERTAKING ANY ACTIVITIES UNDER THIS LICENSE.
+
+5. Limitation of Liability.
+
+EXCEPT AS PROHIBITED BY APPLICABLE LAW, IN NO EVENT AND UNDER NO LEGAL THEORY, WHETHER IN TORT (INCLUDING NEGLIGENCE), CONTRACT, OR OTHERWISE SHALL ANY LICENSOR BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES ARISING OUT OF OR RELATED TO THIS LICENSE, THE USE OR INABILITY TO USE THE WORK (INCLUDING BUT NOT LIMITED TO LOSS OF GOODWILL, BUSINESS INTERRUPTION, LOST PROFITS OR DATA, COMPUTER FAILURE OR MALFUNCTION, OR ANY OTHER DAMAGES OR LOSSES), EVEN IF THE LICENSOR HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH DAMAGES.
+
diff --git a/third_party/GVHMR/README.md b/third_party/GVHMR/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..d89c2fed4218ac2626923fcb8061aea5e243bc88
--- /dev/null
+++ b/third_party/GVHMR/README.md
@@ -0,0 +1,75 @@
+
+
GEM: A Generalist Model for Human Motion
+
+ Jiefeng Li
+ ·
+ Jinkun Cao
+ ·
+ Haotian Zhang
+ ·
+ Davis Rempe
+ ·
+ Jan Kautz
+ ·
+ Umar Iqbal
+ ·
+ Ye Yuan
+
+ ICCV 2025 (Highlight)
+
+

+
+
+
+
+
+
+
+
+**GEM** is a generalist model for human motion that handles multiple tasks with a single model, supporting diverse conditioning signals including video, keypoints, text, audio, and 3D keyframes.
+
+---
+
+## 📰 News
+- **[December 2025]** 📢 GENMO has been renamed to **GEM**.
+- **[October 2025]** 📢 The **GEM** codebase is **released!**
+ Stay tuned for the pretrained models and evaluation scripts.
+ Follow the [project page](https://research.nvidia.com/labs/dair/gem/) for updates and announcements.
+
+
+---
+
+
+## 🚀 Highlights
+
+GEM introduces a **unified generative framework** that connects motion estimation and generation through shared objectives.
+
+- **Unified framework:** Reframes motion estimation as *constrained generation*, allowing a single model to perform both tasks.
+- **Regression × Diffusion synergy:** Combines the accuracy of regression models with the diversity of diffusion-based generation.
+- **Estimation-guided training:** Trains effectively on in-the-wild datasets using only 2D or textual supervision.
+- **Multimodal conditioning:** Supports video, text, audio, 2D/3D keyframes, or even time-varying mixed inputs (e.g., video → text → video).
+- **Arbitrary-length motion:** Generates continuous, coherent sequences of any duration in one diffusion pass.
+- **State-of-the-art performance:** Achieves leading results on diverse motion estimation and generation benchmarks.
+
+For more details, visit the **[GEM project page →](https://research.nvidia.com/labs/dair/gem/)**
+
+---
+
+### Pretrained Models
+You can download pretrained models from [Google Drive](https://drive.google.com/file/d/1b1E84G7S0h2n5o0RmrcmKOhRKukOjgsJ/view?usp=sharing).
+
+## 📖 Paper & Citation
+
+**Paper:**
+[GENMO: A GENeralist Model for Human MOtion](https://arxiv.org/abs/2505.01425)
+*Jiefeng Li, Jinkun Cao, Haotian Zhang, Davis Rempe, Jan Kautz, Umar Iqbal, Ye Yuan*
+ICCV, 2025
+
+**BibTeX:**
+```bibtex
+@inproceedings{genmo2025,
+ title = {GENMO: A GENeralist Model for Human MOtion},
+ author = {Li, Jiefeng and Cao, Jinkun and Zhang, Haotian and Rempe, Davis and Kautz, Jan and Iqbal, Umar and Yuan, Ye},
+ booktitle = {Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)},
+ year = {2025}
+}
diff --git a/third_party/GVHMR/UI/Inter_18pt-Bold.ttf b/third_party/GVHMR/UI/Inter_18pt-Bold.ttf
new file mode 100644
index 0000000000000000000000000000000000000000..a7877923ff1dafa5e986ed143084fffa35ab214e
--- /dev/null
+++ b/third_party/GVHMR/UI/Inter_18pt-Bold.ttf
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:30a5c45ec23a594af2effe8d3b589ad22c2dede27441050a1604d00ff82fd0dc
+size 344152
diff --git a/third_party/GVHMR/UI/Layer 1.png b/third_party/GVHMR/UI/Layer 1.png
new file mode 100644
index 0000000000000000000000000000000000000000..4a0aed267993584f2658889440f0a7b882709704
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 1.png differ
diff --git a/third_party/GVHMR/UI/Layer 10.png b/third_party/GVHMR/UI/Layer 10.png
new file mode 100644
index 0000000000000000000000000000000000000000..a60016e645a8f7e4bd72627960ca3461badbcaa7
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 10.png differ
diff --git a/third_party/GVHMR/UI/Layer 11.png b/third_party/GVHMR/UI/Layer 11.png
new file mode 100644
index 0000000000000000000000000000000000000000..2164f60b04f63c17f39dffc5936f2ef46daf9412
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 11.png differ
diff --git a/third_party/GVHMR/UI/Layer 12.png b/third_party/GVHMR/UI/Layer 12.png
new file mode 100644
index 0000000000000000000000000000000000000000..734479baddc162d1d3e736c912155394cfe64c33
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 12.png differ
diff --git a/third_party/GVHMR/UI/Layer 13.png b/third_party/GVHMR/UI/Layer 13.png
new file mode 100644
index 0000000000000000000000000000000000000000..5be4e995ae51a6398a9193c9c87face10e42338f
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 13.png differ
diff --git a/third_party/GVHMR/UI/Layer 14.png b/third_party/GVHMR/UI/Layer 14.png
new file mode 100644
index 0000000000000000000000000000000000000000..edb013377222d1d60c5e2b46c9f4805ae50b7035
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 14.png differ
diff --git a/third_party/GVHMR/UI/Layer 2.png b/third_party/GVHMR/UI/Layer 2.png
new file mode 100644
index 0000000000000000000000000000000000000000..fc59fee37e5d03190a5be9a7d7bbc85ee5c13832
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 2.png differ
diff --git a/third_party/GVHMR/UI/Layer 3.png b/third_party/GVHMR/UI/Layer 3.png
new file mode 100644
index 0000000000000000000000000000000000000000..456b928c799d596e7186b3eda21bc0d39db5d528
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 3.png differ
diff --git a/third_party/GVHMR/UI/Layer 4.png b/third_party/GVHMR/UI/Layer 4.png
new file mode 100644
index 0000000000000000000000000000000000000000..275e1405aebea4823d46256ca92e7ec7b7c4c3e8
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 4.png differ
diff --git a/third_party/GVHMR/UI/Layer 5.png b/third_party/GVHMR/UI/Layer 5.png
new file mode 100644
index 0000000000000000000000000000000000000000..46325a7c1e1dca2bbf16e2dbd5f6178b3db6b66d
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 5.png differ
diff --git a/third_party/GVHMR/UI/Layer 6.png b/third_party/GVHMR/UI/Layer 6.png
new file mode 100644
index 0000000000000000000000000000000000000000..4786e42a7e79aa2b36a086327db4c1c614be0107
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 6.png differ
diff --git a/third_party/GVHMR/UI/Layer 7.png b/third_party/GVHMR/UI/Layer 7.png
new file mode 100644
index 0000000000000000000000000000000000000000..2ec46e4cc3561c66e9c40b7cdb78b04ce9ad49b9
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 7.png differ
diff --git a/third_party/GVHMR/UI/Layer 8.png b/third_party/GVHMR/UI/Layer 8.png
new file mode 100644
index 0000000000000000000000000000000000000000..473e4742b794464988bba3a4b26a682b76a7ee93
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 8.png differ
diff --git a/third_party/GVHMR/UI/Layer 9.png b/third_party/GVHMR/UI/Layer 9.png
new file mode 100644
index 0000000000000000000000000000000000000000..23ef88518efbabdc885f6ff5183f4bf6ecf6af60
Binary files /dev/null and b/third_party/GVHMR/UI/Layer 9.png differ
diff --git a/third_party/GVHMR/configs/__init__.py b/third_party/GVHMR/configs/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..58e91e0e40b6205baf87e1440cda68b977471f15
--- /dev/null
+++ b/third_party/GVHMR/configs/__init__.py
@@ -0,0 +1,30 @@
+import argparse
+import os
+
+from hydra import compose, initialize_config_module
+from hydra.core.config_store import ConfigStore
+
+os.environ["HYDRA_FULL_ERROR"] = "1"
+
+MainStore = ConfigStore.instance()
+
+
+def parse_args_to_cfg():
+ """
+ Use minimal Hydra API to parse args and return cfg.
+ This function don't do _run_hydra which create log file hierarchy.
+ """
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--config-name", "-cn", default="train")
+ parser.add_argument(
+ "overrides",
+ nargs="*",
+ help="Any key=value arguments to override config values (use dots for.nested=overrides)",
+ )
+ args = parser.parse_args()
+
+ # Cfg
+ with initialize_config_module(version_base="1.3", config_module="configs"):
+ cfg = compose(config_name=args.config_name, overrides=args.overrides)
+
+ return cfg
diff --git a/third_party/GVHMR/configs/callbacks/ckpt_saver/every10000s_top100.yaml b/third_party/GVHMR/configs/callbacks/ckpt_saver/every10000s_top100.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..f379fee9f967423ae13730351ed5b9488377c4f5
--- /dev/null
+++ b/third_party/GVHMR/configs/callbacks/ckpt_saver/every10000s_top100.yaml
@@ -0,0 +1,5 @@
+every10000s_top100:
+ _target_: genmo.callbacks.simple_ckpt_saver.SimpleCkptSaver
+ output_dir: ${output_dir}/checkpoints/
+ every_n_steps: 10000
+ save_top_k: 100
diff --git a/third_party/GVHMR/configs/callbacks/lr_monitor/pl.yaml b/third_party/GVHMR/configs/callbacks/lr_monitor/pl.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..5d5416b7be73bd6207492e53f44e227aedd65094
--- /dev/null
+++ b/third_party/GVHMR/configs/callbacks/lr_monitor/pl.yaml
@@ -0,0 +1,2 @@
+pl:
+ _target_: pytorch_lightning.callbacks.lr_monitor.LearningRateMonitor
diff --git a/third_party/GVHMR/configs/callbacks/metric/metric_3dpw.yaml b/third_party/GVHMR/configs/callbacks/metric/metric_3dpw.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..6f8f71666b797a02584874597fc48834084c32cb
--- /dev/null
+++ b/third_party/GVHMR/configs/callbacks/metric/metric_3dpw.yaml
@@ -0,0 +1,2 @@
+metric_3dpw:
+ _target_: genmo.callbacks.metric.metric_3dpw.MetricMocap
diff --git a/third_party/GVHMR/configs/callbacks/metric/metric_3dpw_occ.yaml b/third_party/GVHMR/configs/callbacks/metric/metric_3dpw_occ.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..5aebfa3eb816153ecd8a6fb05e9e1018c491c310
--- /dev/null
+++ b/third_party/GVHMR/configs/callbacks/metric/metric_3dpw_occ.yaml
@@ -0,0 +1,2 @@
+metric_3dpw_occ:
+ _target_: genmo.callbacks.metric.metric_3dpw_occ.MetricMocap
diff --git a/third_party/GVHMR/dwpose_l_384_config.py b/third_party/GVHMR/dwpose_l_384_config.py
new file mode 100644
index 0000000000000000000000000000000000000000..293efad219b835650bf3659a33826c51b04cfad4
--- /dev/null
+++ b/third_party/GVHMR/dwpose_l_384_config.py
@@ -0,0 +1,257 @@
+#_base_ = ['../../../_base_/default_runtime.py']
+_base_ = ['default_runtime.py']
+
+# runtime
+max_epochs = 270
+stage2_num_epochs = 30
+base_lr = 4e-3
+train_batch_size = 32
+val_batch_size = 32
+
+train_cfg = dict(max_epochs=max_epochs, val_interval=10)
+randomness = dict(seed=21)
+
+# optimizer
+optim_wrapper = dict(
+ type='OptimWrapper',
+ optimizer=dict(type='AdamW', lr=base_lr, weight_decay=0.05),
+ paramwise_cfg=dict(
+ norm_decay_mult=0, bias_decay_mult=0, bypass_duplicate=True))
+
+# learning rate
+param_scheduler = [
+ dict(
+ type='LinearLR',
+ start_factor=1.0e-5,
+ by_epoch=False,
+ begin=0,
+ end=1000),
+ dict(
+ # use cosine lr from 150 to 300 epoch
+ type='CosineAnnealingLR',
+ eta_min=base_lr * 0.05,
+ begin=max_epochs // 2,
+ end=max_epochs,
+ T_max=max_epochs // 2,
+ by_epoch=True,
+ convert_to_iter_based=True),
+]
+
+# automatically scaling LR based on the actual training batch size
+auto_scale_lr = dict(base_batch_size=512)
+
+# codec settings
+codec = dict(
+ type='SimCCLabel',
+ input_size=(288, 384),
+ sigma=(6., 6.93),
+ simcc_split_ratio=2.0,
+ normalize=False,
+ use_dark=False)
+
+# model settings
+model = dict(
+ type='TopdownPoseEstimator',
+ data_preprocessor=dict(
+ type='PoseDataPreprocessor',
+ mean=[123.675, 116.28, 103.53],
+ std=[58.395, 57.12, 57.375],
+ bgr_to_rgb=True),
+ backbone=dict(
+ _scope_='mmdet',
+ type='CSPNeXt',
+ arch='P5',
+ expand_ratio=0.5,
+ deepen_factor=1.,
+ widen_factor=1.,
+ out_indices=(4, ),
+ channel_attention=True,
+ norm_cfg=dict(type='SyncBN'),
+ act_cfg=dict(type='SiLU'),
+ init_cfg=dict(
+ type='Pretrained',
+ prefix='backbone.',
+ checkpoint='https://download.openmmlab.com/mmpose/v1/projects/'
+ 'rtmpose/cspnext-l_udp-aic-coco_210e-256x192-273b7631_20230130.pth' # noqa: E501
+ )),
+ head=dict(
+ type='RTMCCHead',
+ in_channels=1024,
+ out_channels=133,
+ input_size=codec['input_size'],
+ in_featuremap_size=(9, 12),
+ simcc_split_ratio=codec['simcc_split_ratio'],
+ final_layer_kernel_size=7,
+ gau_cfg=dict(
+ hidden_dims=256,
+ s=128,
+ expansion_factor=2,
+ dropout_rate=0.,
+ drop_path=0.,
+ act_fn='SiLU',
+ use_rel_bias=False,
+ pos_enc=False),
+ loss=dict(
+ type='KLDiscretLoss',
+ use_target_weight=True,
+ beta=10.,
+ label_softmax=True),
+ decoder=codec),
+ test_cfg=dict(flip_test=True, ))
+
+# base dataset settings
+dataset_type = 'UBody2dDataset'
+data_mode = 'topdown'
+data_root = 'data/UBody/'
+
+backend_args = dict(backend='local')
+
+scenes = [
+ 'Magic_show', 'Entertainment', 'ConductMusic', 'Online_class', 'TalkShow',
+ 'Speech', 'Fitness', 'Interview', 'Olympic', 'TVShow', 'Singing',
+ 'SignLanguage', 'Movie', 'LiveVlog', 'VideoConference'
+]
+
+train_datasets = [
+ dict(
+ type='CocoWholeBodyDataset',
+ data_root='data/coco/',
+ data_mode=data_mode,
+ ann_file='annotations/coco_wholebody_train_v1.0.json',
+ data_prefix=dict(img='train2017/'),
+ pipeline=[])
+]
+
+for scene in scenes:
+ train_dataset = dict(
+ type=dataset_type,
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file=f'annotations/{scene}/train_annotations.json',
+ data_prefix=dict(img='images/'),
+ pipeline=[],
+ sample_interval=10)
+ train_datasets.append(train_dataset)
+
+# pipelines
+train_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='RandomFlip', direction='horizontal'),
+ dict(type='RandomHalfBody'),
+ dict(
+ type='RandomBBoxTransform', scale_factor=[0.5, 1.5], rotate_factor=90),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='mmdet.YOLOXHSVRandomAug'),
+ dict(
+ type='Albumentation',
+ transforms=[
+ dict(type='Blur', p=0.1),
+ dict(type='MedianBlur', p=0.1),
+ dict(
+ type='CoarseDropout',
+ max_holes=1,
+ max_height=0.4,
+ max_width=0.4,
+ min_holes=1,
+ min_height=0.2,
+ min_width=0.2,
+ p=1.0),
+ ]),
+ dict(type='GenerateTarget', encoder=codec),
+ dict(type='PackPoseInputs')
+]
+val_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='PackPoseInputs')
+]
+
+train_pipeline_stage2 = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='RandomFlip', direction='horizontal'),
+ dict(type='RandomHalfBody'),
+ dict(
+ type='RandomBBoxTransform',
+ shift_factor=0.,
+ scale_factor=[0.5, 1.5],
+ rotate_factor=90),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='mmdet.YOLOXHSVRandomAug'),
+ dict(
+ type='Albumentation',
+ transforms=[
+ dict(type='Blur', p=0.1),
+ dict(type='MedianBlur', p=0.1),
+ dict(
+ type='CoarseDropout',
+ max_holes=1,
+ max_height=0.4,
+ max_width=0.4,
+ min_holes=1,
+ min_height=0.2,
+ min_width=0.2,
+ p=0.5),
+ ]),
+ dict(type='GenerateTarget', encoder=codec),
+ dict(type='PackPoseInputs')
+]
+
+# data loaders
+train_dataloader = dict(
+ batch_size=train_batch_size,
+ num_workers=10,
+ persistent_workers=True,
+ sampler=dict(type='DefaultSampler', shuffle=True),
+ dataset=dict(
+ type='CombinedDataset',
+ metainfo=dict(from_file='configs/_base_/datasets/coco_wholebody.py'),
+ datasets=train_datasets,
+ pipeline=train_pipeline,
+ test_mode=False,
+ ))
+
+val_dataloader = dict(
+ batch_size=val_batch_size,
+ num_workers=10,
+ persistent_workers=True,
+ drop_last=False,
+ sampler=dict(type='DefaultSampler', shuffle=False, round_up=False),
+ dataset=dict(
+ type='CocoWholeBodyDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='data/coco/annotations/coco_wholebody_val_v1.0.json',
+ bbox_file='data/coco/person_detection_results/'
+ 'COCO_val2017_detections_AP_H_56_person.json',
+ data_prefix=dict(img='coco/val2017/'),
+ test_mode=True,
+ pipeline=val_pipeline,
+ ))
+test_dataloader = val_dataloader
+
+# hooks
+default_hooks = dict(
+ checkpoint=dict(
+ save_best='coco-wholebody/AP', rule='greater', max_keep_ckpts=1))
+
+custom_hooks = [
+ dict(
+ type='EMAHook',
+ ema_type='ExpMomentumEMA',
+ momentum=0.0002,
+ update_buffers=True,
+ priority=49),
+ dict(
+ type='mmdet.PipelineSwitchHook',
+ switch_epoch=max_epochs - stage2_num_epochs,
+ switch_pipeline=train_pipeline_stage2)
+]
+
+# evaluators
+val_evaluator = dict(
+ type='CocoWholeBodyMetric',
+ ann_file='data/coco/annotations/coco_wholebody_val_v1.0.json')
+test_evaluator = val_evaluator
diff --git a/third_party/GVHMR/dwpose_l_384_fixed.py b/third_party/GVHMR/dwpose_l_384_fixed.py
new file mode 100644
index 0000000000000000000000000000000000000000..2f39b6b9095d1cd76a2ad9e621bb41a9981f7efe
--- /dev/null
+++ b/third_party/GVHMR/dwpose_l_384_fixed.py
@@ -0,0 +1,254 @@
+
+# runtime
+max_epochs = 270
+stage2_num_epochs = 30
+base_lr = 4e-3
+train_batch_size = 32
+val_batch_size = 32
+
+train_cfg = dict(max_epochs=max_epochs, val_interval=10)
+randomness = dict(seed=21)
+
+# optimizer
+optim_wrapper = dict(
+ type='OptimWrapper',
+ optimizer=dict(type='AdamW', lr=base_lr, weight_decay=0.05),
+ paramwise_cfg=dict(
+ norm_decay_mult=0, bias_decay_mult=0, bypass_duplicate=True))
+
+# learning rate
+param_scheduler = [
+ dict(
+ type='LinearLR',
+ start_factor=1.0e-5,
+ by_epoch=False,
+ begin=0,
+ end=1000),
+ dict(
+ # use cosine lr from 150 to 300 epoch
+ type='CosineAnnealingLR',
+ eta_min=base_lr * 0.05,
+ begin=max_epochs // 2,
+ end=max_epochs,
+ T_max=max_epochs // 2,
+ by_epoch=True,
+ convert_to_iter_based=True),
+]
+
+# automatically scaling LR based on the actual training batch size
+auto_scale_lr = dict(base_batch_size=512)
+
+# codec settings
+codec = dict(
+ type='SimCCLabel',
+ input_size=(288, 384),
+ sigma=(6., 6.93),
+ simcc_split_ratio=2.0,
+ normalize=False,
+ use_dark=False)
+
+# model settings
+model = dict(
+ type='TopdownPoseEstimator',
+ data_preprocessor=dict(
+ type='PoseDataPreprocessor',
+ mean=[123.675, 116.28, 103.53],
+ std=[58.395, 57.12, 57.375],
+ bgr_to_rgb=True),
+ backbone=dict(
+ _scope_='mmdet',
+ type='CSPNeXt',
+ arch='P5',
+ expand_ratio=0.5,
+ deepen_factor=1.,
+ widen_factor=1.,
+ out_indices=(4, ),
+ channel_attention=True,
+ norm_cfg=dict(type='SyncBN'),
+ act_cfg=dict(type='SiLU'),
+ init_cfg=dict(
+ type='Pretrained',
+ prefix='backbone.',
+ checkpoint='https://download.openmmlab.com/mmpose/v1/projects/'
+ 'rtmpose/cspnext-l_udp-aic-coco_210e-256x192-273b7631_20230130.pth' # noqa: E501
+ )),
+ head=dict(
+ type='RTMCCHead',
+ in_channels=1024,
+ out_channels=133,
+ input_size=codec['input_size'],
+ in_featuremap_size=(9, 12),
+ simcc_split_ratio=codec['simcc_split_ratio'],
+ final_layer_kernel_size=7,
+ gau_cfg=dict(
+ hidden_dims=256,
+ s=128,
+ expansion_factor=2,
+ dropout_rate=0.,
+ drop_path=0.,
+ act_fn='SiLU',
+ use_rel_bias=False,
+ pos_enc=False),
+ loss=dict(
+ type='KLDiscretLoss',
+ use_target_weight=True,
+ beta=10.,
+ label_softmax=True),
+ decoder=codec),
+ test_cfg=dict(flip_test=True, ))
+
+# base dataset settings
+dataset_type = 'UBody2dDataset'
+data_mode = 'topdown'
+data_root = 'data/UBody/'
+
+backend_args = dict(backend='local')
+
+scenes = [
+ 'Magic_show', 'Entertainment', 'ConductMusic', 'Online_class', 'TalkShow',
+ 'Speech', 'Fitness', 'Interview', 'Olympic', 'TVShow', 'Singing',
+ 'SignLanguage', 'Movie', 'LiveVlog', 'VideoConference'
+]
+
+train_datasets = [
+ dict(
+ type='CocoWholeBodyDataset',
+ data_root='data/coco/',
+ data_mode=data_mode,
+ ann_file='annotations/coco_wholebody_train_v1.0.json',
+ data_prefix=dict(img='train2017/'),
+ pipeline=[])
+]
+
+for scene in scenes:
+ train_dataset = dict(
+ type=dataset_type,
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file=f'annotations/{scene}/train_annotations.json',
+ data_prefix=dict(img='images/'),
+ pipeline=[],
+ sample_interval=10)
+ train_datasets.append(train_dataset)
+
+# pipelines
+train_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='RandomFlip', direction='horizontal'),
+ dict(type='RandomHalfBody'),
+ dict(
+ type='RandomBBoxTransform', scale_factor=[0.5, 1.5], rotate_factor=90),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='mmdet.YOLOXHSVRandomAug'),
+ dict(
+ type='Albumentation',
+ transforms=[
+ dict(type='Blur', p=0.1),
+ dict(type='MedianBlur', p=0.1),
+ dict(
+ type='CoarseDropout',
+ max_holes=1,
+ max_height=0.4,
+ max_width=0.4,
+ min_holes=1,
+ min_height=0.2,
+ min_width=0.2,
+ p=1.0),
+ ]),
+ dict(type='GenerateTarget', encoder=codec),
+ dict(type='PackPoseInputs')
+]
+val_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='PackPoseInputs')
+]
+
+train_pipeline_stage2 = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='RandomFlip', direction='horizontal'),
+ dict(type='RandomHalfBody'),
+ dict(
+ type='RandomBBoxTransform',
+ shift_factor=0.,
+ scale_factor=[0.5, 1.5],
+ rotate_factor=90),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='mmdet.YOLOXHSVRandomAug'),
+ dict(
+ type='Albumentation',
+ transforms=[
+ dict(type='Blur', p=0.1),
+ dict(type='MedianBlur', p=0.1),
+ dict(
+ type='CoarseDropout',
+ max_holes=1,
+ max_height=0.4,
+ max_width=0.4,
+ min_holes=1,
+ min_height=0.2,
+ min_width=0.2,
+ p=0.5),
+ ]),
+ dict(type='GenerateTarget', encoder=codec),
+ dict(type='PackPoseInputs')
+]
+
+# data loaders
+train_dataloader = dict(
+ batch_size=train_batch_size,
+ num_workers=10,
+ persistent_workers=True,
+ sampler=dict(type='DefaultSampler', shuffle=True),
+ dataset=dict(
+ type='CombinedDataset',
+ datasets=train_datasets,
+ pipeline=train_pipeline,
+ test_mode=False,
+ ))
+
+val_dataloader = dict(
+ batch_size=val_batch_size,
+ num_workers=10,
+ persistent_workers=True,
+ drop_last=False,
+ sampler=dict(type='DefaultSampler', shuffle=False, round_up=False),
+ dataset=dict(
+ type='CocoWholeBodyDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='data/coco/annotations/coco_wholebody_val_v1.0.json',
+ bbox_file='data/coco/person_detection_results/'
+ 'COCO_val2017_detections_AP_H_56_person.json',
+ data_prefix=dict(img='coco/val2017/'),
+ test_mode=True,
+ pipeline=val_pipeline,
+ ))
+test_dataloader = val_dataloader
+
+# hooks
+default_hooks = dict(
+ checkpoint=dict(
+ save_best='coco-wholebody/AP', rule='greater', max_keep_ckpts=1))
+
+custom_hooks = [
+ dict(
+ type='EMAHook',
+ ema_type='ExpMomentumEMA',
+ momentum=0.0002,
+ update_buffers=True,
+ priority=49),
+ dict(
+ type='mmdet.PipelineSwitchHook',
+ switch_epoch=max_epochs - stage2_num_epochs,
+ switch_pipeline=train_pipeline_stage2)
+]
+
+# evaluators
+val_evaluator = dict(
+ type='CocoWholeBodyMetric',
+ ann_file='data/coco/annotations/coco_wholebody_val_v1.0.json')
+test_evaluator = val_evaluator
diff --git a/third_party/GVHMR/hmr4d/__init__.py b/third_party/GVHMR/hmr4d/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..ca7a686241b661a145c0c1041ff9aa311f92e7c2
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/__init__.py
@@ -0,0 +1,9 @@
+import os
+from pathlib import Path
+
+PROJ_ROOT = Path(__file__).resolve().parents[1]
+
+
+def os_chdir_to_proj_root():
+ """useful for running notebooks in different directories."""
+ os.chdir(PROJ_ROOT)
diff --git a/third_party/GVHMR/hmr4d/build_gvhmr.py b/third_party/GVHMR/hmr4d/build_gvhmr.py
new file mode 100644
index 0000000000000000000000000000000000000000..669e3af042b02bfbbf3fcc380d3784ec473b2bbe
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/build_gvhmr.py
@@ -0,0 +1,11 @@
+from omegaconf import OmegaConf
+from hmr4d import PROJ_ROOT
+from hydra.utils import instantiate
+from hmr4d.model.gvhmr.gvhmr_pl_demo import DemoPL
+
+
+def build_gvhmr_demo():
+ cfg = OmegaConf.load(PROJ_ROOT / "hmr4d/configs/demo_gvhmr_model/siga24_release.yaml")
+ gvhmr_demo_pl: DemoPL = instantiate(cfg.model, _recursive_=False)
+ gvhmr_demo_pl.load_pretrained_model(PROJ_ROOT / "inputs/checkpoints/gvhmr/gvhmr_siga24_release.ckpt")
+ return gvhmr_demo_pl.eval()
diff --git a/third_party/GVHMR/hmr4d/configs/__init__.py b/third_party/GVHMR/hmr4d/configs/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..3079c6e28d387296959f9a3c4b2e3adc636f6a69
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/__init__.py
@@ -0,0 +1,37 @@
+from dataclasses import dataclass
+from hydra.core.config_store import ConfigStore
+from hydra_zen import builds
+
+import argparse
+from hydra import compose, initialize_config_module
+import os
+
+os.environ["HYDRA_FULL_ERROR"] = "1"
+
+MainStore = ConfigStore.instance()
+
+
+def register_store_gvhmr():
+ """Register group options to MainStore"""
+ from . import store_gvhmr
+
+
+def parse_args_to_cfg():
+ """
+ Use minimal Hydra API to parse args and return cfg.
+ This function don't do _run_hydra which create log file hierarchy.
+ """
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--config-name", "-cn", default="train")
+ parser.add_argument(
+ "overrides",
+ nargs="*",
+ help="Any key=value arguments to override config values (use dots for.nested=overrides)",
+ )
+ args = parser.parse_args()
+
+ # Cfg
+ with initialize_config_module(version_base="1.3", config_module=f"hmr4d.configs"):
+ cfg = compose(config_name=args.config_name, overrides=args.overrides)
+
+ return cfg
diff --git a/third_party/GVHMR/hmr4d/configs/data/mocap/testY.yaml b/third_party/GVHMR/hmr4d/configs/data/mocap/testY.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..8dbf3c3923e4b6b4a9cb4d8c46f3c838b1a57f76
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/data/mocap/testY.yaml
@@ -0,0 +1,10 @@
+# definition of lightning datamodule (dataset + dataloader)
+_target_: hmr4d.datamodule.mocap_trainX_testY.DataModule
+
+dataset_opts:
+ test: ${test_datasets}
+
+loader_opts:
+ test:
+ batch_size: 1
+ num_workers: 0
diff --git a/third_party/GVHMR/hmr4d/configs/data/mocap/trainX_testY.yaml b/third_party/GVHMR/hmr4d/configs/data/mocap/trainX_testY.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..d512a787d4ffe820f8b686458a2742d2d3f9343a
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/data/mocap/trainX_testY.yaml
@@ -0,0 +1,16 @@
+# definition of lightning datamodule (dataset + dataloader)
+_target_: hmr4d.datamodule.mocap_trainX_testY.DataModule
+
+dataset_opts:
+ train: ${train_datasets}
+ val: ${test_datasets}
+
+loader_opts:
+ train:
+ batch_size: 32
+ num_workers: 8
+ val:
+ batch_size: 1
+ num_workers: 1
+
+limit_each_trainset: null
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/configs/demo.yaml b/third_party/GVHMR/hmr4d/configs/demo.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..a8f84d278f52fb694d87ef3324e710e57f554053
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/demo.yaml
@@ -0,0 +1,44 @@
+defaults:
+ - _self_
+ - model: gvhmr/gvhmr_pl_demo
+ - network: gvhmr/relative_transformer
+ - endecoder: gvhmr/v1_amass_local_bedlam_cam
+
+pipeline:
+ _target_: hmr4d.model.gvhmr.pipeline.gvhmr_pipeline.Pipeline
+ args_denoiser3d: ${network}
+ args:
+ endecoder_opt: ${endecoder}
+ normalize_cam_angvel: True
+ weights: null
+ static_conf: null
+
+ckpt_path: inputs/checkpoints/gvhmr/gvhmr_siga24_release.ckpt
+
+# ================================ #
+# global setting #
+# ================================ #
+
+video_name: ???
+output_root: outputs/demo
+output_dir: "${output_root}/${video_name}"
+preprocess_dir: ${output_dir}/preprocess
+video_path: "${output_dir}/0_input_video.mp4"
+
+# Options
+static_cam: False
+verbose: False
+use_dpvo: False
+f_mm: null # focal length of fullframe camera in mm
+
+paths:
+ bbx: ${preprocess_dir}/bbx.pt
+ bbx_xyxy_video_overlay: ${preprocess_dir}/bbx_xyxy_video_overlay.mp4
+ vit_features: ${preprocess_dir}/vit_features.pt
+ vitpose: ${preprocess_dir}/vitpose.pt
+ vitpose_video_overlay: ${preprocess_dir}/vitpose_video_overlay.mp4
+ hmr4d_results: ${output_dir}/hmr4d_results.pt
+ incam_video: ${output_dir}/1_incam.mp4
+ global_video: ${output_dir}/2_global.mp4
+ incam_global_horiz_video: ${output_dir}/${video_name}_3_incam_global_horiz.mp4
+ slam: ${preprocess_dir}/slam_results.pt
diff --git a/third_party/GVHMR/hmr4d/configs/exp/gvhmr/mixed/mixed.yaml b/third_party/GVHMR/hmr4d/configs/exp/gvhmr/mixed/mixed.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..9a6d36aa6cb4211097c22025cde77181c65b39e8
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/exp/gvhmr/mixed/mixed.yaml
@@ -0,0 +1,71 @@
+# @package _global_
+defaults:
+ - override /data: mocap/trainX_testY
+ - override /model: gvhmr/gvhmr_pl
+ - override /endecoder: gvhmr/v1_amass_local_bedlam_cam
+ - override /optimizer: adamw_2e-4
+ - override /scheduler_cfg: epoch_half_200_350
+ - override /train_datasets:
+ - pure_motion_amass/v11
+ - imgfeat_bedlam/v2
+ - imgfeat_h36m/v1
+ - imgfeat_3dpw/v1
+ - override /test_datasets:
+ - emdb1/v1_fliptest
+ - emdb2/v1_fliptest
+ - rich/all
+ - 3dpw/fliptest
+ - override /callbacks:
+ - simple_ckpt_saver/every10e_top100
+ - prog_bar/prog_reporter_every0.1
+ - train_speed_timer/base
+ - lr_monitor/pl
+ - metric_emdb1
+ - metric_emdb2
+ - metric_rich
+ - metric_3dpw
+ - override /network: gvhmr/relative_transformer
+
+exp_name_base: mixed
+exp_name_var: ""
+exp_name: ${exp_name_base}${exp_name_var}
+data_name: mocap_mixed_v1
+
+pipeline:
+ _target_: hmr4d.model.gvhmr.pipeline.gvhmr_pipeline.Pipeline
+ args_denoiser3d: ${network}
+ args:
+ endecoder_opt: ${endecoder}
+ normalize_cam_angvel: True
+ weights:
+ cr_j3d: 500.
+ transl_c: 1.
+ cr_verts: 500.
+ j2d: 1000.
+ verts2d: 1000.
+
+ transl_w: 1.
+ static_conf_bce: 1.
+
+ static_conf:
+ vel_thr: 0.15
+
+data:
+ loader_opts:
+ train:
+ batch_size: 128
+ num_workers: 12
+
+pl_trainer:
+ precision: 16-mixed
+ log_every_n_steps: 50
+ gradient_clip_val: 0.5
+ max_epochs: 500
+ check_val_every_n_epoch: 10
+ devices: 2
+
+logger:
+ _target_: pytorch_lightning.loggers.TensorBoardLogger
+ save_dir: ${output_dir} # /save_dir/name/version/sub_dir
+ name: ""
+ version: "tb" # merge name and version
diff --git a/third_party/GVHMR/hmr4d/configs/global/debug/debug_train.yaml b/third_party/GVHMR/hmr4d/configs/global/debug/debug_train.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..1545817e14b9322834b7a91bd90f229d042d5d5d
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/global/debug/debug_train.yaml
@@ -0,0 +1,24 @@
+# @package _global_
+
+data_name: debug
+exp_name: debug
+
+# data:
+# limit_each_trainset: 40
+# loader_opts:
+# train:
+# batch_size: 4
+# num_workers: 0
+# val:
+# batch_size: 1
+# num_workers: 0
+
+pl_trainer:
+ limit_train_batches: 32
+ limit_val_batches: 2
+ check_val_every_n_epoch: 3
+ enable_checkpointing: False
+ devices: 1
+
+callbacks:
+ model_checkpoint: null
diff --git a/third_party/GVHMR/hmr4d/configs/global/debug/debug_train_limit_data.yaml b/third_party/GVHMR/hmr4d/configs/global/debug/debug_train_limit_data.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..727cf234bd493c687b3c44a028fa3183250ffec9
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/global/debug/debug_train_limit_data.yaml
@@ -0,0 +1,23 @@
+# @package _global_
+
+data_name: debug
+exp_name: debug
+
+data:
+ limit_each_trainset: 40
+ loader_opts:
+ train:
+ batch_size: 4
+ num_workers: 0
+ val:
+ batch_size: 1
+ num_workers: 0
+
+pl_trainer:
+ limit_val_batches: 2
+ check_val_every_n_epoch: 3
+ enable_checkpointing: False
+ devices: 1
+
+callbacks:
+ model_checkpoint: null
diff --git a/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_3dpw.yaml b/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_3dpw.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..f820f9dec799f9a94d70bd9b312f9e367e3e257f
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_3dpw.yaml
@@ -0,0 +1,17 @@
+# @package _global_
+defaults:
+ - override /data: mocap/testY
+ - override /test_datasets:
+ - 3dpw/fliptest
+ - override /callbacks:
+ - metric_3dpw
+ - _self_
+
+task: test
+data_name: test_mocap
+ckpt_path: ??? # will not override previous setting if already set
+
+# lightning utilities
+pl_trainer:
+ devices: 1
+logger: null
diff --git a/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_3dpw_emdb_rich.yaml b/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_3dpw_emdb_rich.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..50021685ff112dd964fc68a028e5d6f4c164ca17
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_3dpw_emdb_rich.yaml
@@ -0,0 +1,23 @@
+# @package _global_
+defaults:
+ - override /data: mocap/testY
+ - override /test_datasets:
+ - rich/all
+ - emdb1/v1_fliptest
+ - emdb2/v1_fliptest
+ - 3dpw/fliptest
+ - override /callbacks:
+ - metric_rich
+ - metric_emdb1
+ - metric_emdb2
+ - metric_3dpw
+ - _self_
+
+task: test
+data_name: test_mocap
+ckpt_path: ??? # will not override previous setting if already set
+
+# lightning utilities
+pl_trainer:
+ devices: 1
+logger: null
diff --git a/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_emdb.yaml b/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_emdb.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..ff1e1c9cf10bd377ae180ef2d0d88940eea9f92e
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_emdb.yaml
@@ -0,0 +1,19 @@
+# @package _global_
+defaults:
+ - override /data: mocap/testY
+ - override /test_datasets:
+ - emdb1/v1_fliptest
+ - emdb2/v1_fliptest
+ - override /callbacks:
+ - metric_emdb1
+ - metric_emdb2
+ - _self_
+
+task: test
+data_name: test_mocap
+ckpt_path: ??? # will not override previous setting if already set
+
+# lightning utilities
+pl_trainer:
+ devices: 1
+logger: null
diff --git a/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_rich.yaml b/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_rich.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..923511b16b7775cea67d0f69c320a01b9c316a0c
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/global/task/gvhmr/test_rich.yaml
@@ -0,0 +1,17 @@
+# @package _global_
+defaults:
+ - override /data: mocap/testY
+ - override /test_datasets:
+ - rich/all
+ - override /callbacks:
+ - metric_rich
+ - _self_
+
+task: test
+data_name: test_mocap
+ckpt_path: ??? # will not override previous setting if already set
+
+# lightning utilities
+pl_trainer:
+ devices: 1
+logger: null
diff --git a/third_party/GVHMR/hmr4d/configs/hydra/default.yaml b/third_party/GVHMR/hmr4d/configs/hydra/default.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..619a4b46a2eeeee33b4d238a3a7d46411fb6f808
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/hydra/default.yaml
@@ -0,0 +1,19 @@
+# enable color logging
+defaults:
+ - override hydra_logging: colorlog
+ - override job_logging: colorlog
+
+job_logging:
+ formatters:
+ simple:
+ datefmt: '%m/%d %H:%M:%S'
+ format: '[%(asctime)s][%(levelname)s] %(message)s'
+ colorlog:
+ datefmt: '%m/%d %H:%M:%S'
+ format: '[%(cyan)s%(asctime)s%(reset)s][%(log_color)s%(levelname)s%(reset)s] %(message)s'
+ handlers:
+ file:
+ filename: ${output_dir}/${hydra.job.name}.log
+
+run:
+ dir: ${output_dir}
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/configs/siga24_release.yaml b/third_party/GVHMR/hmr4d/configs/siga24_release.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..d664542d9378442f130ce3c9c73629953f6b9db7
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/siga24_release.yaml
@@ -0,0 +1,35 @@
+pipeline:
+ _target_: hmr4d.model.gvhmr.pipeline.gvhmr_pipeline.Pipeline
+ args_denoiser3d: ${network}
+ args:
+ endecoder_opt: ${endecoder}
+ normalize_cam_angvel: true
+ weights: null
+ static_conf: null
+model:
+ _target_: hmr4d.model.gvhmr.gvhmr_pl_demo.DemoPL
+ pipeline: ${pipeline}
+network:
+ _target_: hmr4d.network.gvhmr.relative_transformer.NetworkEncoderRoPEV2
+ output_dim: 151
+ max_len: 120
+ kp2d_mapping: linear_v2
+ cliffcam_dim: 3
+ cam_angvel_dim: 6
+ imgseq_dim: 1024
+ f_imgseq_filter: null
+ cond_ver: v1
+ latent_dim: 512
+ num_layers: 12
+ num_heads: 8
+ mlp_ratio: 4.0
+ pred_cam_ver: v2
+ pred_cam_dim: 3
+ static_conf_dim: 6
+ pred_coco17_dim: 0
+ dropout: 0.1
+ avgbeta: true
+endecoder:
+ _target_: hmr4d.model.gvhmr.utils.endecoder.EnDecoder
+ stats_name: MM_V1_AMASS_LOCAL_BEDLAM_CAM
+ noise_pose_k: 10
diff --git a/third_party/GVHMR/hmr4d/configs/store_gvhmr.py b/third_party/GVHMR/hmr4d/configs/store_gvhmr.py
new file mode 100644
index 0000000000000000000000000000000000000000..b038911e3d065cddfbba57605a5420f9b4387708
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/store_gvhmr.py
@@ -0,0 +1,29 @@
+# Dataset
+import hmr4d.dataset.pure_motion.amass
+import hmr4d.dataset.emdb.emdb_motion_test
+import hmr4d.dataset.rich.rich_motion_test
+import hmr4d.dataset.threedpw.threedpw_motion_test
+import hmr4d.dataset.threedpw.threedpw_motion_train
+import hmr4d.dataset.bedlam.bedlam
+import hmr4d.dataset.h36m.h36m
+
+# Trainer: Model Optimizer Loss
+import hmr4d.model.gvhmr.gvhmr_pl
+import hmr4d.model.gvhmr.utils.endecoder
+import hmr4d.model.common_utils.optimizer
+import hmr4d.model.common_utils.scheduler_cfg
+
+# Metric
+import hmr4d.model.gvhmr.callbacks.metric_emdb
+import hmr4d.model.gvhmr.callbacks.metric_rich
+import hmr4d.model.gvhmr.callbacks.metric_3dpw
+
+
+# PL Callbacks
+import hmr4d.utils.callbacks.simple_ckpt_saver
+import hmr4d.utils.callbacks.train_speed_timer
+import hmr4d.utils.callbacks.prog_bar
+import hmr4d.utils.callbacks.lr_monitor
+
+# Networks
+import hmr4d.network.gvhmr.relative_transformer
diff --git a/third_party/GVHMR/hmr4d/configs/train.yaml b/third_party/GVHMR/hmr4d/configs/train.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..ee1ff159e4244d2a43f63f3e9af1c8cbc7a754f4
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/configs/train.yaml
@@ -0,0 +1,52 @@
+# ================================ #
+# override #
+# ================================ #
+# specify default configuration; the order determines the override order
+defaults:
+ - _self_
+ # pytorch-lightning
+ - data: ???
+ - model: ???
+ - callbacks: null
+
+ # system
+ - hydra: default
+
+ # utility groups that changes a lot
+ - pipeline: null
+ - network: null
+ - optimizer: null
+ - scheduler_cfg: default
+ - train_datasets: null
+ - test_datasets: null
+ - endecoder: null # normalize/unnormalize data
+ - refiner: null
+
+ # global-override
+ - exp: ??? # set "data, model and callbacks" in yaml
+ - global/task: null # dump/test
+ - global/hsearch: null # hyper-param search
+ - global/debug: null # debug mode
+
+# ================================ #
+# global setting #
+# ================================ #
+# expirement information
+task: fit # [fit, predict]
+exp_name: ???
+data_name: ???
+
+# utilities in the entry file
+output_dir: "outputs/${data_name}/${exp_name}"
+ckpt_path: null
+resume_mode: null
+seed: 42
+
+# lightning default settings
+pl_trainer:
+ devices: 1
+ num_sanity_val_steps: 0 # disable sanity check
+ precision: 32
+ inference_mode: False
+
+logger: null
diff --git a/third_party/GVHMR/hmr4d/datamodule/mocap_trainX_testY.py b/third_party/GVHMR/hmr4d/datamodule/mocap_trainX_testY.py
new file mode 100644
index 0000000000000000000000000000000000000000..af04744d3b621b6e5340b6acc2abcc05f20880a0
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/datamodule/mocap_trainX_testY.py
@@ -0,0 +1,130 @@
+import pytorch_lightning as pl
+from pytorch_lightning.utilities.combined_loader import CombinedLoader
+from hydra.utils import instantiate
+from torch.utils.data import DataLoader, ConcatDataset, Subset
+from omegaconf import ListConfig, DictConfig
+from hmr4d.utils.pylogger import Log
+from numpy.random import choice
+from torch.utils.data import default_collate
+
+
+import resource
+
+rlimit = resource.getrlimit(resource.RLIMIT_NOFILE)
+resource.setrlimit(resource.RLIMIT_NOFILE, (4096, rlimit[1]))
+
+
+def collate_fn(batch):
+ """Handle meta and Add batch size to the return dict
+ Args:
+ batch: list of dict, each dict is a data point
+ """
+ # Assume all keys in the batch are the same
+ return_dict = {}
+ for k in batch[0].keys():
+ if k.startswith("meta"): # data information, do not batch
+ return_dict[k] = [d[k] for d in batch]
+ else:
+ return_dict[k] = default_collate([d[k] for d in batch])
+ return_dict["B"] = len(batch)
+ return return_dict
+
+
+class DataModule(pl.LightningDataModule):
+ def __init__(self, dataset_opts: DictConfig, loader_opts: DictConfig, limit_each_trainset=None):
+ """This is a general datamodule that can be used for any dataset.
+ Train uses ConcatDataset
+ Val and Test use CombinedLoader, sequential, completely consumes ecah iterable sequentially, and returns a triplet (data, idx, iterable_idx)
+
+ Args:
+ dataset_opts: the target of the dataset. e.g. dataset_opts.train = {_target_: ..., limit_size: None}
+ loader_opts: the options for the dataset
+ limit_each_trainset: limit the size of each dataset, None means no limit, useful for debugging
+ """
+ super().__init__()
+ self.loader_opts = loader_opts
+ self.limit_each_trainset = limit_each_trainset
+
+ # Train uses concat dataset
+ if "train" in dataset_opts:
+ assert "train" in self.loader_opts, "train not in loader_opts"
+ split_opts = dataset_opts.get("train")
+ assert isinstance(split_opts, DictConfig), "split_opts should be a dict for each dataset"
+ dataset = []
+ dataset_num = len(split_opts)
+ for idx, (k, v) in enumerate(split_opts.items()):
+ dataset_i = instantiate(v)
+ if self.limit_each_trainset:
+ dataset_i = Subset(dataset_i, choice(len(dataset_i), self.limit_each_trainset))
+ dataset.append(dataset_i)
+ Log.info(f"[Train Dataset][{idx+1}/{dataset_num}]: name={k}, size={len(dataset[-1])}, {v._target_}")
+ dataset = ConcatDataset(dataset)
+ self.trainset = dataset
+ Log.info(f"[Train Dataset][All]: ConcatDataset size={len(dataset)}")
+ Log.info(f"")
+
+ # Val and Test use sequential dataset
+ for split in ("val", "test"):
+ if split not in dataset_opts:
+ continue
+ assert split in self.loader_opts, f"split={split} not in loader_opts"
+ split_opts = dataset_opts.get(split)
+ assert isinstance(split_opts, DictConfig), "split_opts should be a dict for each dataset"
+ dataset = []
+ dataset_num = len(split_opts)
+ for idx, (k, v) in enumerate(split_opts.items()):
+ dataset.append(instantiate(v))
+ dataset_type = "Val Dataset" if split == "val" else "Test Dataset"
+ Log.info(f"[{dataset_type}][{idx+1}/{dataset_num}]: name={k}, size={len(dataset[-1])}, {v._target_}")
+ setattr(self, f"{split}sets", dataset)
+ Log.info(f"")
+
+ def train_dataloader(self):
+ if hasattr(self, "trainset"):
+ return DataLoader(
+ self.trainset,
+ shuffle=True,
+ num_workers=self.loader_opts.train.num_workers,
+ persistent_workers=True and self.loader_opts.train.num_workers > 0,
+ batch_size=self.loader_opts.train.batch_size,
+ drop_last=True,
+ collate_fn=collate_fn,
+ )
+ else:
+ return super().train_dataloader()
+
+ def val_dataloader(self):
+ if hasattr(self, "valsets"):
+ loaders = []
+ for valset in self.valsets:
+ loaders.append(
+ DataLoader(
+ valset,
+ shuffle=False,
+ num_workers=self.loader_opts.val.num_workers,
+ persistent_workers=True and self.loader_opts.val.num_workers > 0,
+ batch_size=self.loader_opts.val.batch_size,
+ collate_fn=collate_fn,
+ )
+ )
+ return CombinedLoader(loaders, mode="sequential")
+ else:
+ return None
+
+ def test_dataloader(self):
+ if hasattr(self, "testsets"):
+ loaders = []
+ for testset in self.testsets:
+ loaders.append(
+ DataLoader(
+ testset,
+ shuffle=False,
+ num_workers=self.loader_opts.test.num_workers,
+ persistent_workers=False,
+ batch_size=self.loader_opts.test.batch_size,
+ collate_fn=collate_fn,
+ )
+ )
+ return CombinedLoader(loaders, mode="sequential")
+ else:
+ return super().test_dataloader()
diff --git a/third_party/GVHMR/hmr4d/dataset/bedlam/bedlam.py b/third_party/GVHMR/hmr4d/dataset/bedlam/bedlam.py
new file mode 100644
index 0000000000000000000000000000000000000000..52b75efc6003f56dbe0688df9e364b5552a4fe62
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/bedlam/bedlam.py
@@ -0,0 +1,251 @@
+from pathlib import Path
+import numpy as np
+import torch
+from hmr4d.utils.pylogger import Log
+from pytorch3d.transforms import axis_angle_to_matrix, matrix_to_axis_angle
+from time import time
+
+from hmr4d.configs import MainStore, builds
+from hmr4d.utils.smplx_utils import make_smplx
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines
+from hmr4d.utils.vis.renderer_utils import simple_render_mesh_background
+from hmr4d.utils.video_io_utils import read_video_np, save_video
+
+import hmr4d.utils.matrix as matrix
+from hmr4d.utils.net_utils import get_valid_mask, repeat_to_max_len, repeat_to_max_len_dict
+from hmr4d.dataset.imgfeat_motion.base_dataset import ImgfeatMotionDatasetBase
+from hmr4d.dataset.bedlam.utils import mid2featname, mid2vname
+from hmr4d.utils.geo_transform import compute_cam_angvel, apply_T_on_points
+from hmr4d.utils.geo.hmr_global import get_T_w2c_from_wcparams, get_c_rootparam, get_R_c2gv
+
+
+class BedlamDatasetV2(ImgfeatMotionDatasetBase):
+ """mid_to_valid_range and features are newly generated."""
+
+ MIDINDEX_TO_LOAD = {
+ "all60": ("mid_to_valid_range_all60.pt", "imgfeats/bedlam_all60"),
+ "maxspan60": ("mid_to_valid_range_maxspan60.pt", "imgfeats/bedlam_maxspan60"),
+ }
+
+ def __init__(
+ self,
+ mid_indices=["all60", "maxspan60"],
+ lazy_load=True, # Load from disk when needed
+ random1024=False, # Faster loading for debugging
+ ):
+ self.root = Path("inputs/BEDLAM/hmr4d_support")
+ self.min_motion_frames = 60
+ self.max_motion_frames = 120
+ self.lazy_load = lazy_load
+ self.random1024 = random1024
+
+ # speficify mid_index to handle
+ if not isinstance(mid_indices, list):
+ mid_indices = [mid_indices]
+ self.mid_indices = mid_indices
+ assert all([m in self.MIDINDEX_TO_LOAD for m in mid_indices])
+
+ super().__init__()
+
+ def _load_dataset(self):
+ Log.info(f"[BEDLAM] Loading from {self.root}")
+ tic = time()
+ # Load mid to valid range
+ self.mid_to_valid_range = {}
+ self.mid_to_imgfeat_dir = {}
+ for m in self.mid_indices:
+ fn, feat_dir = self.MIDINDEX_TO_LOAD[m]
+ mid_to_valid_range_ = torch.load(self.root / fn)
+ self.mid_to_valid_range.update(mid_to_valid_range_)
+ self.mid_to_imgfeat_dir.update({mid: self.root / feat_dir for mid in mid_to_valid_range_})
+
+ # Load motionfiles
+ Log.info(f"[BEDLAM] Start loading motion files")
+ if self.random1024: # Debug, faster loading
+ try:
+ Log.info(f"[BEDLAM] Loading 1024 samples for debugging ...")
+ self.motion_files = torch.load(self.root / "smplpose_v2_random1024.pth")
+ except:
+ Log.info(f"[BEDLAM] Not found, saving 1024 samples to disk ...")
+ self.motion_files = torch.load(self.root / "smplpose_v2.pth")
+ keys = list(self.motion_files.keys())
+ keys = np.random.choice(keys, 1024, replace=False)
+ self.motion_files = {k: self.motion_files[k] for k in keys}
+ torch.save(self.motion_files, self.root / "smplpose_v2_random1024.pth")
+ self.mid_to_valid_range = {k: v for k, v in self.mid_to_valid_range.items() if k in self.motion_files}
+ else:
+ self.motion_files = torch.load(self.root / "smplpose_v2.pth")
+ Log.info(f"[BEDLAM] Motion files loaded. Elapsed: {time() - tic:.2f}s")
+
+ def _get_idx2meta(self):
+ # sum_frame = sum([e-s for s, e in self.mid_to_valid_range.values()])
+ self.idx2meta = list(self.mid_to_valid_range.keys())
+ Log.info(f"[BEDLAM] {len(self.idx2meta)} sequences. ")
+
+ def _load_data(self, idx):
+ mid = self.idx2meta[idx]
+ # neutral smplx : "pose": (F, 63), "trans": (F, 3), "beta": (10),
+ # and : "skeleton": (J, 3)
+ data = self.motion_files[mid].copy()
+
+ # Random select a subset
+ range1, range2 = self.mid_to_valid_range[mid] # [range1, range2)
+ mlength = range2 - range1
+ min_motion_len = self.min_motion_frames
+ max_motion_len = self.max_motion_frames
+
+ if mlength < min_motion_len: # the minimal mlength is 30 when generating data
+ start = range1
+ length = mlength
+ else:
+ effect_max_motion_len = min(max_motion_len, mlength)
+ length = np.random.randint(min_motion_len, effect_max_motion_len + 1) # [low, high)
+ start = np.random.randint(range1, range2 - length + 1)
+ end = start + length
+ data["start_end"] = (start, end)
+ data["length"] = length
+
+ # Update data to a subset
+ for k, v in data.items():
+ if isinstance(v, torch.Tensor) and len(v.shape) > 1 and k != "skeleton":
+ data[k] = v[start:end]
+
+ # Load img(as feature) : {mid -> 'features', 'bbx_xys', 'img_wh', 'start_end'}
+ imgfeat_dir = self.mid_to_imgfeat_dir[mid]
+ f_img_dict = torch.load(imgfeat_dir / mid2featname(mid))
+
+ # remap (start, end)
+ start_mapped = start - f_img_dict["start_end"][0]
+ end_mapped = end - f_img_dict["start_end"][0]
+
+ data["f_imgseq"] = f_img_dict["features"][start_mapped:end_mapped].float() # (L, 1024)
+ data["bbx_xys"] = f_img_dict["bbx_xys"][start_mapped:end_mapped].float() # (L, 4)
+ data["img_wh"] = f_img_dict["img_wh"] # (2)
+ data["kp2d"] = torch.zeros((end - start), 17, 3) # (L, 17, 3) # do not provide kp2d
+
+ return data
+
+ def _process_data(self, data, idx):
+ length = data["length"]
+
+ # SMPL params in cam
+ body_pose = data["pose"][:, 3:] # (F, 63)
+ betas = data["beta"].repeat(length, 1) # (F, 10)
+ global_orient = data["global_orient_incam"] # (F, 3)
+ transl = data["trans_incam"] + data["cam_ext"][:, :3, 3] # (F, 3), bedlam convention
+ smpl_params_c = {"body_pose": body_pose, "betas": betas, "transl": transl, "global_orient": global_orient}
+
+ # SMPL params in world
+ global_orient_w = data["pose"][:, :3] # (F, 3)
+ transl_w = data["trans"] # (F, 3)
+ smpl_params_w = {"body_pose": body_pose, "betas": betas, "transl": transl_w, "global_orient": global_orient_w}
+
+ gravity_vec = torch.tensor([0, -1, 0], dtype=torch.float32) # (3), BEDLAM is ay
+ T_w2c = get_T_w2c_from_wcparams(
+ global_orient_w=global_orient_w,
+ transl_w=transl_w,
+ global_orient_c=global_orient,
+ transl_c=transl,
+ offset=data["skeleton"][0],
+ ) # (F, 4, 4)
+ R_c2gv = get_R_c2gv(T_w2c[:, :3, :3], gravity_vec) # (F, 3, 3)
+
+ # cam_angvel (slightly different from WHAM)
+ cam_angvel = compute_cam_angvel(T_w2c[:, :3, :3]) # (F, 6)
+
+ # Returns: do not forget to make it batchable! (last lines)
+ max_len = self.max_motion_frames
+ return_data = {
+ "meta": {"data_name": "bedlam", "idx": idx},
+ "length": length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w,
+ "R_c2gv": R_c2gv, # (F, 3, 3)
+ "gravity_vec": gravity_vec, # (3)
+ "bbx_xys": data["bbx_xys"], # (F, 3)
+ "K_fullimg": data["cam_int"], # (F, 3, 3)
+ "f_imgseq": data["f_imgseq"], # (F, D)
+ "kp2d": data["kp2d"], # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "mask": {
+ "valid": get_valid_mask(max_len, length),
+ "vitpose": False,
+ "bbx_xys": True,
+ "f_imgseq": True,
+ "spv_incam_only": False,
+ },
+ }
+
+ if False: # check transformation, wis3d: sampled motion (global, incam)
+ wis3d = make_wis3d(name="debug-data-bedlam")
+ smplx = make_smplx("supermotion")
+
+ # global
+ smplx_out = smplx(**smpl_params_w)
+ w_gt_joints = smplx_out.joints
+ add_motion_as_lines(w_gt_joints, wis3d, name="w-gt_joints")
+
+ # incam
+ smplx_out = smplx(**smpl_params_c)
+ c_gt_joints = smplx_out.joints
+ add_motion_as_lines(c_gt_joints, wis3d, name="c-gt_joints")
+
+ # Check transformation works correctly
+ print("T_w2c", (apply_T_on_points(w_gt_joints, T_w2c) - c_gt_joints).abs().max())
+ R_c, t_c = get_c_rootparam(
+ smpl_params_w["global_orient"], smpl_params_w["transl"], T_w2c, data["skeleton"][0]
+ )
+ print("transl_c", (t_c - smpl_params_c["transl"]).abs().max())
+ R_diff = matrix_to_axis_angle(
+ (axis_angle_to_matrix(R_c) @ axis_angle_to_matrix(smpl_params_c["global_orient"]).transpose(-1, -2))
+ ).norm(dim=-1)
+ print("global_orient_c", R_diff.abs().max()) # < 1e-6
+
+ skeleton_beta = smplx.get_skeleton(smpl_params_c["betas"])
+ print("Skeleton", (skeleton_beta[0] - data["skeleton"]).abs().max()) # (1.2e-7)
+
+ if False: # cam-overlay
+ smplx = make_smplx("supermotion")
+
+ # *. original bedlam param
+ # mid = self.idx2meta[idx]
+ # video_path = "-".join(mid.replace("bedlam_data/", "inputs/bedlam/").split("-")[:-1])
+ # npz_file = "inputs/bedlam/processed_labels/20221024_3-10_100_batch01handhair_static_highSchoolGym.npz"
+ # params = np.load(npz_file, allow_pickle=True)
+ # mid2index = {}
+ # for j in tqdm(range(len(params["video_name"]))):
+ # k = params["video_name"][j] + "-" + params["sub"][j]
+ # mid2index[k] = j
+ # betas = params['shape'][mid2index[mid]][:length]
+ # global_orient_incam = torch.from_numpy(params['pose_cam'][121][:, :3])
+ # body_pose = torch.from_numpy(params['pose_cam'][121][:, 3:66])
+ # transl_incam = torch.from_numpy(params["trans_cam"][121])
+ smplx_out = smplx(**smpl_params_c)
+
+ # ----- Render Overlay ----- #
+ mid = self.idx2meta[idx]
+ images = read_video_np(self.root / "videos" / mid2vname(mid), data["start_end"][0], data["start_end"][1])
+ render_dict = {
+ "K": data["cam_int"][:1], # only support batch-size 1
+ "faces": smplx.faces,
+ "verts": smplx_out.vertices,
+ "background": images,
+ }
+ img_overlay = simple_render_mesh_background(render_dict)
+ save_video(img_overlay, "tmp.mp4", crf=23)
+
+ # Batchable
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(return_data["smpl_params_c"], max_len)
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(return_data["smpl_params_w"], max_len)
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["bbx_xys"] = repeat_to_max_len(return_data["bbx_xys"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(return_data["K_fullimg"], max_len)
+ return_data["f_imgseq"] = repeat_to_max_len(return_data["f_imgseq"], max_len)
+ return_data["kp2d"] = repeat_to_max_len(return_data["kp2d"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(return_data["cam_angvel"], max_len)
+ return return_data
+
+
+group_name = "train_datasets/imgfeat_bedlam"
+MainStore.store(name="v2", node=builds(BedlamDatasetV2), group=group_name)
+MainStore.store(name="v2_random1024", node=builds(BedlamDatasetV2, random1024=True), group=group_name)
diff --git a/third_party/GVHMR/hmr4d/dataset/bedlam/resource/vname2lwh.pt b/third_party/GVHMR/hmr4d/dataset/bedlam/resource/vname2lwh.pt
new file mode 100644
index 0000000000000000000000000000000000000000..9b73acc4b801b642d63133536204db9ba0bb7053
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/bedlam/resource/vname2lwh.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7b81e1b70f79872a89fb5c3df2296f7d8d7f5f1baf1f03afdb2c2046323486d4
+size 840936
diff --git a/third_party/GVHMR/hmr4d/dataset/bedlam/utils.py b/third_party/GVHMR/hmr4d/dataset/bedlam/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..ca6b8d5a96d1496b9016e2c4a07e00f81f344e3e
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/bedlam/utils.py
@@ -0,0 +1,39 @@
+import torch
+import numpy as np
+from pathlib import Path
+
+resource_dir = Path(__file__).parent / "resource"
+
+
+def mid2vname(mid):
+ """vname = {scene}/{seq}, Note that it ends with .mp4"""
+ # mid example: "inputs/bedlam/bedlam_download/20221011_1_250_batch01hand_closeup_suburb_a/mp4/seq_000001.mp4-rp_emma_posed_008"
+ # -> vname: 20221011_1_250_batch01hand_closeup_suburb_a/seq_000001.mp4
+ scene = mid.split("/")[-3]
+ seq = mid.split("/")[-1].split("-")[0]
+ vname = f"{scene}/{seq}"
+ return vname
+
+
+def mid2featname(mid):
+ """featname = {scene}/{seqsubj}, Note that it ends with .pt (extra)"""
+ # mid example: "inputs/bedlam/bedlam_download/20221011_1_250_batch01hand_closeup_suburb_a/mp4/seq_000001.mp4-rp_emma_posed_008"
+ # -> featname: 20221011_1_250_batch01hand_closeup_suburb_a/seq_000001.mp4-rp_emma_posed_008.pt
+ scene = mid.split("/")[-3]
+ seqsubj = mid.split("/")[-1]
+ featname = f"{scene}/{seqsubj}.pt"
+ return featname
+
+
+def featname2mid(featname):
+ """reverse func of mid2featname, Note that it removes .pt (extra)"""
+ # featname example: 20221011_1_250_batch01hand_closeup_suburb_a/seq_000001.mp4-rp_emma_posed_008.pt
+ # -> mid: inputs/bedlam/bedlam_download/20221011_1_250_batch01hand_closeup_suburb_a/mp4/seq_000001.mp4-rp_emma_posed_008
+ scene = featname.split("/")[0]
+ seqsubj = featname.split("/")[1].strip(".pt")
+ mid = f"inputs/bedlam/bedlam_download/{scene}/mp4/{seqsubj}"
+ return mid
+
+
+def load_vname2lwh():
+ return torch.load(resource_dir / "vname2lwh.pt")
diff --git a/third_party/GVHMR/hmr4d/dataset/emdb/emdb_motion_test.py b/third_party/GVHMR/hmr4d/dataset/emdb/emdb_motion_test.py
new file mode 100644
index 0000000000000000000000000000000000000000..1a2e81c7d7e4d348d8d151d06dd0f1c10f13ce05
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/emdb/emdb_motion_test.py
@@ -0,0 +1,167 @@
+from pathlib import Path
+import numpy as np
+import torch
+from torch.utils import data
+from hmr4d.utils.pylogger import Log
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines
+
+from hmr4d.utils.geo_transform import compute_cam_angvel
+from pytorch3d.transforms import quaternion_to_matrix
+from hmr4d.utils.geo.hmr_cam import estimate_K, resize_K
+from hmr4d.utils.geo.flip_utils import flip_kp2d_coco17
+
+from .utils import EMDB1_NAMES, EMDB2_NAMES
+
+VID_PRESETS = {1: EMDB1_NAMES, 2: EMDB2_NAMES}
+
+
+from hmr4d.configs import MainStore, builds
+
+
+class EmdbSmplFullSeqDataset(data.Dataset):
+ def __init__(self, split=1, flip_test=False):
+ """
+ split: 1 for EMDB-1, 2 for EMDB-2
+ flip_test: if True, extra flip data will be returned
+ """
+ super().__init__()
+ self.dataset_name = "EMDB"
+ self.split = split
+ self.dataset_id = f"EMDB_{split}"
+ Log.info(f"[{self.dataset_name}] Full sequence, split={split}")
+
+ # Load evaluation protocol from WHAM labels
+ tic = Log.time()
+ self.emdb_dir = Path("inputs/EMDB/hmr4d_support")
+ # 'name', 'gender', 'smpl_params', 'mask', 'K_fullimg', 'T_w2c', 'bbx_xys', 'kp2d', 'features'
+ self.labels = torch.load(self.emdb_dir / "emdb_vit_v4.pt")
+ self.cam_traj = torch.load(self.emdb_dir / "emdb_dpvo_traj.pt") # estimated with DPVO
+
+ # Setup dataset index
+ self.idx2meta = []
+ for vid in VID_PRESETS[split]:
+ seq_length = len(self.labels[vid]["mask"])
+ self.idx2meta.append((vid, 0, seq_length)) # start=0, end=seq_length
+ Log.info(f"[{self.dataset_name}] {len(self.idx2meta)} sequences. Elapsed: {Log.time() - tic:.2f}s")
+
+ # If flip_test is enabled, we will return extra data for flipped test
+ self.flip_test = flip_test
+ if self.flip_test:
+ Log.info(f"[{self.dataset_name}] Flip test enabled")
+
+ def __len__(self):
+ return len(self.idx2meta)
+
+ def _load_data(self, idx):
+ data = {}
+
+ # [vid, start, end]
+ vid, start, end = self.idx2meta[idx]
+ length = end - start
+ meta = {"dataset_id": self.dataset_id, "vid": vid, "vid-start-end": (start, end)}
+ data.update({"meta": meta, "length": length})
+
+ label = self.labels[vid]
+
+ # smpl_params in world
+ gender = label["gender"]
+ smpl_params = label["smpl_params"]
+ mask = label["mask"]
+ data.update({"smpl_params": smpl_params, "gender": gender, "mask": mask})
+
+ # camera
+ # K_fullimg = label["K_fullimg"] # We use estimated K
+ width_height = (1440, 1920) if vid != "P0_09_outdoor_walk" else (720, 960)
+ K_fullimg = estimate_K(*width_height)
+ T_w2c = label["T_w2c"]
+ data.update({"K_fullimg": K_fullimg, "T_w2c": T_w2c})
+
+ # R_w2c -> cam_angvel
+ use_DPVO = False
+ if use_DPVO:
+ traj = self.cam_traj[data["meta"]["vid"]] # (L, 7)
+ R_w2c = quaternion_to_matrix(traj[:, [6, 3, 4, 5]]).mT # (L, 3, 3)
+ else: # GT
+ R_w2c = data["T_w2c"][:, :3, :3] # (L, 3, 3)
+ data["cam_angvel"] = compute_cam_angvel(R_w2c) # (L, 6)
+
+ # image bbx, features
+ bbx_xys = label["bbx_xys"]
+ f_imgseq = label["features"]
+ kp2d = label["kp2d"]
+ data.update({"bbx_xys": bbx_xys, "f_imgseq": f_imgseq, "kp2d": kp2d})
+
+ # to render a video
+ video_path = self.emdb_dir / f"videos/{vid}.mp4"
+ frame_id = torch.where(mask)[0].long()
+ resize_factor = 0.5
+ width_height_render = torch.tensor(width_height) * resize_factor
+ K_render = resize_K(K_fullimg, resize_factor)
+ bbx_xys_render = bbx_xys * resize_factor
+ data["meta_render"] = {
+ "split": self.split,
+ "name": vid,
+ "video_path": str(video_path),
+ "resize_factor": resize_factor,
+ "frame_id": frame_id,
+ "width_height": width_height_render.int(),
+ "K": K_render,
+ "bbx_xys": bbx_xys_render,
+ "R_cam_type": "DPVO" if use_DPVO else "GtGyro",
+ }
+
+ # if enable flip_test
+ if self.flip_test:
+ imgfeat_dir = self.emdb_dir / "imgfeats/emdb_flip"
+ f_img_dict = torch.load(imgfeat_dir / f"{vid}.pt")
+
+ flipped_bbx_xys = f_img_dict["bbx_xys"].float() # (L, 3)
+ flipped_features = f_img_dict["features"].float() # (L, 1024)
+ width = width_height[0]
+ flipped_kp2d = flip_kp2d_coco17(kp2d, width) # (L, 17, 3)
+
+ R_flip_x = torch.tensor([[-1, 0, 0], [0, 1, 0], [0, 0, 1]]).float()
+ flipped_R_w2c = R_flip_x @ R_w2c.clone()
+
+ data_flip = {
+ "bbx_xys": flipped_bbx_xys,
+ "f_imgseq": flipped_features,
+ "kp2d": flipped_kp2d,
+ "cam_angvel": compute_cam_angvel(flipped_R_w2c),
+ }
+ data["flip_test"] = data_flip
+
+ return data
+
+ def _process_data(self, data):
+ length = data["length"]
+ data["K_fullimg"] = data["K_fullimg"][None].repeat(length, 1, 1)
+ return data
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data)
+ return data
+
+
+# EMDB-1 and EMDB-2
+MainStore.store(
+ name="v1",
+ node=builds(EmdbSmplFullSeqDataset, populate_full_signature=True),
+ group="test_datasets/emdb1",
+)
+MainStore.store(
+ name="v1_fliptest",
+ node=builds(EmdbSmplFullSeqDataset, flip_test=True, populate_full_signature=True),
+ group="test_datasets/emdb1",
+)
+MainStore.store(
+ name="v1",
+ node=builds(EmdbSmplFullSeqDataset, split=2, populate_full_signature=True),
+ group="test_datasets/emdb2",
+)
+MainStore.store(
+ name="v1_fliptest",
+ node=builds(EmdbSmplFullSeqDataset, split=2, flip_test=True, populate_full_signature=True),
+ group="test_datasets/emdb2",
+)
diff --git a/third_party/GVHMR/hmr4d/dataset/emdb/utils.py b/third_party/GVHMR/hmr4d/dataset/emdb/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..24320e9b2688fbf652cecf719446be511cb0546f
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/emdb/utils.py
@@ -0,0 +1,120 @@
+import torch
+import pickle
+import numpy as np
+from pathlib import Path
+from tqdm import tqdm
+from hmr4d.utils.geo_transform import convert_lurb_to_bbx_xys
+from hmr4d.utils.video_io_utils import get_video_lwh
+
+
+def name_to_subfolder(name):
+ return f"{name[:2]}/{name[3:]}"
+
+
+def name_to_local_pkl_path(name):
+ return f"{name_to_subfolder(name)}/{name}_data.pkl"
+
+
+def load_raw_pkl(fp):
+ annot = pickle.load(open(fp, "rb"))
+ annot["subfolder"] = name_to_subfolder(annot["name"])
+ return annot
+
+
+def load_pkl(fp):
+ annot = pickle.load(open(fp, "rb"))
+ # ['gender', 'name', 'emdb1', 'emdb2', 'n_frames', 'good_frames_mask', 'camera', 'smpl', 'kp2d', 'bboxes', 'subfolder']
+ data = {}
+
+ F = annot["n_frames"]
+ smpl_params = {
+ "body_pose": annot["smpl"]["poses_body"], # (F, 69)
+ "betas": annot["smpl"]["betas"][None].repeat(F, axis=0), # (F, 10)
+ "global_orient": annot["smpl"]["poses_root"], # (F, 3)
+ "transl": annot["smpl"]["trans"], # (F, 3)
+ }
+ smpl_params = {k: torch.from_numpy(v).float() for k, v in smpl_params.items()}
+
+ data["name"] = annot["name"]
+ data["gender"] = annot["gender"]
+ data["smpl_params"] = smpl_params
+ data["mask"] = torch.from_numpy(annot["good_frames_mask"]).bool() # (L,)
+ data["K_fullimg"] = torch.from_numpy(annot["camera"]["intrinsics"]).float() # (3, 3)
+ data["T_w2c"] = torch.from_numpy(annot["camera"]["extrinsics"]).float() # (L, 4, 4)
+ bbx_lurb = torch.from_numpy(annot["bboxes"]["bboxes"]).float()
+ data["bbx_xys"] = convert_lurb_to_bbx_xys(bbx_lurb) # (L, 3)
+
+ return data
+
+
+EMDB1_LIST = [
+ "P1/14_outdoor_climb/P1_14_outdoor_climb_data.pkl",
+ "P2/23_outdoor_hug_tree/P2_23_outdoor_hug_tree_data.pkl",
+ "P3/31_outdoor_workout/P3_31_outdoor_workout_data.pkl",
+ "P3/32_outdoor_soccer_warmup_a/P3_32_outdoor_soccer_warmup_a_data.pkl",
+ "P3/33_outdoor_soccer_warmup_b/P3_33_outdoor_soccer_warmup_b_data.pkl",
+ "P5/42_indoor_dancing/P5_42_indoor_dancing_data.pkl",
+ "P5/44_indoor_rom/P5_44_indoor_rom_data.pkl",
+ "P6/49_outdoor_big_stairs_down/P6_49_outdoor_big_stairs_down_data.pkl", # DUPLICATE
+ "P6/50_outdoor_workout/P6_50_outdoor_workout_data.pkl",
+ "P6/51_outdoor_dancing/P6_51_outdoor_dancing_data.pkl",
+ "P7/57_outdoor_rock_chair/P7_57_outdoor_rock_chair_data.pkl", # DUPLICATE
+ "P7/59_outdoor_rom/P7_59_outdoor_rom_data.pkl",
+ "P7/60_outdoor_workout/P7_60_outdoor_workout_data.pkl",
+ "P8/64_outdoor_skateboard/P8_64_outdoor_skateboard_data.pkl", # DUPLICATE
+ "P8/68_outdoor_handstand/P8_68_outdoor_handstand_data.pkl",
+ "P8/69_outdoor_cartwheel/P8_69_outdoor_cartwheel_data.pkl",
+ "P9/76_outdoor_sitting/P9_76_outdoor_sitting_data.pkl",
+]
+EMDB1_NAMES = ["_".join(p.split("/")[:2]) for p in EMDB1_LIST]
+
+
+EMDB2_LIST = [
+ "P0/09_outdoor_walk/P0_09_outdoor_walk_data.pkl",
+ "P2/19_indoor_walk_off_mvs/P2_19_indoor_walk_off_mvs_data.pkl",
+ "P2/20_outdoor_walk/P2_20_outdoor_walk_data.pkl",
+ "P2/24_outdoor_long_walk/P2_24_outdoor_long_walk_data.pkl",
+ "P3/27_indoor_walk_off_mvs/P3_27_indoor_walk_off_mvs_data.pkl",
+ "P3/28_outdoor_walk_lunges/P3_28_outdoor_walk_lunges_data.pkl",
+ "P3/29_outdoor_stairs_up/P3_29_outdoor_stairs_up_data.pkl",
+ "P3/30_outdoor_stairs_down/P3_30_outdoor_stairs_down_data.pkl",
+ "P4/35_indoor_walk/P4_35_indoor_walk_data.pkl",
+ "P4/36_outdoor_long_walk/P4_36_outdoor_long_walk_data.pkl",
+ "P4/37_outdoor_run_circle/P4_37_outdoor_run_circle_data.pkl",
+ "P5/40_indoor_walk_big_circle/P5_40_indoor_walk_big_circle_data.pkl",
+ "P6/48_outdoor_walk_downhill/P6_48_outdoor_walk_downhill_data.pkl",
+ "P6/49_outdoor_big_stairs_down/P6_49_outdoor_big_stairs_down_data.pkl", # DUPLICATE
+ "P7/55_outdoor_walk/P7_55_outdoor_walk_data.pkl",
+ "P7/56_outdoor_stairs_up_down/P7_56_outdoor_stairs_up_down_data.pkl",
+ "P7/57_outdoor_rock_chair/P7_57_outdoor_rock_chair_data.pkl", # DUPLICATE
+ "P7/58_outdoor_parcours/P7_58_outdoor_parcours_data.pkl",
+ "P7/61_outdoor_sit_lie_walk/P7_61_outdoor_sit_lie_walk_data.pkl",
+ "P8/64_outdoor_skateboard/P8_64_outdoor_skateboard_data.pkl", # DUPLICATE
+ "P8/65_outdoor_walk_straight/P8_65_outdoor_walk_straight_data.pkl",
+ "P9/77_outdoor_stairs_up/P9_77_outdoor_stairs_up_data.pkl",
+ "P9/78_outdoor_stairs_up_down/P9_78_outdoor_stairs_up_down_data.pkl",
+ "P9/79_outdoor_walk_rectangle/P9_79_outdoor_walk_rectangle_data.pkl",
+ "P9/80_outdoor_walk_big_circle/P9_80_outdoor_walk_big_circle_data.pkl",
+]
+EMDB2_NAMES = ["_".join(p.split("/")[:2]) for p in EMDB2_LIST]
+EMDB_NAMES = list(sorted(set(EMDB1_NAMES + EMDB2_NAMES)))
+
+
+def _check_annot(emdb_raw_dir=Path("inputs/EMDB/EMDB")):
+ for pkl_local_path in set(EMDB1_LIST + EMDB2_LIST):
+ annot = load_raw_pkl(emdb_raw_dir / pkl_local_path)
+ if any((annot["bboxes"]["invalid_idxs"] != np.where(~annot["good_frames_mask"])[0])):
+ print(annot["name"])
+
+
+def _check_length(emdb_raw_dir=Path("inputs/EMDB/EMDB"), emdb_hmr4d_support_dir=Path("inputs/EMDB/hmr4d_support")):
+ lengths = []
+ for local_pkl_path in tqdm(set(EMDB1_LIST + EMDB2_LIST)):
+ data = load_pkl(emdb_raw_dir / local_pkl_path)
+ video_path = emdb_hmr4d_support_dir / "videos" / f"{data['name']}.mp4"
+ length, width, height = get_video_lwh(video_path)
+ lengths.append(length)
+ print(sorted(lengths))
+
+ video_ram = length[-1] * (width / 4) * (height / 4) * 3 / 1e6
+ print(f"Video RAM for {lengths[-1]} x {width} x {height}: {video_ram:.2f} MB")
diff --git a/third_party/GVHMR/hmr4d/dataset/h36m/camera-parameters.json b/third_party/GVHMR/hmr4d/dataset/h36m/camera-parameters.json
new file mode 100644
index 0000000000000000000000000000000000000000..6e147975d7a7317e544197ba3a453654230e784f
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/h36m/camera-parameters.json
@@ -0,0 +1,1452 @@
+{
+ "intrinsics": {
+ "54138969": {
+ "calibration_matrix": [
+ [
+ 1145.04940458804,
+ 0.0,
+ 512.541504956548
+ ],
+ [
+ 0.0,
+ 1143.78109572365,
+ 515.4514869776
+ ],
+ [
+ 0.0,
+ 0.0,
+ 1.0
+ ]
+ ],
+ "distortion": [
+ -0.207098910824901,
+ 0.247775183068982,
+ -0.00142447157470321,
+ -0.000975698859470499,
+ -0.00307515035078854
+ ]
+ },
+ "55011271": {
+ "calibration_matrix": [
+ [
+ 1149.67569986785,
+ 0.0,
+ 508.848621645943
+ ],
+ [
+ 0.0,
+ 1147.59161666764,
+ 508.064917088557
+ ],
+ [
+ 0.0,
+ 0.0,
+ 1.0
+ ]
+ ],
+ "distortion": [
+ -0.194213629607385,
+ 0.240408539138292,
+ -0.0027408943961907,
+ -0.001619026613787,
+ 0.00681997559022603
+ ]
+ },
+ "58860488": {
+ "calibration_matrix": [
+ [
+ 1149.14071676148,
+ 0.0,
+ 519.815837182153
+ ],
+ [
+ 0.0,
+ 1148.7989685676,
+ 501.402658888552
+ ],
+ [
+ 0.0,
+ 0.0,
+ 1.0
+ ]
+ ],
+ "distortion": [
+ -0.208338188251856,
+ 0.255488007488945,
+ -0.000759999321030303,
+ 0.00148438698385668,
+ -0.00246049749891915
+ ]
+ },
+ "60457274": {
+ "calibration_matrix": [
+ [
+ 1145.51133842318,
+ 0.0,
+ 514.968197319863
+ ],
+ [
+ 0.0,
+ 1144.77392807652,
+ 501.882018537695
+ ],
+ [
+ 0.0,
+ 0.0,
+ 1.0
+ ]
+ ],
+ "distortion": [
+ -0.198384093827848,
+ 0.218323676298049,
+ -0.00181336200488089,
+ -0.000587205583421232,
+ -0.00894780704152122
+ ]
+ }
+ },
+ "extrinsics": {
+ "S1": {
+ "54138969": {
+ "R": [
+ [
+ -0.9153617321513369,
+ 0.40180836633680234,
+ 0.02574754463350265
+ ],
+ [
+ 0.051548117060134555,
+ 0.1803735689384521,
+ -0.9822464900705729
+ ],
+ [
+ -0.399319034032262,
+ -0.8977836111057917,
+ -0.185819527201491
+ ]
+ ],
+ "t": [
+ [
+ -346.05078140028075
+ ],
+ [
+ 546.9807793144001
+ ],
+ [
+ 5474.481087434061
+ ]
+ ]
+ },
+ "55011271": {
+ "R": [
+ [
+ 0.9281683400814921,
+ 0.3721538354721445,
+ 0.002248380248018696
+ ],
+ [
+ 0.08166409428175585,
+ -0.1977722953267526,
+ -0.976840363061605
+ ],
+ [
+ -0.3630902204349604,
+ 0.9068559102440475,
+ -0.21395758897485287
+ ]
+ ],
+ "t": [
+ [
+ 251.42516271750836
+ ],
+ [
+ 420.9422103702068
+ ],
+ [
+ 5588.195881837821
+ ]
+ ]
+ },
+ "58860488": {
+ "R": [
+ [
+ -0.9141549520542256,
+ -0.4027780222811878,
+ -0.045722952682337906
+ ],
+ [
+ -0.04562341383935875,
+ 0.21430849526487267,
+ -0.9756999400261069
+ ],
+ [
+ 0.40278930937200774,
+ -0.889854894701693,
+ -0.214287280609606
+ ]
+ ],
+ "t": [
+ [
+ 480.482559565337
+ ],
+ [
+ 253.83237471361554
+ ],
+ [
+ 5704.2076793704555
+ ]
+ ]
+ },
+ "60457274": {
+ "R": [
+ [
+ 0.9141562410494211,
+ -0.40060705854636447,
+ 0.061905989962380774
+ ],
+ [
+ -0.05641000739510571,
+ -0.2769531972942539,
+ -0.9592261660183036
+ ],
+ [
+ 0.40141783470104664,
+ 0.8733904688919611,
+ -0.2757767409202658
+ ]
+ ],
+ "t": [
+ [
+ 51.88347637559197
+ ],
+ [
+ 378.4208425426766
+ ],
+ [
+ 4406.149140878431
+ ]
+ ]
+ }
+ },
+ "S2": {
+ "54138969": {
+ "R": [
+ [
+ -0.9072826056858586,
+ 0.4200536513985309,
+ 0.019829356183203237
+ ],
+ [
+ 0.06404223092375372,
+ 0.18462275321422528,
+ -0.9807206695353717
+ ],
+ [
+ -0.4156162485733534,
+ -0.8885208882982778,
+ -0.1944061855483302
+ ]
+ ],
+ "t": [
+ [
+ -253.9473271477662
+ ],
+ [
+ 543.369692173605
+ ],
+ [
+ 5522.981999493327
+ ]
+ ]
+ },
+ "55011271": {
+ "R": [
+ [
+ 0.9195695689704942,
+ 0.3926824530407384,
+ 0.013867187794489123
+ ],
+ [
+ 0.09616327770610274,
+ -0.190692439252443,
+ -0.9769283584955307
+ ],
+ [
+ -0.38097825639298405,
+ 0.8996871037676718,
+ -0.21311659595137136
+ ]
+ ],
+ "t": [
+ [
+ 123.3506735789221
+ ],
+ [
+ 401.02404156275884
+ ],
+ [
+ 5743.522551411228
+ ]
+ ]
+ },
+ "58860488": {
+ "R": [
+ [
+ -0.9231022562305128,
+ -0.3793547679556717,
+ -0.06302526930870815
+ ],
+ [
+ -0.023520852900409527,
+ 0.21928184512961552,
+ -0.9753779994829639
+ ],
+ [
+ 0.3838345920067314,
+ -0.898891223911909,
+ -0.2113423136836923
+ ]
+ ],
+ "t": [
+ [
+ 498.7689000990772
+ ],
+ [
+ 278.0695777621727
+ ],
+ [
+ 5618.721192968872
+ ]
+ ]
+ },
+ "60457274": {
+ "R": [
+ [
+ 0.9239917699501332,
+ -0.37272063182115767,
+ 0.08554846392108466
+ ],
+ [
+ -0.01857104155727153,
+ -0.2671779087245581,
+ -0.9634682566151569
+ ],
+ [
+ 0.38196115703026423,
+ 0.8886480156419687,
+ -0.2537919991167828
+ ]
+ ],
+ "t": [
+ [
+ -55.1478742462578
+ ],
+ [
+ 424.8747833741909
+ ],
+ [
+ 4452.137526291175
+ ]
+ ]
+ }
+ },
+ "S3": {
+ "54138969": {
+ "R": [
+ [
+ -0.909926063968229,
+ 0.4142842734534348,
+ 0.020077322541766036
+ ],
+ [
+ 0.06112258570603725,
+ 0.18181129378483157,
+ -0.9814319553432596
+ ],
+ [
+ -0.41024210855042,
+ -0.891803338310328,
+ -0.19075696094942407
+ ]
+ ],
+ "t": [
+ [
+ -144.30406670344493
+ ],
+ [
+ 546.2767112872957
+ ],
+ [
+ 5569.530692348755
+ ]
+ ]
+ },
+ "55011271": {
+ "R": [
+ [
+ 0.9248703521034336,
+ 0.3800681977315835,
+ 0.012767022876799783
+ ],
+ [
+ 0.093795468138089,
+ -0.1954524371286302,
+ -0.9762175756342618
+ ],
+ [
+ -0.3685339088290622,
+ 0.9040721817938792,
+ -0.21641683887726407
+ ]
+ ],
+ "t": [
+ [
+ -38.93379836342622
+ ],
+ [
+ 375.57502666735104
+ ],
+ [
+ 5759.402838804998
+ ]
+ ]
+ },
+ "58860488": {
+ "R": [
+ [
+ -0.9218827889823751,
+ -0.38260686272952316,
+ -0.061189149122614306
+ ],
+ [
+ -0.02577019492115,
+ 0.21811470471458455,
+ -0.9755828374059251
+ ],
+ [
+ 0.3866109419452632,
+ -0.897796170731164,
+ -0.2109360457310579
+ ]
+ ],
+ "t": [
+ [
+ 596.8162203909545
+ ],
+ [
+ 282.123966506171
+ ],
+ [
+ 5575.726600786697
+ ]
+ ]
+ },
+ "60457274": {
+ "R": [
+ [
+ 0.9244960445794738,
+ -0.37161308683612865,
+ 0.08491629554468147
+ ],
+ [
+ -0.018795005038688972,
+ -0.26693214570791374,
+ -0.963532032354589
+ ],
+ [
+ 0.3807280017840865,
+ 0.88918555053481,
+ -0.2537621827176058
+ ]
+ ],
+ "t": [
+ [
+ -158.57266932864025
+ ],
+ [
+ 433.1881250816
+ ],
+ [
+ 4413.555688648984
+ ]
+ ]
+ }
+ },
+ "S4": {
+ "54138969": {
+ "R": [
+ [
+ -0.906169211683753,
+ 0.422346184383899,
+ 0.021933087625945674
+ ],
+ [
+ 0.06180306305120707,
+ 0.18355044391174938,
+ -0.9810655512947585
+ ],
+ [
+ -0.4183751201899252,
+ -0.8876558652294037,
+ -0.19243004892662768
+ ]
+ ],
+ "t": [
+ [
+ -201.25197932223173
+ ],
+ [
+ 537.4605027947064
+ ],
+ [
+ 5553.966756732112
+ ]
+ ]
+ },
+ "55011271": {
+ "R": [
+ [
+ 0.9205073288493492,
+ 0.39058428754662783,
+ 0.010496278213208041
+ ],
+ [
+ 0.0923916650578188,
+ -0.1914846595009468,
+ -0.9771373523735801
+ ],
+ [
+ -0.3796446203523497,
+ 0.900431862773358,
+ -0.21234976510469855
+ ]
+ ],
+ "t": [
+ [
+ 63.12322044876507
+ ],
+ [
+ 396.6138950755392
+ ],
+ [
+ 5760.7235858284985
+ ]
+ ]
+ },
+ "58860488": {
+ "R": [
+ [
+ -0.9244800422436603,
+ -0.37641653359695837,
+ -0.060392422769829215
+ ],
+ [
+ -0.02551533125481826,
+ 0.2191523935220463,
+ -0.9753569583924211
+ ],
+ [
+ 0.3803756292983513,
+ -0.9001571094250204,
+ -0.21220640656557746
+ ]
+ ],
+ "t": [
+ [
+ 559.4298619884164
+ ],
+ [
+ 278.041710381495
+ ],
+ [
+ 5601.2846874450925
+ ]
+ ]
+ },
+ "60457274": {
+ "R": [
+ [
+ 0.9241606780958346,
+ -0.3729066880542538,
+ 0.0828712439019392
+ ],
+ [
+ -0.021270464031387784,
+ -0.2668345784720987,
+ -0.9635075895349796
+ ],
+ [
+ 0.38141133756265905,
+ 0.8886731174824772,
+ -0.2545299232755129
+ ]
+ ],
+ "t": [
+ [
+ -98.61477305435534
+ ],
+ [
+ 432.68486951797627
+ ],
+ [
+ 4419.390974448715
+ ]
+ ]
+ }
+ },
+ "S5": {
+ "54138969": {
+ "R": [
+ [
+ -0.9042074184788829,
+ 0.42657831374650107,
+ 0.020973473936051274
+ ],
+ [
+ 0.06390493744399675,
+ 0.18368565260974637,
+ -0.9809055713959477
+ ],
+ [
+ -0.4222855708380685,
+ -0.8856017859436166,
+ -0.1933503902128034
+ ]
+ ],
+ "t": [
+ [
+ -219.3059666108619
+ ],
+ [
+ 544.4787497640639
+ ],
+ [
+ 5518.740477016156
+ ]
+ ]
+ },
+ "55011271": {
+ "R": [
+ [
+ 0.9222116004775194,
+ 0.38649075753002626,
+ 0.012274293810989732
+ ],
+ [
+ 0.09333184463870337,
+ -0.19167233853095322,
+ -0.9770111982052265
+ ],
+ [
+ -0.3752531555110883,
+ 0.902156643264318,
+ -0.21283434941998647
+ ]
+ ],
+ "t": [
+ [
+ 103.90282067751986
+ ],
+ [
+ 395.67169468951965
+ ],
+ [
+ 5767.97265758172
+ ]
+ ]
+ },
+ "58860488": {
+ "R": [
+ [
+ -0.9258288614330635,
+ -0.3728674116124112,
+ -0.06173178026768599
+ ],
+ [
+ -0.023578112500148365,
+ 0.220000562347259,
+ -0.9752147584905696
+ ],
+ [
+ 0.3772068291381898,
+ -0.9014264506460582,
+ -0.21247437993123308
+ ]
+ ],
+ "t": [
+ [
+ 520.3272318446208
+ ],
+ [
+ 283.3690958234795
+ ],
+ [
+ 5591.123958858676
+ ]
+ ]
+ },
+ "60457274": {
+ "R": [
+ [
+ 0.9222815489764817,
+ -0.3772688722588351,
+ 0.0840532119677073
+ ],
+ [
+ -0.021177649402562934,
+ -0.26645871124348197,
+ -0.9636136478735888
+ ],
+ [
+ 0.3859381447632816,
+ 0.88694303832152,
+ -0.25373962085111357
+ ]
+ ],
+ "t": [
+ [
+ -79.116431351199
+ ],
+ [
+ 425.59047114848386
+ ],
+ [
+ 4454.481629705836
+ ]
+ ]
+ }
+ },
+ "S6": {
+ "54138969": {
+ "R": [
+ [
+ -0.9149503344107554,
+ 0.4034864343564006,
+ 0.008036345687245266
+ ],
+ [
+ 0.07174776353922047,
+ 0.1822275975157708,
+ -0.9806351824867137
+ ],
+ [
+ -0.3971374371533952,
+ -0.896655898321083,
+ -0.19567845056940925
+ ]
+ ],
+ "t": [
+ [
+ -239.5182864132218
+ ],
+ [
+ 545.8141831785044
+ ],
+ [
+ 5523.931578633363
+ ]
+ ]
+ },
+ "55011271": {
+ "R": [
+ [
+ 0.9197364689900042,
+ 0.39209901596964664,
+ 0.018525368698999664
+ ],
+ [
+ 0.101478073351267,
+ -0.19191459963948,
+ -0.9761511087296542
+ ],
+ [
+ -0.37919260045353465,
+ 0.899681692667386,
+ -0.21630030892357308
+ ]
+ ],
+ "t": [
+ [
+ 169.02510061389722
+ ],
+ [
+ 409.6671223380997
+ ],
+ [
+ 5714.338002825065
+ ]
+ ]
+ },
+ "58860488": {
+ "R": [
+ [
+ -0.916577698818659,
+ -0.39393483656788014,
+ -0.06856140726771254
+ ],
+ [
+ -0.01984531630322392,
+ 0.21607069980297702,
+ -0.9761760169700323
+ ],
+ [
+ 0.3993638509543854,
+ -0.8933805444629346,
+ -0.20586334624209834
+ ]
+ ],
+ "t": [
+ [
+ 521.9864793089763
+ ],
+ [
+ 286.28272817103516
+ ],
+ [
+ 5643.2724406159
+ ]
+ ]
+ },
+ "60457274": {
+ "R": [
+ [
+ 0.9182950552949388,
+ -0.3850769011116475,
+ 0.09192372735651859
+ ],
+ [
+ -0.015534985886560007,
+ -0.26706146429979655,
+ -0.9635542737695438
+ ],
+ [
+ 0.3955917790277871,
+ 0.8833990913037544,
+ -0.25122338635033875
+ ]
+ ],
+ "t": [
+ [
+ -56.29675276801464
+ ],
+ [
+ 420.29579722027506
+ ],
+ [
+ 4499.322693551688
+ ]
+ ]
+ }
+ },
+ "S7": {
+ "54138969": {
+ "R": [
+ [
+ -0.9055764231419416,
+ 0.42392653746206904,
+ 0.014752378956221508
+ ],
+ [
+ 0.06862812683752326,
+ 0.18074371881263407,
+ -0.9811329615890764
+ ],
+ [
+ -0.41859469903024304,
+ -0.8874784498483331,
+ -0.19277053457045695
+ ]
+ ],
+ "t": [
+ [
+ -323.9118424584857
+ ],
+ [
+ 541.7715234126381
+ ],
+ [
+ 5506.569132699328
+ ]
+ ]
+ },
+ "55011271": {
+ "R": [
+ [
+ 0.9212640765077017,
+ 0.3886011826562522,
+ 0.01617473877914905
+ ],
+ [
+ 0.09922277503271489,
+ -0.1946115441987536,
+ -0.9758489574618522
+ ],
+ [
+ -0.3760682680727248,
+ 0.9006194910741931,
+ -0.21784671226815075
+ ]
+ ],
+ "t": [
+ [
+ 178.6238708832376
+ ],
+ [
+ 403.59193467821774
+ ],
+ [
+ 5694.8801003668095
+ ]
+ ]
+ },
+ "58860488": {
+ "R": [
+ [
+ -0.9245069728829368,
+ -0.37555597339631824,
+ -0.06515034871105972
+ ],
+ [
+ -0.018955014220249332,
+ 0.21601110989507338,
+ -0.9762068980691586
+ ],
+ [
+ 0.38069353097569036,
+ -0.9012751584550871,
+ -0.20682244613440448
+ ]
+ ],
+ "t": [
+ [
+ 441.1064712697594
+ ],
+ [
+ 271.91614362573955
+ ],
+ [
+ 5660.120611352617
+ ]
+ ]
+ },
+ "60457274": {
+ "R": [
+ [
+ 0.9228353966173104,
+ -0.3744001545228767,
+ 0.09055029013436408
+ ],
+ [
+ -0.014982084363704698,
+ -0.269786590656035,
+ -0.9628035794752281
+ ],
+ [
+ 0.3849030629889691,
+ 0.8871525910436372,
+ -0.25457791009093983
+ ]
+ ],
+ "t": [
+ [
+ 25.768533743836343
+ ],
+ [
+ 431.05581759025813
+ ],
+ [
+ 4461.872981411145
+ ]
+ ]
+ }
+ },
+ "S8": {
+ "54138969": {
+ "R": [
+ [
+ -0.9115694669712032,
+ 0.4106494283805017,
+ 0.020202818036194434
+ ],
+ [
+ 0.060907749548984036,
+ 0.1834736632003901,
+ -0.9811359034082424
+ ],
+ [
+ -0.40660958293025334,
+ -0.8931430243150293,
+ -0.19226072190306673
+ ]
+ ],
+ "t": [
+ [
+ -82.70216069652597
+ ],
+ [
+ 552.1896311377282
+ ],
+ [
+ 5557.353609418419
+ ]
+ ]
+ },
+ "55011271": {
+ "R": [
+ [
+ 0.931016282525616,
+ 0.3647626932499711,
+ 0.01252434769597448
+ ],
+ [
+ 0.08939715221301257,
+ -0.19463753190599434,
+ -0.9767929055586687
+ ],
+ [
+ -0.35385990285476776,
+ 0.9105297407479727,
+ -0.2138194574051759
+ ]
+ ],
+ "t": [
+ [
+ -209.06289992510443
+ ],
+ [
+ 375.0691429434037
+ ],
+ [
+ 5818.276676972416
+ ]
+ ]
+ },
+ "58860488": {
+ "R": [
+ [
+ -0.9209075762929309,
+ -0.3847355178017309,
+ -0.0625125368875214
+ ],
+ [
+ -0.02568138180824641,
+ 0.21992027027623712,
+ -0.9751797482259595
+ ],
+ [
+ 0.38893405939143305,
+ -0.8964450100611084,
+ -0.21240678280563546
+ ]
+ ],
+ "t": [
+ [
+ 623.0985110132146
+ ],
+ [
+ 290.9053651845054
+ ],
+ [
+ 5534.379001592981
+ ]
+ ]
+ },
+ "60457274": {
+ "R": [
+ [
+ 0.927667052235436,
+ -0.3636062759574404,
+ 0.08499597802942535
+ ],
+ [
+ -0.01666268768012713,
+ -0.26770413351564454,
+ -0.9633570738505596
+ ],
+ [
+ 0.37303645269074087,
+ 0.8922583555131325,
+ -0.2543989622245125
+ ]
+ ],
+ "t": [
+ [
+ -178.36705625795474
+ ],
+ [
+ 423.4669232560848
+ ],
+ [
+ 4421.6448791590965
+ ]
+ ]
+ }
+ },
+ "S9": {
+ "54138969": {
+ "R": [
+ [
+ -0.9033486204435297,
+ 0.4269119782787646,
+ 0.04132109321984796
+ ],
+ [
+ 0.04153061098352977,
+ 0.182951140059007,
+ -0.9822444139329296
+ ],
+ [
+ -0.4268916470184284,
+ -0.8855930460167476,
+ -0.18299857527497945
+ ]
+ ],
+ "t": [
+ [
+ -321.2078335720134
+ ],
+ [
+ 467.13452033013084
+ ],
+ [
+ 5514.330338522134
+ ]
+ ]
+ },
+ "55011271": {
+ "R": [
+ [
+ 0.9315720471487059,
+ 0.36348288012373176,
+ -0.007329176497134756
+ ],
+ [
+ 0.06810069482701912,
+ -0.19426747906725159,
+ -0.9785818524481906
+ ],
+ [
+ -0.35712157080642226,
+ 0.911120377575769,
+ -0.20572758986325015
+ ]
+ ],
+ "t": [
+ [
+ 19.193095609487138
+ ],
+ [
+ 404.22842728571936
+ ],
+ [
+ 5702.169280033924
+ ]
+ ]
+ },
+ "58860488": {
+ "R": [
+ [
+ -0.9269344193869241,
+ -0.3732303525241731,
+ -0.03862235247246717
+ ],
+ [
+ -0.04725991098820678,
+ 0.218240494552814,
+ -0.9747500127472326
+ ],
+ [
+ 0.37223525218497616,
+ -0.901704048173249,
+ -0.21993345934341726
+ ]
+ ],
+ "t": [
+ [
+ 455.40107288876885
+ ],
+ [
+ 273.3589338272866
+ ],
+ [
+ 5657.814488280711
+ ]
+ ]
+ },
+ "60457274": {
+ "R": [
+ [
+ 0.915460708083783,
+ -0.39734606500700814,
+ 0.06362229623477154
+ ],
+ [
+ -0.04940628468469528,
+ -0.26789167566119776,
+ -0.9621814117644814
+ ],
+ [
+ 0.39936288133525055,
+ 0.8776959352388969,
+ -0.26487569589663096
+ ]
+ ],
+ "t": [
+ [
+ -69.271255294384
+ ],
+ [
+ 422.1843366088847
+ ],
+ [
+ 4457.893374979773
+ ]
+ ]
+ }
+ },
+ "S10": {
+ "54138969": {
+ "R": [
+ [
+ -0.9199955359932982,
+ 0.39133749168985454,
+ 0.021521648410310328
+ ],
+ [
+ 0.0555185840851712,
+ 0.18448351869097226,
+ -0.9812662829999691
+ ],
+ [
+ -0.3879766752957989,
+ -0.9015657485337887,
+ -0.19145051709809383
+ ]
+ ],
+ "t": [
+ [
+ -181.4625993368258
+ ],
+ [
+ 543.5199110634021
+ ],
+ [
+ 5582.194377534298
+ ]
+ ]
+ },
+ "55011271": {
+ "R": [
+ [
+ 0.9152587269115653,
+ 0.40266346194010966,
+ 0.01279059148853104
+ ],
+ [
+ 0.09843295287698457,
+ -0.1927270179143742,
+ -0.9763028476624197
+ ],
+ [
+ -0.3906563919867918,
+ 0.8948287171209015,
+ -0.21603043863220686
+ ]
+ ],
+ "t": [
+ [
+ -22.5707386911355
+ ],
+ [
+ 383.7773845053516
+ ],
+ [
+ 5727.149101385447
+ ]
+ ]
+ },
+ "58860488": {
+ "R": [
+ [
+ -0.9117691356172892,
+ -0.4060874893594546,
+ -0.06140027948988781
+ ],
+ [
+ -0.03165845257462336,
+ 0.21854812554171174,
+ -0.9753124931029975
+ ],
+ [
+ 0.4094811176553588,
+ -0.887315990956964,
+ -0.21212153703897946
+ ]
+ ],
+ "t": [
+ [
+ 579.9870562891809
+ ],
+ [
+ 276.09388439709664
+ ],
+ [
+ 5616.656671116378
+ ]
+ ]
+ },
+ "60457274": {
+ "R": [
+ [
+ 0.9374925472123639,
+ -0.3377263929586908,
+ 0.08395598501825681
+ ],
+ [
+ -0.009787543064644189,
+ -0.2667415901136707,
+ -0.9637183863060765
+ ],
+ [
+ 0.3478676873784507,
+ 0.9026570819545717,
+ -0.25337376437829157
+ ]
+ ],
+ "t": [
+ [
+ -72.59483557976097
+ ],
+ [
+ 445.63607020105314
+ ],
+ [
+ 4402.73689876101
+ ]
+ ]
+ }
+ },
+ "S11": {
+ "54138969": {
+ "R": [
+ [
+ -0.9059013006181885,
+ 0.4217144115102914,
+ 0.038727105014486805
+ ],
+ [
+ 0.044493184429779696,
+ 0.1857199061874203,
+ -0.9815948619389944
+ ],
+ [
+ -0.4211450938543295,
+ -0.8875049698848251,
+ -0.1870073216538954
+ ]
+ ],
+ "t": [
+ [
+ -234.7208032216618
+ ],
+ [
+ 464.34018262882194
+ ],
+ [
+ 5536.652631113797
+ ]
+ ]
+ },
+ "55011271": {
+ "R": [
+ [
+ 0.9216646531492915,
+ 0.3879848687925067,
+ -0.0014172943441045224
+ ],
+ [
+ 0.07721054863099915,
+ -0.18699239961454955,
+ -0.979322405373477
+ ],
+ [
+ -0.3802272982247548,
+ 0.9024974149959955,
+ -0.20230080971229314
+ ]
+ ],
+ "t": [
+ [
+ -11.934348472090557
+ ],
+ [
+ 449.4165893644565
+ ],
+ [
+ 5541.113551868937
+ ]
+ ]
+ },
+ "58860488": {
+ "R": [
+ [
+ -0.9063540572469627,
+ -0.42053101768163204,
+ -0.04093880896680188
+ ],
+ [
+ -0.0603212197838846,
+ 0.22468715090881142,
+ -0.9725620980997899
+ ],
+ [
+ 0.4181909532208387,
+ -0.8790161246439863,
+ -0.2290130547809762
+ ]
+ ],
+ "t": [
+ [
+ 781.127357651581
+ ],
+ [
+ 235.3131620173424
+ ],
+ [
+ 5576.37044019807
+ ]
+ ]
+ },
+ "60457274": {
+ "R": [
+ [
+ 0.91754082476548,
+ -0.39226322025776267,
+ 0.06517975852741943
+ ],
+ [
+ -0.04531905395586976,
+ -0.26600517028098103,
+ -0.9629057236990188
+ ],
+ [
+ 0.395050652748768,
+ 0.8805514269006645,
+ -0.2618476013752581
+ ]
+ ],
+ "t": [
+ [
+ -155.13650339749012
+ ],
+ [
+ 422.16256306729633
+ ],
+ [
+ 4435.416222660868
+ ]
+ ]
+ }
+ }
+ }
+ }
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/dataset/h36m/h36m.py b/third_party/GVHMR/hmr4d/dataset/h36m/h36m.py
new file mode 100644
index 0000000000000000000000000000000000000000..5370b0d2f6d37f4c7a82faec9a02203c8b1f578a
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/h36m/h36m.py
@@ -0,0 +1,205 @@
+import torch
+import numpy as np
+from pathlib import Path
+from hmr4d.configs import MainStore, builds
+
+from hmr4d.utils.pylogger import Log
+from hmr4d.dataset.imgfeat_motion.base_dataset import ImgfeatMotionDatasetBase
+from pytorch3d.transforms import axis_angle_to_matrix, matrix_to_axis_angle
+from hmr4d.utils import matrix
+from hmr4d.utils.smplx_utils import make_smplx
+from tqdm import tqdm
+
+from hmr4d.utils.geo_transform import compute_cam_angvel, apply_T_on_points
+from hmr4d.utils.geo.hmr_global import get_tgtcoord_rootparam, get_T_w2c_from_wcparams, get_c_rootparam, get_R_c2gv
+
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines
+from hmr4d.utils.vis.renderer import Renderer
+import imageio
+from hmr4d.utils.video_io_utils import read_video_np
+from hmr4d.utils.net_utils import get_valid_mask, repeat_to_max_len, repeat_to_max_len_dict
+
+
+class H36mSmplDataset(ImgfeatMotionDatasetBase):
+ def __init__(
+ self,
+ root="inputs/H36M/hmr4d_support",
+ original_coord="az",
+ motion_frames=120, # H36M's videos are 25fps and very long
+ lazy_load=False,
+ ):
+ # Path
+ self.root = Path(root)
+
+ # Coord
+ self.original_coord = original_coord
+
+ # Setting
+ self.motion_frames = motion_frames
+ self.lazy_load = lazy_load
+
+ super().__init__()
+
+ def _load_dataset(self):
+ # smplpose
+ tic = Log.time()
+ fn = self.root / "smplxpose_v1.pt"
+ self.smpl_model = make_smplx("supermotion")
+ Log.info(f"[H36M] Loading from {fn} ...")
+ self.motion_files = torch.load(fn)
+ # Dict of {
+ # "smpl_params_glob": {'body_pose', 'global_orient', 'transl', 'betas'}, FxC
+ # "cam_Rt": tensor(F, 3),
+ # "cam_K": tensor(1, 10),
+ # }
+ self.seqs = list(self.motion_files.keys())
+ Log.info(f"[H36M] {len(self.seqs)} sequences. Elapsed: {Log.time() - tic:.2f}s")
+
+ # img(as feature)
+ # vid -> (features, vid, meta {bbx_xys, K_fullimg})
+ if not self.lazy_load:
+ tic = Log.time()
+ fn = self.root / "vitfeat_h36m.pt"
+ Log.info(f"[H36M] Fully Loading to RAM ViT-Feat: {fn}")
+ self.f_img_dicts = torch.load(fn)
+ Log.info(f"[H36M] Finished. Elapsed: {Log.time() - tic:.2f}s")
+ else:
+ raise NotImplementedError # "Check BEDLAM-SMPL for lazy_load"
+
+ def _get_idx2meta(self):
+ # We expect to see the entire sequence during one epoch,
+ # so each sequence will be sampled max(SeqLength // MotionFrames, 1) times
+ seq_lengths = []
+ self.idx2meta = []
+ for vid in self.f_img_dicts:
+ seq_length = self.f_img_dicts[vid]["bbx_xys"].shape[0]
+ num_samples = max(seq_length // self.motion_frames, 1)
+ seq_lengths.append(seq_length)
+ self.idx2meta.extend([vid] * num_samples)
+ hours = sum(seq_lengths) / 25 / 3600
+ Log.info(f"[H36M] has {hours:.1f} hours motion -> Resampled to {len(self.idx2meta)} samples.")
+
+ def _load_data(self, idx):
+ sampled_motion = {}
+ vid = self.idx2meta[idx]
+ motion = self.motion_files[vid]
+ seq_length = self.f_img_dicts[vid]["bbx_xys"].shape[0] # this is a better choice
+ sampled_motion["vid"] = vid
+
+ # Random select a subset
+ target_length = self.motion_frames
+ if target_length > seq_length: # this should not happen
+ start = 0
+ length = seq_length
+ Log.info(f"[H36M] ({idx}) target length < sequence length: {target_length} <= {seq_length}")
+ else:
+ start = np.random.randint(0, seq_length - target_length)
+ length = target_length
+ end = start + length
+ sampled_motion["length"] = length
+ sampled_motion["start_end"] = (start, end)
+
+ # Select motion subset
+ # body_pose, global_orient, transl, betas
+ sampled_motion["smpl_params_global"] = {k: v[start:end] for k, v in motion["smpl_params_glob"].items()}
+
+ # Image as feature
+ f_img_dict = self.f_img_dicts[vid]
+ sampled_motion["f_imgseq"] = f_img_dict["features"][start:end].float() # (L, 1024)
+ sampled_motion["bbx_xys"] = f_img_dict["bbx_xys"][start:end]
+ sampled_motion["K_fullimg"] = f_img_dict["K_fullimg"]
+ # sampled_motion["kp2d"] = self.vitpose[vid][start:end].float() # (L, 17, 3)
+ sampled_motion["kp2d"] = torch.zeros((end - start), 17, 3) # (L, 17, 3)
+
+ # Camera
+ sampled_motion["T_w2c"] = motion["cam_Rt"] # (4, 4)
+
+ return sampled_motion
+
+ def _process_data(self, data, idx):
+ length = data["length"]
+
+ # SMPL params in world
+ smpl_params_w = data["smpl_params_global"].copy() # in az
+
+ # SMPL params in cam
+ T_w2c = data["T_w2c"] # (4, 4)
+ offset = self.smpl_model.get_skeleton(smpl_params_w["betas"][0])[0] # (3)
+ global_orient_c, transl_c = get_c_rootparam(
+ smpl_params_w["global_orient"],
+ smpl_params_w["transl"],
+ T_w2c,
+ offset,
+ )
+ smpl_params_c = {
+ "body_pose": smpl_params_w["body_pose"].clone(), # (F, 63)
+ "betas": smpl_params_w["betas"].clone(), # (F, 10)
+ "global_orient": global_orient_c, # (F, 3)
+ "transl": transl_c, # (F, 3)
+ }
+
+ # World params
+ gravity_vec = torch.tensor([0, 0, -1]).float() # (3), H36M is az
+ T_w2c = T_w2c.repeat(length, 1, 1) # (F, 4, 4)
+ R_c2gv = get_R_c2gv(T_w2c[..., :3, :3], axis_gravity_in_w=gravity_vec) # (F, 3, 3)
+
+ # Image
+ bbx_xys = data["bbx_xys"] # (F, 3)
+ K_fullimg = data["K_fullimg"].repeat(length, 1, 1) # (F, 3, 3)
+ f_imgseq = data["f_imgseq"] # (F, 1024)
+ cam_angvel = compute_cam_angvel(T_w2c[:, :3, :3]) # (F, 6) slightly different from WHAM
+
+ # Returns: do not forget to make it batchable! (last lines)
+ max_len = self.motion_frames
+ return_data = {
+ "meta": {"data_name": "h36m", "idx": idx, "vid": data["vid"]},
+ "length": length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w,
+ "R_c2gv": R_c2gv, # (F, 3, 3)
+ "gravity_vec": gravity_vec, # (3)
+ "bbx_xys": bbx_xys, # (F, 3)
+ "K_fullimg": K_fullimg, # (F, 3, 3)
+ "f_imgseq": f_imgseq, # (F, D)
+ "kp2d": data["kp2d"], # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "mask": {
+ "valid": get_valid_mask(max_len, length),
+ "vitpose": False,
+ "bbx_xys": True,
+ "f_imgseq": True,
+ "spv_incam_only": False,
+ },
+ }
+
+ if False: # Render to image to check
+ smplx_out = self.smplx(**smpl_params_c)
+ # ----- Overlay ----- #
+ mid = return_data["meta"]["mid"]
+ video_path = self.root / f"videos/{mid}.mp4"
+ images = read_video_np(video_path, data["start_end"][0], data["start_end"][1])
+ render_dict = {
+ "K": K_fullimg[:1], # only support batch size 1
+ "faces": self.smplx.faces,
+ "verts": smplx_out.vertices,
+ "background": images,
+ }
+ img_overlay = simple_render_mesh_background(render_dict)
+ save_video(img_overlay, f"tmp.mp4")
+
+ # Batchable
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(return_data["smpl_params_c"], max_len)
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(return_data["smpl_params_w"], max_len)
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["bbx_xys"] = repeat_to_max_len(return_data["bbx_xys"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(return_data["K_fullimg"], max_len)
+ return_data["f_imgseq"] = repeat_to_max_len(return_data["f_imgseq"], max_len)
+ return_data["kp2d"] = repeat_to_max_len(return_data["kp2d"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(return_data["cam_angvel"], max_len)
+
+ return return_data
+
+
+group_name = "train_datasets/imgfeat_h36m"
+node_v1 = builds(H36mSmplDataset)
+MainStore.store(name="v1", node=node_v1, group=group_name)
diff --git a/third_party/GVHMR/hmr4d/dataset/h36m/utils.py b/third_party/GVHMR/hmr4d/dataset/h36m/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..87091268c2b2bd8554aa95dfc6a0cda0a8e791c2
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/h36m/utils.py
@@ -0,0 +1,82 @@
+import json
+import numpy as np
+from pathlib import Path
+from collections import defaultdict
+import pickle
+import torch
+
+RESOURCE_FOLDER = Path(__file__).resolve().parent / "resource"
+
+camera_idx_to_name = {0: "54138969", 1: "55011271", 2: "58860488", 3: "60457274"}
+
+
+def get_vid(pkl_path, cam_id):
+ """.../S6/Posing 1.pkl, 54138969 -> S6@Posing_1@54138969"""
+ sub_id, fn = pkl_path.split("/")[-2:]
+ vid = f"{sub_id}@{fn.split('.')[0].replace(' ', '_')}@{cam_id}"
+ return vid
+
+
+def get_raw_pkl_paths(h36m_raw_root):
+ smpl_param_dir = h36m_raw_root / "neutrSMPL_H3.6"
+ pkl_paths = []
+ for train_sub in ["S1", "S5", "S6", "S7", "S8"]:
+ for pth in (smpl_param_dir / train_sub).glob("*.pkl"):
+ if "aligned" not in str(pth): # Use world sequence only
+ pkl_paths.append(str(pth))
+
+ return pkl_paths
+
+
+def get_cam_KRts():
+ """
+ Returns:
+ Ks (torch.Tensor): {cam_id: 3x3}
+ Rts (torch.Tensor): {subj_id: {cam_id: 4x4}}
+ """
+ # this file is copied from https://github.com/karfly/human36m-camera-parameters
+ cameras_path = RESOURCE_FOLDER / "camera-parameters.json"
+ with open(cameras_path, "r") as f:
+ cameras = json.load(f)
+
+ # 4 camera ids: '54138969', '55011271', '58860488', '60457274'
+ Ks = {}
+ for cam in cameras["intrinsics"]:
+ Ks[cam] = torch.tensor(cameras["intrinsics"][cam]["calibration_matrix"]).float()
+
+ # extrinsics
+ extrinsics = cameras["extrinsics"]
+ Rts = defaultdict(dict)
+ for subj in extrinsics:
+ for cam in extrinsics[subj]:
+ Rt = torch.eye(4)
+ Rt[:3, :3] = torch.tensor(extrinsics[subj][cam]["R"])
+ Rt[:3, [3]] = torch.tensor(extrinsics[subj][cam]["t"]) / 1000
+ Rts[subj][cam] = Rt.float()
+
+ return Ks, Rts
+
+
+def parse_raw_pkl(pkl_path, to_50hz=True):
+ """
+ raw_pkl @ 200Hz, where video @ 50Hz.
+ the frames should be divided by 4, and mannually align with the video.
+ """
+ with open(str(pkl_path), "rb") as f:
+ data = pickle.load(f, encoding="bytes")
+ poses = torch.from_numpy(data[b"poses"]).float()
+ betas = torch.from_numpy(data[b"betas"]).float()
+ trans = torch.from_numpy(data[b"trans"]).float()
+ assert poses.shape[0] == trans.shape[0]
+ if to_50hz:
+ poses = poses[::4]
+ trans = trans[::4]
+
+ seq_length = poses.shape[0] # 50FPS
+ smpl_params = {
+ "body_pose": poses[:, 3:],
+ "betas": betas[None].expand(seq_length, -1),
+ "global_orient": poses[:, :3],
+ "transl": trans,
+ }
+ return smpl_params
diff --git a/third_party/GVHMR/hmr4d/dataset/imgfeat_motion/base_dataset.py b/third_party/GVHMR/hmr4d/dataset/imgfeat_motion/base_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..4df66dfd7239b56ccffa9412599ed1c839a98ecf
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/imgfeat_motion/base_dataset.py
@@ -0,0 +1,32 @@
+import torch
+from torch.utils import data
+import numpy as np
+from pathlib import Path
+from hmr4d.utils.pylogger import Log
+
+
+class ImgfeatMotionDatasetBase(data.Dataset):
+ def __init__(self):
+ super().__init__()
+ self._load_dataset()
+ self._get_idx2meta() # -> Set self.idx2meta
+
+ def __len__(self):
+ return len(self.idx2meta)
+
+ def _load_dataset(self):
+ raise NotImplemented
+
+ def _get_idx2meta(self):
+ raise NotImplemented
+
+ def _load_data(self, idx):
+ raise NotImplemented
+
+ def _process_data(self, data, idx):
+ raise NotImplemented
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data, idx)
+ return data
diff --git a/third_party/GVHMR/hmr4d/dataset/pure_motion/amass.py b/third_party/GVHMR/hmr4d/dataset/pure_motion/amass.py
new file mode 100644
index 0000000000000000000000000000000000000000..5070bcc978788337c4729b785340667cf834777c
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/pure_motion/amass.py
@@ -0,0 +1,119 @@
+import torch
+import torch.nn.functional as F
+import numpy as np
+
+from tqdm import tqdm
+from pathlib import Path
+from hmr4d.utils.pylogger import Log
+from hmr4d.configs import MainStore, builds
+
+from .base_dataset import BaseDataset
+from .utils import *
+from hmr4d.utils.geo.hmr_global import get_tgtcoord_rootparam
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines, convert_motion_as_line_mesh
+
+
+class AmassDataset(BaseDataset):
+ def __init__(
+ self,
+ motion_frames=120,
+ l_factor=1.5, # speed augmentation
+ skip_moyo=True, # not contained in the ICCV19 released version
+ cam_augmentation="v11",
+ random1024=False, # DEBUG
+ limit_size=None,
+ ):
+ self.root = Path("inputs/AMASS/hmr4d_support")
+ self.motion_frames = motion_frames
+ self.l_factor = l_factor
+ self.random1024 = random1024
+ self.skip_moyo = skip_moyo
+ self.dataset_name = "AMASS"
+ super().__init__(cam_augmentation, limit_size)
+
+ def _load_dataset(self):
+ filename = self.root / "smplxpose_v2.pth"
+ Log.info(f"[{self.dataset_name}] Loading from {filename} ...")
+ tic = Log.time()
+ if self.random1024: # Debug, faster loading
+ try:
+ Log.info(f"[{self.dataset_name}] Loading 1024 samples for debugging ...")
+ self.motion_files = torch.load(self.root / "smplxpose_v2_random1024.pth")
+ except:
+ Log.info(f"[{self.dataset_name}] Not found! Saving 1024 samples for debugging ...")
+ self.motion_files = torch.load(filename)
+ keys = list(self.motion_files.keys())
+ keys = np.random.choice(keys, 1024, replace=False)
+ self.motion_files = {k: self.motion_files[k] for k in keys}
+ torch.save(self.motion_files, self.root / "smplxpose_v2_random1024.pth")
+ else:
+ self.motion_files = torch.load(filename)
+ self.seqs = list(self.motion_files.keys())
+ Log.info(f"[{self.dataset_name}] {len(self.seqs)} sequences. Elapsed: {Log.time() - tic:.2f}s")
+
+ def _get_idx2meta(self):
+ # We expect to see the entire sequence during one epoch,
+ # so each sequence will be sampled max(SeqLength // MotionFrames, 1) times
+ seq_lengths = []
+ self.idx2meta = []
+
+ # Skip too-long idle-prefix
+ motion_start_id = {}
+ for vid in self.motion_files:
+ if self.skip_moyo and "moyo_smplxn" in vid:
+ continue
+ seq_length = self.motion_files[vid]["pose"].shape[0]
+ start_id = motion_start_id[vid] if vid in motion_start_id else 0
+ seq_length = seq_length - start_id
+ if seq_length < 25: # Skip clips that are too short
+ continue
+ num_samples = max(seq_length // self.motion_frames, 1)
+ seq_lengths.append(seq_length)
+ self.idx2meta.extend([(vid, start_id)] * num_samples)
+ hours = sum(seq_lengths) / 30 / 3600
+ Log.info(f"[{self.dataset_name}] has {hours:.1f} hours motion -> Resampled to {len(self.idx2meta)} samples.")
+
+ def _load_data(self, idx):
+ """
+ - Load original data
+ - Augmentation: speed-augmentation to L frames
+ """
+ # Load original data
+ mid, start_id = self.idx2meta[idx]
+ raw_data = self.motion_files[mid]
+ raw_len = raw_data["pose"].shape[0] - start_id
+ data = {
+ "body_pose": raw_data["pose"][start_id:, 3:], # (F, 63)
+ "betas": raw_data["beta"].repeat(raw_len, 1), # (10)
+ "global_orient": raw_data["pose"][start_id:, :3], # (F, 3)
+ "transl": raw_data["trans"][start_id:], # (F, 3)
+ }
+
+ # Get {tgt_len} frames from data
+ # Random select a subset with speed augmentation [start, end)
+ tgt_len = self.motion_frames
+ raw_subset_len = np.random.randint(int(tgt_len / self.l_factor), int(tgt_len * self.l_factor))
+ if raw_subset_len <= raw_len:
+ start = np.random.randint(0, raw_len - raw_subset_len + 1)
+ end = start + raw_subset_len
+ else: # interpolation will use all possible frames (results in a slow motion)
+ start = 0
+ end = raw_len
+ data = {k: v[start:end] for k, v in data.items()}
+
+ # Interpolation (vec + r6d)
+ data_interpolated = interpolate_smpl_params(data, tgt_len)
+
+ # AZ -> AY
+ data_interpolated["global_orient"], data_interpolated["transl"], _ = get_tgtcoord_rootparam(
+ data_interpolated["global_orient"],
+ data_interpolated["transl"],
+ tsf="az->ay",
+ )
+
+ data_interpolated["data_name"] = "amass"
+ return data_interpolated
+
+
+group_name = "train_datasets/pure_motion_amass"
+MainStore.store(name="v11", node=builds(AmassDataset, cam_augmentation="v11"), group=group_name)
diff --git a/third_party/GVHMR/hmr4d/dataset/pure_motion/base_dataset.py b/third_party/GVHMR/hmr4d/dataset/pure_motion/base_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..29def088e640657f5a0daa1851a36ee3442fd649
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/pure_motion/base_dataset.py
@@ -0,0 +1,182 @@
+import torch
+from torch.utils.data import Dataset
+from pathlib import Path
+
+from .utils import *
+from .cam_traj_utils import CameraAugmentorV11
+from hmr4d.utils.geo.hmr_cam import create_camera_sensor
+from hmr4d.utils.geo.hmr_global import get_c_rootparam, get_R_c2gv
+from hmr4d.utils.net_utils import get_valid_mask, repeat_to_max_len, repeat_to_max_len_dict
+from hmr4d.utils.geo_transform import compute_cam_angvel, apply_T_on_points, project_p2d, cvt_p2d_from_i_to_c
+
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines, convert_motion_as_line_mesh
+from hmr4d.utils.smplx_utils import make_smplx
+
+
+class BaseDataset(Dataset):
+ def __init__(self, cam_augmentation, limit_size=None):
+ super().__init__()
+ self.cam_augmentation = cam_augmentation
+ self.limit_size = limit_size
+ self.smplx = make_smplx("supermotion")
+ self.smplx_lite = make_smplx("supermotion_smpl24")
+
+ self._load_dataset()
+ self._get_idx2meta()
+
+ def _load_dataset(self):
+ NotImplementedError("_load_dataset is not implemented")
+
+ def _get_idx2meta(self):
+ self.idx2meta = None
+ NotImplementedError("_get_idx2meta is not implemented")
+
+ def __len__(self):
+ if self.limit_size is not None:
+ return min(self.limit_size, len(self.idx2meta))
+ return len(self.idx2meta)
+
+ def _load_data(self, idx):
+ NotImplementedError("_load_data is not implemented")
+
+ def _process_data(self, data, idx):
+ """
+ Args:
+ data: dict {
+ "body_pose": (F, 63),
+ "betas": (F, 10),
+ "global_orient": (F, 3), in the AY coordinates
+ "transl": (F, 3), in the AY coordinates
+ }
+ """
+ data_name = data["data_name"]
+ length = data["body_pose"].shape[0]
+ # Augmentation: betas, SMPL (gravity-axis)
+ body_pose = data["body_pose"]
+ betas = augment_betas(data["betas"], std=0.1)
+ global_orient_w, transl_w = rotate_around_axis(data["global_orient"], data["transl"], axis="y")
+ del data
+
+ # SMPL_params in world
+ smpl_params_w = {
+ "body_pose": body_pose, # (F, 63)
+ "betas": betas, # (F, 10)
+ "global_orient": global_orient_w, # (F, 3)
+ "transl": transl_w, # (F, 3)
+ }
+
+ # Camera trajectory augmentation
+ if self.cam_augmentation == "v11":
+ # interleave repeat to original length (faster)
+ N = 10
+ w_j3d = self.smplx_lite(
+ smpl_params_w["body_pose"][::N],
+ smpl_params_w["betas"][::N],
+ smpl_params_w["global_orient"][::N],
+ None,
+ )
+ w_j3d = w_j3d.repeat_interleave(N, dim=0) + smpl_params_w["transl"][:, None] # (F, 24, 3)
+
+ if False:
+ wis3d = make_wis3d(name="debug_amass")
+ add_motion_as_lines(w_j3d, wis3d, "w_j3d")
+
+ width, height, K_fullimg = create_camera_sensor(1000, 1000, 43.3) # WHAM
+ focal_length = K_fullimg[0, 0]
+ wham_cam_augmentor = CameraAugmentorV11()
+ T_w2c = wham_cam_augmentor(w_j3d, length) # (F, 4, 4)
+
+ else:
+ raise NotImplementedError
+
+ if False: # render
+ for idx_render in range(10):
+ T_w2c = wham_cam_augmentor(smpl_params_w["transl"])
+
+ # targets
+ w_j3d = self.smplx(**smpl_params_w).joints[:, :22]
+ c_j3d = apply_T_on_points(w_j3d, T_w2c)
+ verts, faces, vertex_colors = convert_motion_as_line_mesh(c_j3d)
+ vertex_colors = vertex_colors[None] / 255.0
+ bg = np.ones((height, width, 3), dtype=np.uint8) * 255
+
+ # render
+ renderer = Renderer(width, height, device="cuda", faces=faces, K=K_fullimg)
+ vname = f"{idx_render:02d}"
+ out_fn = Path(f"outputs/dump_render_wham_cam/{vname}.mp4")
+ out_fn.parent.mkdir(exist_ok=True, parents=True)
+ writer = imageio.get_writer(out_fn, fps=30, mode="I", format="FFMPEG", macro_block_size=1)
+ for i in tqdm(range(len(verts)), desc=f"Rendering {vname}"):
+ # incam
+ # img_overlay_pred = renderer.render_mesh(verts[i].cuda(), bg, [0.8, 0.8, 0.8], VI=1)
+ img_overlay_pred = renderer.render_mesh(verts[i].cuda(), bg, vertex_colors, VI=1)
+ # if batch["meta_render"][0].get("bbx_xys", None) is not None: # draw bbox lines
+ # bbx_xys = batch["meta_render"][0]["bbx_xys"][i].cpu().numpy()
+ # lu_point = (bbx_xys[:2] - bbx_xys[2:] / 2).astype(int)
+ # rd_point = (bbx_xys[:2] + bbx_xys[2:] / 2).astype(int)
+ # img_overlay_pred = cv2.rectangle(img_overlay_pred, lu_point, rd_point, (255, 178, 102), 2)
+
+ # write
+ writer.append_data(img_overlay_pred)
+ writer.close()
+ pass
+
+ # SMPL params in cam
+ offset = self.smplx.get_skeleton(smpl_params_w["betas"][0])[0] # (3)
+ global_orient_c, transl_c = get_c_rootparam(
+ smpl_params_w["global_orient"],
+ smpl_params_w["transl"],
+ T_w2c,
+ offset,
+ )
+ smpl_params_c = {
+ "body_pose": smpl_params_w["body_pose"].clone(), # (F, 63)
+ "betas": smpl_params_w["betas"].clone(), # (F, 10)
+ "global_orient": global_orient_c, # (F, 3)
+ "transl": transl_c, # (F, 3)
+ }
+
+ # World params
+ gravity_vec = torch.tensor([0, -1, 0], dtype=torch.float32) # (3), BEDLAM is ay
+ R_c2gv = get_R_c2gv(T_w2c[:, :3, :3], gravity_vec) # (F, 3, 3)
+
+ # Image
+ K_fullimg = K_fullimg.repeat(length, 1, 1) # (F, 3, 3)
+ cam_angvel = compute_cam_angvel(T_w2c[:, :3, :3]) # (F, 6)
+
+ # Returns: do not forget to make it batchable! (last lines)
+ # NOTE: bbx_xys and f_imgseq will be added later
+ max_len = length
+ return_data = {
+ "meta": {"data_name": data_name, "idx": idx, "T_w2c": T_w2c},
+ "length": length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w,
+ "R_c2gv": R_c2gv, # (F, 3, 3)
+ "gravity_vec": gravity_vec, # (3)
+ "bbx_xys": torch.zeros((length, 3)), # (F, 3) # NOTE: a placeholder
+ "K_fullimg": K_fullimg, # (F, 3, 3)
+ "f_imgseq": torch.zeros((length, 1024)), # (F, D) # NOTE: a placeholder
+ "kp2d": torch.zeros(length, 17, 3), # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "mask": {
+ "valid": get_valid_mask(length, length),
+ "vitpose": False,
+ "bbx_xys": False,
+ "f_imgseq": False,
+ "spv_incam_only": False,
+ },
+ }
+
+ # Batchable
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(return_data["smpl_params_c"], max_len)
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(return_data["smpl_params_w"], max_len)
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(return_data["K_fullimg"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(return_data["cam_angvel"], max_len)
+ return return_data
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data, idx)
+ return data
diff --git a/third_party/GVHMR/hmr4d/dataset/pure_motion/cam_traj_utils.py b/third_party/GVHMR/hmr4d/dataset/pure_motion/cam_traj_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..71b0affc64898c659032bcb0fdfac5ce8ac8ebf4
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/pure_motion/cam_traj_utils.py
@@ -0,0 +1,427 @@
+import torch
+import torch.nn.functional as F
+import numpy as np
+from numpy.random import rand, randn
+from pytorch3d.transforms import (
+ axis_angle_to_matrix,
+ matrix_to_axis_angle,
+ matrix_to_rotation_6d,
+ rotation_6d_to_matrix,
+)
+from einops import rearrange
+from hmr4d.utils.geo.hmr_cam import create_camera_sensor
+from hmr4d.utils.geo_transform import transform_mat, apply_T_on_points
+from hmr4d.utils.geo.transforms import axis_rotate_to_matrix
+import hmr4d.utils.matrix as matrix
+
+halfpi = np.pi / 2
+R_y_upsidedown = torch.tensor([[-1, 0, 0], [0, -1, 0], [0, 0, 1]]).float()
+
+
+def noisy_interpolation(x, length, step_noise_perc=0.2):
+ """Non-linear interpolation with noise, although with noise, the jittery is very small
+ Args:
+ x: (2, C)
+ length: scalar
+ step_noise_perc: [x0, x1 +-(step_noise_perc * step), x2], where step = x1-x0
+ """
+ assert x.shape[0] == 2 and len(x.shape) == 2
+ dim = x.shape[-1]
+ output = np.zeros((length, dim))
+
+ # Use linsapce(0, 1) +- noise as reference
+ linspace = np.repeat(np.linspace(0, 1, length)[None], dim, axis=0) # (D, L)
+ noise = (linspace[0, 1] - linspace[0, 0]) * step_noise_perc
+ space_noise = np.random.uniform(-noise, noise, (dim, length - 2)) # (D, L-2)
+ linspace[:, 1:-1] = linspace[:, 1:-1] + space_noise
+
+ # Do 1d interp
+ for i in range(dim):
+ output[:, i] = np.interp(linspace[i], np.array([0.0, 1.0]), x[:, i])
+ return output
+
+
+def noisy_impluse_interpolation(data1, data2, step_noise_perc=0.2):
+ """Non-linear interpolation of impluse with noise"""
+
+ dim = data1.shape[-1]
+ L = data1.shape[0]
+
+ linspace1 = np.stack([np.linspace(0, 1, L // 2) for _ in range(dim)])
+ linspace2 = np.stack([np.linspace(0, 1, L // 2)[::-1] for _ in range(dim)])
+ linspace = np.concatenate([linspace1, linspace2], axis=-1)
+ noise = (linspace[0, 1] - linspace[0, 0]) * step_noise_perc
+ space_noise = np.stack([np.random.uniform(-noise, noise, L - 2) for _ in range(dim)])
+
+ linspace[:, 1:-1] = linspace[:, 1:-1] + space_noise
+ linspace = linspace.T
+ output = data1 * (1 - linspace) + data2 * linspace
+ return output
+
+
+def create_camera(w_root, cfg):
+ """Create static camera pose
+ Args:
+ w_root: (3,), y-up coordinates
+ Returns:
+ R_w2c: (3, 3)
+ t_w2c: (3)
+ """
+ # Parse
+ pitch_std = cfg["pitch_std"]
+ pitch_mean = cfg["pitch_mean"]
+ roll_std = cfg["roll_std"]
+ tz_range1_prob = cfg["tz_range1_prob"]
+ tz_range1 = cfg["tz_range1"]
+ tz_range2 = cfg["tz_range2"]
+ f = cfg["f"]
+ w = cfg["w"]
+
+ # algo
+ yaw = rand() * 2 * np.pi # Look at any direction in xz-plane
+ pitch = np.clip(randn() * pitch_std + pitch_mean, -halfpi, halfpi)
+ roll = np.clip(randn() * roll_std, -halfpi, halfpi) # Normal-dist
+
+ # Note we use OpenCV's camera system by first applying R_y_upsidedown
+ yaw_rm = axis_rotate_to_matrix(yaw, axis="y")
+ pitch_rm = axis_rotate_to_matrix(pitch, axis="x")
+ roll_rm = axis_rotate_to_matrix(roll, axis="z")
+ R_w2c = (roll_rm @ pitch_rm @ yaw_rm @ R_y_upsidedown).squeeze(0) # (3, 3)
+
+ # Place people in the scene
+ if rand() < tz_range1_prob:
+ tz = rand() * (tz_range1[1] - tz_range1[0]) + tz_range1[0]
+ max_dist_in_fov = (w / 2) / f * tz
+ tx = (rand() * 2 - 1) * 0.7 * max_dist_in_fov
+ ty = (rand() * 2 - 1) * 0.5 * max_dist_in_fov
+
+ else:
+ tz = rand() * (tz_range2[1] - tz_range2[0]) + tz_range2[0]
+ max_dist_in_fov = (w / 2) / f * tz
+ max_dist_in_fov *= 0.9 # add a threshold
+ tx = torch.randn(1) * 1.6
+ tx = torch.clamp(tx, -max_dist_in_fov, max_dist_in_fov)
+ ty = torch.randn(1) * 0.8
+ ty = torch.clamp(ty, -max_dist_in_fov, max_dist_in_fov)
+
+ dist = torch.tensor([tx, ty, tz], dtype=torch.float)
+ t_w2c = dist - torch.matmul(R_w2c, w_root)
+
+ return R_w2c, t_w2c
+
+
+def create_rotation_move(R, length, r_xyz_w_std=[np.pi / 8, np.pi / 4, np.pi / 8]):
+ """Create rotational move for the camera
+ Args:
+ R: (3, 3)
+ Return:
+ R_move: (L, 3, 3)
+ """
+ # Create final camera pose
+ assert len(R.size()) == 2
+ r_xyz = (2 * rand(3) - 1) * r_xyz_w_std
+ Rf = R @ axis_angle_to_matrix(torch.from_numpy(r_xyz).float())
+
+ # Inbetweening two poses
+ Rs = torch.stack((R, Rf)) # (2, 3, 3)
+ rs = matrix_to_rotation_6d(Rs).numpy() # (2, 6)
+ rs_move = noisy_interpolation(rs, length) # (L, 6)
+ R_move = rotation_6d_to_matrix(torch.from_numpy(rs_move).float())
+
+ return R_move
+
+
+def create_translation_move(R_w2c, t_w2c, length, t_xyz_w_std=[1.0, 0.25, 1.0]):
+ """Create translational move for the camera
+ Args:
+ R_w2c: (3, 3),
+ t_w2c: (3,),
+ """
+ # Create subject final displacement
+ subj_start_final = np.array([[0, 0, 0], randn(3) * t_xyz_w_std])
+ subj_move = noisy_interpolation(subj_start_final, length)
+ subj_move = torch.from_numpy(subj_move).float() # (L, 3)
+
+ # Equal to camera move
+ t_move = t_w2c + torch.einsum("ij,lj->li", R_w2c, subj_move)
+
+ return t_move
+
+
+class CameraAugmentorV11:
+ cfg_create_camera = {
+ "pitch_mean": np.pi / 36,
+ "pitch_std": np.pi / 8,
+ "roll_std": np.pi / 24,
+ "tz_range1_prob": 0.4,
+ "tz_range1": [1.0, 6.0], # uniform sample
+ "tz_range2": [4.0, 12.0],
+ "tx_scale": 0.7,
+ "ty_scale": 0.3,
+ }
+
+ # r_xyz_w_std = [np.pi / 8, np.pi / 4, np.pi / 8] # in world coords
+ r_xyz_w_std = [np.pi / 6, np.pi / 3, np.pi / 6] # in world coords
+ t_xyz_w_std = [1.0, 0.25, 1.0] # in world coords
+ r_xyz_w_std_half = [x / 2 for x in r_xyz_w_std]
+ t_xyz_w_std_half = [x / 2 for x in t_xyz_w_std]
+
+ t_factor = 1.0
+ tz_bias_factor = 1.0
+
+ rotx_impluse_noise = np.pi / 36
+ roty_impluse_noise = np.pi / 36
+ rotz_impluse_noise = np.pi / 36
+ rot_impluse_n = 1
+
+ tx_step_noise = 0.0025
+ ty_step_noise = 0.0025
+ tz_step_noise = 0.0025
+
+ tx_impluse_noise = 0.15
+ ty_impluse_noise = 0.15
+ tz_impluse_noise = 0.15
+ t_impluse_n = 1
+
+ # === Postprocess === #
+ height_max = 4.0
+ height_min = -2.0 # -1.5 -> -2.0 allow look upside
+ tz_post_min = 0.5
+
+ def __init__(self):
+ self.w = 1000
+ self.f = create_camera_sensor(1000, 1000, 24)[2][0, 0] # use 24mm camera
+ self.half_fov_tol = (self.w / 2) / self.f
+
+ def create_rotation_track(self, cam_mat, root, rx_factor=1.0, ry_factor=1.0, rz_factor=1.0):
+ """Create rotational move for the camera with rotating human"""
+ human_mat = matrix.get_TRS(matrix.identity_mat()[None, :3, :3], root)
+ cam2human_mat = matrix.get_mat_BtoA(human_mat, cam_mat)
+ R = matrix.get_rotation(cam2human_mat)
+
+ # Create final camera pose
+ yaw = np.random.normal(scale=ry_factor)
+ pitch = np.random.normal(scale=rx_factor)
+ roll = np.random.normal(scale=rz_factor)
+
+ yaw_rm = axis_angle_to_matrix(torch.tensor([0, yaw, 0]).float())
+ pitch_rm = axis_angle_to_matrix(torch.tensor([pitch, 0, 0]).float())
+ roll_rm = axis_angle_to_matrix(torch.tensor([0, 0, roll]).float())
+ Rf = roll_rm @ pitch_rm @ yaw_rm @ R[0]
+
+ # Inbetweening two poses
+ Rs = torch.stack((R[0], Rf))
+ rs = matrix_to_rotation_6d(Rs).numpy()
+ rs_move = noisy_interpolation(rs, self.l)
+ R_move = rotation_6d_to_matrix(torch.from_numpy(rs_move).float())
+ R_move = torch.inverse(R_move)
+ return R_move
+
+ def create_translation_track(self, cam_mat, root, t_factor=1.0, tz_bias_factor=0.0):
+ """Create translational move for the camera with tracking human"""
+ delta_T0 = matrix.get_position(cam_mat)[0] - root[0]
+ T_new = matrix.get_position(cam_mat)
+
+ tz_bias = delta_T0.norm(dim=-1) * tz_bias_factor * np.clip(1 + np.random.normal(scale=0.1), 0.67, 1.5)
+
+ T_new[1:] = root[1:] + delta_T0
+ cam_mat = matrix.get_TRS(matrix.get_rotation(cam_mat), T_new)
+ w2c = torch.inverse(cam_mat)
+ T_new = matrix.get_position(w2c)
+
+ # Create final camera position
+ tx = np.random.normal(scale=t_factor)
+ ty = np.random.normal(scale=t_factor)
+ tz = np.random.normal(scale=t_factor) + tz_bias
+ Ts = np.array([[0, 0, 0], [tx, ty, tz]])
+
+ T_move = noisy_interpolation(Ts, self.l)
+ T_move = torch.from_numpy(T_move).float()
+ return T_move + T_new
+
+ def add_stepnoise(self, R, T):
+ w2c = matrix.get_TRS(R, T)
+ cam_mat = torch.inverse(w2c)
+ R_new = matrix.get_rotation(cam_mat)
+ T_new = matrix.get_position(cam_mat)
+
+ L = R_new.shape[0]
+ window = 10
+
+ def add_impulse_rot(R_new):
+ N = np.random.randint(1, self.rot_impluse_n + 1)
+ rx = np.random.normal(scale=self.rotx_impluse_noise, size=N)
+ ry = np.random.normal(scale=self.roty_impluse_noise, size=N)
+ rz = np.random.normal(scale=self.rotz_impluse_noise, size=N)
+ R_impluse_noise = axis_angle_to_matrix(torch.from_numpy(np.array([rx, ry, rz])).float().transpose(0, 1))
+ R_noise = R_new.clone()
+ last_i = 0
+ for i in range(N):
+ n_i = np.random.randint(last_i + window, L - (N - i) * window * 2)
+
+ # make impluse smooth
+ window_R = R_noise[n_i - window : n_i + window].clone()
+ window_r = matrix_to_rotation_6d(window_R).numpy()
+ impluse_R = R_impluse_noise[i] @ window_R[window]
+ window_impluse_R = window_R.clone()
+ window_impluse_R[:] = impluse_R[None]
+ window_impluse_r = matrix_to_rotation_6d(window_impluse_R).numpy()
+
+ window_new_r = noisy_impluse_interpolation(window_r, window_impluse_r)
+ window_new_R = rotation_6d_to_matrix(torch.from_numpy(window_new_r).float())
+ R_noise[n_i - window : n_i + window] = window_new_R
+ last_i = n_i
+ R_new = R_noise
+ return R_new
+
+ def add_impulse_t(T_new):
+ N = np.random.randint(1, self.t_impluse_n + 1)
+ tx = np.random.normal(scale=self.tx_impluse_noise, size=N)
+ ty = np.random.normal(scale=self.ty_impluse_noise, size=N)
+ tz = np.random.normal(scale=self.tz_impluse_noise, size=N)
+ T_impluse_noise = torch.from_numpy(np.array([tx, ty, tz])).float().transpose(0, 1)
+ T_noise = T_new.clone()
+ last_i = 0
+ for i in range(N):
+ n_i = np.random.randint(last_i + window, L - N * window * 2)
+
+ # make impluse smooth
+ window_T = T_noise[n_i - window : n_i + window].clone()
+ window_impluse_T = window_T.clone()
+ window_impluse_T += T_impluse_noise[i : i + 1]
+ window_impluse_T = window_impluse_T.numpy()
+ window_T = window_T.numpy()
+
+ window_new_T = noisy_impluse_interpolation(window_T, window_impluse_T)
+ window_new_T = torch.from_numpy(window_new_T).float()
+ T_noise[n_i - window : n_i + window] = window_new_T
+ last_i = n_i
+ T_new = T_noise
+ return T_new
+
+ impulse_type_prob = {
+ "t": 0.2,
+ "r": 0.2,
+ "both": 0.1,
+ "pass": 0.5,
+ }
+ impulse_type = np.random.choice(list(impulse_type_prob.keys()), p=list(impulse_type_prob.values()))
+ if impulse_type == "t":
+ # impluse translation only
+ T_new = add_impulse_t(T_new)
+ elif impulse_type == "r":
+ # impluse rotation only
+ R_new = add_impulse_rot(R_new)
+ elif impulse_type == "both":
+ # impluse rotation and translation
+ R_new = add_impulse_rot(R_new)
+ T_new = add_impulse_t(T_new)
+ else:
+ assert impulse_type == "pass"
+
+ cam_mat_new = matrix.get_TRS(R_new, T_new)
+ w2c_new = torch.inverse(cam_mat_new)
+ R_new = matrix.get_rotation(w2c_new)
+ T_new = matrix.get_position(w2c_new)
+ tx = np.random.normal(scale=self.tx_step_noise, size=L)
+ ty = np.random.normal(scale=self.ty_step_noise, size=L)
+ tz = np.random.normal(scale=self.tz_step_noise, size=L)
+ T_new = T_new + torch.from_numpy(np.array([tx, ty, tz])).float().transpose(0, 1)
+
+ return R_new, T_new
+
+ def __call__(self, w_j3d, length=120):
+ """
+ Args:
+ w_j3d: (L, J, 3)
+ length: scalar
+ """
+ # Check
+ self.l = length
+ assert w_j3d.size(0) == self.l, "currently, only support fixed length"
+
+ # Setup
+ w_j3d = w_j3d.clone()
+ w_root = w_j3d[:, 0] # (L, 3)
+
+ # Simulate a static camera pose
+ cfg_camera0 = {**self.cfg_create_camera, "w": self.w, "f": self.f}
+ R0_w2c, t0_w2c = create_camera(w_root[0], cfg_camera0) # (3, 3) and (3,)
+
+ # Move camera
+ camera_type_prob = {
+ "random": 0.25,
+ "track": 0.15,
+ "trackrotate": 0.10,
+ "trackpush": 0.05,
+ "trackpull": 0.05,
+ "static": 0.4,
+ }
+ camera_type = np.random.choice(list(camera_type_prob.keys()), p=list(camera_type_prob.values()))
+ if camera_type == "random": # random move + add noise on cam
+ R_w2c = create_rotation_move(R0_w2c, length, self.r_xyz_w_std)
+ t_w2c = create_translation_move(R0_w2c, t0_w2c, length, self.t_xyz_w_std)
+ R_w2c, t_w2c = self.add_stepnoise(R_w2c, t_w2c)
+
+ elif camera_type == "track": # track human
+ R_w2c = create_rotation_move(R0_w2c, length, self.r_xyz_w_std_half)
+ cam_mat = torch.inverse(transform_mat(R0_w2c, t0_w2c)).repeat(length, 1, 1) # (F, 4, 4)
+ t_w2c = self.create_translation_track(cam_mat, w_root, 0.5)
+ R_w2c, t_w2c = self.add_stepnoise(R_w2c, t_w2c)
+
+ elif camera_type == "trackrotate": # track human and rotate
+ cam_mat = torch.inverse(transform_mat(R0_w2c, t0_w2c)).repeat(length, 1, 1) # (F, 4, 4)
+ t_w2c = self.create_translation_track(cam_mat, w_root, 0.5)
+ cam_mat = matrix.get_TRS(matrix.get_rotation(cam_mat), t_w2c)
+ R_w2c = self.create_rotation_track(cam_mat, w_root, np.pi / 16, np.pi, np.pi / 16)
+ R_w2c, t_w2c = self.add_stepnoise(R_w2c, t_w2c)
+
+ elif camera_type == "trackpush": # track human and push close to human
+ R_w2c = create_rotation_move(R0_w2c, length, self.r_xyz_w_std_half)
+ # [1/tz_bias_factor, 1] * dist
+ cam_mat = torch.inverse(transform_mat(R0_w2c, t0_w2c)).repeat(length, 1, 1) # (F, 4, 4)
+ t_w2c = self.create_translation_track(cam_mat, w_root, 0.5, (1.0 / (1 + self.tz_bias_factor) - 1))
+ R_w2c, t_w2c = self.add_stepnoise(R_w2c, t_w2c)
+
+ elif camera_type == "trackpull": # track human and pull far from human
+ R_w2c = create_rotation_move(R0_w2c, length, self.r_xyz_w_std_half)
+ # [1, (tz_bias_factor + 1)] * dist
+ cam_mat = torch.inverse(transform_mat(R0_w2c, t0_w2c)).repeat(length, 1, 1) # (F, 4, 4)
+ t_w2c = self.create_translation_track(cam_mat, w_root, 0.5, self.tz_bias_factor)
+ R_w2c, t_w2c = self.add_stepnoise(R_w2c, t_w2c)
+
+ else:
+ assert camera_type == "static"
+ R_w2c = R0_w2c.repeat(length, 1, 1) # (F, 3, 3)
+ t_w2c = t0_w2c.repeat(length, 1) # (F, 3)
+
+ # Recompute t_w2c for better camera height
+ # cam_w = torch.einsum("lji,lj->li", R_w2c, -t_w2c) # (L, 3), camera center in world: cam_w = - R_w2c^t_w2c @ t
+ # height = cam_w[..., 1] - w_root[:, 1]
+ # height = torch.clamp(height, self.height_min, self.height_max)
+ # new_pos = cam_w.clone()
+ # new_pos[:, 1] = w_root[:, 1] + height
+ # t_w2c = torch.einsum("lij,lj->li", R_w2c, -new_pos) # (L, 3), new t = -R_w2c @ cam_w
+
+ # Recompute t_w2c for better depth and FoV
+ c_j3d = torch.einsum("lij,lkj->lki", R_w2c, w_j3d) + t_w2c[:, None] # (L, J, 3)
+ delta = torch.zeros_like(t_w2c) # (L, 3) this will be later added to t_w2c
+ # - If the person is too close to the camera, push away the person in the z direction
+ c_j3d_min = c_j3d[..., 2].min() # scalar
+ if c_j3d_min < self.tz_post_min:
+ push_away = self.tz_post_min - c_j3d_min
+ delta[..., 2] += push_away
+ c_j3d[..., 2] += push_away
+ # - If the person is not in the FoV, push away the person in the z direction
+ c_root = c_j3d[:, 0] # (L, 3)
+ half_fov = torch.div(c_root[:, :2], c_root[:, 2:]).abs() # (L, 2), [x/z, y/z]
+ if half_fov.max() > self.half_fov_tol:
+ max_idx1, max_idx2 = torch.where(torch.max(half_fov) == half_fov)
+ max_idx1, max_idx2 = max_idx1[0], max_idx2[0]
+ z_trg = c_root[max_idx1, max_idx2].abs() / self.half_fov_tol # extreme fitted z in the fov
+ push_away = z_trg - c_root[max_idx1, 2]
+ delta[..., 2] += push_away
+ t_w2c += delta
+
+ T_w2c = transform_mat(R_w2c, t_w2c) # (F, 4, 4)
+ return T_w2c
diff --git a/third_party/GVHMR/hmr4d/dataset/pure_motion/utils.py b/third_party/GVHMR/hmr4d/dataset/pure_motion/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..9466296b2b2bba6f08693d62ec65652c9dae9537
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/pure_motion/utils.py
@@ -0,0 +1,66 @@
+import torch
+import torch.nn.functional as F
+from pytorch3d.transforms import (
+ axis_angle_to_matrix,
+ matrix_to_axis_angle,
+ matrix_to_rotation_6d,
+ rotation_6d_to_matrix,
+)
+from einops import rearrange
+
+
+def aa_to_r6d(x):
+ return matrix_to_rotation_6d(axis_angle_to_matrix(x))
+
+
+def r6d_to_aa(x):
+ return matrix_to_axis_angle(rotation_6d_to_matrix(x))
+
+
+def interpolate_smpl_params(smpl_params, tgt_len):
+ """
+ smpl_params['body_pose'] (L, 63)
+ tgt_len: L->L'
+ """
+ betas = smpl_params["betas"]
+ body_pose = smpl_params["body_pose"]
+ global_orient = smpl_params["global_orient"] # (L, 3)
+ transl = smpl_params["transl"] # (L, 3)
+
+ # Interpolate
+ body_pose = rearrange(aa_to_r6d(body_pose.reshape(-1, 21, 3)), "l j c -> c j l")
+ body_pose = F.interpolate(body_pose, tgt_len, mode="linear", align_corners=True)
+ body_pose = r6d_to_aa(rearrange(body_pose, "c j l -> l j c")).reshape(-1, 63)
+
+ # although this should be the same as above, we do it for consistency
+ betas = rearrange(betas, "l c -> c 1 l")
+ betas = F.interpolate(betas, tgt_len, mode="linear", align_corners=True)
+ betas = rearrange(betas, "c 1 l -> l c")
+
+ global_orient = rearrange(aa_to_r6d(global_orient.reshape(-1, 1, 3)), "l j c -> c j l")
+ global_orient = F.interpolate(global_orient, tgt_len, mode="linear", align_corners=True)
+ global_orient = r6d_to_aa(rearrange(global_orient, "c j l -> l j c")).reshape(-1, 3)
+
+ transl = rearrange(transl, "l c -> c 1 l")
+ transl = F.interpolate(transl, tgt_len, mode="linear", align_corners=True)
+ transl = rearrange(transl, "c 1 l -> l c")
+
+ return {"body_pose": body_pose, "betas": betas, "global_orient": global_orient, "transl": transl}
+
+
+def rotate_around_axis(global_orient, transl, axis="y"):
+ """Global coordinate augmentation. Random rotation around y-axis"""
+ angle = torch.rand(1) * 2 * torch.pi
+ if axis == "y":
+ aa = torch.tensor([0.0, angle, 0.0]).float().unsqueeze(0)
+ rmat = axis_angle_to_matrix(aa)
+
+ global_orient = matrix_to_axis_angle(rmat @ axis_angle_to_matrix(global_orient))
+ transl = (rmat.squeeze(0) @ transl.T).T
+ return global_orient, transl
+
+
+def augment_betas(betas, std=0.1):
+ noise = torch.normal(mean=torch.zeros(10), std=torch.ones(10) * std)
+ betas_aug = betas + noise[None]
+ return betas_aug
diff --git a/third_party/GVHMR/hmr4d/dataset/rich/resource/cam2params.pt b/third_party/GVHMR/hmr4d/dataset/rich/resource/cam2params.pt
new file mode 100644
index 0000000000000000000000000000000000000000..ed6da4575844b39ae17ef3118421c9166bfc3814
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/rich/resource/cam2params.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e260ae6f8f65e8e0a56af2a1130181ec663f3a382147984272111aec74f7c260
+size 26211
diff --git a/third_party/GVHMR/hmr4d/dataset/rich/resource/seqname2imgrange.json b/third_party/GVHMR/hmr4d/dataset/rich/resource/seqname2imgrange.json
new file mode 100644
index 0000000000000000000000000000000000000000..2bff25c7da579b09a86b5c0aeae29b548e5b28e5
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/rich/resource/seqname2imgrange.json
@@ -0,0 +1 @@
+{"ParkingLot1_002_burpee3": [1, 351], "ParkingLot1_002_overfence1": [1, 268], "ParkingLot1_002_overfence2": [1, 270], "ParkingLot1_002_stretching1": [1, 327], "ParkingLot1_002_pushup1": [1, 220], "ParkingLot1_004_pushup2": [1, 347], "ParkingLot1_004_burpeejump1": [1, 296], "ParkingLot1_004_eating1": [1, 522], "ParkingLot1_004_takingphotos1": [1, 593], "ParkingLot1_004_phonetalk1": [1, 724], "ParkingLot1_005_burpeejump2": [1, 270], "ParkingLot1_005_overfence1": [1, 301], "ParkingLot1_005_pushup2": [1, 262], "ParkingLot1_005_pushup3": [1, 243], "ParkingLot1_004_005_greetingchattingeating1": [275, 849], "ParkingLot1_007_overfence2": [1, 263], "ParkingLot1_007_eating1": [1, 426], "ParkingLot1_007_eating2": [1, 498], "ParkingLot2_008_phonetalk1": [171, 1215], "ParkingLot2_008_burpeejump1": [78, 505], "ParkingLot2_008_overfence1": [161, 459], "ParkingLot2_008_pushup1": [165, 459], "ParkingLot2_008_pushup2": [107, 719], "ParkingLot2_008_overfence2": [138, 632], "ParkingLot2_008_overfence3": [100, 661], "ParkingLot2_008_eating1": [180, 1332], "ParkingLot2_014_pushup2": [80, 420], "ParkingLot2_014_burpeejump1": [50, 348], "ParkingLot2_014_burpeejump2": [50, 248], "ParkingLot2_014_phonetalk2": [121, 1141], "ParkingLot2_014_takingphotos2": [91, 906], "ParkingLot2_014_overfence3": [40, 502], "ParkingLot2_015_overfence1": [170, 692], "ParkingLot2_015_burpeejump2": [344, 678], "ParkingLot2_015_pushup1": [190, 817], "ParkingLot2_015_eating2": [31, 835], "ParkingLot2_016_burpeejump2": [100, 793], "ParkingLot2_016_overfence2": [100, 720], "ParkingLot2_016_pushup1": [61, 680], "ParkingLot2_016_pushup2": [100, 570], "ParkingLot2_016_stretching1": [100, 691], "Pavallion_000_yoga2": [1, 1643], "Pavallion_000_plankjack": [1, 900], "Pavallion_000_phonesiteat": [1, 1157], "Pavallion_000_sidebalancerun": [1, 1091], "Pavallion_002_plankjack": [110, 699], "Pavallion_002_phonesiteat": [1, 1030], "Pavallion_003_plankjack": [1, 764], "Pavallion_003_phonesiteat": [75, 838], "Pavallion_003_sidebalancerun": [1, 942], "Pavallion_006_phonesiteat": [130, 841], "Pavallion_006_sidebalancerun": [1, 798], "Pavallion_006_plankjack": [1, 615], "Pavallion_013_phonesiteat": [1, 1254], "Pavallion_013_plankjack": [1, 641], "Pavallion_013_yoga2": [1, 884], "Pavallion_003_018_tossball": [230, 949], "LectureHall_018_wipingchairs1": [1, 1166], "LectureHall_018_wipingspray1": [1, 904], "LectureHall_020_wipingtable1": [1, 897], "BBQ_001_juggle": [0, 297], "BBQ_001_guitar": [0, 381], "ParkingLot1_002_stretching2": [240, 240], "ParkingLot1_002_burpee1": [1, 286], "ParkingLot1_002_burpee2": [1, 203], "ParkingLot1_004_pushup1": [1, 354], "ParkingLot1_004_eating2": [1, 516], "ParkingLot1_004_phonetalk2": [1, 960], "ParkingLot1_004_takingphotos2": [1, 571], "ParkingLot1_004_stretching2": [1, 399], "ParkingLot1_005_overfence2": [1, 298], "ParkingLot1_005_pushup1": [1, 476], "ParkingLot1_005_burpeejump1": [1, 252], "ParkingLot1_007_burpee2": [1, 349], "ParkingLot2_008_eating2": [160, 1100], "ParkingLot2_008_burpeejump2": [129, 492], "ParkingLot2_014_overfence1": [95, 547], "ParkingLot2_014_eating2": [101, 986], "ParkingLot2_016_phonetalk5": [170, 1259], "Pavallion_002_sidebalancerun": [1, 655], "Pavallion_013_sidebalancerun": [1, 810], "Pavallion_018_sidebalancerun": [1, 873], "LectureHall_018_wipingtable1": [1, 1280], "LectureHall_020_wipingchairs1": [1, 1163], "LectureHall_003_wipingchairs1": [1, 724], "Pavallion_000_yoga1": [1, 1757], "Pavallion_002_yoga1": [1, 613], "Pavallion_003_yoga1": [1, 792], "Pavallion_006_yoga1": [1, 930], "Pavallion_018_yoga1": [1, 880], "ParkingLot2_017_burpeejump2": [118, 612], "ParkingLot2_017_burpeejump1": [40, 817], "ParkingLot2_017_overfence1": [110, 661], "ParkingLot2_017_overfence2": [90, 944], "ParkingLot2_017_eating1": [97, 895], "ParkingLot2_017_pushup1": [191, 719], "ParkingLot2_017_pushup2": [74, 811], "ParkingLot2_009_burpeejump1": [200, 1085], "ParkingLot2_009_burpeejump2": [150, 399], "ParkingLot2_009_overfence1": [140, 601], "ParkingLot2_009_overfence2": [150, 559], "LectureHall_009_sidebalancerun1": [1, 673], "LectureHall_010_plankjack1": [1, 532], "LectureHall_010_sidebalancerun1": [1, 919], "LectureHall_021_plankjack1": [1, 507], "LectureHall_021_sidebalancerun1": [1, 855], "LectureHall_019_wipingchairs1": [1, 978], "LectureHall_009_021_reparingprojector1": [1, 499], "ParkingLot2_009_spray1": [145, 1242], "ParkingLot2_009_impro1": [100, 990], "ParkingLot2_009_impro2": [100, 1140], "ParkingLot2_009_impro5": [100, 649], "Gym_010_pushup1": [1, 475], "Gym_010_pushup2": [1, 407], "Gym_011_pushup1": [1, 346], "Gym_011_pushup2": [1, 540], "Gym_011_burpee2": [1, 479], "Gym_012_pushup2": [1, 291], "Gym_010_mountainclimber1": [0, 0], "Gym_010_mountainclimber2": [1, 471], "Gym_013_dips1": [1, 503], "Gym_013_dips2": [1, 333], "Gym_013_dips3": [1, 502], "Gym_013_lunge1": [1, 690], "Gym_013_lunge2": [1, 834], "Gym_013_pushup1": [1, 861], "Gym_013_pushup2": [1, 477], "Gym_013_burpee4": [1, 320], "Gym_010_lunge1": [1, 337], "Gym_010_lunge2": [1, 312], "Gym_010_dips1": [1, 572], "Gym_010_dips2": [1, 603], "Gym_010_cooking1": [1, 779], "Gym_011_cooking1": [1, 1141], "Gym_011_cooking2": [1, 1145], "Gym_011_dips1": [1, 494], "Gym_011_dips4": [1, 495], "Gym_011_dips3": [1, 320], "Gym_011_dips2": [1, 382], "Gym_012_lunge1": [1, 225], "Gym_012_lunge2": [1, 318], "Gym_012_cooking2": [1, 993]}
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/dataset/rich/resource/test.txt b/third_party/GVHMR/hmr4d/dataset/rich/resource/test.txt
new file mode 100644
index 0000000000000000000000000000000000000000..69e157e37c28e04d90d332386d62b3aa418fc80b
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/rich/resource/test.txt
@@ -0,0 +1,54 @@
+sequence_name capture_name scan_name id moving_cam gender scene action/scene-interaction subjects view_id
+ParkingLot2_017_burpeejump2 ParkingLot2 scan_camcoord 017 V female V V X 0,2,3
+ParkingLot2_017_burpeejump1 ParkingLot2 scan_camcoord 017 V female V V X 0,1,5
+ParkingLot2_017_overfence1 ParkingLot2 scan_camcoord 017 V female V V X 0,3,4
+ParkingLot2_017_overfence2 ParkingLot2 scan_camcoord 017 V female V V X 0,1,4
+ParkingLot2_017_eating1 ParkingLot2 scan_camcoord 017 V female V V X 0,2,4
+ParkingLot2_017_pushup1 ParkingLot2 scan_camcoord 017 X female V V X 0,1,4,5
+ParkingLot2_017_pushup2 ParkingLot2 scan_camcoord 017 V female V V X 0,4,5
+ParkingLot2_009_burpeejump1 ParkingLot2 scan_camcoord 009 X female V V X 0,1,2,3
+ParkingLot2_009_burpeejump2 ParkingLot2 scan_camcoord 009 X female V V X 0,2,3,4
+ParkingLot2_009_overfence1 ParkingLot2 scan_camcoord 009 X female V V X 0,3,4,5
+ParkingLot2_009_overfence2 ParkingLot2 scan_camcoord 009 X female V V X 0,1,4,5
+LectureHall_009_sidebalancerun1 LectureHall scan_yoga_scene_camcoord 009 X female V V X 0,1,4,5
+LectureHall_010_plankjack1 LectureHall scan_yoga_scene_camcoord 010 X female V V X 0,2,4,6
+LectureHall_010_sidebalancerun1 LectureHall scan_yoga_scene_camcoord 010 X female V V X 0,1,2,4
+LectureHall_021_plankjack1 LectureHall scan_yoga_scene_camcoord 021 X female V V X 0,3,5,6
+LectureHall_021_sidebalancerun1 LectureHall scan_yoga_scene_camcoord 021 X female V V X 0,4,5,6
+LectureHall_019_wipingchairs1 LectureHall scan_chair_scene_camcoord 019 X female V V X 0,1,2,3
+LectureHall_009_021_reparingprojector1 LectureHall scan_yoga_scene_camcoord 009 X female V X X 0,3,4,5
+LectureHall_009_021_reparingprojector1 LectureHall scan_yoga_scene_camcoord 021 X female V X X 0,3,4,5
+ParkingLot2_009_spray1 ParkingLot2 scan_camcoord 009 X female V X X 0,1,2,3
+ParkingLot2_009_impro1 ParkingLot2 scan_camcoord 009 X female V X X 0,2,3,4
+ParkingLot2_009_impro2 ParkingLot2 scan_camcoord 009 X female V X X 0,3,4,5
+ParkingLot2_009_impro5 ParkingLot2 scan_camcoord 009 X female V X X 0,2,4,5
+Gym_010_pushup1 Gym scan_camcoord 010 X female X V X 3,4,5,6
+Gym_010_pushup2 Gym scan_camcoord 010 X female X V X 2,3,4,5
+Gym_011_pushup1 Gym scan_camcoord 011 X male X V X 2,3,4,5
+Gym_011_pushup2 Gym scan_camcoord 011 X male X V X 2,3,4,5
+Gym_011_burpee2 Gym scan_camcoord 011 X male X V X 2,3,4,5
+Gym_012_pushup2 Gym scan_camcoord 012 X female X V X 3,4,5,6
+Gym_010_mountainclimber1 Gym scan_camcoord 010 X female X V X 3,4,5,6
+Gym_010_mountainclimber2 Gym scan_camcoord 010 X female X V X 3,4,5,6
+Gym_013_dips1 Gym scan_camcoord 013 X female X X V 0,3,4,5
+Gym_013_dips2 Gym scan_camcoord 013 X female X X V 1,2,4,5
+Gym_013_dips3 Gym scan_camcoord 013 X female X X V 1,2,4,5
+Gym_013_lunge1 Gym scan_camcoord 013 X female X X V 1,4,5,6
+Gym_013_lunge2 Gym scan_camcoord 013 X female X X V 0,4,5,6
+Gym_013_pushup1 Gym scan_camcoord 013 X female X V V 0,3,4,5
+Gym_013_pushup2 Gym scan_camcoord 013 X female X V V 1,2,4,5
+Gym_013_burpee4 Gym scan_camcoord 013 X female X V V 0,4,5,6
+Gym_010_lunge1 Gym scan_camcoord 010 X female X X X 1,4,5,6
+Gym_010_lunge2 Gym scan_camcoord 010 X female X X X 0,2,4,5
+Gym_010_dips1 Gym scan_camcoord 010 X female X X X 0,4,5,6
+Gym_010_dips2 Gym scan_camcoord 010 X female X X X 1,2,4,5
+Gym_010_cooking1 Gym scan_table_camcoord 010 X female X X X 1,3,4,5
+Gym_011_cooking1 Gym scan_table_camcoord 011 V male X X X 4,5,6
+Gym_011_cooking2 Gym scan_table_camcoord 011 V male X X X 2,4,5
+Gym_011_dips1 Gym scan_camcoord 011 X male X X X 1,3,4,5
+Gym_011_dips4 Gym scan_camcoord 011 X male X X X 0,2,4,5
+Gym_011_dips3 Gym scan_camcoord 011 X male X X X 0,3,4,5
+Gym_011_dips2 Gym scan_camcoord 011 X male X X X 1,3,4,5
+Gym_012_lunge1 Gym scan_camcoord 012 X female X X X 0,3,4,5
+Gym_012_lunge2 Gym scan_camcoord 012 X female X X X 0,4,5,6
+Gym_012_cooking2 Gym scan_table_camcoord 012 V female X X X 3,4,5
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/dataset/rich/resource/train.txt b/third_party/GVHMR/hmr4d/dataset/rich/resource/train.txt
new file mode 100644
index 0000000000000000000000000000000000000000..875c79e0d4cb60be778043e5a741ca32144b23c6
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/rich/resource/train.txt
@@ -0,0 +1,65 @@
+sequence_name capture_name scan_name id moving_cam gender view_id
+ParkingLot1_002_burpee3 ParkingLot1 scan_camcoord 002 X male 0,1,2,3,4,5,6,7
+ParkingLot1_002_overfence1 ParkingLot1 scan_camcoord 002 X male 0,1,2,3,4,5,6,7
+ParkingLot1_002_overfence2 ParkingLot1 scan_camcoord 002 X male 0,1,2,3,4,5,6,7
+ParkingLot1_002_stretching1 ParkingLot1 scan_camcoord 002 X male 0,1,2,3,4,5,6,7
+ParkingLot1_002_pushup1 ParkingLot1 scan_camcoord 002 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_pushup2 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_burpeejump1 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_eating1 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_takingphotos1 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_phonetalk1 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_005_burpeejump2 ParkingLot1 scan_camcoord 005 X male 0,1,2,3,4,5,6,7
+ParkingLot1_005_overfence1 ParkingLot1 scan_camcoord 005 X male 0,1,2,3,4,5,6,7
+ParkingLot1_005_pushup2 ParkingLot1 scan_camcoord 005 X male 0,1,2,3,4,5,6,7
+ParkingLot1_005_pushup3 ParkingLot1 scan_camcoord 005 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_005_greetingchattingeating1 ParkingLot1 scan_camcoord 004 X male 0,1,2,3,4,5,6,7
+ParkingLot1_004_005_greetingchattingeating1 ParkingLot1 scan_camcoord 005 X male 0,1,2,3,4,5,6,7
+ParkingLot1_007_overfence2 ParkingLot1 scan_camcoord 007 X male 0,1,2,3,4,5,6,7
+ParkingLot1_007_eating1 ParkingLot1 scan_camcoord 007 X male 0,1,2,3,4,5,6,7
+ParkingLot1_007_eating2 ParkingLot1 scan_camcoord 007 X male 0,1,2,3,4,5,6,7
+ParkingLot2_008_phonetalk1 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_burpeejump1 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_overfence1 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_pushup1 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_pushup2 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_overfence2 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_overfence3 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_008_eating1 ParkingLot2 scan_camcoord 008 V male 0,1,2,3,4,5
+ParkingLot2_014_pushup2 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_014_burpeejump1 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_014_burpeejump2 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_014_phonetalk2 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_014_takingphotos2 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_014_overfence3 ParkingLot2 scan_camcoord 014 X male 0,1,2,3,4,5
+ParkingLot2_015_overfence1 ParkingLot2 scan_camcoord 015 X male 0,1,2,3,4,5
+ParkingLot2_015_burpeejump2 ParkingLot2 scan_camcoord 015 X male 0,1,2,3,4,5
+ParkingLot2_015_pushup1 ParkingLot2 scan_camcoord 015 X male 0,1,2,3,4,5
+ParkingLot2_015_eating2 ParkingLot2 scan_camcoord 015 X male 0,1,2,3,4,5
+ParkingLot2_016_burpeejump2 ParkingLot2 scan_camcoord 016 V female 0,1,2,3,4,5
+ParkingLot2_016_overfence2 ParkingLot2 scan_camcoord 016 V female 0,1,2,3,4,5
+ParkingLot2_016_pushup1 ParkingLot2 scan_camcoord 016 V female 0,1,2,3,4,5
+ParkingLot2_016_pushup2 ParkingLot2 scan_camcoord 016 V female 0,1,2,3,4,5
+ParkingLot2_016_stretching1 ParkingLot2 scan_camcoord 016 V female 0,1,2,3,4,5
+Pavallion_000_yoga2 Pavallion scan_camcoord 000 X male 0,1,2,3,4,5,6
+Pavallion_000_plankjack Pavallion scan_camcoord 000 X male 0,1,2,3,4,5,6
+Pavallion_000_phonesiteat Pavallion scan_camcoord 000 X male 0,1,3,4,6
+Pavallion_000_sidebalancerun Pavallion scan_camcoord 000 X male 0,1,2,3,4,5,6
+Pavallion_002_plankjack Pavallion scan_camcoord 002 V male 0,1,2,3,4,5,6
+Pavallion_002_phonesiteat Pavallion scan_camcoord 002 V male 0,1,3,4,6
+Pavallion_003_plankjack Pavallion scan_camcoord 003 V male 0,1,2,3,4,5,6
+Pavallion_003_phonesiteat Pavallion scan_camcoord 003 V male 0,1,3,4,6
+Pavallion_003_sidebalancerun Pavallion scan_camcoord 003 V male 0,1,2,3,4,5,6
+Pavallion_006_phonesiteat Pavallion scan_camcoord 006 V male 0,1,3,4,6
+Pavallion_006_sidebalancerun Pavallion scan_camcoord 006 V male 0,1,2,3,4,5,6
+Pavallion_006_plankjack Pavallion scan_camcoord 006 V male 0,1,2,3,4,5,6
+Pavallion_013_phonesiteat Pavallion scan_camcoord 013 X female 0,1,3,4,6
+Pavallion_013_plankjack Pavallion scan_camcoord 013 X female 0,1,2,3,4,5,6
+Pavallion_013_yoga2 Pavallion scan_camcoord 013 V female 0,1,2,3,4,5,6
+Pavallion_003_018_tossball Pavallion scan_camcoord 003 X male 0,1,2,3,4,5,6
+Pavallion_003_018_tossball Pavallion scan_camcoord 018 X female 0,1,2,3,4,5,6
+LectureHall_018_wipingchairs1 LectureHall scan_chair_scene_camcoord 018 X female 0,1,2,3,4,5,6
+LectureHall_018_wipingspray1 LectureHall scan_chair_scene_camcoord 018 X female 2,3,4
+LectureHall_020_wipingtable1 LectureHall scan_chair_scene_camcoord 020 X male 0,2,4,5,6
+BBQ_001_juggle BBQ scan_camcoord 001 X male 0,1,2,3,4,5,6,7
+BBQ_001_guitar BBQ scan_camcoord 001 X male 0,1,2,3,4,5,6,7
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/dataset/rich/resource/val.txt b/third_party/GVHMR/hmr4d/dataset/rich/resource/val.txt
new file mode 100644
index 0000000000000000000000000000000000000000..714d8ffaab102a350f891119c3ebdc02f088b7d7
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/rich/resource/val.txt
@@ -0,0 +1,29 @@
+sequence_name capture_name scan_name id moving_cam gender scene action/scene-interaction subjects view_id
+ParkingLot1_002_stretching2 ParkingLot1 scan_camcoord 002 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_002_burpee1 ParkingLot1 scan_camcoord 002 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_002_burpee2 ParkingLot1 scan_camcoord 002 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_004_pushup1 ParkingLot1 scan_camcoord 004 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_004_eating2 ParkingLot1 scan_camcoord 004 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_004_phonetalk2 ParkingLot1 scan_camcoord 004 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_004_takingphotos2 ParkingLot1 scan_camcoord 004 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_004_stretching2 ParkingLot1 scan_camcoord 004 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_005_overfence2 ParkingLot1 scan_camcoord 005 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_005_pushup1 ParkingLot1 scan_camcoord 005 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_005_burpeejump1 ParkingLot1 scan_camcoord 005 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot1_007_burpee2 ParkingLot1 scan_camcoord 007 X male V V V 0,1,2,3,4,5,6,7
+ParkingLot2_008_eating2 ParkingLot2 scan_camcoord 008 V male V V V 0,1,2,3,4,5
+ParkingLot2_008_burpeejump2 ParkingLot2 scan_camcoord 008 V male V V V 0,1,2,3,4,5
+ParkingLot2_014_overfence1 ParkingLot2 scan_camcoord 014 X male V V V 0,1,2,3,4,5
+ParkingLot2_014_eating2 ParkingLot2 scan_camcoord 014 X male V V V 0,1,2,3,4,5
+ParkingLot2_016_phonetalk5 ParkingLot2 scan_camcoord 016 V female V V V 0,1,2,3,4,5
+Pavallion_002_sidebalancerun Pavallion scan_camcoord 002 V male V V V 0,1,2,3,4,5,6
+Pavallion_013_sidebalancerun Pavallion scan_camcoord 013 X female V V V 0,1,2,3,4,5,6
+Pavallion_018_sidebalancerun Pavallion scan_camcoord 018 V female V V V 0,1,2,3,4,5,6
+LectureHall_018_wipingtable1 LectureHall scan_chair_scene_camcoord 018 X female V V V 0,2,4,5,6
+LectureHall_020_wipingchairs1 LectureHall scan_chair_scene_camcoord 020 X male V V V 0,1,2,3,4,5,6
+LectureHall_003_wipingchairs1 LectureHall scan_chair_scene_camcoord 003 X male V V V 0,1,2,3,4,5,6
+Pavallion_000_yoga1 Pavallion scan_camcoord 000 X male V X V 0,1,2,3,4,5,6
+Pavallion_002_yoga1 Pavallion scan_camcoord 002 V male V X V 0,1,2,3,4,5,6
+Pavallion_003_yoga1 Pavallion scan_camcoord 003 V male V X V 0,1,2,3,4,5,6
+Pavallion_006_yoga1 Pavallion scan_camcoord 006 V male V X V 0,1,2,3,4,5,6
+Pavallion_018_yoga1 Pavallion scan_camcoord 018 V female V X V 0,1,2,3,4,5,6
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/dataset/rich/resource/w2az_sahmr.json b/third_party/GVHMR/hmr4d/dataset/rich/resource/w2az_sahmr.json
new file mode 100644
index 0000000000000000000000000000000000000000..7dec2ac894408d56a0981de80841ad3afe2476c8
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/rich/resource/w2az_sahmr.json
@@ -0,0 +1 @@
+{"BBQ_scan_camcoord": [[0.9989829107564298, 0.03367618890797693, -0.029984301180211045, 0.0008183751635392625], [0.03414262169451401, -0.1305975871406019, 0.9908473906797644, -0.005059823133706893], [0.02945208652127451, -0.9908633531086326, -0.13161455111748036, 1.4054905296083466], [0.0, 0.0, 0.0, 1.0]], "Gym_scan_camcoord": [[0.9932599733260449, -0.07628732032461205, 0.0872632233306122, -0.047601130084306706], [-0.10233962102690007, -0.22374853741942266, 0.9692590953768503, -0.04091804681182174], [-0.05441716049582774, -0.9716567484252654, -0.23004768176013274, 1.537911791136788], [0.0, 0.0, 0.0, 1.0]], "Gym_scan_table_camcoord": [[0.9974451989415423, -0.06250743213795668, 0.03458172980064169, 0.02231858470834599], [-0.04804912583358893, -0.22882402250236075, 0.972281259838159, 0.039081886755815726], [-0.05286167435026744, -0.9714588965331274, -0.2312428501197992, 1.5421821446346522], [0.0, 0.0, 0.0, 1.0]], "LectureHall_scan_chair_scene_camcoord": [[0.9992930513998263, 0.030087515976743376, -0.0225419343977731, 0.001998908749589632], [0.030705594681969043, -0.30721111058653017, 0.9511458878570781, -0.025811963513866963], [0.021692484396004613, -0.9511656401040444, -0.307917783192506, 2.060346184503773], [0.0, 0.0, 0.0, 1.0]], "LectureHall_scan_yoga_scene_camcoord": [[0.9993358324246812, 0.03030060260429296, -0.020242715082476024, -0.003510046042036605], [0.028600729415016745, -0.3079667078507395, 0.9509671419836329, -0.01748548118379142], [0.022580795137075255, -0.9509144968594153, -0.3086287856852993, 2.0424701474796567], [0.0, 0.0, 0.0, 1.0]], "ParkingLot1_scan_camcoord": [[0.9989627324729327, -0.03724260727951709, 0.02620013994738054, 0.0070941466745699025], [-0.03091587075252664, -0.13228243926883107, 0.9907298144280939, -0.0274920377236923], [-0.03343154297742938, -0.9905121627037764, -0.13329661462331338, 1.3859200914120975], [0.0, 0.0, 0.0, 1.0]], "ParkingLot2_scan_camcoord": [[0.9989532636786039, -0.04044665659892979, 0.021364572447267097, 0.01646827411554571], [-0.026687287930043047, -0.13600581518076985, 0.9903485279940424, 0.030197722289598695], [-0.03715058073335097, -0.9898820567153364, -0.13694286452455984, 1.4372015171546513], [0.0, 0.0, 0.0, 1.0]], "Pavallion_scan_camcoord": [[0.9971864096076799, 0.05693557331723671, -0.048760690979605295, 0.0012478238054067193], [0.05746407703876882, -0.16289761936471214, 0.9849681443861059, -0.006002953831755452], [0.04813672552068054, -0.9849988355812122, -0.16571104235928033, 1.7638454838942128], [0.0, 0.0, 0.0, 1.0]]}
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/dataset/rich/rich_motion_test.py b/third_party/GVHMR/hmr4d/dataset/rich/rich_motion_test.py
new file mode 100644
index 0000000000000000000000000000000000000000..e21a96b3ac2f19c0a668b81ba58d635a8e8d3e24
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/rich/rich_motion_test.py
@@ -0,0 +1,185 @@
+from pathlib import Path
+import numpy as np
+import torch
+from torch.utils import data
+from hmr4d.utils.pylogger import Log
+
+from .rich_utils import (
+ get_cam2params,
+ get_w2az_sahmr,
+ parse_seqname_info,
+ get_cam_key_wham_vid,
+)
+from hmr4d.utils.geo_transform import apply_T_on_points, transform_mat, compute_cam_angvel
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines
+from hmr4d.utils.smplx_utils import make_smplx
+from pytorch3d.transforms import axis_angle_to_matrix, matrix_to_axis_angle
+from hmr4d.utils.geo.hmr_cam import resize_K
+
+
+from hmr4d.configs import MainStore, builds
+
+
+VID_PRESETS = {
+ "easytohard": [
+ "test/Gym_013_burpee4/cam_06",
+ "test/Gym_011_pushup1/cam_02",
+ "test/LectureHall_019_wipingchairs1/cam_03",
+ "test/ParkingLot2_009_overfence1/cam_04",
+ "test/LectureHall_021_sidebalancerun1/cam_00",
+ "test/Gym_010_dips2/cam_05",
+ ],
+}
+
+
+class RichSmplFullSeqDataset(data.Dataset):
+ def __init__(self, vid_presets=None):
+ """
+ Args:
+ vid_presets is a key in VID_PRESETS
+ """
+ super().__init__()
+ self.dataset_name = "RICH"
+ self.dataset_id = "RICH"
+ Log.info(f"[{self.dataset_name}] Full sequence, Test")
+ tic = Log.time()
+
+ # Load evaluation protocol from WHAM labels
+ self.rich_dir = Path("inputs/RICH/hmr4d_support")
+ self.labels = torch.load(self.rich_dir / "rich_test_labels.pt")
+ self.preproc_data = torch.load(self.rich_dir / "rich_test_preproc.pt")
+ vids = select_subset(self.labels, vid_presets)
+
+ # Setup dataset index
+ self.idx2meta = []
+ for vid in vids:
+ seq_length = len(self.labels[vid]["frame_id"])
+ self.idx2meta.append((vid, 0, seq_length)) # start=0, end=seq_length
+ # print(sum([end - start for _, _, start, end in self.idx2meta]))
+
+ # Prepare ground truth motion in ay-coordinate
+ self.w2az = get_w2az_sahmr() # scan_name -> T_w2az, w-coordinate refers to cam-1-coordinate
+ self.cam2params = get_cam2params() # cam_key -> (T_w2c, K)
+ seqname_info = parse_seqname_info(skip_multi_persons=True) # {k: (scan_name, subject_id, gender, cam_ids)}
+ self.seqname_to_scanname = {k: v[0] for k, v in seqname_info.items()}
+
+ Log.info(f"[RICH] {len(self.idx2meta)} sequences. Elapsed: {Log.time() - tic:.2f}s")
+
+ def __len__(self):
+ return len(self.idx2meta)
+
+ def _load_data(self, idx):
+ data = {}
+
+ # [start, end), when loading data from labels
+ vid, start, end = self.idx2meta[idx]
+ label = self.labels[vid]
+ preproc_data = self.preproc_data[vid]
+
+ length = end - start
+ meta = {"dataset_id": "RICH", "vid": vid, "vid-start-end": (start, end)}
+ data.update({"meta": meta, "length": length})
+
+ # SMPLX
+ data.update({"gt_smpl_params": label["gt_smplx_params"], "gender": label["gender"]})
+
+ # camera
+ cam_key = get_cam_key_wham_vid(vid)
+ scan_name = self.seqname_to_scanname[vid.split("/")[1]]
+ T_w2c, K = self.cam2params[cam_key] # (4, 4) (3, 3)
+ T_w2az = self.w2az[scan_name]
+ data.update({"T_w2c": T_w2c, "T_w2az": T_w2az, "K": K})
+
+ # image features
+ data.update(
+ {
+ "f_imgseq": preproc_data["f_imgseq"],
+ "bbx_xys": preproc_data["bbx_xys"],
+ "img_wh": preproc_data["img_wh"],
+ "kp2d": preproc_data["kp2d"],
+ }
+ )
+
+ # to render a video
+ video_path = self.rich_dir / "video" / vid / "video.mp4"
+ frame_id = label["frame_id"] # (F,)
+ width, height = data["img_wh"] / 4 # Video saved has been downsampled 1/4
+ K_render = resize_K(K, 0.25)
+ bbx_xys_render = data["bbx_xys"] / 4
+ data["meta_render"] = {
+ "name": vid.replace("/", "@"),
+ "video_path": str(video_path),
+ "frame_id": frame_id,
+ "width_height": (width, height),
+ "K": K_render,
+ "bbx_xys": bbx_xys_render,
+ }
+
+ return data
+
+ def _process_data(self, data):
+ # T_w2az is pre-computed by using floor clue. az2zy uses a rotation along x-axis.
+ R_az2ay = axis_angle_to_matrix(torch.tensor([1.0, 0.0, 0.0]) * -torch.pi / 2) # (3, 3)
+ T_w2ay = transform_mat(R_az2ay, R_az2ay.new([0, 0, 0])) @ data["T_w2az"] # (4, 4)
+
+ if False: # Visualize groundtruth and observation
+ self.rich_smplx = {
+ "male": make_smplx("rich-smplx", gender="male"),
+ "female": make_smplx("rich-smplx", gender="female"),
+ }
+ wis3d = make_wis3d(name="debug-rich-smpl_dataset")
+ rich_smplx = make_smplx("rich-smplx", gender=data["gender"])
+ smplx_out = rich_smplx(**data["gt_smpl_params"])
+ smplx_verts_ay = apply_T_on_points(smplx_out.vertices, T_w2ay)
+ for i in range(400):
+ wis3d.set_scene_id(i)
+ wis3d.add_mesh(smplx_out.vertices[i], rich_smplx.bm.faces, name=f"gt-smplx")
+ wis3d.add_mesh(smplx_verts_ay[i], rich_smplx.bm.faces, name=f"gt-smplx-ay")
+
+ # process img feature with xys
+ length = data["length"]
+ f_imgseq = data["f_imgseq"] # (F, 1024)
+ R_w2c = data["T_w2c"][:3, :3].repeat(length, 1, 1) # (L, 4, 4)
+ cam_angvel = compute_cam_angvel(R_w2c) # (L, 6)
+
+ # Return
+ data = {
+ # --- not batched
+ "task": "CAP-Seq",
+ "meta": data["meta"],
+ "meta_render": data["meta_render"],
+ # --- we test on single sequence, so set kv manually
+ "length": length,
+ "f_imgseq": f_imgseq,
+ "cam_angvel": cam_angvel,
+ "bbx_xys": data["bbx_xys"], # (F, 3)
+ "K_fullimg": data["K"][None].expand(length, -1, -1), # (F, 3, 3)
+ "kp2d": data["kp2d"], # (F, 17, 3)
+ # --- dataset specific
+ "model": "smplx",
+ "gender": data["gender"],
+ "gt_smpl_params": data["gt_smpl_params"],
+ "T_w2ay": T_w2ay, # (4, 4)
+ "T_w2c": data["T_w2c"], # (4, 4)
+ }
+ return data
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data)
+ return data
+
+
+def select_subset(labels, vid_presets):
+ vids = list(labels.keys())
+ if vid_presets != None: # Use a subset of the videos
+ vids = VID_PRESETS[vid_presets]
+ return vids
+
+
+#
+group_name = "test_datasets/rich"
+base_node = builds(RichSmplFullSeqDataset, vid_presets=None, populate_full_signature=True)
+MainStore.store(name="all", node=base_node, group=group_name)
+MainStore.store(name="easy_to_hard", node=base_node(vid_presets="easytohard"), group=group_name)
+MainStore.store(name="postproc", node=base_node(vid_presets="postproc"), group=group_name)
diff --git a/third_party/GVHMR/hmr4d/dataset/rich/rich_utils.py b/third_party/GVHMR/hmr4d/dataset/rich/rich_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..868c8b9de34392a646c31d11e038cc193e2cb9b5
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/rich/rich_utils.py
@@ -0,0 +1,370 @@
+import torch
+import cv2
+import numpy as np
+from hmr4d.utils.geo_transform import apply_T_on_points, project_p2d
+from pathlib import Path
+import json
+import time
+
+# ----- Meta sample utils ----- #
+
+
+def sample_idx2meta(idx2meta, sample_interval):
+ """
+ 1. remove frames that < 45
+ 2. sample frames by sample_interval
+ 3. sorted
+ """
+ idx2meta = [
+ v
+ for k, v in idx2meta.items()
+ if int(v["frame_name"]) > 45 and (int(v["frame_name"]) + int(v["cam_id"])) % sample_interval == 0
+ ]
+ idx2meta = sorted(idx2meta, key=lambda meta: meta["img_key"])
+ return idx2meta
+
+
+def remove_bbx_invisible_frame(idx2meta, img2gtbbx):
+ raw_img_lu = np.array([0.0, 0.0])
+ raw_img_rb_type1 = np.array([4112.0, 3008.0]) - 1 # horizontal
+ raw_img_rb_type2 = np.array([3008.0, 4112.0]) - 1 # vertical
+
+ idx2meta_new = []
+ for meta in idx2meta:
+ gtbbx_center = np.array([img2gtbbx[meta["img_key"]][[0, 2]].mean(), img2gtbbx[meta["img_key"]][[1, 3]].mean()])
+ if (gtbbx_center < raw_img_lu).any():
+ continue
+ raw_img_rb = raw_img_rb_type1 if meta["cam_key"] not in ["Pavallion_3", "Pavallion_5"] else raw_img_rb_type2
+ if (gtbbx_center > raw_img_rb).any():
+ continue
+ idx2meta_new.append(meta)
+ return idx2meta_new
+
+
+def remove_extra_rules(idx2meta):
+ multi_person_seqs = ["LectureHall_009_021_reparingprojector1"]
+ idx2meta = [meta for meta in idx2meta if meta["seq_name"] not in multi_person_seqs]
+ return idx2meta
+
+
+# ----- Image utils ----- #
+
+
+def compute_bbx(dataset, data):
+ """
+ Use gt_smplh_params to compute bbx (w.r.t. original image resolution)
+ Args:
+ dataset: rich_pose.RichPose
+ data: dict
+
+ # This function need extra scripts to run
+ from hmr4d.utils.smplx_utils import make_smplx
+ self.smplh_male = make_smplx("rich-smplh", gender="male")
+ self.smplh_female = make_smplx("rich-smplh", gender="female")
+ self.smplh = {
+ "male": self.smplh_male,
+ "female": self.smplh_female,
+ }
+ """
+ gender = data["meta"]["gender"]
+ smplh_params = {k: v.reshape(1, -1) for k, v in data["gt_smplh_params"].items()}
+ smplh_opt = dataset.smplh[gender](**smplh_params)
+ verts_3d_w = smplh_opt.vertices
+ T_w2c, K = data["T_w2c"], data["K"]
+ verts_3d_c = apply_T_on_points(verts_3d_w, T_w2c[None])
+ verts_2d = project_p2d(verts_3d_c, K[None])[0]
+ min_2d = verts_2d.T.min(-1)[0]
+ max_2d = verts_2d.T.max(-1)[0]
+ bbx = torch.stack([min_2d, max_2d]).reshape(-1).numpy()
+ return bbx
+
+
+def get_2d(dataset, data):
+ gender = data["meta"]["gender"]
+ smplh_params = {k: v.reshape(1, -1) for k, v in data["gt_smplh_params"].items()}
+ smplh_opt = dataset.smplh[gender](**smplh_params)
+ joints_3d_w = smplh_opt.joints
+ T_w2c, K = data["T_w2c"], data["K"]
+ joints_3d_c = apply_T_on_points(joints_3d_w, T_w2c[None])
+ joints_2d = project_p2d(joints_3d_c, K[None])[0]
+ conf = torch.ones((73, 1))
+ keypoints = torch.cat([joints_2d, conf], dim=1)
+ return keypoints
+
+
+def squared_crop_and_resize(dataset, img, bbx_lurb, dst_size=224, state=None):
+ if state is not None:
+ np.random.set_state(state)
+ center_rand = dataset.BBX_CENTER * (np.random.random(2) * 2 - 1)
+ center_x = (bbx_lurb[0] + bbx_lurb[2]) / 2 + center_rand[0]
+ center_y = (bbx_lurb[1] + bbx_lurb[3]) / 2 + center_rand[1]
+ ori_half_size = max(bbx_lurb[2] - bbx_lurb[0], bbx_lurb[3] - bbx_lurb[1]) / 2
+ ori_half_size *= 1 + 0.15 + dataset.BBX_ZOOM * np.random.random() # zoom
+
+ src = np.array(
+ [
+ [center_x - ori_half_size, center_y - ori_half_size],
+ [center_x + ori_half_size, center_y - ori_half_size],
+ [center_x, center_y],
+ ],
+ dtype=np.float32,
+ )
+ dst = np.array([[0, 0], [dst_size - 1, 0], [dst_size / 2 - 0.5, dst_size / 2 - 0.5]], dtype=np.float32)
+
+ A = cv2.getAffineTransform(src, dst)
+ img_crop = cv2.warpAffine(img, A, (dst_size, dst_size), flags=cv2.INTER_LINEAR)
+ bbx_new = np.array(
+ [center_x - ori_half_size, center_y - ori_half_size, center_x + ori_half_size, center_y + ori_half_size],
+ dtype=bbx_lurb.dtype,
+ )
+ return img_crop, bbx_new, A
+
+
+# Augment bbx
+def get_augmented_square_bbx(bbx_lurb, per_shift=0.1, per_zoomout=0.2, base_zoomout=0.15, state=None):
+ """
+ Args:
+ per_shift: in percent, maximum random shift
+ per_zoomout: in percent, maximum random zoom
+ """
+ if state is not None:
+ np.random.set_state(state)
+ maxsize_bbx = max(bbx_lurb[2] - bbx_lurb[0], bbx_lurb[3] - bbx_lurb[1])
+ # shift of center
+ shift = maxsize_bbx * per_shift * (np.random.random(2) * 2 - 1)
+ center_x = (bbx_lurb[0] + bbx_lurb[2]) / 2 + shift[0]
+ center_y = (bbx_lurb[1] + bbx_lurb[3]) / 2 + shift[1]
+ # zoomout of half-size
+ halfsize_bbx = maxsize_bbx / 2
+ halfsize_bbx *= 1 + base_zoomout + per_zoomout * np.random.random()
+
+ bbx_lurb = np.array(
+ [
+ center_x - halfsize_bbx,
+ center_y - halfsize_bbx,
+ center_x + halfsize_bbx,
+ center_y + halfsize_bbx,
+ ]
+ )
+ return bbx_lurb
+
+
+def get_squared_bbx_region_and_resize(frames, bbx_xys, dst_size=224):
+ """
+ Args:
+ frames: (F, H, W, 3)
+ bbx_xys: (F, 3), xys
+ """
+ frames_np = frames.numpy() if isinstance(frames, torch.Tensor) else frames
+ bbx_xys = bbx_xys if isinstance(bbx_xys, torch.Tensor) else torch.tensor(bbx_xys) # use tensor
+ srcs = torch.stack(
+ [
+ torch.stack([bbx_xys[:, 0] - bbx_xys[:, 2] / 2, bbx_xys[:, 1] - bbx_xys[:, 2] / 2], dim=-1),
+ torch.stack([bbx_xys[:, 0] + bbx_xys[:, 2] / 2, bbx_xys[:, 1] - bbx_xys[:, 2] / 2], dim=-1),
+ bbx_xys[:, :2],
+ ],
+ dim=1,
+ ) # (F, 3, 2)
+ dst = np.array([[0, 0], [dst_size - 1, 0], [dst_size / 2 - 0.5, dst_size / 2 - 0.5]], dtype=np.float32)
+ As = np.stack([cv2.getAffineTransform(src, dst) for src in srcs.numpy()])
+
+ img_crops = np.stack(
+ [cv2.warpAffine(frames_np[i], As[i], (dst_size, dst_size), flags=cv2.INTER_LINEAR) for i in range(len(As))]
+ )
+ img_crops = torch.from_numpy(img_crops)
+ As = torch.from_numpy(As)
+ return img_crops, As
+
+
+# ----- Camera utils ----- #
+
+
+def extract_cam_xml(xml_path="", dtype=torch.float32):
+ import xml.etree.ElementTree as ET
+
+ tree = ET.parse(xml_path)
+
+ extrinsics_mat = [float(s) for s in tree.find("./CameraMatrix/data").text.split()]
+ intrinsics_mat = [float(s) for s in tree.find("./Intrinsics/data").text.split()]
+ distortion_vec = [float(s) for s in tree.find("./Distortion/data").text.split()]
+
+ return {
+ "ext_mat": torch.tensor(extrinsics_mat).float(),
+ "int_mat": torch.tensor(intrinsics_mat).float(),
+ "dis_vec": torch.tensor(distortion_vec).float(),
+ }
+
+
+def get_cam2params(scene_info_root=None):
+ """
+ Args:
+ scene_info_root: this could be repalced by path to scan_calibration
+ """
+ if scene_info_root is not None:
+ cam_params = {}
+ cam_xml_files = Path(scene_info_root).glob("*/calibration/*.xml")
+ for cam_xml_file in cam_xml_files:
+ cam_param = extract_cam_xml(cam_xml_file)
+ T_w2c = cam_param["ext_mat"].reshape(3, 4)
+ T_w2c = torch.cat([T_w2c, torch.tensor([[0, 0, 0, 1.0]])], dim=0) # (4, 4)
+ K = cam_param["int_mat"].reshape(3, 3)
+ cap_name = cam_xml_file.parts[-3]
+ cam_id = int(cam_xml_file.stem)
+ cam_key = f"{cap_name}_{cam_id}"
+ cam_params[cam_key] = (T_w2c, K)
+ else:
+ cam_params = torch.load(Path(__file__).parent / "resource/cam2params.pt")
+ return cam_params
+
+
+# ----- Parse Raw Resource ----- #
+
+
+def get_w2az_sahmr():
+ """
+ Returns:
+ w2az_sahmr: dict, {scan_name: Tw2az}, Tw2az is a tensor of (4,4)
+ """
+ fn = Path(__file__).parent / "resource/w2az_sahmr.json"
+ with open(fn, "r") as f:
+ kvs = json.load(f).items()
+ w2az_sahmr = {k: torch.tensor(v) for k, v in kvs}
+ return w2az_sahmr
+
+
+def has_multi_persons(seq_name):
+ """
+ Args:
+ seq_name: e.g. LectureHall_009_021_reparingprojector1
+ """
+ return len(seq_name.split("_")) != 3
+
+
+def parse_seqname_info(skip_multi_persons=True):
+ """
+ This function will skip multi-person sequences.
+ Returns:
+ sname_to_info: scan_name, subject_id, gender, cam_ids
+ """
+ fns = [Path(__file__).parent / f"resource/{split}.txt" for split in ["train", "val", "test"]]
+ # Train / Val&Test Header:
+ # sequence_name capture_name scan_name id moving_cam gender view_id
+ # sequence_name capture_name scan_name id moving_cam gender scene action/scene-interaction subjects view_id
+ sname_to_info = {}
+ for fn in fns:
+ with open(fn, "r") as f:
+ for line in f.readlines()[1:]:
+ raw_values = line.strip().split()
+ seq_name = raw_values[0]
+ if skip_multi_persons and has_multi_persons(seq_name):
+ continue
+ scan_name = f"{raw_values[1]}_{raw_values[2]}"
+ subject_id = int(raw_values[3])
+ gender = raw_values[5]
+ cam_ids = [int(c) for c in raw_values[-1].split(",")]
+ sname_to_info[seq_name] = (scan_name, subject_id, gender, cam_ids)
+ return sname_to_info
+
+
+def get_seqnames_of_split(splits=["train"], skip_multi_persons=True):
+ if not isinstance(splits, list):
+ splits = [splits]
+ fns = [Path(__file__).parent / f"resource/{split}.txt" for split in splits]
+ seqnames = []
+ for fn in fns:
+ with open(fn, "r") as f:
+ for line in f.readlines()[1:]:
+ seq_name = line.strip().split()[0]
+ if skip_multi_persons and has_multi_persons(seq_name):
+ continue
+ seqnames.append(seq_name)
+ return seqnames
+
+
+def get_seqname_to_imgrange():
+ """Each sequence has a different range of image ids."""
+ from tqdm import tqdm
+
+ split_seqnames = {split: get_seqnames_of_split(split) for split in ["train", "val", "test"]}
+ seqname_to_imgrange = {}
+ for split in ["train", "val", "test"]:
+ for seqname in tqdm(split_seqnames[split]):
+ img_root = Path("inputs/RICH") / "images_ds4" / split # compressed (not original)
+ img_dir = img_root / seqname
+ img_names = sorted([n.name for n in img_dir.glob("**/*.jpeg")])
+ if len(img_names) == 0:
+ img_range = (0, 0)
+ else:
+ img_range = (int(img_names[0].split("_")[0]), int(img_names[-1].split("_")[0]))
+ seqname_to_imgrange[seqname] = img_range
+ return seqname_to_imgrange
+
+
+# ----- Compose keys ----- #
+
+
+def get_img_key(seq_name, cam_id, f_id):
+ assert len(seq_name.split("_")) == 3
+ subject_id = int(seq_name.split("_")[1])
+ return f"{seq_name}_{int(cam_id)}_{int(f_id):05d}_{subject_id}"
+
+
+def get_seq_cam_fn(img_root, seq_name, cam_id):
+ """
+ Args:
+ img_root: "inputs/RICH/images_ds4/train"
+ """
+ img_root = Path(img_root)
+ cam_id = int(cam_id)
+ return str(img_root / f"{seq_name}/cam_{cam_id:02d}")
+
+
+def get_img_fn(img_root, seq_name, cam_id, f_id):
+ """
+ Args:
+ img_root: "inputs/RICH/images_ds4/train"
+ """
+ img_root = Path(img_root)
+ cam_id = int(cam_id)
+ f_id = int(f_id)
+ return str(img_root / f"{seq_name}/cam_{cam_id:02d}" / f"{f_id:05d}_{cam_id:02d}.jpeg")
+
+
+# ----- WHAM ----- #
+
+
+def get_cam_key_wham_vid(vid):
+ _, sname, cname = vid.split("/")
+ scene = sname.split("_")[0]
+ cid = int(cname.split("_")[1])
+ cam_key = f"{scene}_{cid}"
+ return cam_key
+
+
+def get_K_wham_vid(vid):
+ cam_key = get_cam_key_wham_vid(vid)
+ cam2params = get_cam2params()
+ K = cam2params[cam_key][1]
+ return K
+
+
+class RichVid2Tc2az:
+ def __init__(self) -> None:
+ self.w2az = get_w2az_sahmr() # scan_name: tensor 4,4
+ seqname_info = parse_seqname_info(skip_multi_persons=True) # {k: (scan_name, subject_id, gender, cam_ids)}
+ self.seqname_to_scanname = {k: v[0] for k, v in seqname_info.items()}
+ self.cam2params = get_cam2params() # cam_key -> (T_w2c, K)
+
+ def __call__(self, vid):
+ cam_key = get_cam_key_wham_vid(vid)
+ scan_name = self.seqname_to_scanname[vid.split("/")[1]]
+ T_w2c, K = self.cam2params[cam_key] # (4, 4) (3, 3)
+ T_w2az = self.w2az[scan_name]
+ T_c2az = T_w2az @ T_w2c.inverse()
+ return T_c2az
+
+ def get_T_w2az(self, vid):
+ cam_key = get_cam_key_wham_vid(vid)
+ scan_name = self.seqname_to_scanname[vid.split("/")[1]]
+ T_w2az = self.w2az[scan_name]
+ return T_w2az
diff --git a/third_party/GVHMR/hmr4d/dataset/threedpw/threedpw_motion_test.py b/third_party/GVHMR/hmr4d/dataset/threedpw/threedpw_motion_test.py
new file mode 100644
index 0000000000000000000000000000000000000000..469648acc200a37d42dc721fa3bc806894e45dd5
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/threedpw/threedpw_motion_test.py
@@ -0,0 +1,153 @@
+import torch
+from torch.utils import data
+from pathlib import Path
+
+from hmr4d.utils.pylogger import Log
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines
+from hmr4d.utils.geo_transform import compute_cam_angvel
+from hmr4d.utils.geo.hmr_cam import estimate_K, resize_K
+from hmr4d.utils.geo.flip_utils import flip_kp2d_coco17
+
+from hmr4d.configs import MainStore, builds
+
+VID_HARD = []
+# VID_HARD = ["downtown_bar_00_1"]
+
+
+class ThreedpwSmplFullSeqDataset(data.Dataset):
+ def __init__(self, flip_test=False, skip_invalid=False):
+ super().__init__()
+ self.dataset_name = "3DPW"
+ self.skip_invalid = skip_invalid
+ Log.info(f"[{self.dataset_name}] Full sequence")
+
+ # Load evaluation protocol from WHAM labels
+ self.threedpw_dir = Path("inputs/3DPW/hmr4d_support")
+ # ['vname', 'K_fullimg', 'T_w2c', 'smpl_params', 'gender', 'mask_raw', 'mask_wham', 'img_wh']
+ self.labels = torch.load(self.threedpw_dir / "test_3dpw_gt_labels.pt")
+ self.vid2bbx = torch.load(self.threedpw_dir / "preproc_test_bbx.pt")
+ self.vid2kp2d = torch.load(self.threedpw_dir / "preproc_test_kp2d_v0.pt")
+
+ # Setup dataset index
+ self.idx2meta = list(self.labels)
+ if len(VID_HARD) > 0: # Pick subsets for fast testing
+ self.idx2meta = VID_HARD
+ Log.info(f"[{self.dataset_name}] {len(self.idx2meta)} sequences.")
+
+ # If flip_test is enabled, we will return extra data for flipped test
+ self.flip_test = flip_test
+ if self.flip_test:
+ Log.info(f"[{self.dataset_name}] Flip test enabled")
+
+ def __len__(self):
+ return len(self.idx2meta)
+
+ def _load_data(self, idx):
+ data = {}
+ vid = self.idx2meta[idx]
+ meta = {"dataset_id": self.dataset_name, "vid": vid}
+ data.update({"meta": meta})
+
+ # Add useful data
+ label = self.labels[vid]
+ mask = label["mask_wham"]
+ width_height = label["img_wh"]
+ data.update(
+ {
+ "length": len(mask), # F
+ "smpl_params": label["smpl_params"], # world
+ "gender": label["gender"], # str
+ "T_w2c": label["T_w2c"], # (F, 4, 4)
+ "mask": mask, # (F)
+ }
+ )
+ K_fullimg = label["K_fullimg"] # (3, 3)
+ if False:
+ K_fullimg = estimate_K(*width_height)
+ data["K_fullimg"] = K_fullimg
+
+ # Preprocessed: bbx, kp2d, image as feature
+ bbx_xys = self.vid2bbx[vid]["bbx_xys"] # (F, 3)
+ kp2d = self.vid2kp2d[vid] # (F, 17, 3)
+ cam_angvel = compute_cam_angvel(data["T_w2c"][:, :3, :3]) # (L, 6)
+ data.update({"bbx_xys": bbx_xys, "kp2d": kp2d, "cam_angvel": cam_angvel})
+
+ imgfeat_dir = self.threedpw_dir / "imgfeats/3dpw_test"
+ f_img_dict = torch.load(imgfeat_dir / f"{vid}.pt")
+ f_imgseq = f_img_dict["features"].float()
+ data["f_imgseq"] = f_imgseq # (F, 1024)
+
+ # to render a video
+ vname = label["vname"]
+ video_path = self.threedpw_dir / f"videos/{vname}.mp4"
+ frame_id = torch.where(mask)[0].long()
+ ds = 0.5
+ K_render = resize_K(K_fullimg, ds)
+ bbx_xys_render = bbx_xys * ds
+ kp2d_render = kp2d.clone()
+ kp2d_render[..., :2] *= ds
+ data["meta_render"] = {
+ "name": vid,
+ "video_path": str(video_path),
+ "ds": ds,
+ "frame_id": frame_id,
+ "K": K_render,
+ "bbx_xys": bbx_xys_render,
+ "kp2d": kp2d_render,
+ }
+
+ if self.flip_test:
+ imgfeat_dir = self.threedpw_dir / "imgfeats/3dpw_test_flip"
+ f_img_dict = torch.load(imgfeat_dir / f"{vid}.pt")
+ flipped_bbx_xys = f_img_dict["bbx_xys"].float() # (L, 3)
+ flipped_features = f_img_dict["features"].float() # (L, 1024)
+ flipped_kp2d = flip_kp2d_coco17(kp2d, width_height[0]) # (L, 17, 3)
+
+ R_flip_x = torch.tensor([[-1, 0, 0], [0, 1, 0], [0, 0, 1]]).float()
+ flipped_R_w2c = R_flip_x @ data["T_w2c"][:, :3, :3].clone()
+
+ data_flip = {
+ "bbx_xys": flipped_bbx_xys,
+ "f_imgseq": flipped_features,
+ "kp2d": flipped_kp2d,
+ "cam_angvel": compute_cam_angvel(flipped_R_w2c),
+ }
+ data["flip_test"] = data_flip
+ return data
+
+ def _process_data(self, data):
+ length = data["length"]
+ data["K_fullimg"] = data["K_fullimg"][None].repeat(length, 1, 1)
+
+ if self.skip_invalid: # Drop all invalid frames
+ mask = data["mask"].clone()
+ data["length"] = sum(mask)
+ data["smpl_params"] = {k: v[mask].clone() for k, v in data["smpl_params"].items()}
+ data["T_w2c"] = data["T_w2c"][mask].clone()
+ data["mask"] = data["mask"][mask].clone()
+ data["K_fullimg"] = data["K_fullimg"][mask].clone()
+ data["bbx_xys"] = data["bbx_xys"][mask].clone()
+ data["kp2d"] = data["kp2d"][mask].clone()
+ data["cam_angvel"] = data["cam_angvel"][mask].clone()
+ data["f_imgseq"] = data["f_imgseq"][mask].clone()
+ data["flip_test"] = {k: v[mask].clone() for k, v in data["flip_test"].items()}
+
+ return data
+
+ def __getitem__(self, idx):
+ data = self._load_data(idx)
+ data = self._process_data(data)
+ return data
+
+
+# 3DPW
+MainStore.store(
+ name="fliptest",
+ node=builds(ThreedpwSmplFullSeqDataset, flip_test=True),
+ group="test_datasets/3dpw",
+)
+MainStore.store(
+ name="v1",
+ node=builds(ThreedpwSmplFullSeqDataset, flip_test=False),
+ group="test_datasets/3dpw",
+)
diff --git a/third_party/GVHMR/hmr4d/dataset/threedpw/threedpw_motion_train.py b/third_party/GVHMR/hmr4d/dataset/threedpw/threedpw_motion_train.py
new file mode 100644
index 0000000000000000000000000000000000000000..2c803fd0e4032f85ad48bd9d14bc96bb03498d27
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/threedpw/threedpw_motion_train.py
@@ -0,0 +1,164 @@
+import torch
+from torch.utils import data
+from pathlib import Path
+import numpy as np
+
+from hmr4d.utils.pylogger import Log
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines
+from hmr4d.utils.geo_transform import compute_cam_angvel
+from hmr4d.utils.geo.hmr_cam import estimate_K, resize_K
+from hmr4d.utils.geo.flip_utils import flip_kp2d_coco17
+from hmr4d.dataset.imgfeat_motion.base_dataset import ImgfeatMotionDatasetBase
+from hmr4d.utils.net_utils import get_valid_mask, repeat_to_max_len, repeat_to_max_len_dict
+from hmr4d.utils.smplx_utils import make_smplx
+from hmr4d.utils.video_io_utils import get_video_lwh, read_video_np, save_video
+from hmr4d.utils.vis.renderer_utils import simple_render_mesh_background
+
+from hmr4d.configs import MainStore, builds
+
+
+class ThreedpwSmplDataset(ImgfeatMotionDatasetBase):
+ def __init__(self):
+ # Path
+ self.hmr4d_support_dir = Path("inputs/3DPW/hmr4d_support")
+ self.dataset_name = "3DPW"
+
+ # Setting
+ self.min_motion_frames = 60
+ self.max_motion_frames = 120
+ super().__init__()
+
+ def _load_dataset(self):
+ self.train_labels = torch.load(self.hmr4d_support_dir / "train_3dpw_gt_labels.pt")
+ self.refit_smplx = torch.load(self.hmr4d_support_dir / "train_refit_smplx.pt")
+ if True: # Remove clips that have obvious error
+ update_list = {
+ "courtyard_basketball_00_1": [(0, 300), (340, 468)],
+ "courtyard_laceShoe_00_0": [(0, 620), (780, 931)],
+ "courtyard_rangeOfMotions_00_1": [(0, 370), (410, 601)],
+ "courtyard_shakeHands_00_1": [(0, 100), (120, 391)],
+ }
+ for k, v in update_list.items():
+ self.refit_smplx[k]["valid_range_list"] = v
+
+ self.f_img_folder = self.hmr4d_support_dir / "imgfeats/3dpw_train_smplx_refit"
+ Log.info(f"[{self.dataset_name}] Train")
+
+ def _get_idx2meta(self):
+ # We expect to see the entire sequence during one epoch,
+ # so each sequence will be sampled max(SeqLength // MotionFrames, 1) times
+ seq_lengths = []
+ self.idx2meta = []
+ for vid in self.refit_smplx:
+ valid_range_list = self.refit_smplx[vid]["valid_range_list"]
+ for start, end in valid_range_list:
+ seq_length = end - start
+ num_samples = max(seq_length // self.max_motion_frames, 1)
+ seq_lengths.append(seq_length)
+ self.idx2meta.extend([(vid, start, end)] * num_samples)
+ minutes = sum(seq_lengths) / 25 / 60
+ Log.info(
+ f"[{self.dataset_name}] has {minutes:.1f} minutes motion -> Resampled to {len(self.idx2meta)} samples."
+ )
+
+ def _load_data(self, idx):
+ data = {}
+ vid, range1, range2 = self.idx2meta[idx]
+
+ # Random select a subset
+ mlength = range2 - range1
+ min_motion_len = self.min_motion_frames
+ max_motion_len = self.max_motion_frames
+
+ if mlength < min_motion_len: # this may happen, the minimal mlength is around 30
+ start = range1
+ length = mlength
+ else:
+ effect_max_motion_len = min(max_motion_len, mlength)
+ length = np.random.randint(min_motion_len, effect_max_motion_len + 1) # [low, high)
+ start = np.random.randint(range1, range2 - length + 1)
+ end = start + length
+ data["length"] = length
+ data["meta"] = {"data_name": self.dataset_name, "idx": idx, "vid": vid, "start_end": (start, end)}
+
+ # Select motion subset
+ data["smplx_params_incam"] = {k: v[start:end] for k, v in self.refit_smplx[vid]["smplx_params_incam"].items()}
+ data["K_fullimg"] = self.train_labels[vid]["K_fullimg"]
+ data["T_w2c"] = self.train_labels[vid]["T_w2c"][start:end]
+
+ # Img (as feature):
+ f_img_dict = torch.load(self.f_img_folder / f"{vid}.pt")
+ data["bbx_xys"] = f_img_dict["bbx_xys"][start:end] # (F, 3)
+ data["f_imgseq"] = f_img_dict["features"][start:end].float() # (F, 3)
+ data["img_wh"] = f_img_dict["img_wh"] # (2)
+ data["kp2d"] = torch.zeros((end - start), 17, 3) # (L, 17, 3) # do not provide kp2d
+
+ return data
+
+ def _process_data(self, data, idx):
+ length = data["length"]
+
+ smpl_params_c = data["smplx_params_incam"]
+ smpl_params_w_zero = {k: torch.zeros_like(v) for k, v in smpl_params_c.items()}
+ K_fullimg = data["K_fullimg"][None].repeat(length, 1, 1)
+ cam_angvel = compute_cam_angvel(data["T_w2c"][:, :3, :3])
+
+ max_len = self.max_motion_frames
+ return_data = {
+ "meta": data["meta"],
+ "length": length,
+ "smpl_params_c": smpl_params_c,
+ "smpl_params_w": smpl_params_w_zero,
+ "R_c2gv": torch.zeros(length, 3, 3), # (F, 3, 3)
+ "gravity_vec": torch.zeros(3), # (3)
+ "bbx_xys": data["bbx_xys"], # (F, 3)
+ "K_fullimg": K_fullimg, # (F, 3, 3)
+ "f_imgseq": data["f_imgseq"], # (F, D)
+ "kp2d": data["kp2d"], # (F, 17, 3)
+ "cam_angvel": cam_angvel, # (F, 6)
+ "mask": {
+ "valid": get_valid_mask(max_len, length),
+ "vitpose": False,
+ "bbx_xys": True,
+ "f_imgseq": True,
+ "spv_incam_only": True,
+ },
+ }
+
+ if False: # Debug, render incam
+ start, end = data["meta"]["start_end"]
+ vid = data["meta"]["vid"]
+
+ ds = 0.5
+ faces = smplx.faces
+ smplx = make_smplx("supermotion")
+ smplx_c_verts = smplx(**return_data["smpl_params_c"]).vertices
+ K_render = resize_K(K_fullimg, ds)
+
+ video_path = self.hmr4d_support_dir / f"videos/{vid[:-2]}.mp4"
+ images = read_video_np(video_path, scale=ds, start_frame=start, end_frame=end)
+
+ render_dict = {
+ "K": K_render[:1], # only support batch size 1
+ "faces": faces,
+ "verts": smplx_c_verts,
+ "background": images,
+ }
+ img_overlay = simple_render_mesh_background(render_dict, VI=10)
+ save_video(img_overlay, f"tmp.mp4", crf=28)
+
+ # Batchable
+ return_data["smpl_params_c"] = repeat_to_max_len_dict(return_data["smpl_params_c"], max_len)
+ return_data["smpl_params_w"] = repeat_to_max_len_dict(return_data["smpl_params_w"], max_len)
+ return_data["R_c2gv"] = repeat_to_max_len(return_data["R_c2gv"], max_len)
+ return_data["bbx_xys"] = repeat_to_max_len(return_data["bbx_xys"], max_len)
+ return_data["K_fullimg"] = repeat_to_max_len(return_data["K_fullimg"], max_len)
+ return_data["f_imgseq"] = repeat_to_max_len(return_data["f_imgseq"], max_len)
+ return_data["kp2d"] = repeat_to_max_len(return_data["kp2d"], max_len)
+ return_data["cam_angvel"] = repeat_to_max_len(return_data["cam_angvel"], max_len)
+
+ return return_data
+
+
+# 3DPW
+MainStore.store(name="v1", node=builds(ThreedpwSmplDataset), group="train_datasets/imgfeat_3dpw")
diff --git a/third_party/GVHMR/hmr4d/dataset/threedpw/utils.py b/third_party/GVHMR/hmr4d/dataset/threedpw/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..ca0ac3603af3533eb5f455cad0fc5dfd875d1836
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/dataset/threedpw/utils.py
@@ -0,0 +1,81 @@
+import json
+import numpy as np
+from pathlib import Path
+from collections import defaultdict
+import pickle
+import torch
+import joblib
+
+RESOURCE_FOLDER = Path(__file__).resolve().parent / "resource"
+
+
+def read_raw_pkl(pkl_path):
+ with open(pkl_path, "rb") as f:
+ data = pickle.load(f, encoding="bytes")
+
+ num_subjects = len(data[b"poses"])
+ F = data[b"poses"][0].shape[0]
+ smpl_params = []
+ for i in range(num_subjects):
+ smpl_params.append(
+ {
+ "body_pose": torch.from_numpy(data[b"poses"][i][:, 3:72]).float(), # (F, 69)
+ "betas": torch.from_numpy(data[b"betas"][i][:10]).repeat(F, 1).float(), # (F, 10)
+ "global_orient": torch.from_numpy(data[b"poses"][i][:, :3]).float(), # (F, 3)
+ "transl": torch.from_numpy(data[b"trans"][i]).float(), # (F, 3)
+ }
+ )
+ genders = ["male" if g == "m" else "female" for g in data[b"genders"]]
+ campose_valid = [torch.from_numpy(v).bool() for v in data[b"campose_valid"]]
+
+ seq_name = data[b"sequence"]
+ K_fullimg = torch.from_numpy(data[b"cam_intrinsics"]).float()
+ T_w2c = torch.from_numpy(data[b"cam_poses"]).float()
+
+ return_data = {
+ "sequence": seq_name, # 'courtyard_bodyScannerMotions_00'
+ "K_fullimg": K_fullimg, # (3, 3), not 55FoV
+ "T_w2c": T_w2c, # (F, 4, 4)
+ "smpl_params": smpl_params, # list of dict
+ "genders": genders, # list of str
+ "campose_valid": campose_valid, # list of bool-array
+ # "jointPositions": data[b'jointPositions'], # SMPL, 24x3
+ # "poses2d": data[b"poses2d"], # COCO, 3x18(?)
+ }
+ return return_data
+
+
+def load_and_convert_wham_pth(pth):
+ """
+ Convert to {vid: DataDict} style, Add smpl_params_incam
+ """
+ # load
+ wham_labels_raw = joblib.load(pth)
+ # convert it to {vid: DataDict} style
+ wham_labels = {}
+ for i, vid in enumerate(wham_labels_raw["vid"]):
+ wham_labels[vid] = {k: wham_labels_raw[k][i] for k in wham_labels_raw}
+
+ # convert pose and betas as smpl_params_incam (without transl)
+ for vid in wham_labels:
+ pose = wham_labels[vid]["pose"]
+ global_orient = pose[:, :3] # (F, 3)
+ body_pose = pose[:, 3:] # (F, 69)
+ betas = wham_labels[vid]["betas"] # (F, 10), all frames are the same
+ wham_labels[vid]["smpl_params_incam"] = {
+ "body_pose": body_pose.float(), # (F, 69)
+ "betas": betas.float(), # (F, 10)
+ "global_orient": global_orient.float(), # (F, 3)
+ }
+
+ return wham_labels
+
+
+# Neural-Annot utils
+
+
+def na_cam_param_to_K_fullimg(cam_param):
+ K = torch.eye(3)
+ K[[0, 1], [0, 1]] = torch.tensor(cam_param["focal"])
+ K[[0, 1], [2, 2]] = torch.tensor(cam_param["princpt"])
+ return K
diff --git a/third_party/GVHMR/hmr4d/model/common_utils/optimizer.py b/third_party/GVHMR/hmr4d/model/common_utils/optimizer.py
new file mode 100644
index 0000000000000000000000000000000000000000..b03513faf006286a8a4d099e5fc70a65d8ff7988
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/common_utils/optimizer.py
@@ -0,0 +1,17 @@
+from torch.optim import AdamW, Adam
+from hmr4d.configs import MainStore, builds
+
+
+optimizer_cfgs = {
+ "adam_1e-3": builds(Adam, lr=1e-3, zen_partial=True),
+ "adam_2e-4": builds(Adam, lr=2e-4, zen_partial=True),
+ "adamw_2e-4": builds(AdamW, lr=2e-4, zen_partial=True),
+ "adamw_1e-4": builds(AdamW, lr=1e-4, zen_partial=True),
+ "adamw_5e-5": builds(AdamW, lr=5e-5, zen_partial=True),
+ "adamw_1e-5": builds(AdamW, lr=1e-5, zen_partial=True),
+ # zero-shot text-to-image generation
+ "adamw_1e-3_dalle": builds(AdamW, lr=1e-3, weight_decay=1e-4, zen_partial=True),
+}
+
+for name, cfg in optimizer_cfgs.items():
+ MainStore.store(name=name, node=cfg, group=f"optimizer")
diff --git a/third_party/GVHMR/hmr4d/model/common_utils/scheduler.py b/third_party/GVHMR/hmr4d/model/common_utils/scheduler.py
new file mode 100644
index 0000000000000000000000000000000000000000..8d8bc54e46bf5fcc216824827a34786339a9a2f9
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/common_utils/scheduler.py
@@ -0,0 +1,29 @@
+import torch
+from bisect import bisect_right
+
+
+class WarmupMultiStepLR(torch.optim.lr_scheduler.LRScheduler):
+ def __init__(self, optimizer, milestones, warmup=0, gamma=0.1, last_epoch=-1, verbose="deprecated"):
+ """Assume optimizer does not change lr; Scheduler is called epoch-based"""
+ self.milestones = milestones
+ self.warmup = warmup
+ assert warmup < milestones[0]
+ self.gamma = gamma
+ super().__init__(optimizer, last_epoch, verbose)
+
+ def get_lr(self):
+ base_lrs = self.base_lrs # base lr for each groups
+ n_groups = len(base_lrs)
+ comming_epoch = self.last_epoch # the lr will be set for the comming epoch, starts from 0
+
+ # add extra warmup
+ if comming_epoch < self.warmup:
+ # e.g. comming_epoch [0, 1, 2] for warmup == 3
+ # lr should be base_lr * (last_epoch+1) / (warmup + 1), e.g. [0.25, 0.5, 0.75] * base_lr
+ lr_factor = (self.last_epoch + 1) / (self.warmup + 1)
+ return [base_lrs[i] * lr_factor for i in range(n_groups)]
+ else:
+ # bisect_right([3,5,7], 0) -> 0; bisect_right([3,5,7], 5) -> 2
+ p = bisect_right(self.milestones, comming_epoch)
+ lr_factor = self.gamma**p
+ return [base_lrs[i] * lr_factor for i in range(n_groups)]
diff --git a/third_party/GVHMR/hmr4d/model/common_utils/scheduler_cfg.py b/third_party/GVHMR/hmr4d/model/common_utils/scheduler_cfg.py
new file mode 100644
index 0000000000000000000000000000000000000000..fd16eee23acff52145ac69230545cc2301df141e
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/common_utils/scheduler_cfg.py
@@ -0,0 +1,47 @@
+from omegaconf import DictConfig, ListConfig
+from hmr4d.configs import MainStore, builds
+
+# do not perform scheduling
+default = DictConfig({"scheduler": None})
+MainStore.store(name="default", node=default, group=f"scheduler_cfg")
+
+
+# epoch-based
+def epoch_half_by(milestones=[100, 200, 300]):
+ return DictConfig(
+ {
+ "scheduler": {
+ "_target_": "torch.optim.lr_scheduler.MultiStepLR",
+ "milestones": milestones,
+ "gamma": 0.5,
+ },
+ "interval": "epoch",
+ "frequency": 1,
+ }
+ )
+
+
+MainStore.store(name="epoch_half_100_200_300", node=epoch_half_by([100, 200, 300]), group=f"scheduler_cfg")
+MainStore.store(name="epoch_half_100_200", node=epoch_half_by([100, 200]), group=f"scheduler_cfg")
+MainStore.store(name="epoch_half_200_350", node=epoch_half_by([200, 350]), group=f"scheduler_cfg")
+MainStore.store(name="epoch_half_300", node=epoch_half_by([300]), group=f"scheduler_cfg")
+
+
+# epoch-based
+def warmup_epoch_half_by(warmup=10, milestones=[100, 200, 300]):
+ return DictConfig(
+ {
+ "scheduler": {
+ "_target_": "hmr4d.model.common_utils.scheduler.WarmupMultiStepLR",
+ "milestones": milestones,
+ "warmup": warmup,
+ "gamma": 0.5,
+ },
+ "interval": "epoch",
+ "frequency": 1,
+ }
+ )
+
+
+MainStore.store(name="warmup_5_epoch_half_200_350", node=warmup_epoch_half_by(5, [200, 350]), group=f"scheduler_cfg")
+MainStore.store(name="warmup_10_epoch_half_200_350", node=warmup_epoch_half_by(10, [200, 350]), group=f"scheduler_cfg")
diff --git a/third_party/GVHMR/hmr4d/model/gvhmr/callbacks/metric_3dpw.py b/third_party/GVHMR/hmr4d/model/gvhmr/callbacks/metric_3dpw.py
new file mode 100644
index 0000000000000000000000000000000000000000..6af1750a0a150977d4ae774a6b52b2949db95771
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/gvhmr/callbacks/metric_3dpw.py
@@ -0,0 +1,186 @@
+import torch
+import pytorch_lightning as pl
+import numpy as np
+from pathlib import Path
+from einops import einsum, rearrange
+
+from hmr4d.configs import MainStore, builds
+from hmr4d.utils.pylogger import Log
+from hmr4d.utils.comm.gather import all_gather
+from hmr4d.utils.eval.eval_utils import compute_camcoord_metrics, as_np_array
+from hmr4d.utils.smplx_utils import make_smplx
+from hmr4d.utils.vis.cv2_utils import cv2, draw_bbx_xys_on_image_batch, draw_coco17_skeleton_batch
+from hmr4d.utils.vis.renderer_utils import simple_render_mesh_background
+from hmr4d.utils.video_io_utils import read_video_np, get_video_lwh, save_video
+from hmr4d.utils.geo_transform import apply_T_on_points
+from hmr4d.utils.seq_utils import rearrange_by_mask
+
+
+class MetricMocap(pl.Callback):
+ def __init__(self):
+ super().__init__()
+ # vid->result
+ self.metric_aggregator = {
+ "pa_mpjpe": {},
+ "mpjpe": {},
+ "pve": {},
+ "accel": {},
+ }
+
+ # SMPLX and SMPL
+ self.smplx = make_smplx("supermotion_EVAL3DPW")
+ self.smpl = {"male": make_smplx("smpl", gender="male"), "female": make_smplx("smpl", gender="female")}
+ self.J_regressor = torch.load("hmr4d/utils/body_model/smpl_3dpw14_J_regressor_sparse.pt").to_dense()
+ self.J_regressor24 = torch.load("hmr4d/utils/body_model/smpl_neutral_J_regressor.pt")
+ self.smplx2smpl = torch.load("hmr4d/utils/body_model/smplx2smpl_sparse.pt")
+ self.faces_smplx = self.smplx.faces
+ self.faces_smpl = self.smpl["male"].faces
+
+ # The metrics are calculated similarly for val/test/predict
+ self.on_test_batch_end = self.on_validation_batch_end = self.on_predict_batch_end
+
+ # Only validation record the metrics with logger
+ self.on_test_epoch_end = self.on_validation_epoch_end = self.on_predict_epoch_end
+
+ # ================== Batch-based Computation ================== #
+ def on_predict_batch_end(self, trainer, pl_module, outputs, batch, batch_idx, dataloader_idx=0):
+ """The behaviour is the same for val/test/predict"""
+ assert batch["B"] == 1
+ dataset_id = batch["meta"][0]["dataset_id"]
+ if dataset_id != "3DPW":
+ return
+
+ # Move to cuda if not
+ self.smplx = self.smplx.cuda()
+ for g in ["male", "female"]:
+ self.smpl[g] = self.smpl[g].cuda()
+ self.J_regressor = self.J_regressor.cuda()
+ self.J_regressor24 = self.J_regressor24.cuda()
+ self.smplx2smpl = self.smplx2smpl.cuda()
+
+ vid = batch["meta"][0]["vid"]
+ seq_length = batch["length"][0].item()
+ gender = batch["gender"][0]
+ T_w2c = batch["T_w2c"][0]
+ mask = batch["mask"][0]
+
+ # Groundtruth (cam)
+ target_w_params = {k: v[0] for k, v in batch["smpl_params"].items()}
+ target_w_output = self.smpl[gender](**target_w_params)
+ target_w_verts = target_w_output.vertices
+ target_c_verts = apply_T_on_points(target_w_verts, T_w2c)
+ target_c_j3d = torch.matmul(self.J_regressor, target_c_verts)
+
+ # + Prediction -> Metric
+ smpl_out = self.smplx(**outputs["pred_smpl_params_incam"])
+ pred_c_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in smpl_out.vertices])
+ pred_c_j3d = einsum(self.J_regressor, pred_c_verts, "j v, l v i -> l j i")
+ del smpl_out # Prevent OOM
+
+ # Metric of current sequence
+ batch_eval = {
+ "pred_j3d": pred_c_j3d,
+ "target_j3d": target_c_j3d,
+ "pred_verts": pred_c_verts,
+ "target_verts": target_c_verts,
+ }
+ camcoord_metrics = compute_camcoord_metrics(batch_eval, mask=mask, pelvis_idxs=[2, 3])
+ for k in camcoord_metrics:
+ self.metric_aggregator[k][vid] = as_np_array(camcoord_metrics[k])
+
+ if False: # Render incam (simple)
+ meta_render = batch["meta_render"][0]
+ images = read_video_np(meta_render["video_path"], scale=meta_render["ds"])
+ render_dict = {
+ "K": meta_render["K"][None], # only support batch size 1
+ "faces": self.smpl["male"].faces,
+ "verts": pred_c_verts,
+ "background": images,
+ }
+ img_overlay = simple_render_mesh_background(render_dict)
+ output_fn = Path("outputs/3DPW_render_pred_flip") / f"{vid}.mp4"
+ save_video(img_overlay, output_fn, crf=28)
+
+ if False: # Render incam (with details)
+ meta_render = batch["meta_render"][0]
+ images = read_video_np(meta_render["video_path"], scale=meta_render["ds"])
+ render_dict = {
+ "K": meta_render["K"][None], # only support batch size 1
+ "faces": self.smpl["male"].faces,
+ "verts": pred_c_verts,
+ "background": images,
+ }
+ img_overlay = simple_render_mesh_background(render_dict)
+
+ # Add COCO17 and bbx to image
+ bbx_xys_render = meta_render["bbx_xys"]
+ kp2d_render = meta_render["kp2d"]
+ img_overlay = draw_coco17_skeleton_batch(img_overlay, kp2d_render, conf_thr=0.5)
+ img_overlay = draw_bbx_xys_on_image_batch(bbx_xys_render, img_overlay, mask)
+
+ # Add metric
+ metric_all = rearrange_by_mask(torch.tensor(camcoord_metrics["pa_mpjpe"]), mask)
+ for i in range(len(img_overlay)):
+ m = metric_all[i]
+ if m == 0: # a not evaluated frame
+ continue
+ text = f"PA-MPJPE: {m:.1f}"
+ color = (244, 10, 20) if m > 45 else (0, 205, 0) # red or green
+ cv2.putText(img_overlay[i], text, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1, color, 2)
+
+ output_dir = Path("tmp_pred_details")
+ output_dir.mkdir(exist_ok=True, parents=True)
+ save_video(img_overlay, output_dir / f"{vid}.mp4", crf=24)
+
+ # ================== Epoch Summary ================== #
+ def on_predict_epoch_end(self, trainer, pl_module):
+ """Without logger"""
+ local_rank, world_size = trainer.local_rank, trainer.world_size
+ monitor_metric = "pa_mpjpe"
+
+ # Reduce metric_aggregator across all processes
+ metric_keys = list(self.metric_aggregator.keys())
+ with torch.inference_mode(False): # allow in-place operation of all_gather
+ metric_aggregator_gathered = all_gather(self.metric_aggregator) # list of dict
+ for metric_key in metric_keys:
+ for d in metric_aggregator_gathered:
+ self.metric_aggregator[metric_key].update(d[metric_key])
+
+ if False: # debug to make sure the all_gather is correct
+ print(f"[RANK {local_rank}/{world_size}]: {self.metric_aggregator[monitor_metric].keys()}")
+
+ total = len(self.metric_aggregator[monitor_metric])
+ Log.info(f"{total} sequences evaluated in {self.__class__.__name__}")
+ if total == 0:
+ return
+
+ # print monitored metric per sequence
+ mm_per_seq = {k: v.mean() for k, v in self.metric_aggregator[monitor_metric].items()}
+ if len(mm_per_seq) > 0:
+ sorted_mm_per_seq = sorted(mm_per_seq.items(), key=lambda x: x[1], reverse=True)
+ n_worst = 5 if trainer.state.stage == "validate" else len(sorted_mm_per_seq)
+ if local_rank == 0:
+ Log.info(
+ f"monitored metric {monitor_metric} per sequence\n"
+ + "\n".join([f"{m:5.1f} : {s}" for s, m in sorted_mm_per_seq[:n_worst]])
+ + "\n------"
+ )
+
+ # average over all batches
+ metrics_avg = {k: np.concatenate(list(v.values())).mean() for k, v in self.metric_aggregator.items()}
+ if local_rank == 0:
+ Log.info(f"[Metrics] 3DPW:\n" + "\n".join(f"{k}: {v:.1f}" for k, v in metrics_avg.items()) + "\n------")
+
+ # save to logger if available
+ if pl_module.logger is not None:
+ cur_epoch = pl_module.current_epoch
+ for k, v in metrics_avg.items():
+ pl_module.logger.log_metrics({f"val_metric_3DPW/{k}": v}, step=cur_epoch)
+
+ # reset
+ for k in self.metric_aggregator:
+ self.metric_aggregator[k] = {}
+
+
+node_3dpw = builds(MetricMocap)
+MainStore.store(name="metric_3dpw", node=node_3dpw, group="callbacks", package="callbacks.metric_3dpw")
diff --git a/third_party/GVHMR/hmr4d/model/gvhmr/callbacks/metric_emdb.py b/third_party/GVHMR/hmr4d/model/gvhmr/callbacks/metric_emdb.py
new file mode 100644
index 0000000000000000000000000000000000000000..36631a7ebddb245c9b0115a427c29346d5b8364c
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/gvhmr/callbacks/metric_emdb.py
@@ -0,0 +1,323 @@
+import torch
+import torch.nn.functional as F
+import pytorch_lightning as pl
+from pytorch_lightning.utilities import rank_zero_only
+from hmr4d.configs import MainStore, builds
+
+from hmr4d.utils.comm.gather import all_gather
+from hmr4d.utils.pylogger import Log
+
+from hmr4d.utils.eval.eval_utils import (
+ compute_camcoord_metrics,
+ compute_global_metrics,
+ compute_camcoord_perjoint_metrics,
+ rearrange_by_mask,
+ as_np_array,
+)
+from hmr4d.utils.geo_transform import apply_T_on_points, compute_T_ayfz2ay
+from hmr4d.utils.smplx_utils import make_smplx
+from einops import einsum, rearrange
+
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines
+from hmr4d.utils.vis.renderer import Renderer, get_global_cameras_static
+from hmr4d.utils.geo.hmr_cam import estimate_focal_length
+from hmr4d.utils.video_io_utils import read_video_np, save_video
+import imageio
+from tqdm import tqdm
+from pathlib import Path
+import numpy as np
+import cv2
+
+
+class MetricMocap(pl.Callback):
+ def __init__(self, emdb_split=1):
+ """
+ Args:
+ emdb_split: 1 to evaluate incam, 2 to evaluate global
+ """
+ super().__init__()
+ # vid->result
+ if emdb_split == 1:
+ self.target_dataset_id = "EMDB_1"
+ self.metric_aggregator = {
+ "pa_mpjpe": {},
+ "mpjpe": {},
+ "pve": {},
+ "accel": {},
+ }
+ elif emdb_split == 2:
+ self.target_dataset_id = "EMDB_2"
+ self.metric_aggregator = {
+ "wa2_mpjpe": {},
+ "waa_mpjpe": {},
+ "rte": {},
+ "jitter": {},
+ "fs": {},
+ }
+ else:
+ raise ValueError(f"Unknown emdb_split: {emdb_split}")
+
+ # SMPL
+ self.smplx = make_smplx("supermotion")
+ self.smpl_model = {"male": make_smplx("smpl", gender="male"), "female": make_smplx("smpl", gender="female")}
+
+ self.J_regressor = torch.load("hmr4d/utils/body_model/smpl_neutral_J_regressor.pt")
+ self.smplx2smpl = torch.load("hmr4d/utils/body_model/smplx2smpl_sparse.pt")
+ self.faces_smpl = self.smpl_model["male"].faces
+ self.faces_smplx = self.smplx.faces
+
+ # The metrics are calculated similarly for val/test/predict
+ self.on_test_batch_end = self.on_validation_batch_end = self.on_predict_batch_end
+
+ # Only validation record the metrics with logger
+ self.on_test_epoch_end = self.on_validation_epoch_end = self.on_predict_epoch_end
+
+ # ================== Batch-based Computation ================== #
+ def on_predict_batch_end(self, trainer, pl_module, outputs, batch, batch_idx, dataloader_idx=0):
+ """The behaviour is the same for val/test/predict"""
+ assert batch["B"] == 1
+ dataset_id = batch["meta"][0]["dataset_id"]
+ if dataset_id != self.target_dataset_id:
+ return
+
+ # Move to cuda if not
+ self.smplx = self.smplx.cuda()
+ for g in ["male", "female"]:
+ self.smpl_model[g] = self.smpl_model[g].cuda()
+ self.J_regressor = self.J_regressor.cuda()
+ self.smplx2smpl = self.smplx2smpl.cuda()
+
+ vid = batch["meta"][0]["vid"]
+ seq_length = batch["length"][0].item()
+ gender = batch["gender"][0]
+ T_w2c = batch["T_w2c"][0]
+ mask = batch["mask"][0]
+
+ # Groundtruth (world, cam)
+ target_w_params = {k: v[0] for k, v in batch["smpl_params"].items()}
+ target_w_output = self.smpl_model[gender](**target_w_params)
+ target_w_verts = target_w_output.vertices
+ target_w_j3d = torch.matmul(self.J_regressor, target_w_verts)
+ target_c_verts = apply_T_on_points(target_w_verts, T_w2c)
+ target_c_j3d = apply_T_on_points(target_w_j3d, T_w2c)
+
+ # + Prediction -> Metric
+ if self.target_dataset_id == "EMDB_1": # in camera metrics
+ # 1. cam
+ pred_smpl_params_incam = outputs["pred_smpl_params_incam"]
+ smpl_out = self.smplx(**pred_smpl_params_incam)
+ pred_c_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in smpl_out.vertices])
+ pred_c_j3d = einsum(self.J_regressor, pred_c_verts, "j v, l v i -> l j i")
+ del smpl_out # Prevent OOM
+
+ batch_eval = {
+ "pred_j3d": pred_c_j3d,
+ "target_j3d": target_c_j3d,
+ "pred_verts": pred_c_verts,
+ "target_verts": target_c_verts,
+ }
+ camcoord_metrics = compute_camcoord_metrics(batch_eval, mask=mask)
+ for k in camcoord_metrics:
+ self.metric_aggregator[k][vid] = as_np_array(camcoord_metrics[k])
+
+ elif self.target_dataset_id == "EMDB_2": # global metrics
+ # 2. global (align-y axis)
+ pred_smpl_params_global = outputs["pred_smpl_params_global"]
+ smpl_out = self.smplx(**pred_smpl_params_global)
+ pred_ay_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in smpl_out.vertices])
+ pred_ay_j3d = einsum(self.J_regressor, pred_ay_verts, "j v, l v i -> l j i")
+ del smpl_out # Prevent OOM
+
+ batch_eval = {
+ "pred_j3d_glob": pred_ay_j3d,
+ "target_j3d_glob": target_w_j3d,
+ "pred_verts_glob": pred_ay_verts,
+ "target_verts_glob": target_w_verts,
+ }
+ global_metrics = compute_global_metrics(batch_eval, mask=mask)
+ for k in global_metrics:
+ self.metric_aggregator[k][vid] = as_np_array(global_metrics[k])
+
+ if False: # wis3d debug
+ wis3d = make_wis3d(name="debug-emdb-incam")
+ pred_cr_j3d = pred_c_j3d - pred_c_j3d[:, [0]] # (L, J, 3)
+ target_cr_j3d = target_c_j3d - target_c_j3d[:, [0]] # (L, J, 3)
+ add_motion_as_lines(pred_cr_j3d, wis3d, name="pred_cr_j3d", const_color="blue")
+ add_motion_as_lines(target_cr_j3d, wis3d, name="target_cr_j3d", const_color="green")
+
+ if False: # Dump wis3d
+ vid = batch["meta"][0]["vid"]
+ split = batch["meta_render"][0]["split"]
+ wis3d = make_wis3d(name=f"dump_emdb{split}-{vid}")
+ R_cam_type = batch["meta_render"][0]["R_cam_type"]
+
+ pred_cr_j3d = pred_c_j3d - pred_c_j3d[:, [0]] # (L, J, 3)
+ target_cr_j3d = target_c_j3d - target_c_j3d[:, [0]] # (L, J, 3)
+ add_motion_as_lines(pred_cr_j3d, wis3d, name="pred_cr_j3d", const_color="blue")
+ add_motion_as_lines(target_cr_j3d, wis3d, name="target_cr_j3d", const_color="green")
+ add_motion_as_lines(pred_ay_j3d, wis3d, name=f"pred_ay_j3d@{R_cam_type}")
+ # add_motion_as_lines(target_w_j3d, wis3d, name="target_ay_j3d")
+
+ if False: # Render incam
+ # -- rendering code -- #
+ vname = batch["meta_render"][0]["name"]
+ video_path = batch["meta_render"][0]["video_path"]
+ width, height = batch["meta_render"][0]["width_height"]
+ K = batch["meta_render"][0]["K"]
+ faces = self.faces_smpl
+ split = batch["meta_render"][0]["split"]
+
+ out_fn = f"outputs/dump_render_emdb{split}/{vname}.mp4"
+ Path(out_fn).parent.mkdir(exist_ok=True, parents=True)
+
+ # renderer
+ renderer = Renderer(width, height, device="cuda", faces=faces, K=K)
+ # not skipping invalid frames
+ resize_factor = 0.25
+ images = read_video_np(video_path, scale=resize_factor) # (F, H, W, 3), uint8, numpy
+ frame_id = batch["meta_render"][0]["frame_id"]
+ bbx_xys_render = batch["meta_render"][0]["bbx_xys"]
+ metric_vis = rearrange_by_mask(torch.from_numpy(self.metric_aggregator["mpjpe"][vid]), mask)
+
+ # -- render mesh -- #
+ verts_incam = pred_c_verts
+ output_images = []
+ for i in tqdm(range(len(images)), desc=f"Rendering {vname}"):
+ img = renderer.render_mesh(verts_incam[i].cuda(), images[i], [0.8, 0.8, 0.8])
+ # bbx
+ bbx_xys_ = bbx_xys_render[i].cpu().numpy()
+ lu_point = (bbx_xys_[:2] - bbx_xys_[2:] / 2).astype(int)
+ rd_point = (bbx_xys_[:2] + bbx_xys_[2:] / 2).astype(int)
+ img = cv2.rectangle(img, lu_point, rd_point, (255, 178, 102), 2)
+
+ if metric_vis[i] > 0:
+ text = f"pred mpjpe: {metric_vis[i]:.1f}"
+ text_color = (244, 10, 20) if metric_vis[i] > 80 else (0, 205, 0) # red or green
+ cv2.putText(img, text, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 0.75, text_color, 2)
+
+ output_images.append(img)
+ save_video(output_images, out_fn, quality=5)
+
+ if False: # Visualize incam + global results
+
+ def move_to_start_point_face_z(verts):
+ "XZ to origin, Start from the ground, Face-Z"
+ verts = verts.clone() # (L, V, 3)
+ xz_mean = verts[0].mean(0)[[0, 2]]
+ y_min = verts[0, :, [1]].min()
+ offset = torch.tensor([[[xz_mean[0], y_min, xz_mean[1]]]]).to(verts)
+ verts = verts - offset
+
+ T_ay2ayfz = compute_T_ayfz2ay(einsum(self.J_regressor, verts[[0]], "j v, l v i -> l j i"), inverse=True)
+ verts = apply_T_on_points(verts, T_ay2ayfz)
+ return verts
+
+ verts_incam = pred_c_verts.clone()
+ # verts_glob = move_to_start_point_face_z(target_ay_verts) # gt
+ verts_glob = move_to_start_point_face_z(pred_ay_verts)
+ global_R, global_T, global_lights = get_global_cameras_static(verts_glob.cpu())
+
+ # -- rendering code (global version FOV=55) -- #
+ vname = batch["meta_render"][0]["name"]
+ width, height = batch["meta_render"][0]["width_height"]
+ K = batch["meta_render"][0]["K"]
+ faces = self.faces_smpl
+ out_fn = f"outputs/dump_render_global/{vname}.mp4"
+ Path(out_fn).parent.mkdir(exist_ok=True, parents=True)
+ writer = imageio.get_writer(out_fn, fps=30, mode="I", format="FFMPEG", macro_block_size=1)
+
+ # two renderers
+ renderer_incam = Renderer(width, height, device="cuda", faces=faces, K=K)
+ renderer_glob = Renderer(width, height, estimate_focal_length(width, height), device="cuda", faces=faces)
+
+ # imgs
+ video_path = batch["meta_render"][0]["video_path"]
+ frame_id = batch["meta_render"][0]["frame_id"].cpu().numpy()
+ images = read_video_np(video_path, frame_id=frame_id) # (F, H/4, W/4, 3), uint8, numpy
+
+ # Actual rendering
+ cx, cz = (verts_glob.mean(1).max(0)[0] + verts_glob.mean(1).min(0)[0])[[0, 2]] / 2.0
+ scale = (verts_glob.mean(1).max(0)[0] - verts_glob.mean(1).min(0)[0])[[0, 2]].max() * 1.5
+ renderer_glob.set_ground(scale, cx.item(), cz.item())
+ color = torch.ones(3).float().cuda() * 0.8
+
+ for i in tqdm(range(seq_length), desc=f"Rendering {vname}"):
+ # incam
+ img_overlay_pred = renderer_incam.render_mesh(verts_incam[i].cuda(), images[i], [0.8, 0.8, 0.8])
+ if batch["meta_render"][0].get("bbx_xys", None) is not None: # draw bbox lines
+ bbx_xys = batch["meta_render"][0]["bbx_xys"][i].cpu().numpy()
+ lu_point = (bbx_xys[:2] - bbx_xys[2:] / 2).astype(int)
+ rd_point = (bbx_xys[:2] + bbx_xys[2:] / 2).astype(int)
+ img_overlay_pred = cv2.rectangle(img_overlay_pred, lu_point, rd_point, (255, 178, 102), 2)
+ pred_mpjpe_ = self.metric_aggregator["mpjpe"][vid][i]
+ text = f"pred mpjpe: {pred_mpjpe_:.1f}"
+ cv2.putText(img_overlay_pred, text, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1, (200, 100, 200), 2)
+
+ # glob
+ cameras = renderer_glob.create_camera(global_R[i], global_T[i])
+ img_glob = renderer_glob.render_with_ground(verts_glob[[i]], color[None], cameras, global_lights)
+
+ # write
+ img = np.concatenate([img_overlay_pred, img_glob], axis=1)
+ writer.append_data(img)
+ writer.close()
+ pass
+
+ # ================== Epoch Summary ================== #
+ def on_predict_epoch_end(self, trainer, pl_module):
+ """Without logger"""
+ local_rank, world_size = trainer.local_rank, trainer.world_size
+ if "mpjpe" in self.metric_aggregator:
+ monitor_metric = "mpjpe"
+ else:
+ monitor_metric = list(self.metric_aggregator.keys())[0]
+
+ # Reduce metric_aggregator across all processes
+ metric_keys = list(self.metric_aggregator.keys())
+ with torch.inference_mode(False): # allow in-place operation of all_gather
+ metric_aggregator_gathered = all_gather(self.metric_aggregator) # list of dict
+ for metric_key in metric_keys:
+ for d in metric_aggregator_gathered:
+ self.metric_aggregator[metric_key].update(d[metric_key])
+
+ total = len(self.metric_aggregator[monitor_metric])
+ Log.info(f"{total} sequences evaluated in {self.__class__.__name__}")
+ if total == 0:
+ return
+
+ # print monitored metric per sequence
+ mm_per_seq = {k: v.mean() for k, v in self.metric_aggregator[monitor_metric].items()}
+ if len(mm_per_seq) > 0:
+ sorted_mm_per_seq = sorted(mm_per_seq.items(), key=lambda x: x[1], reverse=True)
+ n_worst = 5 if trainer.state.stage == "validate" else len(sorted_mm_per_seq)
+ if local_rank == 0:
+ Log.info(
+ f"monitored metric {monitor_metric} per sequence\n"
+ + "\n".join([f"{m:5.1f} : {s}" for s, m in sorted_mm_per_seq[:n_worst]])
+ + "\n------"
+ )
+
+ # average over all batches
+ metrics_avg = {k: np.concatenate(list(v.values())).mean() for k, v in self.metric_aggregator.items()}
+ if local_rank == 0:
+ Log.info(
+ f"[Metrics] {self.target_dataset_id}:\n"
+ + "\n".join(f"{k}: {v:.1f}" for k, v in metrics_avg.items())
+ + "\n------"
+ )
+
+ # save to logger if available
+ if pl_module.logger is not None:
+ cur_epoch = pl_module.current_epoch
+ for k, v in metrics_avg.items():
+ pl_module.logger.log_metrics({f"val_metric_{self.target_dataset_id}/{k}": v}, step=cur_epoch)
+
+ # reset
+ for k in self.metric_aggregator:
+ self.metric_aggregator[k] = {}
+
+
+emdb1_node = builds(MetricMocap, emdb_split=1)
+emdb2_node = builds(MetricMocap, emdb_split=2)
+MainStore.store(name="metric_emdb1", node=emdb1_node, group="callbacks", package="callbacks.metric_emdb1")
+MainStore.store(name="metric_emdb2", node=emdb2_node, group="callbacks", package="callbacks.metric_emdb2")
diff --git a/third_party/GVHMR/hmr4d/model/gvhmr/callbacks/metric_rich.py b/third_party/GVHMR/hmr4d/model/gvhmr/callbacks/metric_rich.py
new file mode 100644
index 0000000000000000000000000000000000000000..f60c9b596fedcdc3c6e42965e7800ca750d26485
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/gvhmr/callbacks/metric_rich.py
@@ -0,0 +1,389 @@
+import torch
+import torch.nn.functional as F
+import pytorch_lightning as pl
+from pytorch_lightning.utilities import rank_zero_only
+from hmr4d.configs import MainStore, builds
+
+from hmr4d.utils.comm.gather import all_gather
+from hmr4d.utils.pylogger import Log
+
+from hmr4d.utils.eval.eval_utils import (
+ compute_camcoord_metrics,
+ compute_global_metrics,
+ compute_camcoord_perjoint_metrics,
+ as_np_array,
+)
+from hmr4d.utils.geo_transform import apply_T_on_points, compute_T_ayfz2ay
+from hmr4d.utils.smplx_utils import make_smplx
+from einops import einsum, rearrange
+
+from pytorch3d.transforms import axis_angle_to_matrix, matrix_to_axis_angle
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines, get_colors_by_conf
+from hmr4d.utils.vis.renderer import Renderer, get_global_cameras_static, get_ground_params_from_points
+from hmr4d.utils.geo.hmr_cam import estimate_focal_length
+from hmr4d.utils.video_io_utils import read_video_np, save_video, get_writer
+import imageio
+from tqdm import tqdm
+from pathlib import Path
+import numpy as np
+import cv2
+
+from smplx.joint_names import JOINT_NAMES
+from hmr4d.utils.net_utils import repeat_to_max_len, gaussian_smooth
+from hmr4d.utils.geo.hmr_global import rollout_vel, get_static_joint_mask
+
+
+class MetricMocap(pl.Callback):
+ def __init__(self):
+ super().__init__()
+ # vid->result
+ self.metric_aggregator = {
+ "pa_mpjpe": {},
+ "mpjpe": {},
+ "pve": {},
+ "accel": {},
+ "wa2_mpjpe": {},
+ "waa_mpjpe": {},
+ "rte": {},
+ "jitter": {},
+ "fs": {},
+ }
+
+ self.perjoint_metrics = False
+ if self.perjoint_metrics:
+ body_joint_names = JOINT_NAMES[:22] + ["left_hand", "right_hand"]
+ self.body_joint_names = body_joint_names
+ self.perjoint_metric_aggregator = {
+ "mpjpe": {k: {} for k in body_joint_names},
+ }
+ self.perjoint_obs_metric_aggregator = {
+ "mpjpe": {k: {} for k in body_joint_names},
+ }
+
+ # SMPL
+ self.smplx_model = {
+ "male": make_smplx("rich-smplx", gender="male"),
+ "female": make_smplx("rich-smplx", gender="female"),
+ "neutral": make_smplx("rich-smplx", gender="neutral"),
+ }
+ self.J_regressor = torch.load("hmr4d/utils/body_model/smpl_neutral_J_regressor.pt")
+ self.smplx2smpl = torch.load("hmr4d/utils/body_model/smplx2smpl_sparse.pt")
+ self.faces_smpl = make_smplx("smpl").faces
+ self.faces_smplx = self.smplx_model["neutral"].faces
+
+ # The metrics are calculated similarly for val/test/predict
+ self.on_test_batch_end = self.on_validation_batch_end = self.on_predict_batch_end
+
+ # Only validation record the metrics with logger
+ self.on_test_epoch_end = self.on_validation_epoch_end = self.on_predict_epoch_end
+
+ # ================== Batch-based Computation ================== #
+ def on_predict_batch_end(self, trainer, pl_module, outputs, batch, batch_idx, dataloader_idx=0):
+ """The behaviour is the same for val/test/predict"""
+ assert batch["B"] == 1
+ dataset_id = batch["meta"][0]["dataset_id"]
+ if dataset_id != "RICH":
+ return
+
+ # Move to cuda if not
+ for g in ["male", "female", "neutral"]:
+ self.smplx_model[g] = self.smplx_model[g].cuda()
+ self.J_regressor = self.J_regressor.cuda()
+ self.smplx2smpl = self.smplx2smpl.cuda()
+
+ vid = batch["meta"][0]["vid"]
+ seq_length = batch["length"][0].item()
+ gender = batch["gender"][0]
+ T_w2ay = batch["T_w2ay"][0]
+ T_w2c = batch["T_w2c"][0]
+
+ # Groundtruth (world, cam)
+ target_w_params = {k: v[0] for k, v in batch["gt_smpl_params"].items()}
+ target_w_output = self.smplx_model[gender](**target_w_params)
+ target_w_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in target_w_output.vertices])
+ target_c_verts = apply_T_on_points(target_w_verts, T_w2c)
+ target_c_j3d = torch.matmul(self.J_regressor, target_c_verts)
+ offset = target_c_j3d[..., [1, 2], :].mean(-2, keepdim=True) # (L, 1, 3)
+ target_cr_j3d = target_c_j3d - offset
+ target_cr_verts = target_c_verts - offset
+ # optional: ay for visual comparison
+ target_ay_verts = apply_T_on_points(target_w_verts, T_w2ay)
+ target_ay_j3d = torch.matmul(self.J_regressor, target_ay_verts)
+
+ # + Prediction -> Metric
+ # 1. cam
+ pred_smpl_params_incam = outputs["pred_smpl_params_incam"]
+ smpl_out = self.smplx_model["neutral"](**pred_smpl_params_incam)
+ pred_c_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in smpl_out.vertices])
+ pred_c_j3d = einsum(self.J_regressor, pred_c_verts, "j v, l v i -> l j i")
+ offset = pred_c_j3d[..., [1, 2], :].mean(-2, keepdim=True) # (L, 1, 3)
+
+ # 2. ay
+ pred_smpl_params_global = outputs["pred_smpl_params_global"]
+ smpl_out = self.smplx_model["neutral"](**pred_smpl_params_global)
+ pred_ay_verts = torch.stack([torch.matmul(self.smplx2smpl, v_) for v_ in smpl_out.vertices])
+ pred_ay_j3d = einsum(self.J_regressor, pred_ay_verts, "j v, l v i -> l j i")
+
+ # Metric of current sequence
+ batch_eval = {
+ "pred_j3d": pred_c_j3d,
+ "target_j3d": target_c_j3d,
+ "pred_verts": pred_c_verts,
+ "target_verts": target_c_verts,
+ }
+ camcoord_metrics = compute_camcoord_metrics(batch_eval)
+ for k in camcoord_metrics:
+ self.metric_aggregator[k][vid] = as_np_array(camcoord_metrics[k])
+
+ batch_eval = {
+ "pred_j3d_glob": pred_ay_j3d,
+ "target_j3d_glob": target_ay_j3d,
+ "pred_verts_glob": pred_ay_verts,
+ "target_verts_glob": target_ay_verts,
+ }
+ global_metrics = compute_global_metrics(batch_eval)
+ for k in global_metrics:
+ self.metric_aggregator[k][vid] = as_np_array(global_metrics[k])
+
+ if False: # global wi3d debug
+ wis3d = make_wis3d(name="debug-metric-global")
+ add_motion_as_lines(pred_ay_j3d, wis3d, name="pred_ay_j3d")
+ add_motion_as_lines(target_ay_j3d, wis3d, name="target_ay_j3d")
+
+ if False: # incam visualize debug
+ # Print per-sequence error
+ Log.info(
+ f"seq {vid} metrics:\n"
+ + "\n".join(
+ f"{k}: {self.metric_aggregator[k][vid].mean():.1f} (obs:{camcoord_metrics[k].mean():.1f})"
+ for k in camcoord_metrics.keys()
+ )
+ + "\n------\n"
+ )
+ if self.perjoint_metrics:
+ Log.info(
+ f"\n".join(
+ f"{k}-{j}: {self.perjoint_metric_aggregator[k][j][vid].mean():.1f} (obs:{self.perjoint_obs_metric_aggregator[k][j][vid].mean():.1f})"
+ for j in self.body_joint_names
+ for k in self.perjoint_obs_metric_aggregator.keys()
+ )
+ + "\n------"
+ )
+
+ # -- metric -- #
+ pred_mpjpe = self.metric_aggregator["mpjpe"][vid].mean()
+ obs_mpjpe = camcoord_metrics["mpjpe"].mean()
+
+ # -- render mesh -- #
+ vertices_gt = target_c_verts
+ vertices_cr_gt = target_cr_verts + target_cr_verts.new([0, 0, 3.0]) # move forward +z
+ vertices_pred = pred_c_verts
+ vertices_cr_obs = obs_cr_verts + obs_cr_verts.new([0, 0, 3.0]) # move forward +z
+ vertices_cr_pred = pred_cr_verts + pred_cr_verts.new([0, 0, 3.0]) # move forward +z
+
+ # -- rendering code -- #
+ vname = batch["meta_render"][0]["name"]
+ K = batch["meta_render"][0]["K"]
+ width, height = batch["meta_render"][0]["width_height"]
+ faces = self.faces_smpl
+
+ renderer = Renderer(width, height, device="cuda", faces=faces, K=K)
+ out_fn = f"outputs/dump_render/{vname}.mp4"
+ Path(out_fn).parent.mkdir(exist_ok=True, parents=True)
+ writer = imageio.get_writer(out_fn, fps=30, mode="I", format="FFMPEG", macro_block_size=1)
+
+ # imgs
+ video_path = batch["meta_render"][0]["video_path"]
+ frame_id = batch["meta_render"][0]["frame_id"].cpu().numpy()
+ vr = decord.VideoReader(video_path)
+ images = vr.get_batch(list(frame_id)).numpy() # (F, H/4, W/4, 3), uint8, numpy
+
+ for i in tqdm(range(seq_length), desc=f"Rendering {vname}"):
+ img_overlay_gt = renderer.render_mesh(vertices_gt[i].cuda(), images[i], [39, 194, 128])
+ if batch["meta_render"][0].get("bbx_xys", None) is not None: # draw bbox lines
+ bbx_xys = batch["meta_render"][0]["bbx_xys"][i].cpu().numpy()
+ lu_point = (bbx_xys[:2] - bbx_xys[2:] / 2).astype(int)
+ rd_point = (bbx_xys[:2] + bbx_xys[2:] / 2).astype(int)
+ img_overlay_gt = cv2.rectangle(img_overlay_gt, lu_point, rd_point, (255, 178, 102), 2)
+
+ img_overlay_pred = renderer.render_mesh(vertices_pred[i].cuda(), images[i])
+ # img_overlay_pred = renderer.render_mesh(vertices_pred[i].cuda(), np.zeros_like(images[i]))
+ img = np.concatenate([img_overlay_gt, img_overlay_pred], axis=0)
+
+ ####### overlay gt cr first, then overlay pred cr with error color ########
+ # overlay gt cr first with blue color
+ black_overlay_obs = renderer.render_mesh(
+ vertices_cr_gt[i].cuda(), np.zeros_like(images[i]), colors=[39, 194, 128]
+ )
+ black_overlay_pred = renderer.render_mesh(
+ vertices_cr_gt[i].cuda(), np.zeros_like(images[i]), colors=[39, 194, 128]
+ )
+
+ # get error color
+ obs_error = (vertices_cr_gt[i] - vertices_cr_obs[i]).norm(dim=-1)
+ pred_error = (vertices_cr_gt[i] - vertices_cr_pred[i]).norm(dim=-1)
+ max_error = max(obs_error.max(), pred_error.max())
+ obs_error_color = torch.stack(
+ [obs_error / max_error, torch.ones_like(obs_error) * 0.6, torch.ones_like(obs_error) * 0.6],
+ dim=-1,
+ )
+ obs_error_color = torch.clip(obs_error_color, 0, 1)
+ pred_error_color = torch.stack(
+ [pred_error / max_error, torch.ones_like(pred_error) * 0.6, torch.ones_like(pred_error) * 0.6],
+ dim=-1,
+ )
+ pred_error_color = torch.clip(pred_error_color, 0, 1)
+
+ # overlay cr with error color
+ black_overlay_obs = renderer.render_mesh(
+ vertices_cr_obs[i].cuda(), black_overlay_obs, colors=obs_error_color[None]
+ )
+ black_overlay_pred = renderer.render_mesh(
+ vertices_cr_pred[i].cuda(), black_overlay_pred, colors=pred_error_color[None]
+ )
+
+ # write mpjpe on the img
+ obs_mpjpe_ = camcoord_metrics["mpjpe"][i]
+ text = f"obs mpjpe: {obs_mpjpe_:.1f} ({obs_mpjpe:.1f})"
+ cv2.putText(black_overlay_obs, text, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1, (100, 200, 200), 2)
+ pred_mpjpe_ = self.metric_aggregator["mpjpe"][vid][i]
+ text = f"pred mpjpe: {pred_mpjpe_:.1f} ({pred_mpjpe:.1f})"
+ if pred_mpjpe_ > obs_mpjpe_:
+ # large error -> purple
+ cv2.putText(black_overlay_pred, text, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1, (200, 100, 200), 2)
+ else:
+ # small error -> yellow
+ cv2.putText(black_overlay_pred, text, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1, (200, 200, 100), 2)
+ black = np.concatenate([black_overlay_obs, black_overlay_pred], axis=0)
+ ###########################################
+
+ img = np.concatenate([img, black], axis=1)
+
+ writer.append_data(img)
+ writer.close()
+
+ if False: # Visualize incam + global results
+
+ def move_to_start_point_face_z(verts):
+ "XZ to origin, Start from the ground, Face-Z"
+ # position
+ verts = verts.clone() # (L, V, 3)
+ offset = einsum(self.J_regressor, verts[0], "j v, v i -> j i")[0] # (3)
+ offset[1] = verts[:, :, [1]].min()
+ verts = verts - offset
+ # face direction
+ T_ay2ayfz = compute_T_ayfz2ay(einsum(self.J_regressor, verts[[0]], "j v, l v i -> l j i"), inverse=True)
+ verts = apply_T_on_points(verts, T_ay2ayfz)
+ return verts
+
+ verts_incam = pred_c_verts.clone()
+ # verts_glob = move_to_start_point_face_z(target_ay_verts) # gt
+ verts_glob = move_to_start_point_face_z(pred_ay_verts)
+ joints_glob = einsum(self.J_regressor, verts_glob, "j v, l v i -> l j i") # (L, J, 3)
+ global_R, global_T, global_lights = get_global_cameras_static(
+ verts_glob.cpu(),
+ beta=4.0,
+ cam_height_degree=20,
+ target_center_height=1.0,
+ vec_rot=-45,
+ )
+
+ # -- rendering code (global version FOV=55) -- #
+ vname = batch["meta_render"][0]["name"]
+ width, height = batch["meta_render"][0]["width_height"]
+ K = batch["meta_render"][0]["K"]
+ faces = self.faces_smpl
+ out_fn = f"outputs/dump_render_global/{vname}.mp4"
+ Path(out_fn).parent.mkdir(exist_ok=True, parents=True)
+
+ # two renderers
+ renderer_incam = Renderer(width, height, device="cuda", faces=faces, K=K)
+ renderer_glob = Renderer(width, height, estimate_focal_length(width, height), device="cuda", faces=faces)
+
+ # imgs
+ video_path = batch["meta_render"][0]["video_path"]
+ frame_id = batch["meta_render"][0]["frame_id"].cpu().numpy()
+ images = read_video_np(video_path)[frame_id] # (F, H/4, W/4, 3), uint8, numpy
+
+ # Actual rendering
+ scale, cx, cz = get_ground_params_from_points(joints_glob[:, 0], verts_glob)
+ renderer_glob.set_ground(scale * 1.5, cx, cz)
+ color = torch.ones(3).float().cuda() * 0.8
+
+ writer = get_writer(out_fn, fps=30, crf=23)
+ for i in tqdm(range(seq_length), desc=f"Rendering {vname}"):
+ # incam
+ img_overlay_pred = renderer_incam.render_mesh(verts_incam[i].cuda(), images[i], [0.8, 0.8, 0.8])
+ # if batch["meta_render"][0].get("bbx_xys", None) is not None: # draw bbox lines
+ # bbx_xys = batch["meta_render"][0]["bbx_xys"][i].cpu().numpy()
+ # lu_point = (bbx_xys[:2] - bbx_xys[2:] / 2).astype(int)
+ # rd_point = (bbx_xys[:2] + bbx_xys[2:] / 2).astype(int)
+ # img_overlay_pred = cv2.rectangle(img_overlay_pred, lu_point, rd_point, (255, 178, 102), 2)
+ # pred_mpjpe_ = self.metric_aggregator["mpjpe"][vid][i]
+ # text = f"pred mpjpe: {pred_mpjpe_:.1f}"
+ # cv2.putText(img_overlay_pred, text, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1, (200, 100, 200), 2)
+
+ # glob
+ cameras = renderer_glob.create_camera(global_R[i], global_T[i])
+ # img_glob = renderer_glob.render_with_ground(verts_glob[[i]], color_[None], cameras, global_lights)
+ img_glob = renderer_glob.render_with_ground(
+ verts_glob[[i]], color.clone()[None], cameras, global_lights
+ )
+
+ # write
+ img = np.concatenate([img_overlay_pred, img_glob], axis=1)
+ writer.write_frame(img)
+ writer.close()
+
+ # ================== Epoch Summary ================== #
+ def on_predict_epoch_end(self, trainer, pl_module):
+ """Without logger"""
+ local_rank, world_size = trainer.local_rank, trainer.world_size
+ monitor_metric = "mpjpe"
+
+ # Reduce metric_aggregator across all processes
+ metric_keys = list(self.metric_aggregator.keys())
+ with torch.inference_mode(False): # allow in-place operation of all_gather
+ metric_aggregator_gathered = all_gather(self.metric_aggregator) # list of dict
+ for metric_key in metric_keys:
+ for d in metric_aggregator_gathered:
+ self.metric_aggregator[metric_key].update(d[metric_key])
+
+ if False: # debug to make sure the all_gather is correct
+ print(f"[RANK {local_rank}/{world_size}]: {self.metric_aggregator[monitor_metric].keys()}")
+
+ total = len(self.metric_aggregator[monitor_metric])
+ Log.info(f"{total} sequences evaluated in {self.__class__.__name__}")
+ if total == 0:
+ return
+
+ # print monitored metric per sequence
+ mm_per_seq = {k: v.mean() for k, v in self.metric_aggregator[monitor_metric].items()}
+ if len(mm_per_seq) > 0:
+ sorted_mm_per_seq = sorted(mm_per_seq.items(), key=lambda x: x[1], reverse=True)
+ n_worst = 5 if trainer.state.stage == "validate" else len(sorted_mm_per_seq)
+ if local_rank == 0:
+ Log.info(
+ f"monitored metric {monitor_metric} per sequence\n"
+ + "\n".join([f"{m:5.1f} : {s}" for s, m in sorted_mm_per_seq[:n_worst]])
+ + "\n------"
+ )
+
+ # average over all batches
+ metrics_avg = {k: np.concatenate(list(v.values())).mean() for k, v in self.metric_aggregator.items()}
+ if local_rank == 0:
+ Log.info(f"[Metrics] RICH:\n" + "\n".join(f"{k}: {v:.1f}" for k, v in metrics_avg.items()) + "\n------")
+
+ # save to logger if available
+ if pl_module.logger is not None:
+ cur_epoch = pl_module.current_epoch
+ for k, v in metrics_avg.items():
+ pl_module.logger.log_metrics({f"val_metric_RICH/{k}": v}, step=cur_epoch)
+
+ # reset
+ for k in self.metric_aggregator:
+ self.metric_aggregator[k] = {}
+
+
+rich_node = builds(MetricMocap)
+MainStore.store(name="metric_rich", node=rich_node, group="callbacks", package="callbacks.metric_rich")
diff --git a/third_party/GVHMR/hmr4d/model/gvhmr/gvhmr_pl.py b/third_party/GVHMR/hmr4d/model/gvhmr/gvhmr_pl.py
new file mode 100644
index 0000000000000000000000000000000000000000..af9d3a2b1434cdf98c660bc387f9e8a6f22200ad
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/gvhmr/gvhmr_pl.py
@@ -0,0 +1,324 @@
+from typing import Any, Dict
+import numpy as np
+from pathlib import Path
+import torch
+import pytorch_lightning as pl
+from hydra.utils import instantiate
+from hmr4d.utils.pylogger import Log
+from einops import rearrange, einsum
+from hmr4d.configs import MainStore, builds
+
+from hmr4d.utils.geo_transform import compute_T_ayfz2ay, apply_T_on_points
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines
+from hmr4d.utils.smplx_utils import make_smplx
+from hmr4d.utils.geo.augment_noisy_pose import (
+ get_wham_aug_kp3d,
+ get_visible_mask,
+ get_invisible_legs_mask,
+ randomly_occlude_lower_half,
+ randomly_modify_hands_legs,
+)
+from hmr4d.utils.geo.hmr_cam import perspective_projection, normalize_kp2d, safely_render_x3d_K, get_bbx_xys
+
+from hmr4d.utils.video_io_utils import save_video
+from hmr4d.utils.vis.cv2_utils import draw_bbx_xys_on_image_batch
+from hmr4d.utils.geo.flip_utils import flip_smplx_params, avg_smplx_aa
+from hmr4d.model.gvhmr.utils.postprocess import pp_static_joint, pp_static_joint_cam, process_ik
+
+
+class GvhmrPL(pl.LightningModule):
+ def __init__(
+ self,
+ pipeline,
+ optimizer=None,
+ scheduler_cfg=None,
+ ignored_weights_prefix=["smplx", "pipeline.endecoder"],
+ ):
+ super().__init__()
+ self.pipeline = instantiate(pipeline, _recursive_=False)
+ self.optimizer = instantiate(optimizer)
+ self.scheduler_cfg = scheduler_cfg
+
+ # Options
+ self.ignored_weights_prefix = ignored_weights_prefix
+
+ # The test step is the same as validation
+ self.test_step = self.predict_step = self.validation_step
+
+ # SMPLX
+ self.smplx = make_smplx("supermotion_v437coco17")
+
+ def training_step(self, batch, batch_idx):
+ B, F = batch["smpl_params_c"]["body_pose"].shape[:2]
+
+ # Create augmented noisy-obs : gt_j3d(coco17)
+ with torch.no_grad():
+ gt_verts437, gt_j3d = self.smplx(**batch["smpl_params_c"])
+ root_ = gt_j3d[:, :, [11, 12], :].mean(-2, keepdim=True)
+ batch["gt_j3d"] = gt_j3d
+ batch["gt_cr_coco17"] = gt_j3d - root_
+ batch["gt_c_verts437"] = gt_verts437
+ batch["gt_cr_verts437"] = gt_verts437 - root_
+
+ # bbx_xys
+ i_x2d = safely_render_x3d_K(gt_verts437, batch["K_fullimg"], thr=0.3)
+ bbx_xys = get_bbx_xys(i_x2d, do_augment=True)
+ if False: # trust image bbx_xys seems better
+ batch["bbx_xys"] = bbx_xys
+ else:
+ mask_bbx_xys = batch["mask"]["bbx_xys"]
+ batch["bbx_xys"][~mask_bbx_xys] = bbx_xys[~mask_bbx_xys]
+ if False: # visualize bbx_xys from an iPhone view
+ render_w, render_h = 120, 160 # iphone main-lens 24mm 3:4
+ ratio = render_w / 1528
+ offset = torch.tensor([764 - 500, 1019 - 500]).to(i_x2d)
+ i_x2d_render = (i_x2d + offset).clone()
+ i_x2d_render = (i_x2d_render * ratio).long().clone()
+ torch.clamp_(i_x2d_render[..., 0], 0, render_w - 1)
+ torch.clamp_(i_x2d_render[..., 1], 0, render_h - 1)
+ bbx_xys_render = bbx_xys.clone()
+ bbx_xys_render[..., :2] += offset
+ bbx_xys_render *= ratio
+
+ output_dir = Path("outputs/simulated_bbx_xys")
+ output_dir.mkdir(parents=True, exist_ok=True)
+ video_list = []
+ for bid in range(B):
+ images = torch.zeros(F, render_h, render_w, 3, device=i_x2d.device)
+ for fid in range(F):
+ images[fid, i_x2d_render[bid, fid, :, 1], i_x2d_render[bid, fid, :, 0]] = 255
+
+ images = draw_bbx_xys_on_image_batch(bbx_xys_render[bid].cpu().numpy(), images.cpu().numpy())
+ images = np.stack(images).astype("uint8") # (L, H, W, 3)
+ images[:, 0, :] = np.array([255, 255, 255])
+ images[:, -1, :] = np.array([255, 255, 255])
+ images[:, :, 0] = np.array([255, 255, 255])
+ images[:, :, -1] = np.array([255, 255, 255])
+ video_list.append(images)
+
+ # stack videos
+ video_output = []
+ for i in range(0, len(video_list), 4):
+ if i + 4 <= len(video_list):
+ video_output.append(np.concatenate(video_list[i : i + 4], axis=2))
+ video_output = np.concatenate(video_output, axis=1)
+ save_video(video_output, output_dir / f"{batch_idx}.mp4", fps=30, quality=5)
+
+ # noisy_j3d -> project to i_j2d -> compute a bbx -> normalized kp2d [-1, 1]
+ noisy_j3d = gt_j3d + get_wham_aug_kp3d(gt_j3d.shape[:2])
+ if True:
+ noisy_j3d = randomly_modify_hands_legs(noisy_j3d)
+ obs_i_j2d = perspective_projection(noisy_j3d, batch["K_fullimg"]) # (B, L, J, 2)
+ j2d_visible_mask = get_visible_mask(gt_j3d.shape[:2]).cuda() # (B, L, J)
+ j2d_visible_mask[noisy_j3d[..., 2] < 0.3] = False # Set close-to-image-plane points as invisible
+ if True: # Set both legs as invisible for a period
+ legs_invisible_mask = get_invisible_legs_mask(gt_j3d.shape[:2]).cuda() # (B, L, J)
+ j2d_visible_mask[legs_invisible_mask] = False
+ obs_kp2d = torch.cat([obs_i_j2d, j2d_visible_mask[:, :, :, None].float()], dim=-1) # (B, L, J, 3)
+ obs = normalize_kp2d(obs_kp2d, batch["bbx_xys"]) # (B, L, J, 3)
+ obs[~j2d_visible_mask] = 0 # if not visible, set to (0,0,0)
+ batch["obs"] = obs
+
+ if True: # Use some detected vitpose (presave data)
+ prob = 0.5
+ mask_real_vitpose = (torch.rand(B).to(obs_kp2d) < prob) * batch["mask"]["vitpose"]
+ batch["obs"][mask_real_vitpose] = normalize_kp2d(batch["kp2d"], batch["bbx_xys"])[mask_real_vitpose]
+
+ # Set untrusted frames to False
+ batch["obs"][~batch["mask"]["valid"]] = 0
+
+ if False: # wis3d
+ wis3d = make_wis3d(name="debug-aug-kp3d")
+ add_motion_as_lines(gt_j3d[0], wis3d, name="gt_j3d", skeleton_type="coco17")
+ add_motion_as_lines(noisy_j3d[0], wis3d, name="noisy_j3d", skeleton_type="coco17")
+
+ # f_imgseq: apply random aug on offline extracted features
+ # f_imgseq = batch["f_imgseq"] + torch.randn_like(batch["f_imgseq"]) * 0.1
+ # f_imgseq[~batch["mask"]["f_imgseq"]] = 0
+ # batch["f_imgseq"] = f_imgseq.clone()
+
+ # Forward and get loss
+ outputs = self.pipeline.forward(batch, train=True)
+
+ # Log
+ log_kwargs = {
+ "on_epoch": True,
+ "prog_bar": True,
+ "logger": True,
+ "batch_size": B,
+ "sync_dist": True,
+ }
+ self.log("train/loss", outputs["loss"], **log_kwargs)
+ for k, v in outputs.items():
+ if "_loss" in k:
+ self.log(f"train/{k}", v, **log_kwargs)
+
+ return outputs
+
+ def validation_step(self, batch, batch_idx, dataloader_idx=0):
+ # Options & Check
+ do_postproc = self.trainer.state.stage == "test" # Only apply postproc in test
+ do_flip_test = "flip_test" in batch
+ do_postproc_not_flip_test = do_postproc and not do_flip_test # later pp when flip_test
+ assert batch["B"] == 1, "Only support batch size 1 in evalution."
+
+ # ROPE inference
+ obs = normalize_kp2d(batch["kp2d"], batch["bbx_xys"])
+ if "mask" in batch:
+ obs[0, ~batch["mask"][0]] = 0
+
+ batch_ = {
+ "length": batch["length"],
+ "obs": obs,
+ "bbx_xys": batch["bbx_xys"],
+ "K_fullimg": batch["K_fullimg"],
+ "cam_angvel": batch["cam_angvel"],
+ "f_imgseq": batch["f_imgseq"],
+ }
+ outputs = self.pipeline.forward(batch_, train=False, postproc=do_postproc_not_flip_test)
+ outputs["pred_smpl_params_global"] = {k: v[0] for k, v in outputs["pred_smpl_params_global"].items()}
+ outputs["pred_smpl_params_incam"] = {k: v[0] for k, v in outputs["pred_smpl_params_incam"].items()}
+
+ if do_flip_test:
+ flip_test = batch["flip_test"]
+ obs = normalize_kp2d(flip_test["kp2d"], flip_test["bbx_xys"])
+ if "mask" in batch:
+ obs[0, ~batch["mask"][0]] = 0
+
+ batch_ = {
+ "length": batch["length"],
+ "obs": obs,
+ "bbx_xys": flip_test["bbx_xys"],
+ "K_fullimg": batch["K_fullimg"],
+ "cam_angvel": flip_test["cam_angvel"],
+ "f_imgseq": flip_test["f_imgseq"],
+ }
+ flipped_outputs = self.pipeline.forward(batch_, train=False)
+
+ # First update incam results
+ flipped_outputs["pred_smpl_params_incam"] = {
+ k: v[0] for k, v in flipped_outputs["pred_smpl_params_incam"].items()
+ }
+ smpl_params1 = outputs["pred_smpl_params_incam"]
+ smpl_params2 = flip_smplx_params(flipped_outputs["pred_smpl_params_incam"])
+
+ smpl_params_avg = smpl_params1.copy()
+ smpl_params_avg["betas"] = (smpl_params1["betas"] + smpl_params2["betas"]) / 2
+ smpl_params_avg["body_pose"] = avg_smplx_aa(smpl_params1["body_pose"], smpl_params2["body_pose"])
+ smpl_params_avg["global_orient"] = avg_smplx_aa(
+ smpl_params1["global_orient"], smpl_params2["global_orient"]
+ )
+ outputs["pred_smpl_params_incam"] = smpl_params_avg
+
+ # Then update global results
+ outputs["pred_smpl_params_global"]["betas"] = smpl_params_avg["betas"]
+ outputs["pred_smpl_params_global"]["body_pose"] = smpl_params_avg["body_pose"]
+
+ # Finally, apply postprocess
+ if do_postproc:
+ # temporarily recover the original batch-dim
+ outputs["pred_smpl_params_global"] = {k: v[None] for k, v in outputs["pred_smpl_params_global"].items()}
+ outputs["pred_smpl_params_global"]["transl"] = pp_static_joint(outputs, self.pipeline.endecoder)
+ body_pose = process_ik(outputs, self.pipeline.endecoder)
+ outputs["pred_smpl_params_global"] = {k: v[0] for k, v in outputs["pred_smpl_params_global"].items()}
+
+ outputs["pred_smpl_params_global"]["body_pose"] = body_pose[0]
+ # outputs["pred_smpl_params_incam"]["body_pose"] = body_pose[0]
+
+ if False: # wis3d
+ wis3d = make_wis3d(name="debug-rich-cap")
+ smplx_model = make_smplx("rich-smplx", gender="neutral").cuda()
+ gender = batch["gender"][0]
+ T_w2ay = batch["T_w2ay"][0]
+
+ # Prediction
+ # add_motion_as_lines(outputs_window["pred_ayfz_motion"][bid], wis3d, name="pred_ayfz_motion")
+
+ smplx_out = smplx_model(**pred_smpl_params_global)
+ for i in range(len(smplx_out.vertices)):
+ wis3d.set_scene_id(i)
+ wis3d.add_mesh(smplx_out.vertices[i], smplx_model.bm.faces, name=f"pred-smplx-global")
+
+ # GT (w)
+ smplx_models = {
+ "male": make_smplx("rich-smplx", gender="male").cuda(),
+ "female": make_smplx("rich-smplx", gender="female").cuda(),
+ }
+ gt_smpl_params = {k: v[0, windows[0]] for k, v in batch["gt_smpl_params"].items()}
+ gt_smplx_out = smplx_models[gender](**gt_smpl_params)
+
+ # GT (ayfz)
+ smplx_verts_ay = apply_T_on_points(gt_smplx_out.vertices, T_w2ay)
+ smplx_joints_ay = apply_T_on_points(gt_smplx_out.joints, T_w2ay)
+ T_ay2ayfz = compute_T_ayfz2ay(smplx_joints_ay[:1], inverse=True)[0] # (4, 4)
+ smplx_verts_ayfz = apply_T_on_points(smplx_verts_ay, T_ay2ayfz) # (F, 22, 3)
+
+ for i in range(len(smplx_verts_ayfz)):
+ wis3d.set_scene_id(i)
+ wis3d.add_mesh(smplx_verts_ayfz[i], smplx_models[gender].bm.faces, name=f"gt-smplx-ayfz")
+
+ breakpoint()
+
+ if False: # o3d
+ prog_keys = [
+ "pred_smpl_progress",
+ "pred_localjoints_progress",
+ "pred_incam_localjoints_progress",
+ ]
+ for k in prog_keys:
+ if k in outputs_window:
+ seq_out = torch.cat(
+ [v[:, :l] for v, l in zip(outputs_window[k], length)], dim=1
+ ) # (B, P, L, J, 3) -> (P, L, J, 3) -> (P, CL, J, 3)
+ outputs[k] = seq_out[None]
+
+ return outputs
+
+ def configure_optimizers(self):
+ params = []
+ for k, v in self.pipeline.named_parameters():
+ if v.requires_grad:
+ params.append(v)
+ optimizer = self.optimizer(params=params)
+
+ if self.scheduler_cfg["scheduler"] is None:
+ return optimizer
+
+ scheduler_cfg = dict(self.scheduler_cfg)
+ scheduler_cfg["scheduler"] = instantiate(scheduler_cfg["scheduler"], optimizer=optimizer)
+ return [optimizer], [scheduler_cfg]
+
+ # ============== Utils ================= #
+ def on_save_checkpoint(self, checkpoint) -> None:
+ for ig_keys in self.ignored_weights_prefix:
+ for k in list(checkpoint["state_dict"].keys()):
+ if k.startswith(ig_keys):
+ # Log.info(f"Remove key `{ig_keys}' from checkpoint.")
+ checkpoint["state_dict"].pop(k)
+
+ def load_pretrained_model(self, ckpt_path):
+ """Load pretrained checkpoint, and assign each weight to the corresponding part."""
+ Log.info(f"[PL-Trainer] Loading ckpt: {ckpt_path}")
+
+ state_dict = torch.load(ckpt_path, "cpu")["state_dict"]
+ missing, unexpected = self.load_state_dict(state_dict, strict=False)
+ real_missing = []
+ for k in missing:
+ ignored_when_saving = any(k.startswith(ig_keys) for ig_keys in self.ignored_weights_prefix)
+ if not ignored_when_saving:
+ real_missing.append(k)
+
+ if len(real_missing) > 0:
+ Log.warn(f"Missing keys: {real_missing}")
+ if len(unexpected) > 0:
+ Log.warn(f"Unexpected keys: {unexpected}")
+
+
+gvhmr_pl = builds(
+ GvhmrPL,
+ pipeline="${pipeline}",
+ optimizer="${optimizer}",
+ scheduler_cfg="${scheduler_cfg}",
+ populate_full_signature=True, # Adds all the arguments to the signature
+)
+MainStore.store(name="gvhmr_pl", node=gvhmr_pl, group="model/gvhmr")
diff --git a/third_party/GVHMR/hmr4d/model/gvhmr/gvhmr_pl_demo.py b/third_party/GVHMR/hmr4d/model/gvhmr/gvhmr_pl_demo.py
new file mode 100644
index 0000000000000000000000000000000000000000..95bec101b3e9be163224ff6d130d7310586721ca
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/gvhmr/gvhmr_pl_demo.py
@@ -0,0 +1,60 @@
+import torch
+import pytorch_lightning as pl
+from hydra.utils import instantiate
+from hmr4d.utils.pylogger import Log
+from hmr4d.configs import MainStore, builds
+
+from hmr4d.utils.geo.hmr_cam import normalize_kp2d
+
+
+class DemoPL(pl.LightningModule):
+ def __init__(self, pipeline):
+ super().__init__()
+ self.pipeline = instantiate(pipeline, _recursive_=False)
+
+ @torch.no_grad()
+ def predict(self, data, static_cam=False):
+ """auto add batch dim
+ data: {
+ "length": int, or Torch.Tensor,
+ "kp2d": (F, 3)
+ "bbx_xys": (F, 3)
+ "K_fullimg": (F, 3, 3)
+ "cam_angvel": (F, 3)
+ "f_imgseq": (F, 3, 256, 256)
+ }
+
+ """
+ # ROPE inference
+ batch = {
+ "length": data["length"][None],
+ "obs": normalize_kp2d(data["kp2d"], data["bbx_xys"])[None],
+ "bbx_xys": data["bbx_xys"][None],
+ "K_fullimg": data["K_fullimg"][None],
+ "cam_angvel": data["cam_angvel"][None],
+ "f_imgseq": data["f_imgseq"][None],
+ }
+ batch = {k: v.cuda() for k, v in batch.items()}
+ outputs = self.pipeline.forward(batch, train=False, postproc=False, static_cam=static_cam)
+
+ pred = {
+ "smpl_params_global": {k: v[0] for k, v in outputs["pred_smpl_params_global"].items()},
+ "smpl_params_incam": {k: v[0] for k, v in outputs["pred_smpl_params_incam"].items()},
+ "K_fullimg": data["K_fullimg"],
+ "net_outputs": outputs, # intermediate outputs
+ }
+ return pred
+
+ def load_pretrained_model(self, ckpt_path):
+ """Load pretrained checkpoint, and assign each weight to the corresponding part."""
+ Log.info(f"[PL-Trainer] Loading ckpt type: {ckpt_path}")
+
+ state_dict = torch.load(ckpt_path, "cpu")["state_dict"]
+ missing, unexpected = self.load_state_dict(state_dict, strict=False)
+ if len(missing) > 0:
+ Log.warn(f"Missing keys: {missing}")
+ if len(unexpected) > 0:
+ Log.warn(f"Unexpected keys: {unexpected}")
+
+
+MainStore.store(name="gvhmr_pl_demo", node=builds(DemoPL, pipeline="${pipeline}"), group="model/gvhmr")
diff --git a/third_party/GVHMR/hmr4d/model/gvhmr/pipeline/gvhmr_pipeline.py b/third_party/GVHMR/hmr4d/model/gvhmr/pipeline/gvhmr_pipeline.py
new file mode 100644
index 0000000000000000000000000000000000000000..9e99b22016f834f17e61cfc6d49f42d7ad3fb1b7
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/gvhmr/pipeline/gvhmr_pipeline.py
@@ -0,0 +1,384 @@
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from torch.cuda.amp import autocast
+import numpy as np
+from einops import einsum, rearrange, repeat
+from hydra.utils import instantiate
+from hmr4d.utils.pylogger import Log
+from hmr4d.utils.net_utils import gaussian_smooth
+
+from hmr4d.model.gvhmr.utils.endecoder import EnDecoder
+from hmr4d.model.gvhmr.utils.postprocess import (
+ pp_static_joint,
+ process_ik,
+ pp_static_joint_cam,
+)
+from hmr4d.model.gvhmr.utils import stats_compose
+
+from pytorch3d.transforms import (
+ matrix_to_rotation_6d,
+ rotation_6d_to_matrix,
+ axis_angle_to_matrix,
+ matrix_to_axis_angle,
+)
+from hmr4d.utils.geo.hmr_cam import compute_bbox_info_bedlam, compute_transl_full_cam, get_a_pred_cam, project_to_bi01
+from hmr4d.utils.geo.hmr_global import (
+ rollout_local_transl_vel,
+ get_static_joint_mask,
+ get_tgtcoord_rootparam,
+)
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines
+from hmr4d.utils.smplx_utils import make_smplx
+
+
+class Pipeline(nn.Module):
+ def __init__(self, args, args_denoiser3d, **kwargs):
+ super().__init__()
+ self.args = args
+ self.weights = args.weights # loss weights
+
+ # Networks
+ self.denoiser3d = instantiate(args_denoiser3d, _recursive_=False)
+ # Log.info(self.denoiser3d)
+
+ # Normalizer
+ self.endecoder: EnDecoder = instantiate(args.endecoder_opt, _recursive_=False)
+ if self.args.normalize_cam_angvel:
+ cam_angvel_stats = stats_compose.cam_angvel["manual"]
+ self.register_buffer("cam_angvel_mean", torch.tensor(cam_angvel_stats["mean"]), persistent=False)
+ self.register_buffer("cam_angvel_std", torch.tensor(cam_angvel_stats["std"]), persistent=False)
+
+ # ========== Training ========== #
+
+ def forward(self, inputs, train=False, postproc=False, static_cam=False):
+ outputs = dict()
+ length = inputs["length"] # (B,) effective length of each sample
+
+ # *. Conditions
+ cliff_cam = compute_bbox_info_bedlam(inputs["bbx_xys"], inputs["K_fullimg"]) # (B, L, 3)
+ f_cam_angvel = inputs["cam_angvel"]
+ if self.args.normalize_cam_angvel:
+ f_cam_angvel = (f_cam_angvel - self.cam_angvel_mean) / self.cam_angvel_std
+ f_condition = {
+ "obs": inputs["obs"], # (B, L, J, 3)
+ "f_cliffcam": cliff_cam, # (B, L, 3)
+ "f_cam_angvel": f_cam_angvel, # (B, L, C=6)
+ "f_imgseq": inputs["f_imgseq"], # (B, L, C=1024)
+ }
+ if train:
+ f_condition = randomly_set_null_condition(f_condition, 0.1)
+
+ # Forward & output
+ model_output = self.denoiser3d(length=length, **f_condition) # pred_x, pred_cam, static_conf_logits
+ decode_dict = self.endecoder.decode(model_output["pred_x"]) # (B, L, C) -> dict
+ outputs.update({"model_output": model_output, "decode_dict": decode_dict})
+
+ # Post-processing
+ outputs["pred_smpl_params_incam"] = {
+ "body_pose": decode_dict["body_pose"], # (B, L, 63)
+ "betas": decode_dict["betas"], # (B, L, 10)
+ "global_orient": decode_dict["global_orient"], # (B, L, 3)
+ "transl": compute_transl_full_cam(model_output["pred_cam"], inputs["bbx_xys"], inputs["K_fullimg"]),
+ }
+ if not train:
+ pred_smpl_params_global = get_smpl_params_w_Rt_v2( # This function has for-loop
+ global_orient_gv=decode_dict["global_orient_gv"],
+ local_transl_vel=decode_dict["local_transl_vel"],
+ global_orient_c=decode_dict["global_orient"],
+ cam_angvel=inputs["cam_angvel"],
+ )
+ outputs["pred_smpl_params_global"] = {
+ "body_pose": decode_dict["body_pose"],
+ "betas": decode_dict["betas"],
+ **pred_smpl_params_global,
+ }
+ outputs["static_conf_logits"] = model_output["static_conf_logits"]
+
+ if postproc: # apply post-processing
+ if static_cam: # extra post-processing to utilize static camera prior
+ outputs["pred_smpl_params_global"]["transl"] = pp_static_joint_cam(outputs, self.endecoder)
+ else:
+ outputs["pred_smpl_params_global"]["transl"] = pp_static_joint(outputs, self.endecoder)
+ body_pose = process_ik(outputs, self.endecoder)
+ decode_dict["body_pose"] = body_pose
+ outputs["pred_smpl_params_global"]["body_pose"] = body_pose
+ outputs["pred_smpl_params_incam"]["body_pose"] = body_pose
+
+ return outputs
+
+ # ========== Compute Loss ========== #
+ total_loss = 0
+ mask = inputs["mask"]["valid"] # (B, L)
+
+ # 1. Simple loss: MSE
+ pred_x = model_output["pred_x"] # (B, L, C)
+ target_x = self.endecoder.encode(inputs) # (B, L, C)
+ simple_loss = F.mse_loss(pred_x, target_x, reduction="none")
+ mask_simple = mask[:, :, None].expand(-1, -1, pred_x.size(2)).clone() # (B, L, C)
+ mask_simple[inputs["mask"]["spv_incam_only"], :, 142:] = False # 3dpw training
+ simple_loss = (simple_loss * mask_simple).mean()
+ total_loss += simple_loss
+ outputs["simple_loss"] = simple_loss
+
+ # 2. Extra loss
+ extra_funcs = [
+ compute_extra_incam_loss,
+ compute_extra_global_loss,
+ ]
+ for extra_func in extra_funcs:
+ extra_loss, extra_loss_dict = extra_func(inputs, outputs, self)
+ total_loss += extra_loss
+ outputs.update(extra_loss_dict)
+
+ outputs["loss"] = total_loss
+ return outputs
+
+
+def randomly_set_null_condition(f_condition, uncond_prob=0.1):
+ """Conditions are in shape (B, L, *)"""
+ keys = list(f_condition.keys())
+ for k in keys:
+ if f_condition[k] is None:
+ continue
+ f_condition[k] = f_condition[k].clone()
+ mask = torch.rand(f_condition[k].shape[:2]) < uncond_prob
+ f_condition[k][mask] = 0.0
+ return f_condition
+
+
+def compute_extra_incam_loss(inputs, outputs, ppl):
+ model_output = outputs["model_output"]
+ decode_dict = outputs["decode_dict"]
+ endecoder = ppl.endecoder
+ weights = ppl.weights
+ args = ppl.args
+
+ extra_loss_dict = {}
+ extra_loss = 0
+ mask = inputs["mask"]["valid"] # effective length mask
+ mask_reproj = ~inputs["mask"]["spv_incam_only"] # do not supervise reproj for 3DPW
+
+ # Incam FK
+ # prediction
+ pred_c_j3d = endecoder.fk_v2(**outputs["pred_smpl_params_incam"])
+ pred_cr_j3d = pred_c_j3d - pred_c_j3d[:, :, :1] # (B, L, J, 3)
+
+ # gt
+ gt_c_j3d = endecoder.fk_v2(**inputs["smpl_params_c"]) # (B, L, J, 3)
+ gt_cr_j3d = gt_c_j3d - gt_c_j3d[:, :, :1] # (B, L, J, 3)
+
+ # Root aligned C-MPJPE Loss
+ if weights.cr_j3d > 0.0:
+ cr_j3d_loss = F.mse_loss(pred_cr_j3d, gt_cr_j3d, reduction="none")
+ cr_j3d_loss = (cr_j3d_loss * mask[..., None, None]).mean()
+ extra_loss += cr_j3d_loss * weights.cr_j3d
+ extra_loss_dict["cr_j3d_loss"] = cr_j3d_loss
+
+ # Reprojection (to align with image)
+ if weights.transl_c > 0.0:
+ # pred_transl = decode_dict["transl"] # (B, L, 3)
+ # gt_transl = inputs["smpl_params_c"]["transl"]
+ # transl_c_loss = F.l1_loss(pred_transl, gt_transl, reduction="none")
+ # transl_c_loss = (transl_c_loss * mask[..., None]).mean()
+
+ # Instead of supervising transl, we convert gt to pred_cam (prevent divide 0)
+ pred_cam = model_output["pred_cam"] # (B, L, 3)
+ gt_transl = inputs["smpl_params_c"]["transl"] # (B, L, 3)
+ gt_pred_cam = get_a_pred_cam(gt_transl, inputs["bbx_xys"], inputs["K_fullimg"]) # (B, L, 3)
+ gt_pred_cam[gt_pred_cam.isinf()] = -1 # this will be handled by valid_mask
+ # (compute_transl_full_cam(gt_pred_cam, inputs["bbx_xys"], inputs["K_fullimg"]) - gt_transl).abs().max()
+
+ # Skip gts that are not good during random construction
+ gt_j3d_z_min = inputs["gt_j3d"][..., 2].min(dim=-1)[0]
+ valid_mask = (
+ (gt_j3d_z_min > 0.3)
+ * (gt_pred_cam[..., 0] > 0.3)
+ * (gt_pred_cam[..., 0] < 5.0)
+ * (gt_pred_cam[..., 1] > -3.0)
+ * (gt_pred_cam[..., 1] < 3.0)
+ * (gt_pred_cam[..., 2] > -3.0)
+ * (gt_pred_cam[..., 2] < 3.0)
+ * (inputs["bbx_xys"][..., 2] > 0)
+ )[..., None]
+ transl_c_loss = F.mse_loss(pred_cam, gt_pred_cam, reduction="none")
+ transl_c_loss = (transl_c_loss * mask[..., None] * valid_mask).mean()
+
+ extra_loss_dict["transl_c_loss"] = transl_c_loss
+ extra_loss += transl_c_loss * weights.transl_c
+
+ if weights.j2d > 0.0:
+ # prevent divide 0 or small value to overflow(fp16)
+ reproj_z_thr = 0.3
+ pred_c_j3d_z0_mask = pred_c_j3d[..., 2].abs() <= reproj_z_thr
+ pred_c_j3d[pred_c_j3d_z0_mask] = reproj_z_thr
+ gt_c_j3d_z0_mask = gt_c_j3d[..., 2].abs() <= reproj_z_thr
+ gt_c_j3d[gt_c_j3d_z0_mask] = reproj_z_thr
+
+ pred_j2d_01 = project_to_bi01(pred_c_j3d, inputs["bbx_xys"], inputs["K_fullimg"])
+ gt_j2d_01 = project_to_bi01(gt_c_j3d, inputs["bbx_xys"], inputs["K_fullimg"]) # (B, L, J, 2)
+
+ valid_mask = (
+ (gt_c_j3d[..., 2] > reproj_z_thr)
+ * (pred_c_j3d[..., 2] > reproj_z_thr) # Be safe
+ * (gt_j2d_01[..., 0] > 0.0)
+ * (gt_j2d_01[..., 0] < 1.0)
+ * (gt_j2d_01[..., 1] > 0.0)
+ * (gt_j2d_01[..., 1] < 1.0)
+ )[..., None]
+ valid_mask[~mask_reproj] = False # Do not supervise on 3dpw
+ j2d_loss = F.mse_loss(pred_j2d_01, gt_j2d_01, reduction="none")
+ j2d_loss = (j2d_loss * mask[..., None, None] * valid_mask).mean()
+
+ extra_loss += j2d_loss * weights.j2d
+ extra_loss_dict["j2d_loss"] = j2d_loss
+
+ if weights.cr_verts > 0:
+ # SMPL forward
+ pred_c_verts437, pred_c_j17 = endecoder.smplx_model(**outputs["pred_smpl_params_incam"])
+ root_ = pred_c_j17[:, :, [11, 12], :].mean(-2, keepdim=True)
+ pred_cr_verts437 = pred_c_verts437 - root_
+
+ gt_cr_verts437 = inputs["gt_cr_verts437"] # (B, L, 437, 3)
+ cr_vert_loss = F.mse_loss(pred_cr_verts437, gt_cr_verts437, reduction="none")
+ cr_vert_loss = (cr_vert_loss * mask[:, :, None, None]).mean()
+ extra_loss += cr_vert_loss * weights.cr_verts
+ extra_loss_dict["cr_vert_loss"] = cr_vert_loss
+
+ if weights.verts2d > 0:
+ gt_c_verts437 = inputs["gt_c_verts437"] # (B, L, 437, 3)
+
+ # prevent divide 0 or small value to overflow(fp16)
+ reproj_z_thr = 0.3
+ pred_c_verts437_z0_mask = pred_c_verts437[..., 2].abs() <= reproj_z_thr
+ pred_c_verts437[pred_c_verts437_z0_mask] = reproj_z_thr
+ gt_c_verts437_z0_mask = gt_c_verts437[..., 2].abs() <= reproj_z_thr
+ gt_c_verts437[gt_c_verts437_z0_mask] = reproj_z_thr
+
+ pred_verts2d_01 = project_to_bi01(pred_c_verts437, inputs["bbx_xys"], inputs["K_fullimg"])
+ gt_verts2d_01 = project_to_bi01(gt_c_verts437, inputs["bbx_xys"], inputs["K_fullimg"]) # (B, L, 437, 2)
+
+ valid_mask = (
+ (gt_c_verts437[..., 2] > reproj_z_thr)
+ * (pred_c_verts437[..., 2] > reproj_z_thr) # Be safe
+ * (gt_verts2d_01[..., 0] > 0.0)
+ * (gt_verts2d_01[..., 0] < 1.0)
+ * (gt_verts2d_01[..., 1] > 0.0)
+ * (gt_verts2d_01[..., 1] < 1.0)
+ )[..., None]
+ valid_mask[~mask_reproj] = False # Do not supervise on 3dpw
+ verts2d_loss = F.mse_loss(pred_verts2d_01, gt_verts2d_01, reduction="none")
+ verts2d_loss = (verts2d_loss * mask[..., None, None] * valid_mask).mean()
+
+ extra_loss += verts2d_loss * weights.verts2d
+ extra_loss_dict["verts2d_loss"] = verts2d_loss
+
+ return extra_loss, extra_loss_dict
+
+
+def compute_extra_global_loss(inputs, outputs, ppl):
+ decode_dict = outputs["decode_dict"]
+ endecoder = ppl.endecoder
+ weights = ppl.weights
+ args = ppl.args
+
+ extra_loss_dict = {}
+ extra_loss = 0
+ mask = inputs["mask"]["valid"].clone() # (B, L)
+ mask[inputs["mask"]["spv_incam_only"]] = False
+
+ if weights.transl_w > 0:
+ # compute pred_transl_w by rollout
+ gt_transl_w = inputs["smpl_params_w"]["transl"]
+ gt_global_orient_w = inputs["smpl_params_w"]["global_orient"]
+ local_transl_vel = decode_dict["local_transl_vel"]
+ pred_transl_w = rollout_local_transl_vel(local_transl_vel, gt_global_orient_w, gt_transl_w[:, [0]])
+
+ trans_w_loss = F.l1_loss(pred_transl_w, gt_transl_w, reduction="none")
+ trans_w_loss = (trans_w_loss * mask[..., None]).mean()
+ extra_loss += trans_w_loss * weights.transl_w
+ extra_loss_dict["transl_w_loss"] = trans_w_loss
+
+ # Static-Conf loss
+ if weights.static_conf_bce > 0:
+ # Compute gt by thresholding velocity
+ vel_thr = args.static_conf.vel_thr
+ assert vel_thr > 0
+ joint_ids = [7, 10, 8, 11, 20, 21] # [L_Ankle, L_foot, R_Ankle, R_foot, L_wrist, R_wrist]
+ gt_w_j3d = endecoder.fk_v2(**inputs["smpl_params_w"]) # (B, L, J=22, 3)
+ static_gt = get_static_joint_mask(gt_w_j3d, vel_thr=vel_thr, repeat_last=True) # (B, L, J)
+ static_gt = static_gt[:, :, joint_ids].float() # (B, L, J')
+ pred_static_conf_logits = outputs["model_output"]["static_conf_logits"]
+
+ static_conf_loss = F.binary_cross_entropy_with_logits(pred_static_conf_logits, static_gt, reduction="none")
+ static_conf_loss = (static_conf_loss * mask[..., None]).mean()
+ extra_loss += static_conf_loss * weights.static_conf_bce
+ extra_loss_dict["static_conf_loss"] = static_conf_loss
+
+ return extra_loss, extra_loss_dict
+
+
+@autocast(enabled=False)
+def get_smpl_params_w_Rt_v2(
+ global_orient_gv,
+ local_transl_vel,
+ global_orient_c,
+ cam_angvel,
+):
+ """Get global R,t in GV0(ay)
+ Args:
+ cam_angvel: (B, L, 6), defined as R @ R_{w2c}^{t} = R_{w2c}^{t+1}
+ """
+
+ # Get R_ct_to_c0 from cam_angvel
+ def as_identity(R):
+ is_I = matrix_to_axis_angle(R).norm(dim=-1) < 1e-5
+ R[is_I] = torch.eye(3)[None].expand(is_I.sum(), -1, -1).to(R)
+ return R
+
+ B = cam_angvel.shape[0]
+ R_t_to_tp1 = rotation_6d_to_matrix(cam_angvel) # (B, L, 3, 3)
+ R_t_to_tp1 = as_identity(R_t_to_tp1)
+
+ # Get R_c2gv
+ R_gv = axis_angle_to_matrix(global_orient_gv) # (B, L, 3, 3)
+ R_c = axis_angle_to_matrix(global_orient_c) # (B, L, 3, 3)
+
+ # Camera view direction in GV coordinate: Rc2gv @ [0,0,1]
+ R_c2gv = R_gv @ R_c.mT
+ view_axis_gv = R_c2gv[:, :, :, 2] # (B, L, 3) Rc2gv is estimated, so the x-axis is not accurate, i.e. != 0
+
+ # Rotate axis use camera relative rotation
+ R_cnext2gv = R_c2gv @ R_t_to_tp1.mT
+ view_axis_gv_next = R_cnext2gv[..., 2]
+
+ vec1_xyz = view_axis_gv.clone()
+ vec1_xyz[..., 1] = 0
+ vec1_xyz = F.normalize(vec1_xyz, dim=-1)
+ vec2_xyz = view_axis_gv_next.clone()
+ vec2_xyz[..., 1] = 0
+ vec2_xyz = F.normalize(vec2_xyz, dim=-1)
+
+ aa_tp1_to_t = vec2_xyz.cross(vec1_xyz, dim=-1)
+ aa_tp1_to_t_angle = torch.acos(torch.clamp((vec1_xyz * vec2_xyz).sum(dim=-1, keepdim=True), -1.0, 1.0))
+ aa_tp1_to_t = F.normalize(aa_tp1_to_t, dim=-1) * aa_tp1_to_t_angle
+
+ aa_tp1_to_t = gaussian_smooth(aa_tp1_to_t, dim=-2) # Smooth
+ R_tp1_to_t = axis_angle_to_matrix(aa_tp1_to_t).mT # (B, L, 3)
+
+ # Get R_t_to_0
+ R_t_to_0 = [torch.eye(3)[None].expand(B, -1, -1).to(R_t_to_tp1)]
+ for i in range(1, R_t_to_tp1.shape[1]):
+ R_t_to_0.append(R_t_to_0[-1] @ R_tp1_to_t[:, i])
+ R_t_to_0 = torch.stack(R_t_to_0, dim=1) # (B, L, 3, 3)
+ R_t_to_0 = as_identity(R_t_to_0)
+
+ global_orient = matrix_to_axis_angle(R_t_to_0 @ R_gv)
+
+ # Rollout to global transl
+ # Start from transl0, in gv0 -> flip y-axis of gv0
+ transl = rollout_local_transl_vel(local_transl_vel, global_orient)
+ global_orient, transl, _ = get_tgtcoord_rootparam(global_orient, transl, tsf="any->ay")
+
+ smpl_params_w_Rt = {"global_orient": global_orient, "transl": transl}
+ return smpl_params_w_Rt
diff --git a/third_party/GVHMR/hmr4d/model/gvhmr/utils/endecoder.py b/third_party/GVHMR/hmr4d/model/gvhmr/utils/endecoder.py
new file mode 100644
index 0000000000000000000000000000000000000000..223cb6d636121d400ebe924b7761d7715bae8682
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/gvhmr/utils/endecoder.py
@@ -0,0 +1,199 @@
+import torch
+import torch.nn as nn
+from pytorch3d.transforms import (
+ rotation_6d_to_matrix,
+ matrix_to_axis_angle,
+ axis_angle_to_matrix,
+ matrix_to_rotation_6d,
+ matrix_to_quaternion,
+ quaternion_to_matrix,
+)
+from hmr4d.configs import MainStore, builds
+from hmr4d.utils.geo.augment_noisy_pose import gaussian_augment
+import hmr4d.utils.matrix as matrix
+from hmr4d.utils.pylogger import Log
+from hmr4d.utils.geo.hmr_global import get_local_transl_vel, rollout_local_transl_vel
+from hmr4d.utils.smplx_utils import make_smplx
+from . import stats_compose
+
+
+class EnDecoder(nn.Module):
+ def __init__(self, stats_name="DEFAULT_01", noise_pose_k=10):
+ super().__init__()
+ # Load mean, std
+ stats = getattr(stats_compose, stats_name)
+ Log.info(f"[EnDecoder] Use {stats_name} for statistics!")
+ self.register_buffer("mean", torch.tensor(stats["mean"]).float(), False)
+ self.register_buffer("std", torch.tensor(stats["std"]).float(), False)
+
+ # option
+ self.noise_pose_k = noise_pose_k
+
+ # smpl
+ self.smplx_model = make_smplx("supermotion_v437coco17")
+ parents = self.smplx_model.parents[:22]
+ self.register_buffer("parents_tensor", parents, False)
+ self.parents = parents.tolist()
+
+ def get_noisyobs(self, data, return_type="r6d"):
+ """
+ Noisy observation contains local pose with noise
+ Args:
+ data (dict):
+ body_pose: (B, L, J*3) or (B, L, J, 3)
+ Returns:
+ noisy_bosy_pose: (B, L, J, 6) or (B, L, J, 3) or (B, L, 3, 3) depends on return_type
+ """
+ body_pose = data["body_pose"] # (B, L, 63)
+ B, L, _ = body_pose.shape
+ body_pose = body_pose.reshape(B, L, -1, 3)
+
+ # (B, L, J, C)
+ return_mapping = {"R": 0, "r6d": 1, "aa": 2}
+ return_id = return_mapping[return_type]
+ noisy_bosy_pose = gaussian_augment(body_pose, self.noise_pose_k, to_R=True)[return_id]
+ return noisy_bosy_pose
+
+ def normalize_body_pose_r6d(self, body_pose_r6d):
+ """body_pose_r6d: (B, L, {J*6}/{J, 6}) -> (B, L, J*6)"""
+ B, L = body_pose_r6d.shape[:2]
+ body_pose_r6d = body_pose_r6d.reshape(B, L, -1)
+ if self.mean.shape[-1] == 1: # no mean, std provided
+ return body_pose_r6d
+ body_pose_r6d = (body_pose_r6d - self.mean[:126]) / self.std[:126] # (B, L, C)
+ return body_pose_r6d
+
+ def fk_v2(self, body_pose, betas, global_orient=None, transl=None, get_intermediate=False):
+ """
+ Args:
+ body_pose: (B, L, 63)
+ betas: (B, L, 10)
+ global_orient: (B, L, 3)
+ Returns:
+ joints: (B, L, 22, 3)
+ """
+ B, L = body_pose.shape[:2]
+ if global_orient is None:
+ global_orient = torch.zeros((B, L, 3), device=body_pose.device)
+ aa = torch.cat([global_orient, body_pose], dim=-1).reshape(B, L, -1, 3)
+ rotmat = axis_angle_to_matrix(aa) # (B, L, 22, 3, 3)
+
+ skeleton = self.smplx_model.get_skeleton(betas)[..., :22, :] # (B, L, 22, 3)
+ local_skeleton = skeleton - skeleton[:, :, self.parents_tensor]
+ local_skeleton = torch.cat([skeleton[:, :, :1], local_skeleton[:, :, 1:]], dim=2)
+
+ if transl is not None:
+ local_skeleton[..., 0, :] += transl # B, L, 22, 3
+
+ mat = matrix.get_TRS(rotmat, local_skeleton) # B, L, 22, 4, 4
+ fk_mat = matrix.forward_kinematics(mat, self.parents) # B, L, 22, 4, 4
+ joints = matrix.get_position(fk_mat) # B, L, 22, 3
+ if not get_intermediate:
+ return joints
+ else:
+ return joints, mat, fk_mat
+
+ def get_local_pos(self, betas):
+ skeleton = self.smplx_model.get_skeleton(betas)[..., :22, :] # (B, L, 22, 3)
+ local_skeleton = skeleton - skeleton[:, :, self.parents_tensor]
+ local_skeleton = torch.cat([skeleton[:, :, :1], local_skeleton[:, :, 1:]], dim=2)
+ return local_skeleton
+
+ def encode(self, inputs):
+ """
+ definition: {
+ body_pose_r6d, # (B, L, (J-1)*6) -> 0:126
+ betas, # (B, L, 10) -> 126:136
+ global_orient_r6d, # (B, L, 6) -> 136:142 incam
+ global_orient_gv_r6d: # (B, L, 6) -> 142:148 gv
+ local_transl_vel, # (B, L, 3) -> 148:151, smpl-coord
+ }
+ """
+ B, L = inputs["smpl_params_c"]["body_pose"].shape[:2]
+ # cam
+ smpl_params_c = inputs["smpl_params_c"]
+ body_pose = smpl_params_c["body_pose"].reshape(B, L, 21, 3)
+ body_pose_r6d = matrix_to_rotation_6d(axis_angle_to_matrix(body_pose)).flatten(-2)
+ betas = smpl_params_c["betas"]
+ global_orient_R = axis_angle_to_matrix(smpl_params_c["global_orient"])
+ global_orient_r6d = matrix_to_rotation_6d(global_orient_R)
+
+ # global
+ R_c2gv = inputs["R_c2gv"] # (B, L, 3, 3)
+ global_orient_gv_r6d = matrix_to_rotation_6d(R_c2gv @ global_orient_R)
+
+ # local_transl_vel
+ smpl_params_w = inputs["smpl_params_w"]
+ local_transl_vel = get_local_transl_vel(smpl_params_w["transl"], smpl_params_w["global_orient"])
+ if False: # debug
+ transl_recover = rollout_local_transl_vel(
+ local_transl_vel, smpl_params_w["global_orient"], smpl_params_w["transl"][:, [0]]
+ )
+ print((transl_recover - smpl_params_w["transl"]).abs().max())
+
+ # returns
+ x = torch.cat([body_pose_r6d, betas, global_orient_r6d, global_orient_gv_r6d, local_transl_vel], dim=-1)
+ x_norm = (x - self.mean) / self.std
+ return x_norm
+
+ def encode_translw(self, inputs):
+ """
+ definition: {
+ body_pose_r6d, # (B, L, (J-1)*6) -> 0:126
+ betas, # (B, L, 10) -> 126:136
+ global_orient_r6d, # (B, L, 6) -> 136:142 incam
+ global_orient_gv_r6d: # (B, L, 6) -> 142:148 gv
+ local_transl_vel, # (B, L, 3) -> 148:151, smpl-coord
+ }
+ """
+ # local_transl_vel
+ smpl_params_w = inputs["smpl_params_w"]
+ local_transl_vel = get_local_transl_vel(smpl_params_w["transl"], smpl_params_w["global_orient"])
+
+ # returns
+ x = local_transl_vel
+ x_norm = (x - self.mean[-3:]) / self.std[-3:]
+ return x_norm
+
+ def decode_translw(self, x_norm):
+ return x_norm * self.std[-3:] + self.mean[-3:]
+
+ def decode(self, x_norm):
+ """x_norm: (B, L, C)"""
+ B, L, C = x_norm.shape
+ x = (x_norm * self.std) + self.mean
+
+ body_pose_r6d = x[:, :, :126]
+ betas = x[:, :, 126:136]
+ global_orient_r6d = x[:, :, 136:142]
+ global_orient_gv_r6d = x[:, :, 142:148]
+ local_transl_vel = x[:, :, 148:151]
+
+ body_pose = matrix_to_axis_angle(rotation_6d_to_matrix(body_pose_r6d.reshape(B, L, -1, 6)))
+ body_pose = body_pose.flatten(-2)
+ global_orient_c = matrix_to_axis_angle(rotation_6d_to_matrix(global_orient_r6d))
+ global_orient_gv = matrix_to_axis_angle(rotation_6d_to_matrix(global_orient_gv_r6d))
+
+ output = {
+ "body_pose": body_pose,
+ "betas": betas,
+ "global_orient": global_orient_c,
+ "global_orient_gv": global_orient_gv,
+ "local_transl_vel": local_transl_vel,
+ }
+
+ return output
+
+
+group_name = "endecoder/gvhmr"
+cfg_base = builds(EnDecoder, populate_full_signature=True)
+MainStore.store(name="v1_no_stdmean", node=cfg_base, group=group_name)
+MainStore.store(name="v1", node=cfg_base(stats_name="MM_V1"), group=group_name)
+MainStore.store(
+ name="v1_amass_local_bedlam_cam",
+ node=cfg_base(stats_name="MM_V1_AMASS_LOCAL_BEDLAM_CAM"),
+ group=group_name,
+)
+
+MainStore.store(name="v2", node=cfg_base(stats_name="MM_V2"), group=group_name)
+MainStore.store(name="v2_1", node=cfg_base(stats_name="MM_V2_1"), group=group_name)
diff --git a/third_party/GVHMR/hmr4d/model/gvhmr/utils/postprocess.py b/third_party/GVHMR/hmr4d/model/gvhmr/utils/postprocess.py
new file mode 100644
index 0000000000000000000000000000000000000000..7ac4f5094e79a2a34cb217855b463454bdc12af4
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/gvhmr/utils/postprocess.py
@@ -0,0 +1,172 @@
+import torch
+from torch.cuda.amp import autocast
+from pytorch3d.transforms import (
+ matrix_to_rotation_6d,
+ rotation_6d_to_matrix,
+ axis_angle_to_matrix,
+ matrix_to_axis_angle,
+)
+
+import hmr4d.utils.matrix as matrix
+from hmr4d.utils.ik.ccd_ik import CCD_IK
+from hmr4d.utils.geo_transform import get_sequence_cammat, transform_mat, apply_T_on_points
+from hmr4d.utils.net_utils import gaussian_smooth
+from hmr4d.model.gvhmr.utils.endecoder import EnDecoder
+
+from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines
+
+
+@autocast(enabled=False)
+def pp_static_joint(outputs, endecoder: EnDecoder):
+ # Global FK
+ pred_w_j3d = endecoder.fk_v2(**outputs["pred_smpl_params_global"])
+ L = pred_w_j3d.shape[1]
+ joint_ids = [7, 10, 8, 11, 20, 21] # [L_Ankle, L_foot, R_Ankle, R_foot, L_wrist, R_wrist]
+ pred_j3d_static = pred_w_j3d.clone()[:, :, joint_ids] # (B, L, J, 3)
+
+ ######## update overall movement with static info, and make displacement ~[0,0,0]
+ pred_j_disp = pred_j3d_static[:, 1:] - pred_j3d_static[:, :-1] # (B, L-1, J, 3)
+
+ static_conf_logits = outputs["static_conf_logits"][:, :-1].clone()
+ static_label_ = static_conf_logits > 0 # (B, L-1, J) # avoid non-contact frame
+ static_conf_logits = static_conf_logits.float() - (~static_label_ * 1e6) # fp16 cannot go through softmax
+ is_static = static_label_.sum(dim=-1) > 0 # (B, L-1)
+
+ pred_disp = pred_j_disp * static_conf_logits[..., None].softmax(dim=-2) # (B, L-1, J, 3)
+ pred_disp = pred_disp * is_static[..., None, None] # (B, L-1, J, 3)
+ pred_disp = pred_disp.sum(-2) # (B, L-1, 3)
+ ####################
+
+ # Overwrite results:
+ if False: # for-loop
+ post_w_transl = outputs["pred_smpl_params_global"]["transl"].clone() # (B, L, 3)
+ for i in range(1, L):
+ post_w_transl[:, i:] -= pred_disp[:, i - 1 : i]
+ else: # vectorized
+ pred_w_transl = outputs["pred_smpl_params_global"]["transl"].clone() # (B, L, 3)
+ pred_w_disp = pred_w_transl[:, 1:] - pred_w_transl[:, :-1] # (B, L-1, 3)
+ pred_w_disp_new = pred_w_disp - pred_disp
+ post_w_transl = torch.cumsum(torch.cat([pred_w_transl[:, :1], pred_w_disp_new], dim=1), dim=1)
+ post_w_transl[..., 0] = gaussian_smooth(post_w_transl[..., 0], dim=-1)
+ post_w_transl[..., 2] = gaussian_smooth(post_w_transl[..., 2], dim=-1)
+
+ # Optional: put the sequence on the ground by -min(y).
+ # NOTE: Some datasets (e.g., Unity exports with explicit ground normalization) want to keep the
+ # absolute vertical position. They can disable this via `outputs["disable_pp_ground"]=True`.
+ if not bool(outputs.get("disable_pp_ground", False)):
+ post_w_j3d = pred_w_j3d - pred_w_transl.unsqueeze(-2) + post_w_transl.unsqueeze(-2)
+ ground_y = post_w_j3d[..., 1].flatten(-2).min(dim=-1)[0] # (B,)
+ post_w_transl[..., 1] -= ground_y
+
+ return post_w_transl
+
+
+@autocast(enabled=False)
+def pp_static_joint_cam(outputs, endecoder: EnDecoder):
+ """Use static joint and static camera assumption to postprocess the global transl"""
+ # input
+ pred_smpl_params_incam = outputs["pred_smpl_params_incam"].copy()
+ pred_smpl_params_global = outputs["pred_smpl_params_global"]
+ static_conf_logits = outputs["static_conf_logits"].clone()[:, :-1] # (B, L-1, J)
+ joint_ids = [7, 10, 8, 11, 20, 21] # [L_Ankle, L_foot, R_Ankle, R_foot, L_wrist, R_wrist]
+ B, L = pred_smpl_params_incam["transl"].shape[:2]
+ assert B == 1
+
+ # FK
+ pred_w_j3d = endecoder.fk_v2(**pred_smpl_params_global) # (B, L, J, 3)
+ # smooth incam results, as this could be noisy
+ pred_smpl_params_incam["transl"] = gaussian_smooth(pred_smpl_params_incam["transl"], sigma=5, dim=-2)
+ pred_c_j3d = endecoder.fk_v2(**pred_smpl_params_incam) # (B, L, J, 3)
+
+ # compute T_c2w (static) from first frame
+ R_gv = axis_angle_to_matrix(pred_smpl_params_global["global_orient"][:, 0]) # (B, 3, 3)
+ R_c = axis_angle_to_matrix(pred_smpl_params_incam["global_orient"][:, 0]) # (B, 3, 3)
+ R_c2w = R_gv @ R_c.mT # (B, 3, 3)
+ t_c2w = pred_w_j3d[:, 0, 0] - torch.einsum("bij,bj->bi", R_c2w, pred_c_j3d[:, 0, 0]) # (B, 3)
+ T_c2w = transform_mat(R_c2w, t_c2w) # (B, 4, 4)
+ pred_c_j3d_in_w = apply_T_on_points(pred_c_j3d, T_c2w[:, None])
+
+ # 1. Make transl similar to incam
+ post_w_transl = pred_smpl_params_global["transl"].clone() # (B, L, 3)
+ post_w_j3d = pred_w_j3d.clone() # (B, L, J, 3)
+ cp_thr = torch.tensor([0.25, 0.25, 0.25]).to(post_w_j3d) # Only update very bad pred
+ for i in range(1, L):
+ cp_diff = post_w_j3d[:, i, 0] - pred_c_j3d_in_w[:, i, 0] # (B, 3)
+ cp_diff = cp_diff * ~((cp_diff > -cp_thr) * (cp_diff < cp_thr))
+ cp_diff = torch.clamp(cp_diff, -0.02, 0.02)
+ post_w_transl[:, i:] -= cp_diff
+ post_w_j3d[:, i:] -= (cp_diff)[:, None, None]
+
+ # 1. Make stationary joint stay stationary
+ # pred_j3d_static = pred_w_j3d.clone()[:, :, joint_ids] # (B, L, J, 3)
+ pred_j3d_static = post_w_j3d[:, :, joint_ids] # (B, L, J, 3)
+ pred_j_disp = pred_j3d_static[:, 1:] - pred_j3d_static[:, :-1] # (B, L-1, J, 3)
+
+ static_label = static_conf_logits.sigmoid() > 0.8 # (B, L-1, J)
+ static_label_sumJ = static_label.sum(-1, keepdim=True) # (B, L-1, 1)
+ static_label_sumJ = torch.clamp_min(static_label_sumJ, 1) # replace 0 with 1
+ pred_disp_sumJ = (pred_j_disp * static_label[..., None]).sum(-2) # (B, L-1, 3)
+ pred_disp = pred_disp_sumJ / static_label_sumJ # (B, L-1, 3)
+ pred_disp[:, :, 1] = 0 # do not modify y
+
+ # Overwrite results (for-loop)
+ for i in range(1, L):
+ post_w_transl[:, i:] -= pred_disp[:, [i - 1]]
+ post_w_j3d[:, i:] -= pred_disp[:, [i - 1], None]
+
+ # Optional: put the sequence on the ground by -min(y).
+ if not bool(outputs.get("disable_pp_ground", False)):
+ ground_y = post_w_j3d[..., 1].flatten(-2).min(dim=-1)[0] # (B,) Minimum y value
+ post_w_transl[..., 1] -= ground_y
+
+ return post_w_transl
+
+
+@autocast(enabled=False)
+def process_ik(outputs, endecoder):
+ static_conf = outputs["static_conf_logits"].sigmoid() # (B, L, J)
+ post_w_j3d, local_mat, post_w_mat = endecoder.fk_v2(**outputs["pred_smpl_params_global"], get_intermediate=True)
+
+ # sebas rollout merge
+ joint_ids = [7, 10, 8, 11, 20, 21] # [L_Ankle, L_foot, R_Ankle, R_foot, L_wrist, R_wrist]
+ post_target_j3d = post_w_j3d.clone()
+ for i in range(1, post_w_j3d.size(1)):
+ prev = post_target_j3d[:, i - 1, joint_ids]
+ this = post_w_j3d[:, i, joint_ids]
+ c_prev = static_conf[:, i - 1, :, None]
+ post_target_j3d[:, i, joint_ids] = prev * c_prev + this * (1 - c_prev)
+
+ # ik
+ global_rot = matrix.get_rotation(post_w_mat)
+ parents = [-1, 0, 0, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 9, 9, 12, 13, 14, 16, 17, 18, 19]
+ left_leg_chain = [0, 1, 4, 7, 10]
+ right_leg_chain = [0, 2, 5, 8, 11]
+ left_hand_chain = [9, 13, 16, 18, 20]
+ right_hand_chain = [9, 14, 17, 19, 21]
+
+ def ik(local_mat, target_pos, target_rot, target_ind, chain):
+ local_mat = local_mat.clone()
+ IK_solver = CCD_IK(
+ local_mat,
+ parents,
+ target_ind,
+ target_pos,
+ target_rot,
+ kinematic_chain=chain,
+ max_iter=2,
+ )
+
+ chain_local_mat = IK_solver.solve()
+ chain_rotmat = matrix.get_rotation(chain_local_mat)
+ local_mat[:, :, chain[1:], :-1, :-1] = chain_rotmat[:, :, 1:] # (B, L, J, 3, 3)
+ return local_mat
+
+ local_mat = ik(local_mat, post_target_j3d[:, :, [7, 10]], global_rot[:, :, [7, 10]], [3, 4], left_leg_chain)
+ local_mat = ik(local_mat, post_target_j3d[:, :, [8, 11]], global_rot[:, :, [8, 11]], [3, 4], right_leg_chain)
+ local_mat = ik(local_mat, post_target_j3d[:, :, [20]], global_rot[:, :, [20]], [4], left_hand_chain)
+ local_mat = ik(local_mat, post_target_j3d[:, :, [21]], global_rot[:, :, [21]], [4], right_hand_chain)
+
+ body_pose = matrix_to_axis_angle(matrix.get_rotation(local_mat[:, :, 1:])) # (B, L, J-1, 3, 3)
+ body_pose = body_pose.flatten(2) # (B, L, (J-1)*3)
+
+ return body_pose
diff --git a/third_party/GVHMR/hmr4d/model/gvhmr/utils/stats_compose.py b/third_party/GVHMR/hmr4d/model/gvhmr/utils/stats_compose.py
new file mode 100644
index 0000000000000000000000000000000000000000..80e4d07140260d568fd7710907d3730b8b78d41f
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/model/gvhmr/utils/stats_compose.py
@@ -0,0 +1,259 @@
+# fmt:off
+body_pose_r6d = {
+ "bedlam": {
+ "count": 5417929,
+ "mean": [ 0.9772, -0.0925, 0.0028, 0.1058, 0.9111, 0.1373, 0.9796, 0.0711,
+ -0.0193, -0.0816, 0.8910, 0.1953, 0.9935, 0.0072, 0.0270, -0.0046,
+ 0.9200, -0.2511, 0.9752, 0.0477, -0.0990, -0.0613, 0.8242, -0.2730,
+ 0.9836, -0.0400, 0.0067, 0.0148, 0.7836, -0.3471, 0.9931, -0.0300,
+ -0.0469, 0.0244, 0.9825, -0.0513, 0.9777, 0.0206, 0.1444, -0.0470,
+ 0.9603, 0.1521, 0.9804, -0.0362, -0.0902, 0.0500, 0.9546, 0.1337,
+ 0.9969, -0.0105, 0.0076, 0.0090, 0.9914, 0.0150, 0.9953, -0.0607,
+ 0.0089, 0.0602, 0.9942, 0.0146, 0.9934, -0.0682, -0.0171, 0.0680,
+ 0.9932, -0.0017, 0.9790, 0.0294, 0.0065, -0.0338, 0.9706, -0.0456,
+ 0.9056, 0.2457, -0.1029, -0.2279, 0.9262, 0.0145, 0.9233, -0.1301,
+ 0.1550, 0.1140, 0.9476, 0.0534, 0.9769, -0.0572, -0.0095, 0.0569,
+ 0.9690, 0.0472, 0.6782, 0.5746, -0.2378, -0.5546, 0.7212, 0.0917,
+ 0.6489, -0.5955, 0.2424, 0.5821, 0.6797, 0.0563, 0.5562, -0.1252,
+ -0.5860, 0.0937, 0.9176, -0.1287, 0.4453, 0.1421, 0.6119, -0.1427,
+ 0.8996, -0.1136, 0.9186, -0.0881, -0.1463, 0.1087, 0.8692, 0.0845,
+ 0.9175, 0.0257, 0.0663, -0.0385, 0.8603, 0.1020],
+ "std": [0.0429, 0.1392, 0.1236, 0.1323, 0.1645, 0.3086, 0.0375, 0.1406, 0.1172,
+ 0.1275, 0.1934, 0.3280, 0.0119, 0.0835, 0.0716, 0.0741, 0.1528, 0.2484,
+ 0.0349, 0.0947, 0.1633, 0.0924, 0.3469, 0.3370, 0.0273, 0.1009, 0.1411,
+ 0.0680, 0.3876, 0.3323, 0.0103, 0.0735, 0.0712, 0.0690, 0.0246, 0.1617,
+ 0.0216, 0.1097, 0.1016, 0.0924, 0.0509, 0.2035, 0.0245, 0.1188, 0.1212,
+ 0.1056, 0.0634, 0.2308, 0.0054, 0.0579, 0.0517, 0.0575, 0.0124, 0.1158,
+ 0.0076, 0.0654, 0.0367, 0.0644, 0.0118, 0.0592, 0.0116, 0.0829, 0.0361,
+ 0.0832, 0.0124, 0.0422, 0.0343, 0.1060, 0.1680, 0.1075, 0.0473, 0.2023,
+ 0.0701, 0.2344, 0.2213, 0.2632, 0.0589, 0.1318, 0.0767, 0.2456, 0.2009,
+ 0.2666, 0.0542, 0.1106, 0.0347, 0.1080, 0.1718, 0.1117, 0.0459, 0.2025,
+ 0.1882, 0.2769, 0.2032, 0.3072, 0.1447, 0.2204, 0.2018, 0.2820, 0.2126,
+ 0.3213, 0.1760, 0.2486, 0.4749, 0.1677, 0.2791, 0.2239, 0.0963, 0.2705,
+ 0.5540, 0.1846, 0.2572, 0.2411, 0.1287, 0.2878, 0.1151, 0.2993, 0.1557,
+ 0.2812, 0.1880, 0.3334, 0.1286, 0.3355, 0.1553, 0.3216, 0.1880, 0.3306]
+ },
+ "amass": {
+ "count": 7114038,
+ "mean": [ 9.6969e-01, -5.9719e-02, -3.7700e-02, 5.8256e-02, 9.0800e-01,
+ 1.0972e-01, 9.7636e-01, 4.3401e-02, 4.3110e-03, -4.3032e-02,
+ 9.0261e-01, 1.4478e-01, 9.9288e-01, 3.5673e-03, 1.6264e-02,
+ -2.2260e-03, 9.3470e-01, -2.3495e-01, 9.7147e-01, 5.2553e-02,
+ -9.3666e-02, -5.4550e-02, 8.3321e-01, -2.4246e-01, 9.7971e-01,
+ -3.8429e-02, 5.3575e-03, 1.5537e-02, 8.1449e-01, -3.0926e-01,
+ 9.9532e-01, -9.4398e-03, -3.8328e-02, 8.5141e-03, 9.8880e-01,
+ 1.9976e-04, 9.5602e-01, -3.9528e-02, 2.0017e-01, 1.0363e-02,
+ 9.5965e-01, 1.3770e-01, 9.6223e-01, -4.6278e-02, -1.5177e-01,
+ 6.6705e-02, 9.5545e-01, 1.2519e-01, 9.9767e-01, -1.2616e-02,
+ -2.5442e-04, 1.1661e-02, 9.9376e-01, -3.6222e-02, 9.9511e-01,
+ -1.0583e-02, 1.2130e-02, 7.6461e-03, 9.9137e-01, 2.0029e-02,
+ 9.9295e-01, 7.2917e-03, 4.9454e-03, -8.0286e-03, 9.9137e-01,
+ 2.3707e-03, 9.7698e-01, 1.9943e-02, 1.3808e-03, -2.2006e-02,
+ 9.7375e-01, -6.7936e-02, 9.2804e-01, 2.5005e-01, -5.7167e-02,
+ -2.4047e-01, 9.4246e-01, 2.5863e-02, 9.2957e-01, -2.1329e-01,
+ 1.1112e-01, 2.0741e-01, 9.4876e-01, 2.9901e-02, 9.7683e-01,
+ -4.1210e-02, 2.3248e-03, 4.0967e-02, 9.7365e-01, 5.7309e-03,
+ 6.4513e-01, 6.1999e-01, -2.5469e-01, -6.2342e-01, 6.8177e-01,
+ 3.5524e-02, 6.6192e-01, -5.9341e-01, 2.7136e-01, 5.9269e-01,
+ 6.8966e-01, 3.1309e-02, 6.8946e-01, -1.1676e-01, -4.9859e-01,
+ 4.0969e-02, 9.3656e-01, -1.4875e-01, 6.2787e-01, 1.3793e-01,
+ 5.4289e-01, -9.1946e-02, 9.2868e-01, -1.1927e-01, 9.3012e-01,
+ -8.3810e-02, -1.1951e-01, 9.7211e-02, 8.9118e-01, 5.9887e-02,
+ 9.3033e-01, 7.1047e-02, 7.5264e-02, -8.0679e-02, 8.8562e-01,
+ 4.8960e-02],
+ "std": [0.0612, 0.1390, 0.1779, 0.1415, 0.1826, 0.3268, 0.0440, 0.1382, 0.1542,
+ 0.1348, 0.1930, 0.3272, 0.0132, 0.0801, 0.0855, 0.0729, 0.1255, 0.2238,
+ 0.0554, 0.1088, 0.1727, 0.0939, 0.3294, 0.3559, 0.0532, 0.1082, 0.1554,
+ 0.0768, 0.3446, 0.3407, 0.0120, 0.0650, 0.0584, 0.0632, 0.0198, 0.1335,
+ 0.0631, 0.1250, 0.1574, 0.1047, 0.0730, 0.2091, 0.0759, 0.1241, 0.1667,
+ 0.1112, 0.0831, 0.2185, 0.0060, 0.0441, 0.0502, 0.0441, 0.0102, 0.0946,
+ 0.0237, 0.0722, 0.0610, 0.0738, 0.0479, 0.0949, 0.0369, 0.0943, 0.0610,
+ 0.0966, 0.0498, 0.0729, 0.0425, 0.1001, 0.1824, 0.0972, 0.0408, 0.1887,
+ 0.0594, 0.1842, 0.1884, 0.2020, 0.0457, 0.1018, 0.0640, 0.1990, 0.1854,
+ 0.2133, 0.0467, 0.0910, 0.0392, 0.1049, 0.1776, 0.1037, 0.0413, 0.1945,
+ 0.1733, 0.2612, 0.1905, 0.2963, 0.1512, 0.1861, 0.1710, 0.2663, 0.1896,
+ 0.3135, 0.1568, 0.2219, 0.3976, 0.1594, 0.2810, 0.1855, 0.0845, 0.2398,
+ 0.4398, 0.1629, 0.2685, 0.1990, 0.0998, 0.2556, 0.1137, 0.2837, 0.1419,
+ 0.2761, 0.1678, 0.2973, 0.1172, 0.3010, 0.1394, 0.2910, 0.1724, 0.3039]
+ }
+}
+
+betas = {
+ "bedlam": {
+ "count": 37855, # so many subjects?
+ "mean": [ 0.0378, -0.3562, 0.1185, 0.2245, 0.0204, 0.0929, 0.0537, 0.1006,
+ -0.1180, 0.0936],
+ "std":[0.8070, 1.3480, 0.8964, 0.7390, 0.6433, 0.6089, 0.5374, 0.6984, 0.7263,
+ 0.5395],
+ },
+ "amass": {
+ "count": 18086,
+ "mean": [ 0.2310, 0.1750, 0.2931, -0.1859, -1.1163, -1.1028, -0.2573, 0.3555,
+ 0.3732, 0.2852],
+ "std": [0.8831, 0.7965, 1.0899, 1.1788, 1.2128, 1.1081, 0.9780, 1.1434, 0.8498,
+ 1.1462],
+ }
+}
+
+global_orient_c_r6d = {
+ "bedlam": {
+ "count": 5417929,
+ "mean": [-4.9862e-03, -8.7136e-04, -1.4187e-03, 1.4825e-02, -9.4419e-01,
+ -5.1653e-02],
+ "std": [0.7048, 0.1713, 0.6884, 0.1548, 0.1546, 0.2403],
+ },
+}
+
+global_orient_gv_r6d = {
+ "bedlam": {
+ "count": 5134187,
+ "mean": [ 3.6018e-04, -2.2327e-04, 2.2316e-03, -4.4879e-02, -9.7435e-01,
+ 1.0021e-01],
+ "std": [0.6070, 0.5355, 0.5873, 0.6285, 0.2336, 0.7675],
+ },
+}
+
+local_transl_vel = {
+ "none":{
+ "mean": [0., 0., 0.],
+ "std": [1., 1., 1.]
+ },
+ "1e-2":{
+ "mean": [0., 0., 0.],
+ "std": [1e-2, 1e-2, 1e-2]
+ },
+ "bedlam": {
+ "count": 5417929,
+ "mean": [7.3057e-05, -2.2142e-04, 3.2444e-03],
+ "std": [0.0065, 0.0091, 0.0114],
+ },
+ "amass": {
+ "count": 7113068,
+ "mean": [-0.0002, -0.0006, 0.0069],
+ "std": [0.0064, 0.0070, 0.0138],
+ },
+ "alignhead":{
+ "count": 7113068,
+ "mean":[-2.0822e-04, -1.7966e-06, 6.9816e-03],
+ "std":[0.0065, 0.0066, 0.0139],
+ },
+ "alignhead_absy":{
+ "count": 7113068,
+ "mean":[-0.0002, -0.0316, 0.0070],
+ "std":[0.0065, 0.1351, 0.0139],
+ },
+ "alignhead_absgy":{
+ "count": 7113068,
+ "mean":[[-2.0822e-04, 1.2627e+00, 6.9816e-03]],
+ "std":[0.0065, 0.1516, 0.0139],
+ }
+
+}
+
+pred_cam = {
+ "bedlam": {
+ "count": 5096332,
+ "mean": [1.0606, -0.0027, 0.2702],
+ "std": [0.1784, 0.0956, 0.0764],
+ }
+}
+
+vitfeat = {
+ "bedlam": {
+ "count": 5546332,
+ "mean": [-1.3772, 0.2490, 0.0602, -0.1834, 0.2458, 0.5372, 0.3343, -0.3476, -0.1017, -0.0362, -0.0678, 0.2150, -0.2534, 0.1029, 0.8199, -0.4676, 0.6259, -0.3350, 0.0549, -0.4469, 0.2751, -0.1763, 0.1114, -0.2115, -0.0264, 0.5294, 0.8212, -0.4562, 0.4147, -0.0256, -0.1019, 0.2798, 0.9284, 0.4652, 0.6365, 0.6785, -0.0765, 0.0337, -0.2566, -0.0335, -0.1799, 0.7426, 0.2810, -0.7121, -0.0893, 0.1608, -0.2483, 1.5094, -1.4395, -0.3682, -0.4157, -0.0032, -0.0376, -0.0043, 0.2092, 0.3038, -0.2077, -0.4868, -0.1534, 0.2668, 1.2773, 0.2838, -0.4863, -1.2300, 0.0581, -0.3041, 0.1518, 0.7955, -0.4293, 1.4666, 0.3077, 0.3918, 0.1418, 0.1590, 0.8671, -0.3527, 0.5629, 0.1414, 0.0964, -0.1094, -0.0211, -0.0937, 0.1606, -0.7900, 0.0397, 0.0570, 0.7083, -0.5732, 0.1430, -0.2571, 0.5275, 0.6603, 0.3265, 0.4574, -0.3361, -0.1267, 0.3841, 0.1758, -0.6207, -0.3673, 0.8914, 0.4297, -0.8118, 0.2229, -0.2876, 0.2460, 0.4856, -0.1446, -0.2416, 0.1229, 0.2865, 0.7023, -0.2883, 0.3940, -1.5496, 0.4456, 0.6445, 0.2058, -0.4265, 0.3724, 0.1557, -1.4208, -0.1246, 0.1237, -0.3965, 0.0105, -0.0780, 0.6448, -0.1132, 0.8500, -0.2828, 0.4447, 0.6257, -0.2664, -0.8384, -1.8091, -0.2769, 0.1866, 0.6051, -0.2548, 0.9823, -0.2985, -0.2773, -0.4383, 0.1886, 0.2411, 0.2546, 0.2195, -0.0041, 0.1038, -0.6804, 1.2364, 0.5393, 0.0351, 0.4537, -0.8044, -0.1993, -2.1097, -0.8458, 0.1497, 1.6042, 0.6458, -0.5455, 0.0778, 0.0504, -0.5242, -0.3215, -0.0199, 1.1461, -0.3355, -0.3421, -0.3951, 0.0184, -0.0261, 0.2048, 0.0080, 0.6553, -1.3221, 0.5140, 0.5958, -0.2523, 0.9434, -0.0727, 0.1978, 1.1105, -0.4992, 0.3990, 0.2074, 0.3843, -0.0444, 0.0624, -0.8442, -0.0724, -0.5328, 1.1723, 0.8043, 0.6674, 1.5283, 4.2502, 0.0935, 0.3733, 0.1569, 0.0154, 0.0674, 0.0862, -0.2744, -0.4537, 0.1588, -1.9156, 0.0149, -1.0498, -0.0790, 0.0851, -0.5007, 0.3323, -0.1065, 0.0782, 0.0725, -0.5921, -0.1876, 0.0094, -0.3631, 0.0951, 0.1318, 0.0936, 0.5668, -0.0875, -0.4576, -0.4306, 0.5458, 1.0761, 1.1740, -0.0337, 1.3718, -0.2913, -0.3433, 0.5338, -0.4577, -0.4966, 0.2704, 0.3236, 0.4053, 0.0360, 1.1616, -0.2012, 0.7373, 0.0779, -0.0280, -0.4426, 0.0450, 0.2923, 0.0161, -0.4788, 0.1924, -0.3012, 0.0298, -0.7776, -0.2215, 0.4494, -0.1677, 0.2214, 0.0762, -0.3088, 0.4230, 0.0673, -1.0233, 0.0748, -0.4358, -0.2497, -0.0066, 0.1679, -0.1077, -0.4290, 2.5254, -0.8819, -0.8073, 0.2535, 2.0680, -0.4715, 0.3614, -2.9281, 3.1536, 0.3118, -0.0239, 0.7064, -0.6935, -1.1070, -0.1715, -0.0920, -0.2133, -1.0173, 0.0084, -0.1721, 0.2605, -0.6607, -0.0788, -0.3479, -0.2187, 1.0605, 0.2857, 0.7464, 0.9612, -1.1332, 1.5708, -1.0264, 0.6070, 0.4103, -0.1950, -0.0629, -0.0958, -0.2199, -0.2198, -0.4019, 0.2478, -0.3576, 0.0191, -5.8435, 0.0145, -0.2312, 0.9872, 1.1159, 0.3775, 0.1960, -0.5968, -0.2611, -0.0634, -0.1003, 0.7411, -0.8298, -0.1743, 1.8418, 0.3692, -0.4321, 0.0613, -1.9046, 0.5812, 0.2805, 0.1703, -0.2212, -0.0740, -0.2737, -0.3084, 2.9787, -0.1392, 0.3347, 0.0866, -0.8654, -0.4564, -0.7839, 0.1033, -0.0204, 0.1558, -0.1469, 0.2850, -0.1139, 0.8253, 0.7352, -0.6132, 0.0566, 0.3087, -0.1189, 0.1640, 0.2511, 0.5230, -0.0972, -0.5621, -2.5404, 0.3529, -0.2543, -0.6757, 0.2045, -0.0511, -0.2204, 0.1023, 0.0143, 0.4191, -0.3946, -1.0912, 0.8555, 1.0751, -0.0184, -0.3162, 0.1910, 0.6522, -0.5801, 0.2091, -0.8254, -0.3425, 0.3368, -0.0384, -0.4570, 2.5288, -0.3513, -0.1630, 0.1096, -0.5936, 1.5303, -0.4135, -0.2418, -0.0564, -2.6344, -0.1054, 0.8866, -0.2946, -0.4564, -0.6220, 0.2672, -0.9012, 0.3535, 0.2344, -0.0718, 0.0782, 0.0133, 0.2032, -1.2768, 0.1271, -0.5114, -0.0584, -0.8219, -0.1069, 1.5577, -0.1432, -0.6794, 0.9101, 0.6390, 0.3547, -0.6126, -0.1885, 0.2462, -1.1864, 0.0653, -0.7940, 0.5204, 0.5372, 0.5353, -0.4268, -0.2003, -0.2496, -0.0405, 0.3615, -0.1635, 0.1908, -0.0467, 0.7167, 0.1465, 0.4621, 0.1190, -1.6899, 0.6512, 1.3150, -0.1273, 0.0507, 0.2058, -0.1855, 0.1316, 0.1280, 0.5049, 0.0262, -0.0329, 2.0327, -0.6410, 0.4536, 0.0609, 0.1883, -0.5454, -0.5247, 0.1856, 0.7238, 1.4886, -0.1068, 1.7239, -0.8228, -0.2155, 0.5159, 0.2941, -0.0782, -0.0159, 0.1844, -0.1808, -0.1132, 0.4861, 4.0106, 0.0130, 0.2455, -0.1101, 0.0792, 0.4720, -0.1022, 2.0154, -0.4013, 0.5604, 1.3600, -0.5614, 0.3793, -0.1245, 0.2444, 0.1657, 1.7616, 0.6198, 0.1761, -0.6036, -0.1931, 0.4449, 0.2574, -0.2360, 1.1118, 0.0804, 1.1533, 0.2549, 0.3386, 0.2463, 0.0930, -0.6093, -0.1464, 0.2889, 0.2294, -0.5943, 0.1323, 0.5119, 0.1093, -1.0178, 0.4735, 0.3068, 0.3213, -0.0585, -0.3682, -0.6105, -0.7776, 0.1999, 0.9439, -0.4209, 0.1488, 1.3119, -0.4679, -0.3882, 0.2677, -0.1673, -0.5921, -1.2811, -1.0972, 0.3873, 0.0798, -0.0538, 0.0659, -0.1439, -1.3106, -0.5175, 0.4538, -1.0376, -0.9015, 0.7454, -0.0714, -0.4641, 0.2083, 0.0596, -2.9637, 0.3057, 0.2121, -0.2399, 0.6963, 0.1400, 1.7446, 0.9707, -0.3118, -0.3371, 0.0130, 1.0006, -0.2740, 0.1100, -0.9666, 0.7636, 1.2002, -0.0018, -0.3380, 0.1262, 0.5829, -0.0374, 0.0689, 0.2022, -2.0056, -0.2051, -0.4549, 0.0519, 0.4217, -0.7413, 0.0601, 0.4385, 2.8503, -2.7656, 1.2281, -0.1280, 0.6028, 0.4995, 0.0638, -0.3376, 0.2527, -0.1572, -0.4385, -0.6372, 0.2569, 0.4115, 0.4507, 0.6063, -0.1051, 1.2529, 0.2453, -0.7905, -0.3797, -0.2674, 0.2662, 1.5347, -0.3908, 0.8839, -0.6054, -0.4827, -0.3495, 1.2107, -0.4419, -0.6177, 0.1054, 1.0132, -0.3246, -0.1776, 1.1740, -0.0252, 0.0368, -0.7937, -0.9988, -0.0228, 0.0742, -2.4925, 0.5785, 2.3900, 1.2726, -0.3682, -0.8625, -0.3299, 0.3934, 1.4045, -0.6200, -0.0024, 0.2348, -0.1827, -0.5913, -0.6982, 0.2648, 0.2601, 0.9986, 0.1636, 0.8982, -0.4269, 1.7454, -1.9136, -0.9865, -0.0451, 0.2851, -0.5938, -0.3066, 0.0910, -0.3150, -0.4002, 0.4789, 0.0337, -0.6997, -0.2555, -0.6602, -3.0103, 0.2491, -1.0346, 0.3651, 0.2319, 1.0224, -0.2613, 1.6970, 0.7515, 2.1477, 0.1310, 0.2060, 0.1372, 1.0049, -0.8758, -0.3804, -2.1513, 0.8010, -0.2271, -0.2108, 0.3728, -1.7321, -1.0250, -0.2584, -0.2513, 0.2418, -0.7641, 0.2084, -1.3560, 0.5803, 0.1556, -0.3612, 1.3099, -0.2673, 0.4371, -0.8022, 0.1776, -0.5019, 0.1880, -0.2093, 0.0750, -0.7228, -1.3950, 0.1944, -1.5994, -0.2832, 0.0507, 0.1917, 1.2954, 0.0471, 0.3115, -2.2382, -0.3891, -0.0704, 0.3897, 0.0347, 0.9186, -0.8407, 0.9456, 0.5629, 0.3474, -0.4869, 0.4696, -0.4438, 0.0860, -0.8313, -0.0383, 0.2055, 0.4822, -0.1455, -0.1719, -0.2346, -0.4606, 0.8018, 0.3767, -0.0613, 1.9429, -0.6558, -0.0772, -0.1592, -0.1413, 0.4759, -0.0686, 0.9243, -0.2413, -0.1084, -0.2248, -0.0776, 1.4193, -0.0605, 0.1305, -0.2055, 0.0917, 0.6884, -0.0152, 0.1215, 0.2920, -0.0781, -0.0256, 0.3789, -0.1933, 0.1759, 2.3899, 1.0915, -0.7082, -0.4519, -0.2648, -1.2404, -0.2485, 1.0713, 0.1662, -0.1268, 0.3338, -0.0319, 0.1692, -0.5161, 0.9351, 0.1996, -0.2743, 0.0492, -0.0171, 0.1546, 0.2533, -0.0102, 0.6147, 0.0035, -0.2468, -0.2116, -1.7912, 0.2735, 0.4147, 0.4458, 0.6123, 0.0860, 0.2098, -0.3691, -0.2297, -0.6086, -1.0407, -0.7736, -0.3087, -0.0900, -0.1007, -0.3801, -0.3408, -0.4853, -0.3101, -0.8812, 0.0187, -0.9697, -0.2393, 0.1129, -0.5682, 0.4349, 0.1017, 0.2173, -0.0644, -0.9307, 0.9754, 0.2189, 0.2966, -0.4089, -0.2471, -0.7549, 0.3300, 0.7856, 0.1262, 0.2097, -0.5872, 0.9896, 0.5100, 1.0608, -0.7974, 0.1549, -0.1020, 0.4286, 0.0603, -0.6836, -0.4662, -1.2350, -0.0858, -0.5552, 0.0383, 0.2145, -0.4324, -0.5896, 0.9709, -0.0827, -0.2574, 0.2436, -0.1460, 0.5862, 0.4329, -1.2421, 0.0497, -0.0034, 0.2385, -0.1346, 2.0652, 0.8790, -0.2033, -2.6427, 0.3654, -0.1929, -0.0753, -0.9107, 0.9437, 0.3717, -0.7058, -0.2487, -1.0937, -0.7612, 0.9516, -0.7426, -0.0736, 1.2167, 0.6336, 0.2707, -0.7666, -0.1272, -0.8960, 0.3748, 0.7344, 0.7257, 0.3686, -0.5036, -0.2829, 0.0548, 0.3034, -0.2335, -0.3215, 0.0566, -0.2733, -0.3644, 0.0467, -0.0924, -0.5145, -1.7089, 0.4896, 0.0074, 0.2840, 0.1140, -0.0409, -0.3251, 1.0805, 3.0856, -0.3409, 1.2684, -0.0245, -0.0636, -0.0090, 0.1293, -0.3410, -0.0482, 0.1482, 0.2027, 0.5623, 0.0566, 0.6453, -0.0126, 0.0720, -0.0277, 0.0531, 0.1860, -0.1044, -0.6973, 0.3026, 0.4733, -0.1590, 0.4727, 0.8486, 0.4478, 0.1814, 1.0862, 0.0478, 0.2437, -0.5269, -0.0796, -0.4291, 0.4937, -0.0407, -0.6961, -0.0412, 0.6865, 0.0457, 0.1085, -0.4717, -0.1339, 0.8600, 0.6718, -0.3542, -0.5655, 1.3711, 0.0034, 0.3077, 0.0903, 0.3618, 0.3287, -0.1007, 0.0332, -0.3841, -0.3981, 0.1079, -0.4399, 0.1836, 0.0939, -0.1425, -0.2531, -1.2103, 0.0234, -1.3023, -0.0570, -0.0587, 1.1733, 0.0079, 1.0809, 0.4697, -0.1427, 3.3793, -0.1503, 0.4354, 0.0274, 0.3112, -0.3816, 0.0187, -0.1282, -0.4136, 0.3684, 0.6930, 1.3605, 0.4949, 0.4162, -2.2398, 0.4104, 0.6839, 0.4519, 0.0546, -0.0816, 0.0357, 0.1977, -0.8450, 0.1481, 0.1588, -0.1392, -0.3304, -0.3499, -0.8669, 0.1510, 0.1127, 0.9853, -0.3019, -0.3493, -0.0783, -0.8491, 0.0696, 0.7295, -1.0612, 0.1232],
+ "std": [0.9277, 0.7470, 0.6154, 0.8520, 0.8682, 0.7121, 0.7048, 0.6865, 0.7543, 0.6952, 0.6186, 0.4204, 0.4614, 0.4731, 0.4421, 0.4068, 0.6927, 0.6540, 0.4717, 0.4993, 0.5945, 0.5480, 0.4898, 0.6438, 0.5551, 0.5686, 0.7287, 0.6033, 0.5590, 0.3768, 0.5304, 0.6748, 0.5559, 0.5265, 0.6214, 0.6490, 0.4639, 0.6465, 0.5575, 0.6202, 0.5369, 1.2466, 0.7340, 0.5462, 0.6508, 0.5766, 0.5405, 0.5581, 0.5687, 0.7549, 0.5743, 0.4748, 0.6308, 0.6292, 0.6391, 0.6284, 0.4202, 0.5970, 0.5587, 0.5364, 0.4655, 0.5201, 0.7140, 0.6220, 0.4978, 0.4479, 0.5452, 0.7489, 0.5866, 0.4592, 0.7493, 0.6548, 0.5497, 0.4658, 0.8663, 0.4574, 0.5351, 0.5595, 0.4579, 0.5141, 0.4824, 0.5504, 0.5468, 0.5726, 0.5155, 0.6679, 0.8433, 0.5278, 0.5666, 0.7699, 0.5682, 0.9431, 0.5344, 0.6562, 0.4749, 0.5241, 0.6869, 0.4117, 0.5839, 0.5115, 0.8811, 0.5335, 0.6476, 0.4883, 0.6034, 0.5778, 0.4764, 0.8787, 0.8589, 0.5168, 0.4548, 0.8146, 0.5860, 0.6087, 0.6758, 0.7049, 0.8292, 0.6547, 0.6043, 0.7242, 0.6158, 0.6435, 0.5219, 0.6148, 0.7738, 0.4871, 0.7944, 0.7605, 0.6120, 0.5482, 0.6107, 0.6106, 0.4295, 0.4549, 0.4167, 0.6142, 0.6368, 0.5432, 0.5412, 0.6568, 0.9641, 0.6413, 0.6634, 0.4222, 0.6917, 0.5664, 0.5554, 0.4098, 0.6949, 0.5890, 0.4995, 0.5475, 0.6446, 0.5599, 0.6439, 0.6220, 0.5761, 0.5862, 0.5126, 0.6037, 0.5377, 0.5817, 0.6216, 0.5986, 0.4834, 0.6929, 0.5819, 0.6781, 0.6088, 0.5425, 0.7211, 0.6253, 0.5408, 0.6826, 0.5454, 0.7614, 0.9767, 0.8721, 0.7527, 0.4022, 0.5061, 0.5921, 0.5945, 0.6048, 0.7206, 0.5533, 0.5506, 0.6816, 0.6116, 0.6424, 0.7484, 0.6350, 0.5953, 0.4941, 0.7675, 0.8244, 0.6885, 0.5751, 0.9304, 0.5252, 0.5741, 0.4537, 0.5610, 0.9873, 0.5155, 0.7180, 0.4421, 0.5171, 0.5343, 0.5225, 0.7952, 0.6149, 0.6401, 0.5667, 0.6946, 0.8172, 0.5188, 0.5082, 0.6298, 0.6904, 0.4820, 0.5600, 0.5584, 0.5600, 0.4776, 0.5008, 0.7215, 0.6071, 0.5571, 0.6174, 0.4049, 0.7368, 0.5996, 0.7888, 0.7609, 0.5913, 0.8778, 0.4462, 0.7460, 0.7240, 0.5705, 0.6267, 0.5684, 0.5707, 0.6560, 0.5310, 0.5278, 0.6833, 0.6420, 0.6696, 0.8815, 0.4767, 0.7171, 0.4826, 0.6736, 0.5483, 0.4913, 0.5840, 0.5242, 0.4310, 0.5846, 0.4389, 0.5164, 0.6203, 0.5625, 0.8495, 0.5091, 0.6904, 0.5490, 0.5467, 0.4746, 0.8446, 0.6030, 0.6563, 1.0108, 0.5633, 0.6324, 0.6339, 0.6269, 1.2128, 0.6877, 0.5998, 0.4763, 0.4979, 0.7968, 0.6549, 1.0234, 0.5385, 0.6164, 0.5485, 0.8526, 0.5776, 0.5292, 0.5716, 0.5458, 0.5332, 0.5264, 0.6239, 0.6668, 0.7481, 0.3929, 0.5932, 0.5741, 0.4433, 0.7519, 0.4940, 0.7438, 0.5315, 0.3895, 0.5528, 0.6656, 0.6665, 0.9897, 0.8098, 0.6000, 0.5226, 1.2953, 0.5624, 0.6416, 0.5880, 0.5828, 0.4779, 0.6721, 0.6273, 0.7918, 0.5498, 0.5262, 0.6396, 0.6185, 0.6117, 0.8871, 0.5688, 0.5335, 0.6402, 0.5994, 0.9472, 0.5072, 0.7688, 0.6257, 0.6548, 0.6070, 0.7646, 0.5362, 0.5151, 0.6852, 0.4533, 0.6976, 0.6170, 0.5700, 0.5819, 0.4350, 0.5755, 0.4902, 0.9396, 0.5110, 0.5461, 0.6380, 1.0192, 0.5009, 0.8211, 0.6223, 0.5970, 0.5465, 0.8314, 0.4997, 0.5066, 0.5824, 0.6241, 0.4910, 0.4849, 0.5292, 0.5357, 0.4856, 0.6120, 0.4212, 0.6712, 0.4599, 0.4625, 0.7568, 0.8765, 0.8095, 0.7385, 0.5748, 0.7405, 0.6474, 0.6466, 0.6481, 0.5660, 0.6876, 0.9852, 0.5923, 0.6319, 0.6818, 0.4716, 0.6599, 0.5343, 0.5384, 0.9786, 0.4421, 0.5543, 1.0386, 0.5640, 0.5990, 0.5060, 0.6141, 0.3880, 0.6767, 0.5753, 0.4797, 0.4623, 0.5802, 0.6813, 0.5792, 0.4790, 0.6855, 0.5186, 0.4890, 0.5740, 0.6117, 0.5177, 0.5032, 0.6367, 0.4555, 0.6749, 0.6680, 0.6878, 0.7425, 0.8106, 0.5460, 1.0575, 0.5022, 0.7639, 0.5132, 0.5433, 0.7702, 0.4572, 0.4274, 0.6779, 0.5277, 0.5634, 0.4814, 0.5491, 0.5790, 0.5750, 0.5573, 0.4652, 0.5240, 0.6244, 0.6247, 0.7397, 0.7107, 0.5964, 0.4891, 0.7089, 0.6531, 0.6979, 0.4630, 0.5348, 0.4308, 0.8983, 0.5416, 0.4521, 0.6261, 0.4931, 0.7247, 0.5689, 0.5254, 0.4913, 0.6307, 0.5586, 0.5804, 0.5692, 0.5211, 0.6549, 0.6069, 0.5216, 0.4617, 0.7538, 0.4234, 0.4868, 0.7661, 1.1726, 0.8879, 0.4984, 0.6142, 0.4203, 0.5944, 0.6758, 0.5682, 0.6554, 0.7316, 0.5552, 0.7454, 0.3907, 0.7559, 0.4752, 0.5638, 0.7824, 0.7995, 0.5728, 0.8546, 0.5663, 0.5545, 0.4785, 1.0497, 0.7177, 0.5461, 0.5134, 0.5432, 0.5964, 0.5879, 0.7046, 0.7501, 0.5707, 0.9907, 0.9337, 0.5682, 0.4887, 0.5970, 0.6229, 0.6501, 0.7529, 0.7062, 0.6775, 0.7286, 0.6250, 0.4521, 0.5357, 0.5479, 0.7957, 0.4596, 0.6440, 0.8665, 0.6024, 0.7485, 0.6478, 0.6483, 0.5785, 0.5500, 0.4802, 0.4465, 0.6829, 0.6890, 0.6180, 0.8767, 0.7419, 0.6193, 0.3918, 0.5888, 0.5440, 0.5146, 0.4297, 0.4410, 0.4894, 0.4422, 0.9614, 0.6290, 0.6717, 0.5415, 0.5442, 0.5862, 0.4967, 0.7102, 1.1356, 0.4818, 0.4557, 0.6403, 0.4971, 0.7491, 0.8534, 0.8754, 0.5308, 0.5591, 0.6415, 0.7715, 0.8137, 0.4898, 0.5460, 0.5476, 0.9199, 0.6195, 0.5949, 0.7990, 0.4444, 0.6199, 0.5166, 0.4646, 0.9060, 0.6261, 0.5149, 0.6533, 0.7420, 0.4830, 0.5314, 0.5503, 0.5777, 0.6284, 0.7288, 0.5743, 0.6041, 0.5674, 0.4661, 0.6211, 0.6172, 0.4094, 0.5787, 0.8089, 0.6061, 0.5882, 0.5498, 0.7239, 0.6387, 0.7910, 0.5267, 0.5569, 0.6382, 0.5492, 0.5444, 0.6476, 0.8666, 0.9807, 0.5594, 0.6814, 0.5467, 0.8900, 0.5321, 0.5516, 1.0188, 0.7193, 0.5044, 0.5717, 0.9741, 0.7856, 0.6849, 0.5604, 1.0236, 0.8399, 0.5065, 0.6475, 0.4055, 0.7975, 0.4454, 0.5726, 0.4489, 0.6851, 0.6504, 0.4737, 0.5995, 0.6226, 0.5917, 0.5394, 0.5240, 0.7863, 0.6008, 0.5330, 0.4760, 0.6163, 0.4679, 0.5712, 0.7180, 0.4908, 1.0175, 0.5942, 0.5170, 0.7534, 0.5569, 0.8764, 0.7314, 0.5474, 0.9083, 0.6677, 0.6286, 0.6759, 0.5397, 0.5748, 0.6215, 0.4800, 0.5206, 0.5591, 0.5884, 0.6291, 0.6633, 0.7693, 0.5104, 0.6564, 0.5489, 0.6270, 0.5935, 0.6236, 0.6108, 0.4794, 0.5974, 0.7061, 0.6686, 0.6512, 0.4998, 0.5933, 0.4956, 0.6610, 0.7542, 0.5869, 0.8418, 0.9938, 0.9021, 0.6323, 0.5777, 0.4343, 0.6098, 0.5338, 0.5906, 0.7783, 0.7423, 0.6426, 0.6236, 0.9643, 0.5780, 1.0100, 1.1266, 0.7556, 0.5229, 0.8272, 0.6900, 0.5175, 0.4124, 0.5741, 0.4516, 0.6266, 0.5630, 0.5275, 0.5692, 0.5075, 0.7549, 0.6359, 0.5804, 0.6680, 0.7558, 0.6250, 0.4314, 0.6496, 0.5479, 0.7524, 0.7088, 0.6644, 0.7214, 0.6450, 0.4467, 0.7789, 0.5168, 0.6297, 0.6242, 0.4410, 0.8372, 0.5758, 0.4997, 0.8915, 0.6473, 0.5974, 0.5293, 0.7941, 0.4605, 0.9110, 0.5919, 0.5139, 0.5003, 0.4500, 0.6182, 0.5807, 0.4562, 0.5618, 0.6794, 0.7201, 0.6143, 0.8797, 0.8171, 0.6225, 0.7453, 0.7611, 0.4696, 1.0906, 0.8825, 0.7207, 0.5523, 0.7120, 0.5194, 0.5321, 1.0233, 0.5618, 0.5410, 0.4300, 0.7191, 0.5373, 0.4795, 0.4450, 0.6546, 0.7965, 0.7454, 0.6264, 0.5576, 0.7710, 0.5527, 0.6586, 0.5177, 0.4858, 0.5005, 0.5372, 0.5766, 0.4508, 0.5238, 0.8275, 0.4104, 0.5535, 0.8077, 0.4460, 0.7125, 0.7166, 0.6107, 0.4561, 0.6620, 0.4635, 0.6397, 0.4391, 0.6880, 0.6801, 0.5627, 0.8076, 0.7918, 1.0309, 0.5832, 0.6152, 0.7971, 0.4539, 0.5846, 0.7248, 0.4455, 0.6318, 0.6118, 0.4552, 0.6757, 0.5354, 0.6566, 0.6728, 0.4383, 0.6899, 1.0565, 0.6028, 0.6937, 0.5518, 0.8039, 0.4296, 0.6068, 0.5736, 0.4923, 0.7643, 0.7391, 0.4975, 0.5006, 0.5674, 0.5170, 0.4835, 0.4286, 0.5667, 0.6109, 0.6465, 0.6281, 0.7791, 0.5174, 0.5058, 0.6196, 0.6593, 0.5999, 0.5012, 0.5414, 0.7151, 0.6546, 0.6790, 0.5412, 0.4801, 0.6561, 1.0082, 0.5567, 0.6362, 0.4540, 0.8812, 0.6893, 0.6420, 0.6078, 0.5117, 0.7079, 0.8240, 0.7587, 0.6344, 0.6848, 0.4633, 0.5352, 0.6077, 0.5436, 0.7223, 0.5001, 0.9734, 0.5155, 0.5549, 0.4711, 0.9038, 0.5415, 1.0173, 0.5001, 0.5290, 0.5228, 0.5619, 0.9670, 0.7854, 0.5350, 0.5183, 0.9770, 0.5547, 0.9710, 0.5050, 0.4584, 0.6438, 0.4854, 0.5949, 0.6611, 0.4676, 0.4815, 0.8837, 0.6425, 0.6257, 0.6896, 0.4465, 0.7492, 0.6293, 0.7096, 0.5578, 0.5117, 0.4909, 0.5773, 0.4800, 0.5488, 0.6336, 0.6863, 0.5035, 0.6682, 0.7245, 0.5524, 0.4594, 0.5816, 0.5698, 0.6140, 0.5816, 0.5242, 0.4088, 0.4358, 0.6426, 0.4777, 0.6115, 0.4383, 0.5957, 0.8423, 0.5353, 0.5407, 0.8497, 0.6962, 0.7542, 0.5981, 0.5121, 0.6232, 0.5306, 0.5416, 0.5217, 0.5437, 0.5349, 0.5111, 0.8627, 0.6092, 0.5850, 0.5851, 0.7203, 0.3688, 0.5063, 0.5650, 0.5444, 0.5657, 0.7461, 0.4447, 0.7153, 0.4738, 0.5730, 0.4605, 0.4905, 0.6253, 0.8114, 0.8273, 0.5052, 0.6180, 0.6496, 0.4037, 0.5635, 0.5212, 0.7652, 0.4872, 0.5764, 0.7834, 0.6888, 0.5313, 0.5379, 0.5710, 0.7474, 0.6535, 0.9660, 0.5257, 0.7157, 0.7150, 0.5430, 0.5331, 0.6820, 0.6872, 0.4904, 0.6592, 0.6256, 0.6107, 0.4939, 0.5986, 0.5172, 0.4583],
+ },
+ "emdb": {
+ "count": 62707,
+ "mean": [-1.1869, 0.1485, 0.1933, -0.6247, 0.0793, 0.5762, 0.1835, -0.2564, 0.1285, 0.3221, 0.0577, 0.1154, -0.0818, -0.2512, 0.9673, -0.5680, 0.5968, -0.2124, -0.0112, -0.5576, 0.5339, -0.1490, 0.3102, -0.4012, -0.0570, 0.6416, 0.9359, -0.2932, 0.8544, 0.1719, -0.4534, 0.1316, 0.8625, 0.3806, 0.4884, 1.0853, -0.3872, -0.2403, -0.4274, 0.1319, -0.3334, 0.6352, 0.5748, -0.8850, -0.4331, 0.3662, -0.3324, 1.3993, -1.5142, -0.3082, -0.5491, -0.1847, 0.0145, -0.0726, 0.0015, -0.0358, -0.2815, -0.4356, -0.3842, 0.1150, 1.1513, 0.6343, -0.7336, -1.1613, 0.1020, -0.1291, 0.1560, 0.4854, -0.4191, 1.6794, 0.4274, 0.4792, 0.3570, 0.0811, 1.0886, 0.0670, 0.5227, 0.1891, 0.1121, 0.1495, -0.2090, -0.2156, -0.2512, -0.9291, 0.1287, -0.0481, 0.6701, -0.4579, 0.2352, -0.1056, 0.5551, 0.4357, 0.8168, 0.6344, -0.6445, -0.1965, 0.5587, 0.3860, -0.2466, -0.1542, 0.6825, 0.5875, -0.5208, 0.1500, -0.3980, 0.2157, 0.8368, -0.1356, -0.3387, 0.1747, 0.1467, 0.2282, -0.1412, 0.6216, -1.8406, 0.0150, 0.2891, 0.0280, 0.0461, 0.8558, 0.2929, -1.3753, -0.5792, 0.2089, -0.3524, -0.1849, -0.0157, 0.4454, -0.5306, 0.8238, -0.3160, 0.3760, 0.8978, -0.1943, -0.9474, -1.7321, -0.0149, 0.2338, 0.6087, -0.4851, 0.5210, -0.4042, -0.5368, -0.6220, 0.1245, 0.3112, 0.6360, -0.1522, 0.0540, -0.2380, -0.8354, 1.7591, 0.5687, 0.1732, 0.7923, -0.5383, -0.3271, -2.0050, -0.5563, 0.2979, 1.6609, 0.7108, -1.0155, 0.3591, 0.0136, -0.4743, -0.5401, -0.0176, 1.3333, -0.2973, -0.1114, -0.1616, 0.1160, 0.1152, 0.0057, 0.2067, 0.3876, -1.5311, 0.0636, 0.4566, -0.2653, 1.0534, -0.4638, 0.2166, 0.8686, -0.1447, 0.5605, -0.3841, 0.7015, 0.0418, 0.0811, -0.6406, -0.2929, -0.6821, 1.3678, 0.7574, 0.8315, 2.0377, 4.9034, -0.0097, 0.0165, 0.3248, 0.2994, 0.0210, 0.2276, -0.6580, -0.6899, 0.1981, -2.3205, 0.0059, -0.9412, -0.3191, 0.0389, -0.4170, 0.3391, -0.1346, 0.1567, 0.1838, -0.4176, -0.2758, 0.1495, -0.2977, 0.0929, 0.7186, 0.1230, 0.8780, -0.1240, -0.7370, -0.7551, 0.3830, 1.0824, 1.4500, -0.1040, 1.4225, 0.0929, 0.4612, 0.5167, -0.7093, -0.4729, 0.2321, 0.4156, -0.0696, -0.0626, 1.3341, -0.2398, 0.8453, 0.4048, 0.1690, 0.0074, -0.0474, 0.4134, 0.2043, -0.5962, 0.1643, -0.3821, 0.3012, -0.5690, 0.0133, 0.1876, -0.0727, 0.2896, 0.3253, 0.0313, 0.5141, -0.0055, -1.2889, -0.0983, -0.3212, -0.4173, -0.0804, 0.2591, -0.4160, -0.4815, 2.2822, -1.0033, -0.9814, 0.5290, 1.7943, -0.4217, -0.0373, -3.3970, 3.3067, 0.1174, -0.1369, 0.3847, -0.6960, -0.8867, -0.3825, -0.0134, -0.4367, -1.0273, -0.0623, 0.1520, 0.3816, -0.6543, -0.0118, -0.3019, -0.1190, 1.0490, 0.6255, 0.8503, 0.9500, -1.1942, 1.6886, -1.3958, 0.9389, 0.2318, -0.0460, 0.1140, -0.2352, -0.5648, 0.0363, -0.5636, 0.0661, -0.8680, -0.1223, -6.5336, 0.2139, -0.2734, 1.1739, 0.6003, 0.2183, 0.2154, -0.5902, -0.2916, -0.2748, 0.0787, 0.9065, -0.9764, -0.2278, 1.6248, 0.7941, -0.5014, 0.2422, -2.1474, 0.7818, 0.4370, 0.1361, -0.3936, -0.7724, 0.0941, -0.5762, 3.2182, -0.1101, 0.2677, -0.0101, -1.1798, -0.0122, -0.8163, 0.1115, -0.1697, -0.1466, -0.3549, 0.5360, -0.5183, 0.7519, 0.7093, -0.5946, 0.2787, 0.4822, -0.2680, 0.0934, 0.1483, 0.6706, -0.1150, -0.1945, -2.6643, 0.2194, -0.5014, -0.5869, 0.1022, 0.1988, -0.2558, 0.3732, -0.0644, 0.6440, -0.7403, -1.0228, 0.8158, 0.9543, -0.1226, -0.0929, 0.2716, 0.7962, -0.5293, 0.1538, -1.2074, -0.5093, 0.2037, 0.2156, -0.4407, 2.6976, -0.3653, 0.0458, -0.0899, -0.7584, 1.8329, -0.5082, -0.4776, -0.0265, -2.9437, -0.1675, 1.2358, 0.1571, -0.5022, -0.6370, 0.4087, -0.9664, 0.3533, 0.0928, -0.5308, 0.4462, 0.2476, 0.0976, -1.8347, 0.0468, -0.9309, -0.3712, -0.8578, -0.0568, 1.7377, -0.1299, -0.7187, 0.9764, 0.6858, 0.4272, -0.9588, 0.1038, 0.2520, -1.3775, 0.1491, -0.8507, 0.7052, 0.6483, 0.2818, -0.3305, -0.5913, -0.0907, -0.2438, -0.1932, -0.0564, -0.0777, -0.0748, 0.6530, 0.2393, 0.4476, 0.3941, -1.7061, 0.8876, 1.1888, 0.1423, 0.1737, 0.1330, 0.1115, 0.1525, -0.3715, 0.4657, -0.4010, -0.3089, 2.0455, -0.9555, 0.5093, 0.1502, -0.0865, -0.7851, -0.5175, 0.1613, 0.8113, 1.1943, 0.0612, 1.7087, -1.1616, -0.3204, 0.4428, 0.6120, -0.2282, 0.0174, -0.3141, -0.0045, 0.2204, 0.3966, 4.1174, -0.1531, 0.4325, -0.0245, -0.0310, 0.6541, 0.2904, 1.9309, -0.5405, 0.8576, 1.0352, -0.3592, -0.1056, -0.0047, 0.7218, 0.2350, 1.8817, 0.7558, -0.1575, -0.0544, 0.0234, 0.5841, 0.0996, -0.0503, 1.4150, 0.2260, 0.9152, 0.0688, 0.5286, 0.5885, 0.4606, -0.9186, 0.0441, 0.5233, 0.5305, -0.9086, 0.3728, 0.6752, 0.5453, -1.1360, 0.0613, -0.2365, 0.8856, -0.0512, -0.2589, -0.7055, -0.8111, 0.1787, 1.0393, -0.2469, -0.0922, 1.1790, -0.3284, 0.0402, 0.0746, -0.1033, -0.7248, -1.3859, -1.0511, 0.2797, 0.2777, -0.0877, 0.0271, 0.0740, -1.5863, -0.7014, 0.3677, -1.6786, -1.0769, 0.5594, 0.2428, -0.2664, 0.3454, -0.0490, -3.3762, 0.2004, 0.1913, -0.6461, 0.7643, -0.1239, 1.6487, 0.4942, -0.3305, -0.5069, -0.2183, 1.1533, -0.4380, 0.0219, -0.6319, 0.6743, 1.0648, 0.0587, -0.0989, -0.0995, 0.3757, 0.1813, 0.2854, 0.4345, -2.2154, 0.3601, -0.6406, -0.1099, 0.3583, -0.3726, 0.2892, 0.5897, 3.4282, -2.8781, 0.8985, 0.1550, 0.1102, 0.8008, -0.0811, -0.4199, 0.3145, -0.3236, -0.2425, -0.4502, 0.2431, 0.8504, 0.4597, 0.6396, 0.0902, 1.3885, 0.1297, -1.1721, -0.3227, -0.4472, 0.2575, 1.6201, -0.5444, 0.8665, -0.9622, 0.0035, -0.5908, 1.6270, 0.0351, -0.3419, 0.0039, 1.1001, -0.3767, -0.2270, 1.3332, 0.3555, 0.0667, -0.5392, -1.3500, -0.0842, 0.2591, -2.8862, 0.3166, 2.3757, 1.1254, -0.5208, -0.7074, -0.8110, 0.3715, 1.3720, -0.7236, -0.0665, 0.2772, -0.2840, -0.3515, -0.4777, 0.3030, 0.5417, 0.7752, -0.0182, 1.1569, -0.1614, 1.6521, -2.2844, -0.9332, -0.1472, 0.6151, -0.5020, -0.0719, 0.3361, -0.2722, -0.1500, 0.5092, -0.0348, -0.6530, -0.4159, -0.6603, -3.6738, 0.1421, -1.1267, 0.4267, 0.0699, 1.6415, 0.1451, 1.3309, 0.7792, 2.1801, -0.0886, 0.4233, 0.2828, 1.3708, -1.2021, -0.2627, -2.1505, 0.7701, -0.0167, -0.0247, 0.4665, -1.5951, -0.9997, -0.1568, -0.1108, 0.1543, -1.0055, 0.0001, -1.0355, 0.8421, -0.0485, -0.3064, 1.2358, -0.0448, 0.4038, -0.7671, 0.3624, -0.6197, 0.7966, -0.2266, 0.1130, -0.5302, -1.5468, 0.0700, -1.1711, -0.3307, 0.0086, -0.0416, 1.2763, -0.0574, 0.0121, -2.6334, -0.3180, -0.1954, 0.3944, 0.0076, 1.2025, -0.5634, 0.9271, 0.4198, 0.3251, -0.0041, 0.5236, -0.5314, 0.0639, -0.8840, -0.2680, 0.4958, 0.7804, 0.2942, -0.1935, -0.1405, -0.5670, 0.9489, 0.5726, -0.2529, 1.8878, -0.7204, -0.0050, -0.2448, 0.1725, 0.4253, 0.0058, 1.0247, -0.2908, -0.3978, -0.0963, 0.2107, 1.3576, 0.3074, 0.5527, -0.0927, 0.1521, 0.6300, -0.1377, -0.0497, 0.0425, -0.2248, -0.1534, 0.5778, 0.0033, 0.1789, 2.4935, 1.3225, -0.8038, -0.8864, 0.1176, -1.0532, -0.2375, 1.4582, -0.1168, 0.0548, 0.4221, -0.3585, 0.4043, -0.4371, 1.3289, -0.3674, -0.4286, -0.1730, 0.0535, 0.1441, 0.2703, 0.3826, 0.5123, -0.0401, -0.1230, -0.3143, -1.7583, 0.2582, 0.3484, 0.5722, 0.8621, 0.4420, 0.4442, -0.2445, 0.0532, -0.8102, -1.4058, -0.6382, -0.5799, -0.2456, -0.0906, -0.3191, -0.3395, -0.4364, -0.5810, -0.7970, 0.0831, -1.1570, -0.2573, -0.0644, -0.7106, 0.1313, 0.1944, -0.2329, 0.1409, -1.2096, 1.0822, 0.5523, 0.2151, -0.1106, -0.1034, -0.4873, 0.6932, 1.0196, -0.0521, 0.0569, -0.8759, 1.0084, 0.6800, 1.0768, -1.2878, -0.1161, 0.0447, 0.1888, -0.2371, -1.0470, -0.4027, -1.4363, 0.1606, -0.8026, -0.0244, -0.2893, -0.4938, -0.6921, 1.0140, -0.4158, -0.5957, 0.3313, -0.2462, 0.7703, 0.3403, -1.5113, -0.1231, -0.3776, 0.3326, 0.1634, 2.1520, 0.7302, -0.0300, -2.8234, 0.4553, -0.4652, -0.3331, -1.0286, 1.2882, -0.2797, -0.4759, 0.1470, -1.0253, -0.8175, 0.6936, -0.3728, -0.4594, 1.0876, 0.6229, -0.0461, -0.4342, -0.1686, -1.3960, 0.5283, 0.4002, 0.8179, 0.4787, -0.7147, -0.5052, -0.2552, 0.2817, -0.4022, -0.5289, 0.0815, -0.4814, -0.5451, -0.1384, -0.4303, -0.4506, -1.9036, 0.6884, 0.1361, 0.2678, -0.0052, 0.0119, -0.1882, 1.0507, 3.1094, -0.5746, 1.3087, -0.1831, -0.1917, 0.0633, 0.5083, -0.1448, -0.0134, 0.5002, 0.2579, 0.7755, 0.1579, 0.4157, -0.2610, -0.4953, 0.1709, 0.4063, 0.2068, 0.2666, -0.7872, 0.5325, 0.4910, -0.1599, 0.4387, 0.9262, 0.9245, 0.5763, 0.9292, -0.4531, -0.5367, -0.4911, 0.2302, -0.4182, 0.7188, 0.0342, -0.2079, 0.1310, 0.5718, -0.0331, 0.1861, -0.1287, -0.0427, 0.8478, 0.7278, -0.5664, -0.5335, 1.3976, 0.1697, 0.6063, -0.0220, 0.4921, -0.1349, -0.0531, -0.2408, -0.3858, -0.2741, 0.2285, -0.5532, 0.2704, -0.2687, -0.2161, -0.1179, -1.5228, -0.3683, -1.3004, 0.2431, -0.3305, 1.6118, -0.0328, 1.1503, 0.5712, -0.0423, 3.4830, -0.2760, 0.6307, -0.0419, 0.1553, -0.5602, 0.2106, -0.2213, -0.4543, 0.3034, 0.9189, 1.5738, 0.5071, 0.2238, -2.2069, 0.4104, 0.6224, 0.2836, -0.1620, -0.3043, -0.4012, 0.2410, -0.6261, -0.2435, 0.0211, -0.2227, -0.2392, -0.3634, -0.9207, 0.2260, 0.0929, 0.8206, -0.3214, -0.2296, 0.1274, -0.8615, 0.2329, 1.1085, -1.0565, 0.2258],
+ "std": [0.9963, 0.6391, 0.4956, 0.6280, 0.7591, 0.5610, 0.8236, 0.7139, 0.7494, 0.5686, 0.5042, 0.3464, 0.4228, 0.4171, 0.3526, 0.3710, 0.6288, 0.4674, 0.4413, 0.4741, 0.6553, 0.4882, 0.3697, 0.5507, 0.4961, 0.3683, 0.5604, 0.5302, 0.6027, 0.3023, 0.4882, 0.5746, 0.5314, 0.5031, 0.6145, 0.5994, 0.4285, 0.6399, 0.5362, 0.5403, 0.4677, 1.2902, 0.6126, 0.4145, 0.5068, 0.4667, 0.4825, 0.4275, 0.4381, 0.6758, 0.4866, 0.4136, 0.5262, 0.5698, 0.6550, 0.6492, 0.3450, 0.5948, 0.4219, 0.4973, 0.4483, 0.4336, 0.7440, 0.4595, 0.4366, 0.3634, 0.4430, 0.6587, 0.5073, 0.3533, 0.7036, 0.7039, 0.5312, 0.4701, 0.7512, 0.4102, 0.4227, 0.4488, 0.4158, 0.4676, 0.4521, 0.4560, 0.3917, 0.4757, 0.4348, 0.6013, 0.6715, 0.5179, 0.4834, 0.7451, 0.4845, 0.8893, 0.4188, 0.5963, 0.4306, 0.4551, 0.6417, 0.2886, 0.5378, 0.4316, 0.7568, 0.4818, 0.5494, 0.4736, 0.5841, 0.5043, 0.4265, 0.6994, 0.7652, 0.4344, 0.3931, 0.7198, 0.4169, 0.5794, 0.6720, 0.5694, 0.8603, 0.5307, 0.5893, 0.5763, 0.5292, 0.5228, 0.4156, 0.4901, 0.8334, 0.4574, 0.7241, 0.5346, 0.4063, 0.4147, 0.4979, 0.6599, 0.4173, 0.3715, 0.3828, 0.4492, 0.5576, 0.4060, 0.4353, 0.5315, 0.9834, 0.5548, 0.5679, 0.3506, 0.5419, 0.4256, 0.4187, 0.3570, 0.6316, 0.5870, 0.4832, 0.4862, 0.6072, 0.6781, 0.6152, 0.6708, 0.5008, 0.4435, 0.4229, 0.4973, 0.4301, 0.5363, 0.5478, 0.5388, 0.3952, 0.5961, 0.4721, 0.6389, 0.4450, 0.4841, 0.5594, 0.5234, 0.5224, 0.6326, 0.4469, 0.7397, 0.9551, 0.8426, 0.7576, 0.3893, 0.4382, 0.5222, 0.5234, 0.6035, 0.5764, 0.4043, 0.4741, 0.5471, 0.4229, 0.5962, 0.7127, 0.6205, 0.5671, 0.3766, 0.7455, 0.7315, 0.5891, 0.5372, 0.5957, 0.5342, 0.4010, 0.4453, 0.4609, 0.8789, 0.4353, 0.6297, 0.4126, 0.4149, 0.4597, 0.4859, 0.6733, 0.6096, 0.5719, 0.4494, 0.6353, 0.7537, 0.4643, 0.4577, 0.6485, 0.6069, 0.3603, 0.5821, 0.4807, 0.5192, 0.5329, 0.4153, 0.7329, 0.5444, 0.5742, 0.4593, 0.4003, 0.6770, 0.5428, 0.6781, 0.7920, 0.5037, 0.7615, 0.4537, 0.5931, 0.7333, 0.4880, 0.5469, 0.4698, 0.4917, 0.6256, 0.4947, 0.3974, 0.7559, 0.5916, 0.6547, 0.7502, 0.4682, 0.4517, 0.4888, 0.6472, 0.4755, 0.3927, 0.5845, 0.4135, 0.4091, 0.5860, 0.4544, 0.4051, 0.5547, 0.5322, 0.7200, 0.4595, 0.5484, 0.4758, 0.5259, 0.4137, 0.7149, 0.5638, 0.6221, 0.9309, 0.5637, 0.5657, 0.5711, 0.5651, 1.0484, 0.4435, 0.4587, 0.3716, 0.4108, 0.8114, 0.5531, 1.0675, 0.5825, 0.3841, 0.4500, 0.7335, 0.4767, 0.4162, 0.5679, 0.4880, 0.4614, 0.5118, 0.5198, 0.5619, 0.6869, 0.3536, 0.5128, 0.4722, 0.3722, 0.7705, 0.4556, 0.5365, 0.4999, 0.3254, 0.5268, 0.7580, 0.5932, 0.9908, 0.6171, 0.4912, 0.4439, 0.9135, 0.4658, 0.6566, 0.5500, 0.5423, 0.4725, 0.5415, 0.5550, 0.7519, 0.4220, 0.6024, 0.4821, 0.5268, 0.4583, 0.7421, 0.5200, 0.4541, 0.5197, 0.4562, 0.8381, 0.4423, 0.7400, 0.6578, 0.6459, 0.5316, 0.6877, 0.5362, 0.4215, 0.6455, 0.4363, 0.6716, 0.5795, 0.5587, 0.5234, 0.4456, 0.4991, 0.4244, 0.8959, 0.4744, 0.4440, 0.4437, 0.8485, 0.4237, 0.6907, 0.5582, 0.4315, 0.5458, 0.7341, 0.4731, 0.5065, 0.6181, 0.5643, 0.4407, 0.4353, 0.4732, 0.3769, 0.4162, 0.5028, 0.3689, 0.6656, 0.4598, 0.3735, 0.6801, 0.7902, 0.7101, 0.6292, 0.5732, 0.7452, 0.6803, 0.5065, 0.5261, 0.4644, 0.5021, 0.6714, 0.5226, 0.4455, 0.7599, 0.4380, 0.5468, 0.4595, 0.5308, 0.8445, 0.4413, 0.5196, 0.9241, 0.5414, 0.5018, 0.3832, 0.4950, 0.3185, 0.5330, 0.4844, 0.4481, 0.4517, 0.5104, 0.6092, 0.5712, 0.4164, 0.6590, 0.4888, 0.3930, 0.5419, 0.5486, 0.5165, 0.4390, 0.5542, 0.3883, 0.4074, 0.6213, 0.6185, 0.7711, 0.6565, 0.4925, 1.0624, 0.4690, 0.7498, 0.5333, 0.5290, 0.6258, 0.4473, 0.3862, 0.6571, 0.4873, 0.5240, 0.4127, 0.4445, 0.5094, 0.4754, 0.5769, 0.4786, 0.4510, 0.5130, 0.4897, 0.7568, 0.7398, 0.5718, 0.4229, 0.4929, 0.7470, 0.5901, 0.3772, 0.4914, 0.4074, 0.9471, 0.4967, 0.4323, 0.5259, 0.3591, 0.7202, 0.6012, 0.4573, 0.4296, 0.5578, 0.5218, 0.4640, 0.4522, 0.4029, 0.8071, 0.6086, 0.4832, 0.4202, 0.6781, 0.3862, 0.3920, 0.7543, 1.0257, 0.8849, 0.4181, 0.4722, 0.4069, 0.4854, 0.5405, 0.4676, 0.5547, 0.6282, 0.4275, 0.8011, 0.3308, 0.7135, 0.4315, 0.4915, 0.6616, 0.7376, 0.5742, 0.7461, 0.5443, 0.4749, 0.4906, 1.0020, 0.6306, 0.4435, 0.4559, 0.4360, 0.4047, 0.5802, 0.6109, 0.7836, 0.5163, 0.9777, 0.9272, 0.4618, 0.3534, 0.5218, 0.4479, 0.6498, 0.7145, 0.6224, 0.5671, 0.5042, 0.3885, 0.4079, 0.4481, 0.5406, 0.6944, 0.3744, 0.5942, 0.6770, 0.5934, 0.7417, 0.5662, 0.4753, 0.5063, 0.5003, 0.4510, 0.4358, 0.6455, 0.7740, 0.4780, 0.8687, 0.5533, 0.5700, 0.3518, 0.4868, 0.4154, 0.4798, 0.3266, 0.3536, 0.3789, 0.3805, 0.7909, 0.5760, 0.5784, 0.4993, 0.5787, 0.5324, 0.4496, 0.8483, 1.0794, 0.4820, 0.4135, 0.6231, 0.4668, 0.6684, 0.7052, 0.7616, 0.4881, 0.4150, 0.5793, 0.8068, 0.7793, 0.4721, 0.5230, 0.4810, 0.9577, 0.5537, 0.5583, 0.6645, 0.4334, 0.6398, 0.5011, 0.4081, 0.6255, 0.5372, 0.4846, 0.6125, 0.6509, 0.4413, 0.4762, 0.4917, 0.5940, 0.4950, 0.6753, 0.6653, 0.5210, 0.5599, 0.4678, 0.4868, 0.5985, 0.4160, 0.4874, 0.8380, 0.5382, 0.5701, 0.5448, 0.6131, 0.5674, 0.7120, 0.4070, 0.4434, 0.5725, 0.4919, 0.4805, 0.5997, 0.7108, 0.9824, 0.4765, 0.7575, 0.4452, 0.8892, 0.4639, 0.4962, 1.0346, 0.7584, 0.4312, 0.4835, 0.8968, 0.4799, 0.6864, 0.5641, 1.0694, 0.6750, 0.4288, 0.5159, 0.3649, 0.7699, 0.4386, 0.4449, 0.3923, 0.6499, 0.5612, 0.4541, 0.6261, 0.5444, 0.4369, 0.4124, 0.4174, 0.6129, 0.5005, 0.4779, 0.3929, 0.4865, 0.4338, 0.4114, 0.6266, 0.3669, 1.0147, 0.4856, 0.4867, 0.6250, 0.5368, 0.6699, 0.6411, 0.5296, 0.7614, 0.5643, 0.5843, 0.6846, 0.3923, 0.3928, 0.4964, 0.4490, 0.4755, 0.4104, 0.5468, 0.6040, 0.5808, 0.6283, 0.4316, 0.6127, 0.4635, 0.5303, 0.4261, 0.4668, 0.6121, 0.4063, 0.5571, 0.6130, 0.5874, 0.4987, 0.4113, 0.5401, 0.4028, 0.6598, 0.7740, 0.5384, 0.7890, 0.9379, 0.8801, 0.6222, 0.5356, 0.3990, 0.4802, 0.4107, 0.5475, 0.6936, 0.6865, 0.4776, 0.5211, 0.8844, 0.6517, 1.0729, 0.9252, 0.6953, 0.4177, 0.7587, 0.6628, 0.3629, 0.3685, 0.3758, 0.4439, 0.5236, 0.4905, 0.5290, 0.4184, 0.3940, 0.6498, 0.5411, 0.5662, 0.5519, 0.6107, 0.6385, 0.4127, 0.6277, 0.5255, 0.5926, 0.5653, 0.6570, 0.6034, 0.5312, 0.4128, 0.7292, 0.3620, 0.5067, 0.5314, 0.3908, 0.7561, 0.4494, 0.4501, 0.7682, 0.4939, 0.4198, 0.5256, 0.6339, 0.5123, 0.9018, 0.5054, 0.4879, 0.4567, 0.4145, 0.6046, 0.3835, 0.4289, 0.5254, 0.6191, 0.6610, 0.5933, 0.7890, 0.7817, 0.6299, 0.5977, 0.7094, 0.3737, 1.0318, 0.7045, 0.7785, 0.5376, 0.5861, 0.4233, 0.5538, 1.0604, 0.5690, 0.5249, 0.3747, 0.6036, 0.4707, 0.3617, 0.3665, 0.6184, 0.4878, 0.6193, 0.5311, 0.6187, 0.6748, 0.4493, 0.6137, 0.4601, 0.3855, 0.4183, 0.4986, 0.4832, 0.4192, 0.4416, 0.7202, 0.3724, 0.4899, 0.6939, 0.4272, 0.7122, 0.6950, 0.5565, 0.4417, 0.6186, 0.4753, 0.5919, 0.3763, 0.5643, 0.5347, 0.5454, 0.9336, 0.6594, 0.9747, 0.4970, 0.4725, 0.7820, 0.4113, 0.4942, 0.6699, 0.4159, 0.6766, 0.6564, 0.3947, 0.5381, 0.3874, 0.6686, 0.5628, 0.3904, 0.6647, 0.9821, 0.4343, 0.5455, 0.4879, 0.8165, 0.4153, 0.5544, 0.5179, 0.3821, 0.6678, 0.7883, 0.3372, 0.4702, 0.5044, 0.4584, 0.4769, 0.3787, 0.4377, 0.5435, 0.5899, 0.5378, 0.5986, 0.4887, 0.5390, 0.5464, 0.6330, 0.5010, 0.4244, 0.5249, 0.6770, 0.6314, 0.6404, 0.4605, 0.3649, 0.6489, 1.0657, 0.5497, 0.5357, 0.3651, 0.8484, 0.8126, 0.4873, 0.6711, 0.4401, 0.6181, 0.8585, 0.6000, 0.5654, 0.5416, 0.3504, 0.4671, 0.5499, 0.4409, 0.7650, 0.4980, 0.9734, 0.3568, 0.6037, 0.4361, 0.7880, 0.4726, 0.9902, 0.5020, 0.5178, 0.5065, 0.4543, 0.9039, 0.8296, 0.4451, 0.4436, 0.8518, 0.5201, 0.8668, 0.5122, 0.3412, 0.5849, 0.4815, 0.5795, 0.5664, 0.4384, 0.4593, 0.7974, 0.6570, 0.6522, 0.5490, 0.4195, 0.6821, 0.6133, 0.5692, 0.4780, 0.4574, 0.5090, 0.4488, 0.4269, 0.4153, 0.5143, 0.6560, 0.4480, 0.5482, 0.6997, 0.4377, 0.4166, 0.6103, 0.4671, 0.4449, 0.5672, 0.3296, 0.3898, 0.3778, 0.6572, 0.5555, 0.4047, 0.3720, 0.5728, 0.6867, 0.5435, 0.5001, 0.6808, 0.6373, 0.6849, 0.4826, 0.4767, 0.3736, 0.5070, 0.4442, 0.4302, 0.4339, 0.4614, 0.4735, 0.7977, 0.5657, 0.4047, 0.5261, 0.6204, 0.3413, 0.3996, 0.4236, 0.3303, 0.4193, 0.6074, 0.3941, 0.4802, 0.4114, 0.3880, 0.3460, 0.3767, 0.6491, 0.6893, 0.8560, 0.4244, 0.4307, 0.5702, 0.3635, 0.5170, 0.3975, 0.6187, 0.5012, 0.4976, 0.7149, 0.7001, 0.4834, 0.3844, 0.5179, 0.6909, 0.5862, 1.0062, 0.5099, 0.6410, 0.7432, 0.4219, 0.4655, 0.6067, 0.6674, 0.4618, 0.7115, 0.5300, 0.5284, 0.4208, 0.4955, 0.4561, 0.3723],
+ }
+}
+
+cam_angvel = {
+ "emdb_none_test": {
+ "count": 42622,
+ "mean": [1., 0., 0., 0., 1., 0.],
+ "std": [5.5702e-05, 3.2200e-03, 5.6530e-03, 3.2191e-03, 2.4738e-05, 3.3406e-03],
+ },
+ "manual": {
+ "mean": [1., 0., 0., 0., 1., 0.],
+ "std": [0.001, 0.1, 0.1, 0.1, 0.001, 0.1], # manually
+ }
+}
+# fmt:on
+
+# ====== Compose ====== #
+
+
+def compose(targets, sources):
+ if len(sources) == 1:
+ sources = sources * len(targets)
+ mean = []
+ std = []
+ for t, s in zip(targets, sources):
+ mean.extend(t[s]["mean"])
+ std.extend(t[s]["std"])
+ return {"mean": mean, "std": std}
+
+
+DEFAULT_01 = {"mean": [0.0], "std": [1.0]}
+
+MM_V1 = compose(
+ [body_pose_r6d, betas, global_orient_c_r6d, global_orient_gv_r6d, local_transl_vel],
+ ["bedlam"] * 5,
+)
+MM_V1_AMASS_LOCAL_BEDLAM_CAM = compose(
+ [body_pose_r6d, betas, global_orient_c_r6d, global_orient_gv_r6d, local_transl_vel],
+ ["amass", "amass", "bedlam", "bedlam", "amass"],
+)
+
+MM_V2 = compose(
+ [body_pose_r6d, betas, global_orient_c_r6d, global_orient_gv_r6d, local_transl_vel],
+ ["bedlam", "bedlam", "bedlam", "bedlam", "none"],
+)
+
+MM_V2_1 = compose(
+ [body_pose_r6d, betas, global_orient_c_r6d, global_orient_gv_r6d, local_transl_vel],
+ ["bedlam", "bedlam", "bedlam", "bedlam", "1e-2"],
+)
+
+# ====== Unity Dataset Statistics ====== #
+# Computed from Unity processed data (11997 frames)
+
+body_pose_r6d["unity"] = {
+ "count": 11997,
+ "mean": [0.9924, -0.0491, 0.0246, 0.0435, 0.9702, 0.1913, 0.9851, 0.1239, -0.0341, -0.1161, 0.9608, 0.1993, 0.9954, 0.0867, -7.9529e-03, -0.0873, 0.9926, -0.0668, 0.9999, -5.4245e-03, 7.6656e-04, 5.3467e-03, 0.9432, -0.2797, 0.9997, 0.0115, -2.7577e-03, -0.0112, 0.9211, -0.2825, 0.9915, -0.1267, -2.3587e-03, 0.1266, 0.9911, 0.0400, 0.9931, 0.0408, 5.0597e-03, -0.0424, 0.9783, 0.1149, 0.9894, -0.0670, -0.0346, 0.0696, 0.9813, 0.0514, 1.0000, 0.0, 0.0, 0.0, 1.0000, 0.0, 0.9999, -1.3002e-03, -6.7745e-04, 1.4068e-03, 0.9982, 8.8789e-03, 0.9999, 1.9599e-03, 1.0788e-03, -2.1213e-03, 0.9978, 0.0112, 0.9888, 0.0920, 0.0708, -0.1028, 0.9804, 0.1672, 0.9925, 0.0972, 0.0164, -0.0949, 0.9880, -0.0987, 0.9870, -0.1341, -0.0418, 0.1288, 0.9825, -0.1130, 0.9872, -0.0796, 0.0647, 0.0944, 0.9582, -0.2484, 0.3496, 0.8362, -0.1972, -0.7904, 0.4130, 0.2523, 0.4682, -0.7962, 0.2015, 0.7959, 0.4985, 0.1363, 0.8203, -0.0135, -0.2049, 0.0620, 0.9244, -0.0711, 0.7697, 0.0526, 0.2749, -0.0567, 0.9009, -0.1330, 0.9366, -0.0967, -0.1415, 0.0674, 0.9297, -0.1863, 0.9387, 0.0489, 0.0141, -0.0374, 0.9005, -0.2217],
+ "std": [8.3272e-03, 0.0892, 0.0633, 0.0896, 0.0294, 0.1068, 0.0163, 0.0940, 0.0636, 0.0956, 0.0362, 0.1152, 2.0401e-03, 0.0291, 0.0261, 0.0294, 4.3569e-03, 0.0415, 1.8386e-04, 0.0127, 2.9381e-03, 0.0125, 0.0786, 0.1604, 1.1208e-03, 0.0184, 0.0117, 0.0178, 0.1785, 0.1988, 1.3900e-03, 8.0863e-03, 0.0296, 8.1536e-03, 1.1073e-03, 0.0112, 9.6823e-03, 0.0614, 0.0908, 0.0616, 0.0491, 0.1477, 0.0104, 0.0523, 0.1120, 0.0527, 0.0554, 0.1543, 1.0e-06, 1.0e-06, 1.0e-06, 1.0e-06, 1.0e-06, 1.0e-06, 6.9881e-04, 8.9390e-03, 6.1391e-03, 9.0626e-03, 0.0207, 0.0552, 9.8994e-04, 0.0112, 7.4486e-03, 0.0118, 0.0227, 0.0600, 0.0117, 0.0116, 0.0924, 8.1376e-03, 2.2220e-03, 0.0166, 3.8557e-03, 0.0571, 0.0442, 0.0545, 5.3657e-03, 0.0453, 6.2895e-03, 0.0605, 0.0482, 0.0568, 6.1017e-03, 0.0454, 0.0192, 0.0763, 0.0929, 0.0627, 0.0280, 0.0800, 0.2001, 0.2204, 0.2262, 0.2535, 0.2111, 0.1795, 0.1490, 0.1964, 0.2131, 0.2117, 0.1377, 0.1887, 0.3686, 0.1906, 0.3356, 0.2142, 0.1650, 0.2519, 0.4057, 0.2444, 0.3239, 0.1914, 0.1934, 0.3058, 0.0861, 0.2387, 0.1705, 0.2379, 0.0873, 0.1796, 0.1005, 0.2807, 0.1655, 0.2843, 0.1099, 0.2136],
+}
+
+betas["unity"] = {
+ "count": 11997,
+ "mean": [0.1385, -1.9147, -0.5287, 0.6716, 0.0760, 0.8861, -0.3723, 0.7269, -0.4031, 1.0523],
+ "std": [0.01, 0.01, 0.01, 0.01, 0.01, 0.01, 0.01, 0.01, 0.01, 0.01], # Use small std since all same character
+}
+
+global_orient_c_r6d["unity"] = {
+ "count": 11997,
+ "mean": [0.6875, -0.0190, -0.0269, -0.0195, -0.9795, 0.0464],
+ "std": [0.3081, 0.0465, 0.6552, 0.1071, 0.0323, 0.1595],
+}
+
+global_orient_gv_r6d["unity"] = {
+ "count": 11997,
+ "mean": [0.6875, -0.0189, -0.0269, -0.0215, -0.9971, 0.0325],
+ "std": [0.3081, 0.0466, 0.6552, 0.0476, 4.8359e-03, 0.0439],
+}
+
+local_transl_vel["unity"] = {
+ "count": 11997,
+ "mean": [7.4442e-05, -2.7914e-05, 9.1121e-05],
+ "std": [5.7417e-03, 5.2172e-03, 2.4900e-03],
+}
+
+# Unity-specific composition
+MM_UNITY = compose(
+ [body_pose_r6d, betas, global_orient_c_r6d, global_orient_gv_r6d, local_transl_vel],
+ ["unity", "unity", "unity", "unity", "unity"],
+)
diff --git a/third_party/GVHMR/hmr4d/network/base_arch/embeddings/rotary_embedding.py b/third_party/GVHMR/hmr4d/network/base_arch/embeddings/rotary_embedding.py
new file mode 100644
index 0000000000000000000000000000000000000000..78c8f20f15441131dd3c9e60073d9b5832be7e3b
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/base_arch/embeddings/rotary_embedding.py
@@ -0,0 +1,74 @@
+import torch
+import torch.nn as nn
+from einops import repeat, rearrange
+from torch.cuda.amp import autocast
+
+
+def rotate_half(x):
+ x = rearrange(x, "... (d r) -> ... d r", r=2)
+ x1, x2 = x.unbind(dim=-1)
+ x = torch.stack((-x2, x1), dim=-1)
+ return rearrange(x, "... d r -> ... (d r)")
+
+
+@autocast(enabled=False)
+def apply_rotary_emb(freqs, t, start_index=0, scale=1.0, seq_dim=-2):
+ if t.ndim == 3:
+ seq_len = t.shape[seq_dim]
+ freqs = freqs[-seq_len:].to(t)
+
+ rot_dim = freqs.shape[-1]
+ end_index = start_index + rot_dim
+
+ assert (
+ rot_dim <= t.shape[-1]
+ ), f"feature dimension {t.shape[-1]} is not of sufficient size to rotate in all the positions {rot_dim}"
+
+ t_left, t, t_right = t[..., :start_index], t[..., start_index:end_index], t[..., end_index:]
+ t = (t * freqs.cos() * scale) + (rotate_half(t) * freqs.sin() * scale)
+ return torch.cat((t_left, t, t_right), dim=-1)
+
+
+def get_encoding(d_model, max_seq_len=4096):
+ """Return: (L, D)"""
+ t = torch.arange(max_seq_len).float()
+ freqs = 1.0 / (10000 ** (torch.arange(0, d_model, 2).float() / d_model))
+ freqs = torch.einsum("i, j -> i j", t, freqs)
+ freqs = repeat(freqs, "i j -> i (j r)", r=2)
+ return freqs
+
+
+class ROPE(nn.Module):
+ """Minimal impl of a lang-style positional encoding."""
+
+ def __init__(self, d_model, max_seq_len=4096):
+ super().__init__()
+ self.d_model = d_model
+ self.max_seq_len = max_seq_len
+
+ # Pre-cache a freqs tensor
+ encoding = get_encoding(d_model, max_seq_len)
+ self.register_buffer("encoding", encoding, False)
+
+ def rotate_queries_or_keys(self, x):
+ """
+ Args:
+ x : (B, H, L, D)
+ Returns:
+ rotated_x: (B, H, L, D)
+ """
+
+ seq_len, d_model = x.shape[-2:]
+ assert d_model == self.d_model
+
+ # encoding: (L, D)s
+ if seq_len > self.max_seq_len:
+ encoding = get_encoding(d_model, seq_len).to(x)
+ else:
+ encoding = self.encoding[:seq_len]
+
+ # encoding: (L, D)
+ # x: (B, H, L, D)
+ rotated_x = apply_rotary_emb(encoding, x, seq_dim=-2)
+
+ return rotated_x
diff --git a/third_party/GVHMR/hmr4d/network/base_arch/transformer/encoder_rope.py b/third_party/GVHMR/hmr4d/network/base_arch/transformer/encoder_rope.py
new file mode 100644
index 0000000000000000000000000000000000000000..7f1666062a1433ac690fce1f6411c645c9b65d84
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/base_arch/transformer/encoder_rope.py
@@ -0,0 +1,83 @@
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import math
+from timm.models.vision_transformer import Mlp
+from typing import Optional, Tuple
+from einops import einsum, rearrange, repeat
+from hmr4d.network.base_arch.embeddings.rotary_embedding import ROPE
+
+
+class RoPEAttention(nn.Module):
+ def __init__(self, embed_dim, num_heads, dropout=0.1):
+ super().__init__()
+ self.embed_dim = embed_dim
+ self.num_heads = num_heads
+ self.head_dim = embed_dim // num_heads
+
+ self.rope = ROPE(self.head_dim, max_seq_len=4096)
+
+ self.query = nn.Linear(embed_dim, embed_dim)
+ self.key = nn.Linear(embed_dim, embed_dim)
+ self.value = nn.Linear(embed_dim, embed_dim)
+ self.dropout = nn.Dropout(dropout)
+ self.proj = nn.Linear(embed_dim, embed_dim)
+
+ def forward(self, x, attn_mask=None, key_padding_mask=None):
+ # x: (B, L, C)
+ # attn_mask: (L, L)
+ # key_padding_mask: (B, L)
+ B, L, _ = x.shape
+ xq, xk, xv = self.query(x), self.key(x), self.value(x)
+
+ xq = xq.reshape(B, L, self.num_heads, -1).transpose(1, 2)
+ xk = xk.reshape(B, L, self.num_heads, -1).transpose(1, 2)
+ xv = xv.reshape(B, L, self.num_heads, -1).transpose(1, 2)
+
+ xq = self.rope.rotate_queries_or_keys(xq) # B, N, L, C
+ xk = self.rope.rotate_queries_or_keys(xk) # B, N, L, C
+
+ attn_score = einsum(xq, xk, "b n i c, b n j c -> b n i j") / math.sqrt(self.head_dim)
+ if attn_mask is not None:
+ attn_mask = attn_mask.reshape(1, 1, L, L).expand(B, self.num_heads, -1, -1)
+ attn_score = attn_score.masked_fill(attn_mask, float("-inf"))
+ if key_padding_mask is not None:
+ key_padding_mask = key_padding_mask.reshape(B, 1, 1, L).expand(-1, self.num_heads, L, -1)
+ attn_score = attn_score.masked_fill(key_padding_mask, float("-inf"))
+
+ attn_score = torch.softmax(attn_score, dim=-1)
+ attn_score = self.dropout(attn_score)
+ output = einsum(attn_score, xv, "b n i j, b n j c -> b n i c") # B, N, L, C
+ output = output.transpose(1, 2).reshape(B, L, -1) # B, L, C
+ output = self.proj(output) # B, L, C
+ return output
+
+
+class EncoderRoPEBlock(nn.Module):
+ def __init__(self, hidden_size, num_heads, mlp_ratio=4.0, dropout=0.1, **block_kwargs):
+ super().__init__()
+ self.norm1 = nn.LayerNorm(hidden_size, elementwise_affine=True, eps=1e-6)
+ self.attn = RoPEAttention(hidden_size, num_heads, dropout)
+ self.norm2 = nn.LayerNorm(hidden_size, elementwise_affine=True, eps=1e-6)
+ mlp_hidden_dim = int(hidden_size * mlp_ratio)
+ approx_gelu = lambda: nn.GELU(approximate="tanh")
+ self.mlp = Mlp(in_features=hidden_size, hidden_features=mlp_hidden_dim, act_layer=approx_gelu, drop=dropout)
+
+ self.gate_msa = nn.Parameter(torch.zeros(1, 1, hidden_size))
+ self.gate_mlp = nn.Parameter(torch.zeros(1, 1, hidden_size))
+
+ # Zero-out adaLN modulation layers
+ nn.init.constant_(self.gate_msa, 0)
+ nn.init.constant_(self.gate_mlp, 0)
+
+ def forward(self, x, attn_mask=None, tgt_key_padding_mask=None):
+ x = x + self.gate_msa * self._sa_block(
+ self.norm1(x), attn_mask=attn_mask, key_padding_mask=tgt_key_padding_mask
+ )
+ x = x + self.gate_mlp * self.mlp(self.norm2(x))
+ return x
+
+ def _sa_block(self, x, attn_mask=None, key_padding_mask=None):
+ # x: (B, L, C)
+ x = self.attn(x, attn_mask=attn_mask, key_padding_mask=key_padding_mask)
+ return x
diff --git a/third_party/GVHMR/hmr4d/network/base_arch/transformer/layer.py b/third_party/GVHMR/hmr4d/network/base_arch/transformer/layer.py
new file mode 100644
index 0000000000000000000000000000000000000000..1d4e75c812318ab042507ee06a453d07ad782162
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/base_arch/transformer/layer.py
@@ -0,0 +1,12 @@
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+
+
+def zero_module(module):
+ """
+ Zero out the parameters of a module and return it.
+ """
+ for p in module.parameters():
+ p.detach().zero_()
+ return module
diff --git a/third_party/GVHMR/hmr4d/network/gvhmr/relative_transformer.py b/third_party/GVHMR/hmr4d/network/gvhmr/relative_transformer.py
new file mode 100644
index 0000000000000000000000000000000000000000..e21df80993835408aa6443fb904a0192157f01be
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/gvhmr/relative_transformer.py
@@ -0,0 +1,194 @@
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from einops import einsum, rearrange, repeat
+from hmr4d.configs import MainStore, builds
+
+from hmr4d.network.base_arch.transformer.encoder_rope import EncoderRoPEBlock
+from hmr4d.network.base_arch.transformer.layer import zero_module
+
+from hmr4d.utils.net_utils import length_to_mask
+from timm.models.vision_transformer import Mlp
+
+
+class NetworkEncoderRoPE(nn.Module):
+ def __init__(
+ self,
+ # x
+ output_dim=151,
+ max_len=120,
+ # condition
+ cliffcam_dim=3,
+ cam_angvel_dim=6,
+ imgseq_dim=1024,
+ # intermediate
+ latent_dim=512,
+ num_layers=12,
+ num_heads=8,
+ mlp_ratio=4.0,
+ # output
+ pred_cam_dim=3,
+ static_conf_dim=6,
+ # training
+ dropout=0.1,
+ # other
+ avgbeta=True,
+ ):
+ super().__init__()
+
+ # input
+ self.output_dim = output_dim
+ self.max_len = max_len
+
+ # condition
+ self.cliffcam_dim = cliffcam_dim
+ self.cam_angvel_dim = cam_angvel_dim
+ self.imgseq_dim = imgseq_dim
+
+ # intermediate
+ self.latent_dim = latent_dim
+ self.num_layers = num_layers
+ self.num_heads = num_heads
+ self.dropout = dropout
+
+ # ===== build model ===== #
+ # Input (Kp2d)
+ # Main token: map d_obs 2 to 32
+ self.learned_pos_linear = nn.Linear(2, 32)
+ self.learned_pos_params = nn.Parameter(torch.randn(17, 32), requires_grad=True)
+ self.embed_noisyobs = Mlp(
+ 17 * 32, hidden_features=self.latent_dim * 2, out_features=self.latent_dim, drop=dropout
+ )
+
+ self._build_condition_embedder()
+
+ # Transformer
+ self.blocks = nn.ModuleList(
+ [
+ EncoderRoPEBlock(self.latent_dim, self.num_heads, mlp_ratio=mlp_ratio, dropout=dropout)
+ for _ in range(self.num_layers)
+ ]
+ )
+
+ # Output heads
+ self.final_layer = Mlp(self.latent_dim, out_features=self.output_dim)
+ self.pred_cam_head = pred_cam_dim > 0 # keep extra_output for easy-loading old ckpt
+ if self.pred_cam_head:
+ self.pred_cam_head = Mlp(self.latent_dim, out_features=pred_cam_dim)
+ self.register_buffer("pred_cam_mean", torch.tensor([1.0606, -0.0027, 0.2702]), False)
+ self.register_buffer("pred_cam_std", torch.tensor([0.1784, 0.0956, 0.0764]), False)
+
+ self.static_conf_head = static_conf_dim > 0
+ if self.static_conf_head:
+ self.static_conf_head = Mlp(self.latent_dim, out_features=static_conf_dim)
+
+ self.avgbeta = avgbeta
+
+ def _build_condition_embedder(self):
+ latent_dim = self.latent_dim
+ dropout = self.dropout
+ self.cliffcam_embedder = nn.Sequential(
+ nn.Linear(self.cliffcam_dim, latent_dim),
+ nn.SiLU(),
+ nn.Dropout(dropout),
+ zero_module(nn.Linear(latent_dim, latent_dim)),
+ )
+ if self.cam_angvel_dim > 0:
+ self.cam_angvel_embedder = nn.Sequential(
+ nn.Linear(self.cam_angvel_dim, latent_dim),
+ nn.SiLU(),
+ nn.Dropout(dropout),
+ zero_module(nn.Linear(latent_dim, latent_dim)),
+ )
+ if self.imgseq_dim > 0:
+ self.imgseq_embedder = nn.Sequential(
+ nn.LayerNorm(self.imgseq_dim),
+ zero_module(nn.Linear(self.imgseq_dim, latent_dim)),
+ )
+
+ def forward(self, length, obs=None, f_cliffcam=None, f_cam_angvel=None, f_imgseq=None):
+ """
+ Args:
+ x: None we do not use it
+ timesteps: (B,)
+ length: (B), valid length of x, if None then use x.shape[2]
+ f_imgseq: (B, L, C)
+ f_cliffcam: (B, L, 3), CLIFF-Cam parameters (bbx-detection in the full-image)
+ f_noisyobs: (B, L, C), nosiy pose observation
+ f_cam_angvel: (B, L, 6), Camera angular velocity
+ """
+ B, L, J, C = obs.shape
+ assert J == 17 and C == 3
+
+ # Main token from observation (2D pose)
+ obs = obs.clone()
+ visible_mask = obs[..., [2]] > 0.5 # (B, L, J, 1)
+ obs[~visible_mask[..., 0]] = 0 # set low-conf to all zeros
+ f_obs = self.learned_pos_linear(obs[..., :2]) # (B, L, J, 32)
+ f_obs = f_obs * visible_mask + self.learned_pos_params.repeat(B, L, 1, 1) * ~visible_mask
+ x = self.embed_noisyobs(f_obs.view(B, L, -1)) # (B, L, J*32) -> (B, L, C)
+
+ # Condition
+ f_to_add = []
+ f_to_add.append(self.cliffcam_embedder(f_cliffcam))
+ if hasattr(self, "cam_angvel_embedder"):
+ f_to_add.append(self.cam_angvel_embedder(f_cam_angvel))
+ if f_imgseq is not None and hasattr(self, "imgseq_embedder"):
+ f_to_add.append(self.imgseq_embedder(f_imgseq))
+
+ for f_delta in f_to_add:
+ x = x + f_delta
+
+ # Setup length and make padding mask
+ assert B == length.size(0)
+ pmask = ~length_to_mask(length, L) # (B, L)
+
+ if L > self.max_len:
+ attnmask = torch.ones((L, L), device=x.device, dtype=torch.bool)
+ for i in range(L):
+ min_ind = max(0, i - self.max_len // 2)
+ max_ind = min(L, i + self.max_len // 2)
+ max_ind = max(self.max_len, max_ind)
+ min_ind = min(L - self.max_len, min_ind)
+ attnmask[i, min_ind:max_ind] = False
+ else:
+ attnmask = None
+
+ # Transformer
+ for block in self.blocks:
+ x = block(x, attn_mask=attnmask, tgt_key_padding_mask=pmask)
+
+ # Output
+ sample = self.final_layer(x) # (B, L, C)
+ if self.avgbeta:
+ betas = (sample[..., 126:136] * (~pmask[..., None])).sum(1) / length[:, None] # (B, C)
+ betas = repeat(betas, "b c -> b l c", l=L)
+ sample = torch.cat([sample[..., :126], betas, sample[..., 136:]], dim=-1)
+
+ # Output (extra)
+ pred_cam = None
+ if self.pred_cam_head:
+ pred_cam = self.pred_cam_head(x)
+ pred_cam = pred_cam * self.pred_cam_std + self.pred_cam_mean
+ torch.clamp_min_(pred_cam[..., 0], 0.25) # min_clamp s to 0.25 (prevent negative prediction)
+
+ static_conf_logits = None
+ if self.static_conf_head:
+ static_conf_logits = self.static_conf_head(x) # (B, L, C')
+
+ output = {
+ "pred_context": x,
+ "pred_x": sample,
+ "pred_cam": pred_cam,
+ "static_conf_logits": static_conf_logits,
+ }
+ return output
+
+
+# Add to MainStore
+group_name = "network/gvhmr"
+MainStore.store(
+ name="relative_transformer",
+ node=builds(NetworkEncoderRoPE, populate_full_signature=True),
+ group=group_name,
+)
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/__init__.py b/third_party/GVHMR/hmr4d/network/hmr2/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..4be998b4311d88eed89b1a78997ca6eaf0415846
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/__init__.py
@@ -0,0 +1,32 @@
+import torch
+from .hmr2 import HMR2
+from pathlib import Path
+from .configs import get_config
+from hmr4d import PROJ_ROOT
+
+HMR2A_CKPT = PROJ_ROOT / f"inputs/checkpoints/hmr2/epoch=10-step=25000.ckpt" # this is HMR2.0a, follow WHAM
+
+
+def load_hmr2(checkpoint_path=HMR2A_CKPT):
+ model_cfg = str((Path(__file__).parent / "configs/model_config.yaml").resolve())
+ model_cfg = get_config(model_cfg)
+
+ # Override some config values, to crop bbox correctly
+ if (model_cfg.MODEL.BACKBONE.TYPE == "vit") and ("BBOX_SHAPE" not in model_cfg.MODEL):
+ model_cfg.defrost()
+ assert (
+ model_cfg.MODEL.IMAGE_SIZE == 256
+ ), f"MODEL.IMAGE_SIZE ({model_cfg.MODEL.IMAGE_SIZE}) should be 256 for ViT backbone"
+ model_cfg.MODEL.BBOX_SHAPE = [192, 256] # (W, H)
+ model_cfg.freeze()
+
+ # Setup model and Load weights.
+ # model = HMR2.load_from_checkpoint(checkpoint_path, strict=False, cfg=model_cfg)
+ model = HMR2(model_cfg)
+
+ state_dict = torch.load(checkpoint_path, map_location="cpu")["state_dict"]
+ keys = [k for k in state_dict.keys() if k.split(".")[0] in ["backbone", "smpl_head"]]
+ state_dict = {k: v for k, v in state_dict.items() if k in keys}
+ model.load_state_dict(state_dict, strict=True)
+
+ return model
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/components/__init__.py b/third_party/GVHMR/hmr4d/network/hmr2/components/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/components/pose_transformer.py b/third_party/GVHMR/hmr4d/network/hmr2/components/pose_transformer.py
new file mode 100644
index 0000000000000000000000000000000000000000..ac04971407cb59637490cc4842f048b9bc4758be
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/components/pose_transformer.py
@@ -0,0 +1,358 @@
+from inspect import isfunction
+from typing import Callable, Optional
+
+import torch
+from einops import rearrange
+from einops.layers.torch import Rearrange
+from torch import nn
+
+from .t_cond_mlp import (
+ AdaptiveLayerNorm1D,
+ FrequencyEmbedder,
+ normalization_layer,
+)
+# from .vit import Attention, FeedForward
+
+
+def exists(val):
+ return val is not None
+
+
+def default(val, d):
+ if exists(val):
+ return val
+ return d() if isfunction(d) else d
+
+
+class PreNorm(nn.Module):
+ def __init__(self, dim: int, fn: Callable, norm: str = "layer", norm_cond_dim: int = -1):
+ super().__init__()
+ self.norm = normalization_layer(norm, dim, norm_cond_dim)
+ self.fn = fn
+
+ def forward(self, x: torch.Tensor, *args, **kwargs):
+ if isinstance(self.norm, AdaptiveLayerNorm1D):
+ return self.fn(self.norm(x, *args), **kwargs)
+ else:
+ return self.fn(self.norm(x), **kwargs)
+
+
+class FeedForward(nn.Module):
+ def __init__(self, dim, hidden_dim, dropout=0.0):
+ super().__init__()
+ self.net = nn.Sequential(
+ nn.Linear(dim, hidden_dim),
+ nn.GELU(),
+ nn.Dropout(dropout),
+ nn.Linear(hidden_dim, dim),
+ nn.Dropout(dropout),
+ )
+
+ def forward(self, x):
+ return self.net(x)
+
+
+class Attention(nn.Module):
+ def __init__(self, dim, heads=8, dim_head=64, dropout=0.0):
+ super().__init__()
+ inner_dim = dim_head * heads
+ project_out = not (heads == 1 and dim_head == dim)
+
+ self.heads = heads
+ self.scale = dim_head**-0.5
+
+ self.attend = nn.Softmax(dim=-1)
+ self.dropout = nn.Dropout(dropout)
+
+ self.to_qkv = nn.Linear(dim, inner_dim * 3, bias=False)
+
+ self.to_out = (
+ nn.Sequential(nn.Linear(inner_dim, dim), nn.Dropout(dropout))
+ if project_out
+ else nn.Identity()
+ )
+
+ def forward(self, x):
+ qkv = self.to_qkv(x).chunk(3, dim=-1)
+ q, k, v = map(lambda t: rearrange(t, "b n (h d) -> b h n d", h=self.heads), qkv)
+
+ dots = torch.matmul(q, k.transpose(-1, -2)) * self.scale
+
+ attn = self.attend(dots)
+ attn = self.dropout(attn)
+
+ out = torch.matmul(attn, v)
+ out = rearrange(out, "b h n d -> b n (h d)")
+ return self.to_out(out)
+
+
+class CrossAttention(nn.Module):
+ def __init__(self, dim, context_dim=None, heads=8, dim_head=64, dropout=0.0):
+ super().__init__()
+ inner_dim = dim_head * heads
+ project_out = not (heads == 1 and dim_head == dim)
+
+ self.heads = heads
+ self.scale = dim_head**-0.5
+
+ self.attend = nn.Softmax(dim=-1)
+ self.dropout = nn.Dropout(dropout)
+
+ context_dim = default(context_dim, dim)
+ self.to_kv = nn.Linear(context_dim, inner_dim * 2, bias=False)
+ self.to_q = nn.Linear(dim, inner_dim, bias=False)
+
+ self.to_out = (
+ nn.Sequential(nn.Linear(inner_dim, dim), nn.Dropout(dropout))
+ if project_out
+ else nn.Identity()
+ )
+
+ def forward(self, x, context=None):
+ context = default(context, x)
+ k, v = self.to_kv(context).chunk(2, dim=-1)
+ q = self.to_q(x)
+ q, k, v = map(lambda t: rearrange(t, "b n (h d) -> b h n d", h=self.heads), [q, k, v])
+
+ dots = torch.matmul(q, k.transpose(-1, -2)) * self.scale
+
+ attn = self.attend(dots)
+ attn = self.dropout(attn)
+
+ out = torch.matmul(attn, v)
+ out = rearrange(out, "b h n d -> b n (h d)")
+ return self.to_out(out)
+
+
+class Transformer(nn.Module):
+ def __init__(
+ self,
+ dim: int,
+ depth: int,
+ heads: int,
+ dim_head: int,
+ mlp_dim: int,
+ dropout: float = 0.0,
+ norm: str = "layer",
+ norm_cond_dim: int = -1,
+ ):
+ super().__init__()
+ self.layers = nn.ModuleList([])
+ for _ in range(depth):
+ sa = Attention(dim, heads=heads, dim_head=dim_head, dropout=dropout)
+ ff = FeedForward(dim, mlp_dim, dropout=dropout)
+ self.layers.append(
+ nn.ModuleList(
+ [
+ PreNorm(dim, sa, norm=norm, norm_cond_dim=norm_cond_dim),
+ PreNorm(dim, ff, norm=norm, norm_cond_dim=norm_cond_dim),
+ ]
+ )
+ )
+
+ def forward(self, x: torch.Tensor, *args):
+ for attn, ff in self.layers:
+ x = attn(x, *args) + x
+ x = ff(x, *args) + x
+ return x
+
+
+class TransformerCrossAttn(nn.Module):
+ def __init__(
+ self,
+ dim: int,
+ depth: int,
+ heads: int,
+ dim_head: int,
+ mlp_dim: int,
+ dropout: float = 0.0,
+ norm: str = "layer",
+ norm_cond_dim: int = -1,
+ context_dim: Optional[int] = None,
+ ):
+ super().__init__()
+ self.layers = nn.ModuleList([])
+ for _ in range(depth):
+ sa = Attention(dim, heads=heads, dim_head=dim_head, dropout=dropout)
+ ca = CrossAttention(
+ dim, context_dim=context_dim, heads=heads, dim_head=dim_head, dropout=dropout
+ )
+ ff = FeedForward(dim, mlp_dim, dropout=dropout)
+ self.layers.append(
+ nn.ModuleList(
+ [
+ PreNorm(dim, sa, norm=norm, norm_cond_dim=norm_cond_dim),
+ PreNorm(dim, ca, norm=norm, norm_cond_dim=norm_cond_dim),
+ PreNorm(dim, ff, norm=norm, norm_cond_dim=norm_cond_dim),
+ ]
+ )
+ )
+
+ def forward(self, x: torch.Tensor, *args, context=None, context_list=None):
+ if context_list is None:
+ context_list = [context] * len(self.layers)
+ if len(context_list) != len(self.layers):
+ raise ValueError(f"len(context_list) != len(self.layers) ({len(context_list)} != {len(self.layers)})")
+
+ for i, (self_attn, cross_attn, ff) in enumerate(self.layers):
+ x = self_attn(x, *args) + x
+ x = cross_attn(x, *args, context=context_list[i]) + x
+ x = ff(x, *args) + x
+ return x
+
+
+class DropTokenDropout(nn.Module):
+ def __init__(self, p: float = 0.1):
+ super().__init__()
+ if p < 0 or p > 1:
+ raise ValueError(
+ "dropout probability has to be between 0 and 1, " "but got {}".format(p)
+ )
+ self.p = p
+
+ def forward(self, x: torch.Tensor):
+ # x: (batch_size, seq_len, dim)
+ if self.training and self.p > 0:
+ zero_mask = torch.full_like(x[0, :, 0], self.p).bernoulli().bool()
+ # TODO: permutation idx for each batch using torch.argsort
+ if zero_mask.any():
+ x = x[:, ~zero_mask, :]
+ return x
+
+
+class ZeroTokenDropout(nn.Module):
+ def __init__(self, p: float = 0.1):
+ super().__init__()
+ if p < 0 or p > 1:
+ raise ValueError(
+ "dropout probability has to be between 0 and 1, " "but got {}".format(p)
+ )
+ self.p = p
+
+ def forward(self, x: torch.Tensor):
+ # x: (batch_size, seq_len, dim)
+ if self.training and self.p > 0:
+ zero_mask = torch.full_like(x[:, :, 0], self.p).bernoulli().bool()
+ # Zero-out the masked tokens
+ x[zero_mask, :] = 0
+ return x
+
+
+class TransformerEncoder(nn.Module):
+ def __init__(
+ self,
+ num_tokens: int,
+ token_dim: int,
+ dim: int,
+ depth: int,
+ heads: int,
+ mlp_dim: int,
+ dim_head: int = 64,
+ dropout: float = 0.0,
+ emb_dropout: float = 0.0,
+ emb_dropout_type: str = "drop",
+ emb_dropout_loc: str = "token",
+ norm: str = "layer",
+ norm_cond_dim: int = -1,
+ token_pe_numfreq: int = -1,
+ ):
+ super().__init__()
+ if token_pe_numfreq > 0:
+ token_dim_new = token_dim * (2 * token_pe_numfreq + 1)
+ self.to_token_embedding = nn.Sequential(
+ Rearrange("b n d -> (b n) d", n=num_tokens, d=token_dim),
+ FrequencyEmbedder(token_pe_numfreq, token_pe_numfreq - 1),
+ Rearrange("(b n) d -> b n d", n=num_tokens, d=token_dim_new),
+ nn.Linear(token_dim_new, dim),
+ )
+ else:
+ self.to_token_embedding = nn.Linear(token_dim, dim)
+ self.pos_embedding = nn.Parameter(torch.randn(1, num_tokens, dim))
+ if emb_dropout_type == "drop":
+ self.dropout = DropTokenDropout(emb_dropout)
+ elif emb_dropout_type == "zero":
+ self.dropout = ZeroTokenDropout(emb_dropout)
+ else:
+ raise ValueError(f"Unknown emb_dropout_type: {emb_dropout_type}")
+ self.emb_dropout_loc = emb_dropout_loc
+
+ self.transformer = Transformer(
+ dim, depth, heads, dim_head, mlp_dim, dropout, norm=norm, norm_cond_dim=norm_cond_dim
+ )
+
+ def forward(self, inp: torch.Tensor, *args, **kwargs):
+ x = inp
+
+ if self.emb_dropout_loc == "input":
+ x = self.dropout(x)
+ x = self.to_token_embedding(x)
+
+ if self.emb_dropout_loc == "token":
+ x = self.dropout(x)
+ b, n, _ = x.shape
+ x += self.pos_embedding[:, :n]
+
+ if self.emb_dropout_loc == "token_afterpos":
+ x = self.dropout(x)
+ x = self.transformer(x, *args)
+ return x
+
+
+class TransformerDecoder(nn.Module):
+ def __init__(
+ self,
+ num_tokens: int,
+ token_dim: int,
+ dim: int,
+ depth: int,
+ heads: int,
+ mlp_dim: int,
+ dim_head: int = 64,
+ dropout: float = 0.0,
+ emb_dropout: float = 0.0,
+ emb_dropout_type: str = 'drop',
+ norm: str = "layer",
+ norm_cond_dim: int = -1,
+ context_dim: Optional[int] = None,
+ skip_token_embedding: bool = False,
+ ):
+ super().__init__()
+ if not skip_token_embedding:
+ self.to_token_embedding = nn.Linear(token_dim, dim)
+ else:
+ self.to_token_embedding = nn.Identity()
+ if token_dim != dim:
+ raise ValueError(
+ f"token_dim ({token_dim}) != dim ({dim}) when skip_token_embedding is True"
+ )
+
+ self.pos_embedding = nn.Parameter(torch.randn(1, num_tokens, dim))
+ if emb_dropout_type == "drop":
+ self.dropout = DropTokenDropout(emb_dropout)
+ elif emb_dropout_type == "zero":
+ self.dropout = ZeroTokenDropout(emb_dropout)
+ elif emb_dropout_type == "normal":
+ self.dropout = nn.Dropout(emb_dropout)
+
+ self.transformer = TransformerCrossAttn(
+ dim,
+ depth,
+ heads,
+ dim_head,
+ mlp_dim,
+ dropout,
+ norm=norm,
+ norm_cond_dim=norm_cond_dim,
+ context_dim=context_dim,
+ )
+
+ def forward(self, inp: torch.Tensor, *args, context=None, context_list=None):
+ x = self.to_token_embedding(inp)
+ b, n, _ = x.shape
+
+ x = self.dropout(x)
+ x += self.pos_embedding[:, :n]
+
+ x = self.transformer(x, *args, context=context, context_list=context_list)
+ return x
+
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/components/t_cond_mlp.py b/third_party/GVHMR/hmr4d/network/hmr2/components/t_cond_mlp.py
new file mode 100644
index 0000000000000000000000000000000000000000..44d5a09bf54f67712a69953039b7b5af41c3f029
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/components/t_cond_mlp.py
@@ -0,0 +1,199 @@
+import copy
+from typing import List, Optional
+
+import torch
+
+
+class AdaptiveLayerNorm1D(torch.nn.Module):
+ def __init__(self, data_dim: int, norm_cond_dim: int):
+ super().__init__()
+ if data_dim <= 0:
+ raise ValueError(f"data_dim must be positive, but got {data_dim}")
+ if norm_cond_dim <= 0:
+ raise ValueError(f"norm_cond_dim must be positive, but got {norm_cond_dim}")
+ self.norm = torch.nn.LayerNorm(
+ data_dim
+ ) # TODO: Check if elementwise_affine=True is correct
+ self.linear = torch.nn.Linear(norm_cond_dim, 2 * data_dim)
+ torch.nn.init.zeros_(self.linear.weight)
+ torch.nn.init.zeros_(self.linear.bias)
+
+ def forward(self, x: torch.Tensor, t: torch.Tensor) -> torch.Tensor:
+ # x: (batch, ..., data_dim)
+ # t: (batch, norm_cond_dim)
+ # return: (batch, data_dim)
+ x = self.norm(x)
+ alpha, beta = self.linear(t).chunk(2, dim=-1)
+
+ # Add singleton dimensions to alpha and beta
+ if x.dim() > 2:
+ alpha = alpha.view(alpha.shape[0], *([1] * (x.dim() - 2)), alpha.shape[1])
+ beta = beta.view(beta.shape[0], *([1] * (x.dim() - 2)), beta.shape[1])
+
+ return x * (1 + alpha) + beta
+
+
+class SequentialCond(torch.nn.Sequential):
+ def forward(self, input, *args, **kwargs):
+ for module in self:
+ if isinstance(module, (AdaptiveLayerNorm1D, SequentialCond, ResidualMLPBlock)):
+ # print(f'Passing on args to {module}', [a.shape for a in args])
+ input = module(input, *args, **kwargs)
+ else:
+ # print(f'Skipping passing args to {module}', [a.shape for a in args])
+ input = module(input)
+ return input
+
+
+def normalization_layer(norm: Optional[str], dim: int, norm_cond_dim: int = -1):
+ if norm == "batch":
+ return torch.nn.BatchNorm1d(dim)
+ elif norm == "layer":
+ return torch.nn.LayerNorm(dim)
+ elif norm == "ada":
+ assert norm_cond_dim > 0, f"norm_cond_dim must be positive, got {norm_cond_dim}"
+ return AdaptiveLayerNorm1D(dim, norm_cond_dim)
+ elif norm is None:
+ return torch.nn.Identity()
+ else:
+ raise ValueError(f"Unknown norm: {norm}")
+
+
+def linear_norm_activ_dropout(
+ input_dim: int,
+ output_dim: int,
+ activation: torch.nn.Module = torch.nn.ReLU(),
+ bias: bool = True,
+ norm: Optional[str] = "layer", # Options: ada/batch/layer
+ dropout: float = 0.0,
+ norm_cond_dim: int = -1,
+) -> SequentialCond:
+ layers = []
+ layers.append(torch.nn.Linear(input_dim, output_dim, bias=bias))
+ if norm is not None:
+ layers.append(normalization_layer(norm, output_dim, norm_cond_dim))
+ layers.append(copy.deepcopy(activation))
+ if dropout > 0.0:
+ layers.append(torch.nn.Dropout(dropout))
+ return SequentialCond(*layers)
+
+
+def create_simple_mlp(
+ input_dim: int,
+ hidden_dims: List[int],
+ output_dim: int,
+ activation: torch.nn.Module = torch.nn.ReLU(),
+ bias: bool = True,
+ norm: Optional[str] = "layer", # Options: ada/batch/layer
+ dropout: float = 0.0,
+ norm_cond_dim: int = -1,
+) -> SequentialCond:
+ layers = []
+ prev_dim = input_dim
+ for hidden_dim in hidden_dims:
+ layers.extend(
+ linear_norm_activ_dropout(
+ prev_dim, hidden_dim, activation, bias, norm, dropout, norm_cond_dim
+ )
+ )
+ prev_dim = hidden_dim
+ layers.append(torch.nn.Linear(prev_dim, output_dim, bias=bias))
+ return SequentialCond(*layers)
+
+
+class ResidualMLPBlock(torch.nn.Module):
+ def __init__(
+ self,
+ input_dim: int,
+ hidden_dim: int,
+ num_hidden_layers: int,
+ output_dim: int,
+ activation: torch.nn.Module = torch.nn.ReLU(),
+ bias: bool = True,
+ norm: Optional[str] = "layer", # Options: ada/batch/layer
+ dropout: float = 0.0,
+ norm_cond_dim: int = -1,
+ ):
+ super().__init__()
+ if not (input_dim == output_dim == hidden_dim):
+ raise NotImplementedError(
+ f"input_dim {input_dim} != output_dim {output_dim} is not implemented"
+ )
+
+ layers = []
+ prev_dim = input_dim
+ for i in range(num_hidden_layers):
+ layers.append(
+ linear_norm_activ_dropout(
+ prev_dim, hidden_dim, activation, bias, norm, dropout, norm_cond_dim
+ )
+ )
+ prev_dim = hidden_dim
+ self.model = SequentialCond(*layers)
+ self.skip = torch.nn.Identity()
+
+ def forward(self, x: torch.Tensor, *args, **kwargs) -> torch.Tensor:
+ return x + self.model(x, *args, **kwargs)
+
+
+class ResidualMLP(torch.nn.Module):
+ def __init__(
+ self,
+ input_dim: int,
+ hidden_dim: int,
+ num_hidden_layers: int,
+ output_dim: int,
+ activation: torch.nn.Module = torch.nn.ReLU(),
+ bias: bool = True,
+ norm: Optional[str] = "layer", # Options: ada/batch/layer
+ dropout: float = 0.0,
+ num_blocks: int = 1,
+ norm_cond_dim: int = -1,
+ ):
+ super().__init__()
+ self.input_dim = input_dim
+ self.model = SequentialCond(
+ linear_norm_activ_dropout(
+ input_dim, hidden_dim, activation, bias, norm, dropout, norm_cond_dim
+ ),
+ *[
+ ResidualMLPBlock(
+ hidden_dim,
+ hidden_dim,
+ num_hidden_layers,
+ hidden_dim,
+ activation,
+ bias,
+ norm,
+ dropout,
+ norm_cond_dim,
+ )
+ for _ in range(num_blocks)
+ ],
+ torch.nn.Linear(hidden_dim, output_dim, bias=bias),
+ )
+
+ def forward(self, x: torch.Tensor, *args, **kwargs) -> torch.Tensor:
+ return self.model(x, *args, **kwargs)
+
+
+class FrequencyEmbedder(torch.nn.Module):
+ def __init__(self, num_frequencies, max_freq_log2):
+ super().__init__()
+ frequencies = 2 ** torch.linspace(0, max_freq_log2, steps=num_frequencies)
+ self.register_buffer("frequencies", frequencies)
+
+ def forward(self, x):
+ # x should be of size (N,) or (N, D)
+ N = x.size(0)
+ if x.dim() == 1: # (N,)
+ x = x.unsqueeze(1) # (N, D) where D=1
+ x_unsqueezed = x.unsqueeze(-1) # (N, D, 1)
+ scaled = self.frequencies.view(1, 1, -1) * x_unsqueezed # (N, D, num_frequencies)
+ s = torch.sin(scaled)
+ c = torch.cos(scaled)
+ embedded = torch.cat([s, c, x_unsqueezed], dim=-1).view(
+ N, -1
+ ) # (N, D * 2 * num_frequencies + D)
+ return embedded
+
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/configs/__init__.py b/third_party/GVHMR/hmr4d/network/hmr2/configs/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..68a18564a5063172acbd8a63dc57099865307bc4
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/configs/__init__.py
@@ -0,0 +1,119 @@
+import os
+from typing import Dict
+from yacs.config import CfgNode as CN
+from pathlib import Path
+
+# CACHE_DIR = os.path.join(os.environ.get("HOME"), "Code/4D-Humans/cache")
+# CACHE_DIR_4DHUMANS = os.path.join(CACHE_DIR, "4DHumans")
+
+
+def to_lower(x: Dict) -> Dict:
+ """
+ Convert all dictionary keys to lowercase
+ Args:
+ x (dict): Input dictionary
+ Returns:
+ dict: Output dictionary with all keys converted to lowercase
+ """
+ return {k.lower(): v for k, v in x.items()}
+
+
+_C = CN(new_allowed=True)
+
+_C.GENERAL = CN(new_allowed=True)
+_C.GENERAL.RESUME = True
+_C.GENERAL.TIME_TO_RUN = 3300
+_C.GENERAL.VAL_STEPS = 100
+_C.GENERAL.LOG_STEPS = 100
+_C.GENERAL.CHECKPOINT_STEPS = 20000
+_C.GENERAL.CHECKPOINT_DIR = "checkpoints"
+_C.GENERAL.SUMMARY_DIR = "tensorboard"
+_C.GENERAL.NUM_GPUS = 1
+_C.GENERAL.NUM_WORKERS = 4
+_C.GENERAL.MIXED_PRECISION = True
+_C.GENERAL.ALLOW_CUDA = True
+_C.GENERAL.PIN_MEMORY = False
+_C.GENERAL.DISTRIBUTED = False
+_C.GENERAL.LOCAL_RANK = 0
+_C.GENERAL.USE_SYNCBN = False
+_C.GENERAL.WORLD_SIZE = 1
+
+_C.TRAIN = CN(new_allowed=True)
+_C.TRAIN.NUM_EPOCHS = 100
+_C.TRAIN.BATCH_SIZE = 32
+_C.TRAIN.SHUFFLE = True
+_C.TRAIN.WARMUP = False
+_C.TRAIN.NORMALIZE_PER_IMAGE = False
+_C.TRAIN.CLIP_GRAD = False
+_C.TRAIN.CLIP_GRAD_VALUE = 1.0
+_C.LOSS_WEIGHTS = CN(new_allowed=True)
+
+_C.DATASETS = CN(new_allowed=True)
+
+_C.MODEL = CN(new_allowed=True)
+_C.MODEL.IMAGE_SIZE = 224
+
+_C.EXTRA = CN(new_allowed=True)
+_C.EXTRA.FOCAL_LENGTH = 5000
+
+_C.DATASETS.CONFIG = CN(new_allowed=True)
+_C.DATASETS.CONFIG.SCALE_FACTOR = 0.3
+_C.DATASETS.CONFIG.ROT_FACTOR = 30
+_C.DATASETS.CONFIG.TRANS_FACTOR = 0.02
+_C.DATASETS.CONFIG.COLOR_SCALE = 0.2
+_C.DATASETS.CONFIG.ROT_AUG_RATE = 0.6
+_C.DATASETS.CONFIG.TRANS_AUG_RATE = 0.5
+_C.DATASETS.CONFIG.DO_FLIP = True
+_C.DATASETS.CONFIG.FLIP_AUG_RATE = 0.5
+_C.DATASETS.CONFIG.EXTREME_CROP_AUG_RATE = 0.10
+
+
+def default_config() -> CN:
+ """
+ Get a yacs CfgNode object with the default config values.
+ """
+ # Return a clone so that the defaults will not be altered
+ # This is for the "local variable" use pattern
+ return _C.clone()
+
+
+def dataset_config(name="datasets_tar.yaml") -> CN:
+ """
+ Get dataset config file
+ Returns:
+ CfgNode: Dataset config as a yacs CfgNode object.
+ """
+ cfg = CN(new_allowed=True)
+ config_file = os.path.join(os.path.dirname(os.path.realpath(__file__)), name)
+ cfg.merge_from_file(config_file)
+ cfg.freeze()
+ return cfg
+
+
+def dataset_eval_config() -> CN:
+ return dataset_config("datasets_eval.yaml")
+
+
+def get_config(config_file: str, merge: bool = True) -> CN:
+ """
+ Read a config file and optionally merge it with the default config file.
+ Args:
+ config_file (str): Path to config file.
+ merge (bool): Whether to merge with the default config or not.
+ Returns:
+ CfgNode: Config as a yacs CfgNode object.
+ """
+ if merge:
+ cfg = default_config()
+ else:
+ cfg = CN(new_allowed=True)
+ cfg.merge_from_file(config_file)
+
+ # ---- Update ---- #
+ cfg.SMPL.MODEL_PATH = cfg.SMPL.MODEL_PATH # Not used
+ cfg.SMPL.JOINT_REGRESSOR_EXTRA = cfg.SMPL.JOINT_REGRESSOR_EXTRA # Not Used
+ cfg.SMPL.MEAN_PARAMS = str(Path(__file__).parent / "smpl_mean_params.npz")
+ # ---------------- #
+
+ cfg.freeze()
+ return cfg
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/configs/model_config.yaml b/third_party/GVHMR/hmr4d/network/hmr2/configs/model_config.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..1374229df273969d7f372657fdd7abd3c9999006
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/configs/model_config.yaml
@@ -0,0 +1,131 @@
+task_name: train
+tags:
+- dev
+train: true
+test: false
+ckpt_path: null
+seed: null
+DATASETS:
+ TRAIN:
+ H36M-TRAIN:
+ WEIGHT: 0.3
+ MPII-TRAIN:
+ WEIGHT: 0.1
+ COCO-TRAIN-2014:
+ WEIGHT: 0.4
+ MPI-INF-TRAIN:
+ WEIGHT: 0.2
+ VAL:
+ COCO-VAL:
+ WEIGHT: 1.0
+ MOCAP: CMU-MOCAP
+ CONFIG:
+ SCALE_FACTOR: 0.3
+ ROT_FACTOR: 30
+ TRANS_FACTOR: 0.02
+ COLOR_SCALE: 0.2
+ ROT_AUG_RATE: 0.6
+ TRANS_AUG_RATE: 0.5
+ DO_FLIP: true
+ FLIP_AUG_RATE: 0.5
+ EXTREME_CROP_AUG_RATE: 0.1
+trainer:
+ _target_: pytorch_lightning.Trainer
+ default_root_dir: ${paths.output_dir}
+ accelerator: gpu
+ devices: 8
+ deterministic: false
+ num_sanity_val_steps: 0
+ log_every_n_steps: ${GENERAL.LOG_STEPS}
+ val_check_interval: ${GENERAL.VAL_STEPS}
+ precision: 16
+ max_steps: ${GENERAL.TOTAL_STEPS}
+ move_metrics_to_cpu: true
+ limit_val_batches: 1
+ track_grad_norm: 2
+ strategy: ddp
+ num_nodes: 1
+ sync_batchnorm: true
+paths:
+ root_dir: ${oc.env:PROJECT_ROOT}
+ data_dir: ${paths.root_dir}/data/
+ log_dir: /fsx/shubham/code/hmr2023/logs_hydra/
+ output_dir: ${hydra:runtime.output_dir}
+ work_dir: ${hydra:runtime.cwd}
+extras:
+ ignore_warnings: false
+ enforce_tags: true
+ print_config: true
+exp_name: 3001d
+SMPL:
+ MODEL_PATH: data/smpl
+ GENDER: neutral
+ NUM_BODY_JOINTS: 23
+ JOINT_REGRESSOR_EXTRA: data/SMPL_to_J19.pkl
+ MEAN_PARAMS: data/smpl_mean_params.npz
+EXTRA:
+ FOCAL_LENGTH: 5000
+ NUM_LOG_IMAGES: 4
+ NUM_LOG_SAMPLES_PER_IMAGE: 8
+ PELVIS_IND: 39
+MODEL:
+ IMAGE_SIZE: 256
+ IMAGE_MEAN:
+ - 0.485
+ - 0.456
+ - 0.406
+ IMAGE_STD:
+ - 0.229
+ - 0.224
+ - 0.225
+ BACKBONE:
+ TYPE: vit
+ FREEZE: true
+ NUM_LAYERS: 50
+ OUT_CHANNELS: 2048
+ ADD_NECK: false
+ FLOW:
+ DIM: 144
+ NUM_LAYERS: 4
+ CONTEXT_FEATURES: 2048
+ LAYER_HIDDEN_FEATURES: 1024
+ LAYER_DEPTH: 2
+ FC_HEAD:
+ NUM_FEATURES: 1024
+ SMPL_HEAD:
+ TYPE: transformer_decoder
+ IN_CHANNELS: 2048
+ TRANSFORMER_DECODER:
+ depth: 6
+ heads: 8
+ mlp_dim: 1024
+ dim_head: 64
+ dropout: 0.0
+ emb_dropout: 0.0
+ norm: layer
+ context_dim: 1280
+GENERAL:
+ TOTAL_STEPS: 100000
+ LOG_STEPS: 100
+ VAL_STEPS: 100
+ CHECKPOINT_STEPS: 1000
+ CHECKPOINT_SAVE_TOP_K: -1
+ NUM_WORKERS: 6
+ PREFETCH_FACTOR: 2
+TRAIN:
+ LR: 0.0001
+ WEIGHT_DECAY: 0.0001
+ BATCH_SIZE: 512
+ LOSS_REDUCTION: mean
+ NUM_TRAIN_SAMPLES: 2
+ NUM_TEST_SAMPLES: 64
+ POSE_2D_NOISE_RATIO: 0.01
+ SMPL_PARAM_NOISE_RATIO: 0.005
+LOSS_WEIGHTS:
+ KEYPOINTS_3D: 0.05
+ KEYPOINTS_2D: 0.01
+ GLOBAL_ORIENT: 0.001
+ BODY_POSE: 0.001
+ BETAS: 0.0005
+ ADVERSARIAL: 0.0005
+local: {}
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/configs/smpl_mean_params.npz b/third_party/GVHMR/hmr4d/network/hmr2/configs/smpl_mean_params.npz
new file mode 100644
index 0000000000000000000000000000000000000000..c6f60a76976b877cbc08345b2977c6ddd83ced87
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/configs/smpl_mean_params.npz
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6fd6dd687800da946d0a0492383f973b92ec20f166a0b829775882868c35fcdd
+size 1310
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/hmr2.py b/third_party/GVHMR/hmr4d/network/hmr2/hmr2.py
new file mode 100644
index 0000000000000000000000000000000000000000..b018fb839711885d7886164ffc55b447df8fba94
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/hmr2.py
@@ -0,0 +1,55 @@
+import torch
+import pytorch_lightning as pl
+from yacs.config import CfgNode
+from .vit import ViT
+from .smpl_head import SMPLTransformerDecoderHead
+
+from pytorch3d.transforms import matrix_to_axis_angle
+from hmr4d.utils.geo.hmr_cam import compute_transl_full_cam
+
+
+class HMR2(pl.LightningModule):
+ def __init__(self, cfg: CfgNode):
+ super().__init__()
+ self.cfg = cfg
+ self.backbone = ViT(
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=1280,
+ depth=32,
+ num_heads=16,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.55,
+ )
+ self.smpl_head = SMPLTransformerDecoderHead(cfg)
+
+ def forward(self, batch, feat_mode=True):
+ """this file has been modified
+ Args:
+ feat_mode: default True, as we only need the feature token output for the HMR4D project;
+ when False, the full process of HMR2 will be executed.
+ """
+ # Backbone
+ x = batch["img"][:, :, :, 32:-32]
+ vit_feats = self.backbone(x)
+
+ # Output head
+ if feat_mode:
+ token_out = self.smpl_head(vit_feats, only_return_token_out=True) # (B, 1024)
+ return token_out
+
+ # return full process
+ pred_smpl_params, pred_cam, _, token_out = self.smpl_head(vit_feats, only_return_token_out=False)
+ output = {}
+ output["token_out"] = token_out
+ output["smpl_params"] = {
+ "body_pose": matrix_to_axis_angle(pred_smpl_params["body_pose"]).flatten(-2), # (B, 23, 3)
+ "betas": pred_smpl_params["betas"], # (B, 10)
+ "global_orient": matrix_to_axis_angle(pred_smpl_params["global_orient"])[:, 0], # (B, 3)
+ "transl": compute_transl_full_cam(pred_cam, batch["bbx_xys"], batch["K_fullimg"]), # (B, 3)
+ }
+
+ return output
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/smpl_head.py b/third_party/GVHMR/hmr4d/network/hmr2/smpl_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..5af5cac1f0afae8c89078cc731f0d5a54623c5dc
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/smpl_head.py
@@ -0,0 +1,109 @@
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import numpy as np
+import einops
+
+from .utils.geometry import rot6d_to_rotmat, aa_to_rotmat
+from .components.pose_transformer import TransformerDecoder
+
+
+class SMPLTransformerDecoderHead(nn.Module):
+ """Cross-attention based SMPL Transformer decoder"""
+
+ def __init__(self, cfg):
+ super().__init__()
+ self.cfg = cfg
+ self.joint_rep_type = cfg.MODEL.SMPL_HEAD.get("JOINT_REP", "6d")
+ self.joint_rep_dim = {"6d": 6, "aa": 3}[self.joint_rep_type]
+ npose = self.joint_rep_dim * (cfg.SMPL.NUM_BODY_JOINTS + 1)
+ self.npose = npose
+ self.input_is_mean_shape = cfg.MODEL.SMPL_HEAD.get("TRANSFORMER_INPUT", "zero") == "mean_shape"
+ transformer_args = dict(
+ num_tokens=1,
+ token_dim=(npose + 10 + 3) if self.input_is_mean_shape else 1,
+ dim=1024,
+ )
+ transformer_args.update(**dict(cfg.MODEL.SMPL_HEAD.TRANSFORMER_DECODER))
+ self.transformer = TransformerDecoder(**transformer_args)
+ dim = transformer_args["dim"]
+ self.decpose = nn.Linear(dim, npose)
+ self.decshape = nn.Linear(dim, 10)
+ self.deccam = nn.Linear(dim, 3)
+
+ if cfg.MODEL.SMPL_HEAD.get("INIT_DECODER_XAVIER", False):
+ # True by default in MLP. False by default in Transformer
+ nn.init.xavier_uniform_(self.decpose.weight, gain=0.01)
+ nn.init.xavier_uniform_(self.decshape.weight, gain=0.01)
+ nn.init.xavier_uniform_(self.deccam.weight, gain=0.01)
+
+ mean_params = np.load(cfg.SMPL.MEAN_PARAMS)
+ init_body_pose = torch.from_numpy(mean_params["pose"].astype(np.float32)).unsqueeze(0)
+ init_betas = torch.from_numpy(mean_params["shape"].astype("float32")).unsqueeze(0)
+ init_cam = torch.from_numpy(mean_params["cam"].astype(np.float32)).unsqueeze(0)
+ self.register_buffer("init_body_pose", init_body_pose)
+ self.register_buffer("init_betas", init_betas)
+ self.register_buffer("init_cam", init_cam)
+
+ def forward(self, x, only_return_token_out=False):
+ batch_size = x.shape[0]
+ # vit pretrained backbone is channel-first. Change to token-first
+ x = einops.rearrange(x, "b c h w -> b (h w) c")
+
+ init_body_pose = self.init_body_pose.expand(batch_size, -1)
+ init_betas = self.init_betas.expand(batch_size, -1)
+ init_cam = self.init_cam.expand(batch_size, -1)
+
+ # TODO: Convert init_body_pose to aa rep if needed
+ if self.joint_rep_type == "aa":
+ raise NotImplementedError
+
+ pred_body_pose = init_body_pose
+ pred_betas = init_betas
+ pred_cam = init_cam
+ pred_body_pose_list = []
+ pred_betas_list = []
+ pred_cam_list = []
+ for i in range(self.cfg.MODEL.SMPL_HEAD.get("IEF_ITERS", 1)):
+ assert i == 0, "Only support 1 iteration for now"
+
+ # Input token to transformer is zero token
+ if self.input_is_mean_shape:
+ token = torch.cat([pred_body_pose, pred_betas, pred_cam], dim=1)[:, None, :]
+ else:
+ token = torch.zeros(batch_size, 1, 1).to(x.device)
+
+ # Pass through transformer
+ token_out = self.transformer(token, context=x)
+ token_out = token_out.squeeze(1) # (B, C)
+
+ if only_return_token_out:
+ return token_out
+ else:
+ # Readout from token_out
+ pred_body_pose = self.decpose(token_out) + pred_body_pose
+ pred_betas = self.decshape(token_out) + pred_betas
+ pred_cam = self.deccam(token_out) + pred_cam
+ pred_body_pose_list.append(pred_body_pose)
+ pred_betas_list.append(pred_betas)
+ pred_cam_list.append(pred_cam)
+
+ # Convert self.joint_rep_type -> rotmat
+ joint_conversion_fn = {"6d": rot6d_to_rotmat, "aa": lambda x: aa_to_rotmat(x.view(-1, 3).contiguous())}[
+ self.joint_rep_type
+ ]
+
+ pred_smpl_params_list = {}
+ pred_smpl_params_list["body_pose"] = torch.cat(
+ [joint_conversion_fn(pbp).view(batch_size, -1, 3, 3)[:, 1:, :, :] for pbp in pred_body_pose_list], dim=0
+ )
+ pred_smpl_params_list["betas"] = torch.cat(pred_betas_list, dim=0)
+ pred_smpl_params_list["cam"] = torch.cat(pred_cam_list, dim=0)
+ pred_body_pose = joint_conversion_fn(pred_body_pose).view(batch_size, self.cfg.SMPL.NUM_BODY_JOINTS + 1, 3, 3)
+
+ pred_smpl_params = {
+ "global_orient": pred_body_pose[:, [0]],
+ "body_pose": pred_body_pose[:, 1:],
+ "betas": pred_betas,
+ }
+ return pred_smpl_params, pred_cam, pred_smpl_params_list, token_out
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/utils/geometry.py b/third_party/GVHMR/hmr4d/network/hmr2/utils/geometry.py
new file mode 100644
index 0000000000000000000000000000000000000000..e128ba80ef36c90d221e482f3d66f4aa6fe190e5
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/utils/geometry.py
@@ -0,0 +1,118 @@
+from typing import Optional
+import torch
+from torch.nn import functional as F
+
+
+def aa_to_rotmat(theta: torch.Tensor):
+ """
+ Convert axis-angle representation to rotation matrix.
+ Works by first converting it to a quaternion.
+ Args:
+ theta (torch.Tensor): Tensor of shape (B, 3) containing axis-angle representations.
+ Returns:
+ torch.Tensor: Corresponding rotation matrices with shape (B, 3, 3).
+ """
+ norm = torch.norm(theta + 1e-8, p=2, dim=1)
+ angle = torch.unsqueeze(norm, -1)
+ normalized = torch.div(theta, angle)
+ angle = angle * 0.5
+ v_cos = torch.cos(angle)
+ v_sin = torch.sin(angle)
+ quat = torch.cat([v_cos, v_sin * normalized], dim=1)
+ return quat_to_rotmat(quat)
+
+
+def quat_to_rotmat(quat: torch.Tensor) -> torch.Tensor:
+ """
+ Convert quaternion representation to rotation matrix.
+ Args:
+ quat (torch.Tensor) of shape (B, 4); 4 <===> (w, x, y, z).
+ Returns:
+ torch.Tensor: Corresponding rotation matrices with shape (B, 3, 3).
+ """
+ norm_quat = quat
+ norm_quat = norm_quat / norm_quat.norm(p=2, dim=1, keepdim=True)
+ w, x, y, z = norm_quat[:, 0], norm_quat[:, 1], norm_quat[:, 2], norm_quat[:, 3]
+
+ B = quat.size(0)
+
+ w2, x2, y2, z2 = w.pow(2), x.pow(2), y.pow(2), z.pow(2)
+ wx, wy, wz = w * x, w * y, w * z
+ xy, xz, yz = x * y, x * z, y * z
+
+ rotMat = torch.stack(
+ [
+ w2 + x2 - y2 - z2,
+ 2 * xy - 2 * wz,
+ 2 * wy + 2 * xz,
+ 2 * wz + 2 * xy,
+ w2 - x2 + y2 - z2,
+ 2 * yz - 2 * wx,
+ 2 * xz - 2 * wy,
+ 2 * wx + 2 * yz,
+ w2 - x2 - y2 + z2,
+ ],
+ dim=1,
+ ).view(B, 3, 3)
+ return rotMat
+
+
+def rot6d_to_rotmat(x: torch.Tensor) -> torch.Tensor:
+ """
+ Convert 6D rotation representation to 3x3 rotation matrix.
+ Based on Zhou et al., "On the Continuity of Rotation Representations in Neural Networks", CVPR 2019
+ Args:
+ x (torch.Tensor): (B,6) Batch of 6-D rotation representations.
+ Returns:
+ torch.Tensor: Batch of corresponding rotation matrices with shape (B,3,3).
+ """
+ x = x.reshape(-1, 2, 3).permute(0, 2, 1).contiguous()
+ a1 = x[:, :, 0]
+ a2 = x[:, :, 1]
+ b1 = F.normalize(a1)
+ b2 = F.normalize(a2 - torch.einsum("bi,bi->b", b1, a2).unsqueeze(-1) * b1)
+ b3 = torch.cross(b1, b2)
+ return torch.stack((b1, b2, b3), dim=-1)
+
+
+def perspective_projection(
+ points: torch.Tensor,
+ translation: torch.Tensor,
+ focal_length: torch.Tensor,
+ camera_center: Optional[torch.Tensor] = None,
+ rotation: Optional[torch.Tensor] = None,
+) -> torch.Tensor:
+ """
+ Computes the perspective projection of a set of 3D points.
+ Args:
+ points (torch.Tensor): Tensor of shape (B, N, 3) containing the input 3D points.
+ translation (torch.Tensor): Tensor of shape (B, 3) containing the 3D camera translation.
+ focal_length (torch.Tensor): Tensor of shape (B, 2) containing the focal length in pixels.
+ camera_center (torch.Tensor): Tensor of shape (B, 2) containing the camera center in pixels.
+ rotation (torch.Tensor): Tensor of shape (B, 3, 3) containing the camera rotation.
+ Returns:
+ torch.Tensor: Tensor of shape (B, N, 2) containing the projection of the input points.
+ """
+ batch_size = points.shape[0]
+ if rotation is None:
+ rotation = torch.eye(3, device=points.device, dtype=points.dtype).unsqueeze(0).expand(batch_size, -1, -1)
+ if camera_center is None:
+ camera_center = torch.zeros(batch_size, 2, device=points.device, dtype=points.dtype)
+ # Populate intrinsic camera matrix K.
+ K = torch.zeros([batch_size, 3, 3], device=points.device, dtype=points.dtype)
+ K[:, 0, 0] = focal_length[:, 0]
+ K[:, 1, 1] = focal_length[:, 1]
+ K[:, 2, 2] = 1.0
+ K[:, :-1, -1] = camera_center
+
+ # Transform points
+ points = torch.einsum("bij,bkj->bki", rotation, points)
+ points = points + translation.unsqueeze(1)
+
+ # Apply perspective distortion
+ projected_points = points / points[:, :, -1].unsqueeze(-1)
+
+ # Apply camera intrinsics
+ projected_points = torch.einsum("bij,bkj->bki", K, projected_points)
+
+ return projected_points[:, :, :-1]
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/utils/preproc.py b/third_party/GVHMR/hmr4d/network/hmr2/utils/preproc.py
new file mode 100644
index 0000000000000000000000000000000000000000..3db9dcf309b8d5a6b18f91e657b201af3662019a
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/utils/preproc.py
@@ -0,0 +1,52 @@
+import cv2
+import numpy as np
+import torch
+from pathlib import Path
+
+IMAGE_MEAN = torch.tensor([0.485, 0.456, 0.406])
+IMAGE_STD = torch.tensor([0.229, 0.224, 0.225])
+
+
+def expand_to_aspect_ratio(input_shape, target_aspect_ratio=[192, 256]):
+ """Increase the size of the bounding box to match the target shape."""
+ if target_aspect_ratio is None:
+ return input_shape
+
+ try:
+ w, h = input_shape
+ except (ValueError, TypeError):
+ return input_shape
+
+ w_t, h_t = target_aspect_ratio
+ if h / w < h_t / w_t:
+ h_new = max(w * h_t / w_t, h)
+ w_new = w
+ else:
+ h_new = h
+ w_new = max(h * w_t / h_t, w)
+ if h_new < h or w_new < w:
+ breakpoint()
+ return np.array([w_new, h_new])
+
+
+def crop_and_resize(img, bbx_xy, bbx_s, dst_size=256, enlarge_ratio=1.2):
+ """
+ Args:
+ img: (H, W, 3)
+ bbx_xy: (2,)
+ bbx_s: scalar
+ """
+ hs = bbx_s * enlarge_ratio / 2
+ src = np.stack(
+ [
+ bbx_xy - hs, # left-up corner
+ bbx_xy + np.array([hs, -hs]), # right-up corner
+ bbx_xy, # center
+ ]
+ ).astype(np.float32)
+ dst = np.array([[0, 0], [dst_size - 1, 0], [dst_size / 2 - 0.5, dst_size / 2 - 0.5]], dtype=np.float32)
+ A = cv2.getAffineTransform(src, dst)
+
+ img_crop = cv2.warpAffine(img, A, (dst_size, dst_size), flags=cv2.INTER_LINEAR)
+ bbx_xys_final = np.array([*bbx_xy, bbx_s * enlarge_ratio])
+ return img_crop, bbx_xys_final
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/utils/smpl_wrapper.py b/third_party/GVHMR/hmr4d/network/hmr2/utils/smpl_wrapper.py
new file mode 100644
index 0000000000000000000000000000000000000000..839a83d1e6d35bb440666fe1557ae6689db269db
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/utils/smpl_wrapper.py
@@ -0,0 +1,45 @@
+import torch
+import numpy as np
+import pickle
+from typing import Optional
+import smplx
+from smplx.lbs import vertices2joints
+from smplx.utils import SMPLOutput
+
+
+class SMPL(smplx.SMPLLayer):
+ def __init__(self, *args, joint_regressor_extra: Optional[str] = None, update_hips: bool = False, **kwargs):
+ """
+ Extension of the official SMPL implementation to support more joints.
+ Args:
+ Same as SMPLLayer.
+ joint_regressor_extra (str): Path to extra joint regressor.
+ """
+ super(SMPL, self).__init__(*args, **kwargs)
+ smpl_to_openpose = [24, 12, 17, 19, 21, 16, 18, 20, 0, 2, 5, 8, 1, 4, 7, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34]
+
+ if joint_regressor_extra is not None:
+ self.register_buffer(
+ "joint_regressor_extra",
+ torch.tensor(pickle.load(open(joint_regressor_extra, "rb"), encoding="latin1"), dtype=torch.float32),
+ )
+ self.register_buffer("joint_map", torch.tensor(smpl_to_openpose, dtype=torch.long))
+ self.update_hips = update_hips
+
+ def forward(self, *args, **kwargs) -> SMPLOutput:
+ """
+ Run forward pass. Same as SMPL and also append an extra set of joints if joint_regressor_extra is specified.
+ """
+ smpl_output = super(SMPL, self).forward(*args, **kwargs)
+ joints = smpl_output.joints[:, self.joint_map, :]
+ if self.update_hips:
+ joints[:, [9, 12]] = (
+ joints[:, [9, 12]]
+ + 0.25 * (joints[:, [9, 12]] - joints[:, [12, 9]])
+ + 0.5 * (joints[:, [8]] - 0.5 * (joints[:, [9, 12]] + joints[:, [12, 9]]))
+ )
+ if hasattr(self, "joint_regressor_extra"):
+ extra_joints = vertices2joints(self.joint_regressor_extra, smpl_output.vertices)
+ joints = torch.cat([joints, extra_joints], dim=1)
+ smpl_output.joints = joints
+ return smpl_output
diff --git a/third_party/GVHMR/hmr4d/network/hmr2/vit.py b/third_party/GVHMR/hmr4d/network/hmr2/vit.py
new file mode 100644
index 0000000000000000000000000000000000000000..c56c71889cd441294f57ad687d0678d2443d1eed
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/network/hmr2/vit.py
@@ -0,0 +1,348 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import math
+
+import torch
+from functools import partial
+import torch.nn as nn
+import torch.nn.functional as F
+import torch.utils.checkpoint as checkpoint
+
+from timm.models.layers import drop_path, to_2tuple, trunc_normal_
+
+def vit(cfg):
+ return ViT(
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=1280,
+ depth=32,
+ num_heads=16,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.55,
+ )
+
+def get_abs_pos(abs_pos, h, w, ori_h, ori_w, has_cls_token=True):
+ """
+ Calculate absolute positional embeddings. If needed, resize embeddings and remove cls_token
+ dimension for the original embeddings.
+ Args:
+ abs_pos (Tensor): absolute positional embeddings with (1, num_position, C).
+ has_cls_token (bool): If true, has 1 embedding in abs_pos for cls token.
+ hw (Tuple): size of input image tokens.
+
+ Returns:
+ Absolute positional embeddings after processing with shape (1, H, W, C)
+ """
+ cls_token = None
+ B, L, C = abs_pos.shape
+ if has_cls_token:
+ cls_token = abs_pos[:, 0:1]
+ abs_pos = abs_pos[:, 1:]
+
+ if ori_h != h or ori_w != w:
+ new_abs_pos = F.interpolate(
+ abs_pos.reshape(1, ori_h, ori_w, -1).permute(0, 3, 1, 2),
+ size=(h, w),
+ mode="bicubic",
+ align_corners=False,
+ ).permute(0, 2, 3, 1).reshape(B, -1, C)
+
+ else:
+ new_abs_pos = abs_pos
+
+ if cls_token is not None:
+ new_abs_pos = torch.cat([cls_token, new_abs_pos], dim=1)
+ return new_abs_pos
+
+class DropPath(nn.Module):
+ """Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks).
+ """
+ def __init__(self, drop_prob=None):
+ super(DropPath, self).__init__()
+ self.drop_prob = drop_prob
+
+ def forward(self, x):
+ return drop_path(x, self.drop_prob, self.training)
+
+ def extra_repr(self):
+ return 'p={}'.format(self.drop_prob)
+
+class Mlp(nn.Module):
+ def __init__(self, in_features, hidden_features=None, out_features=None, act_layer=nn.GELU, drop=0.):
+ super().__init__()
+ out_features = out_features or in_features
+ hidden_features = hidden_features or in_features
+ self.fc1 = nn.Linear(in_features, hidden_features)
+ self.act = act_layer()
+ self.fc2 = nn.Linear(hidden_features, out_features)
+ self.drop = nn.Dropout(drop)
+
+ def forward(self, x):
+ x = self.fc1(x)
+ x = self.act(x)
+ x = self.fc2(x)
+ x = self.drop(x)
+ return x
+
+class Attention(nn.Module):
+ def __init__(
+ self, dim, num_heads=8, qkv_bias=False, qk_scale=None, attn_drop=0.,
+ proj_drop=0., attn_head_dim=None,):
+ super().__init__()
+ self.num_heads = num_heads
+ head_dim = dim // num_heads
+ self.dim = dim
+
+ if attn_head_dim is not None:
+ head_dim = attn_head_dim
+ all_head_dim = head_dim * self.num_heads
+
+ self.scale = qk_scale or head_dim ** -0.5
+
+ self.qkv = nn.Linear(dim, all_head_dim * 3, bias=qkv_bias)
+
+ self.attn_drop = nn.Dropout(attn_drop)
+ self.proj = nn.Linear(all_head_dim, dim)
+ self.proj_drop = nn.Dropout(proj_drop)
+
+ def forward(self, x):
+ B, N, C = x.shape
+ qkv = self.qkv(x)
+ qkv = qkv.reshape(B, N, 3, self.num_heads, -1).permute(2, 0, 3, 1, 4)
+ q, k, v = qkv[0], qkv[1], qkv[2] # make torchscript happy (cannot use tensor as tuple)
+
+ q = q * self.scale
+ attn = (q @ k.transpose(-2, -1))
+
+ attn = attn.softmax(dim=-1)
+ attn = self.attn_drop(attn)
+
+ x = (attn @ v).transpose(1, 2).reshape(B, N, -1)
+ x = self.proj(x)
+ x = self.proj_drop(x)
+
+ return x
+
+class Block(nn.Module):
+
+ def __init__(self, dim, num_heads, mlp_ratio=4., qkv_bias=False, qk_scale=None,
+ drop=0., attn_drop=0., drop_path=0., act_layer=nn.GELU,
+ norm_layer=nn.LayerNorm, attn_head_dim=None
+ ):
+ super().__init__()
+
+ self.norm1 = norm_layer(dim)
+ self.attn = Attention(
+ dim, num_heads=num_heads, qkv_bias=qkv_bias, qk_scale=qk_scale,
+ attn_drop=attn_drop, proj_drop=drop, attn_head_dim=attn_head_dim
+ )
+
+ # NOTE: drop path for stochastic depth, we shall see if this is better than dropout here
+ self.drop_path = DropPath(drop_path) if drop_path > 0. else nn.Identity()
+ self.norm2 = norm_layer(dim)
+ mlp_hidden_dim = int(dim * mlp_ratio)
+ self.mlp = Mlp(in_features=dim, hidden_features=mlp_hidden_dim, act_layer=act_layer, drop=drop)
+
+ def forward(self, x):
+ x = x + self.drop_path(self.attn(self.norm1(x)))
+ x = x + self.drop_path(self.mlp(self.norm2(x)))
+ return x
+
+
+class PatchEmbed(nn.Module):
+ """ Image to Patch Embedding
+ """
+ def __init__(self, img_size=224, patch_size=16, in_chans=3, embed_dim=768, ratio=1):
+ super().__init__()
+ img_size = to_2tuple(img_size)
+ patch_size = to_2tuple(patch_size)
+ num_patches = (img_size[1] // patch_size[1]) * (img_size[0] // patch_size[0]) * (ratio ** 2)
+ self.patch_shape = (int(img_size[0] // patch_size[0] * ratio), int(img_size[1] // patch_size[1] * ratio))
+ self.origin_patch_shape = (int(img_size[0] // patch_size[0]), int(img_size[1] // patch_size[1]))
+ self.img_size = img_size
+ self.patch_size = patch_size
+ self.num_patches = num_patches
+
+ self.proj = nn.Conv2d(in_chans, embed_dim, kernel_size=patch_size, stride=(patch_size[0] // ratio), padding=4 + 2 * (ratio//2-1))
+
+ def forward(self, x, **kwargs):
+ B, C, H, W = x.shape
+ x = self.proj(x)
+ Hp, Wp = x.shape[2], x.shape[3]
+
+ x = x.flatten(2).transpose(1, 2)
+ return x, (Hp, Wp)
+
+
+class HybridEmbed(nn.Module):
+ """ CNN Feature Map Embedding
+ Extract feature map from CNN, flatten, project to embedding dim.
+ """
+ def __init__(self, backbone, img_size=224, feature_size=None, in_chans=3, embed_dim=768):
+ super().__init__()
+ assert isinstance(backbone, nn.Module)
+ img_size = to_2tuple(img_size)
+ self.img_size = img_size
+ self.backbone = backbone
+ if feature_size is None:
+ with torch.no_grad():
+ training = backbone.training
+ if training:
+ backbone.eval()
+ o = self.backbone(torch.zeros(1, in_chans, img_size[0], img_size[1]))[-1]
+ feature_size = o.shape[-2:]
+ feature_dim = o.shape[1]
+ backbone.train(training)
+ else:
+ feature_size = to_2tuple(feature_size)
+ feature_dim = self.backbone.feature_info.channels()[-1]
+ self.num_patches = feature_size[0] * feature_size[1]
+ self.proj = nn.Linear(feature_dim, embed_dim)
+
+ def forward(self, x):
+ x = self.backbone(x)[-1]
+ x = x.flatten(2).transpose(1, 2)
+ x = self.proj(x)
+ return x
+
+
+class ViT(nn.Module):
+
+ def __init__(self,
+ img_size=224, patch_size=16, in_chans=3, num_classes=80, embed_dim=768, depth=12,
+ num_heads=12, mlp_ratio=4., qkv_bias=False, qk_scale=None, drop_rate=0., attn_drop_rate=0.,
+ drop_path_rate=0., hybrid_backbone=None, norm_layer=None, use_checkpoint=False,
+ frozen_stages=-1, ratio=1, last_norm=True,
+ patch_padding='pad', freeze_attn=False, freeze_ffn=False,
+ ):
+ # Protect mutable default arguments
+ super(ViT, self).__init__()
+ norm_layer = norm_layer or partial(nn.LayerNorm, eps=1e-6)
+ self.num_classes = num_classes
+ self.num_features = self.embed_dim = embed_dim # num_features for consistency with other models
+ self.frozen_stages = frozen_stages
+ self.use_checkpoint = use_checkpoint
+ self.patch_padding = patch_padding
+ self.freeze_attn = freeze_attn
+ self.freeze_ffn = freeze_ffn
+ self.depth = depth
+
+ if hybrid_backbone is not None:
+ self.patch_embed = HybridEmbed(
+ hybrid_backbone, img_size=img_size, in_chans=in_chans, embed_dim=embed_dim)
+ else:
+ self.patch_embed = PatchEmbed(
+ img_size=img_size, patch_size=patch_size, in_chans=in_chans, embed_dim=embed_dim, ratio=ratio)
+ num_patches = self.patch_embed.num_patches
+
+ # since the pretraining model has class token
+ self.pos_embed = nn.Parameter(torch.zeros(1, num_patches + 1, embed_dim))
+
+ dpr = [x.item() for x in torch.linspace(0, drop_path_rate, depth)] # stochastic depth decay rule
+
+ self.blocks = nn.ModuleList([
+ Block(
+ dim=embed_dim, num_heads=num_heads, mlp_ratio=mlp_ratio, qkv_bias=qkv_bias, qk_scale=qk_scale,
+ drop=drop_rate, attn_drop=attn_drop_rate, drop_path=dpr[i], norm_layer=norm_layer,
+ )
+ for i in range(depth)])
+
+ self.last_norm = norm_layer(embed_dim) if last_norm else nn.Identity()
+
+ if self.pos_embed is not None:
+ trunc_normal_(self.pos_embed, std=.02)
+
+ self._freeze_stages()
+
+ def _freeze_stages(self):
+ """Freeze parameters."""
+ if self.frozen_stages >= 0:
+ self.patch_embed.eval()
+ for param in self.patch_embed.parameters():
+ param.requires_grad = False
+
+ for i in range(1, self.frozen_stages + 1):
+ m = self.blocks[i]
+ m.eval()
+ for param in m.parameters():
+ param.requires_grad = False
+
+ if self.freeze_attn:
+ for i in range(0, self.depth):
+ m = self.blocks[i]
+ m.attn.eval()
+ m.norm1.eval()
+ for param in m.attn.parameters():
+ param.requires_grad = False
+ for param in m.norm1.parameters():
+ param.requires_grad = False
+
+ if self.freeze_ffn:
+ self.pos_embed.requires_grad = False
+ self.patch_embed.eval()
+ for param in self.patch_embed.parameters():
+ param.requires_grad = False
+ for i in range(0, self.depth):
+ m = self.blocks[i]
+ m.mlp.eval()
+ m.norm2.eval()
+ for param in m.mlp.parameters():
+ param.requires_grad = False
+ for param in m.norm2.parameters():
+ param.requires_grad = False
+
+ def init_weights(self):
+ """Initialize the weights in backbone.
+ Args:
+ pretrained (str, optional): Path to pre-trained weights.
+ Defaults to None.
+ """
+ def _init_weights(m):
+ if isinstance(m, nn.Linear):
+ trunc_normal_(m.weight, std=.02)
+ if isinstance(m, nn.Linear) and m.bias is not None:
+ nn.init.constant_(m.bias, 0)
+ elif isinstance(m, nn.LayerNorm):
+ nn.init.constant_(m.bias, 0)
+ nn.init.constant_(m.weight, 1.0)
+
+ self.apply(_init_weights)
+
+ def get_num_layers(self):
+ return len(self.blocks)
+
+ @torch.jit.ignore
+ def no_weight_decay(self):
+ return {'pos_embed', 'cls_token'}
+
+ def forward_features(self, x):
+ B, C, H, W = x.shape
+ x, (Hp, Wp) = self.patch_embed(x)
+
+ if self.pos_embed is not None:
+ # fit for multiple GPU training
+ # since the first element for pos embed (sin-cos manner) is zero, it will cause no difference
+ x = x + self.pos_embed[:, 1:] + self.pos_embed[:, :1]
+
+ for blk in self.blocks:
+ if self.use_checkpoint:
+ x = checkpoint.checkpoint(blk, x)
+ else:
+ x = blk(x)
+
+ x = self.last_norm(x)
+
+ xp = x.permute(0, 2, 1).reshape(B, -1, Hp, Wp).contiguous()
+
+ return xp
+
+ def forward(self, x):
+ x = self.forward_features(x)
+ return x
+
+ def train(self, mode=True):
+ """Convert the model into training mode."""
+ super().train(mode)
+ self._freeze_stages()
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/README.md b/third_party/GVHMR/hmr4d/utils/body_model/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..03397aa52552f0694f3a346b80c3338a485303e7
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/README.md
@@ -0,0 +1,3 @@
+# README
+
+Contents of this folder are modified from HuMoR repository.
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/__init__.py b/third_party/GVHMR/hmr4d/utils/body_model/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..c8a3598de7d522e8262fde0a516ac06a93ccef4d
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/__init__.py
@@ -0,0 +1,3 @@
+from .body_model import BodyModel
+from .body_model_smplh import BodyModelSMPLH
+from .body_model_smplx import BodyModelSMPLX
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/body_model.py b/third_party/GVHMR/hmr4d/utils/body_model/body_model.py
new file mode 100644
index 0000000000000000000000000000000000000000..5f8ac4783df852945667307d0587be7755de5b72
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/body_model.py
@@ -0,0 +1,127 @@
+from turtle import forward
+import numpy as np
+
+import torch
+import torch.nn as nn
+
+from smplx import SMPL, SMPLH, SMPLX
+from smplx.vertex_ids import vertex_ids
+from smplx.utils import Struct
+
+
+class BodyModel(nn.Module):
+ """
+ Wrapper around SMPLX body model class.
+ modified by Zehong Shen
+ """
+
+ def __init__(self,
+ bm_path,
+ num_betas=16,
+ use_vtx_selector=False,
+ model_type='smplh'):
+ super().__init__()
+ '''
+ Creates the body model object at the given path.
+
+ :param bm_path: path to the body model pkl file
+ :param model_type: one of [smpl, smplh, smplx]
+ :param use_vtx_selector: if true, returns additional vertices as joints that correspond to OpenPose joints
+ '''
+ self.use_vtx_selector = use_vtx_selector
+ cur_vertex_ids = None
+ if self.use_vtx_selector:
+ cur_vertex_ids = vertex_ids[model_type]
+ data_struct = None
+ if '.npz' in bm_path:
+ # smplx does not support .npz by default, so have to load in manually
+ smpl_dict = np.load(bm_path, encoding='latin1')
+ data_struct = Struct(**smpl_dict)
+ # print(smpl_dict.files)
+ if model_type == 'smplh':
+ data_struct.hands_componentsl = np.zeros((0))
+ data_struct.hands_componentsr = np.zeros((0))
+ data_struct.hands_meanl = np.zeros((15 * 3))
+ data_struct.hands_meanr = np.zeros((15 * 3))
+ V, D, B = data_struct.shapedirs.shape
+ data_struct.shapedirs = np.concatenate([data_struct.shapedirs, np.zeros(
+ (V, D, SMPL.SHAPE_SPACE_DIM-B))], axis=-1) # super hacky way to let smplh use 16-size beta
+ kwargs = {
+ 'model_type': model_type,
+ 'data_struct': data_struct,
+ 'num_betas': num_betas,
+ 'vertex_ids': cur_vertex_ids,
+ 'use_pca': False,
+ 'flat_hand_mean': True,
+ # - enable variable batchsize, since we don't need module variable - #
+ 'create_body_pose': False,
+ 'create_betas': False,
+ 'create_global_orient': False,
+ 'create_transl': False,
+ 'create_left_hand_pose': False,
+ 'create_right_hand_pose': False,
+ }
+ assert(model_type in ['smpl', 'smplh', 'smplx'])
+ if model_type == 'smpl':
+ self.bm = SMPL(bm_path, **kwargs)
+ self.num_joints = SMPL.NUM_JOINTS
+ elif model_type == 'smplh':
+ self.bm = SMPLH(bm_path, **kwargs)
+ self.num_joints = SMPLH.NUM_JOINTS
+ elif model_type == 'smplx':
+ self.bm = SMPLX(bm_path, **kwargs)
+ self.num_joints = SMPLX.NUM_JOINTS
+
+ self.model_type = model_type
+
+ def forward(self, root_orient=None, pose_body=None, pose_hand=None, pose_jaw=None, pose_eye=None, betas=None,
+ trans=None, dmpls=None, expression=None, return_dict=False, **kwargs):
+ '''
+ Note dmpls are not supported.
+ '''
+ assert(dmpls is None)
+ B = pose_body.shape[0]
+ if pose_hand is None:
+ pose_hand = torch.zeros((B, 2*SMPLH.NUM_HAND_JOINTS*3), device=pose_body.device)
+ if len(betas.shape) == 1:
+ betas = betas.reshape((1, -1)).expand(B, -1)
+
+ out_obj = self.bm(
+ betas=betas,
+ global_orient=root_orient,
+ body_pose=pose_body,
+ left_hand_pose=pose_hand[:, :(SMPLH.NUM_HAND_JOINTS*3)],
+ right_hand_pose=pose_hand[:, (SMPLH.NUM_HAND_JOINTS*3):],
+ transl=trans,
+ expression=expression,
+ jaw_pose=pose_jaw,
+ leye_pose=None if pose_eye is None else pose_eye[:, :3],
+ reye_pose=None if pose_eye is None else pose_eye[:, 3:],
+ return_full_pose=True,
+ **kwargs
+ )
+
+ out = {
+ 'v': out_obj.vertices,
+ 'f': self.bm.faces_tensor,
+ 'Jtr': out_obj.joints,
+ }
+
+ if not self.use_vtx_selector:
+ # don't need extra joints
+ out['Jtr'] = out['Jtr'][:, :self.num_joints+1] # add one for the root
+
+ if not return_dict:
+ out = Struct(**out)
+
+ return out
+
+ def forward_motion(self, **kwargs):
+ B, W, _ = kwargs['pose_body'].shape
+ kwargs = {k: v.reshape(B*W, v.shape[-1]) for k, v in kwargs.items()}
+
+ smpl_opt = self.forward(**kwargs)
+ smpl_opt.v = smpl_opt.v.reshape(B, W, -1, 3)
+ smpl_opt.Jtr = smpl_opt.Jtr.reshape(B, W, -1, 3)
+
+ return smpl_opt
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/body_model_smplh.py b/third_party/GVHMR/hmr4d/utils/body_model/body_model_smplh.py
new file mode 100644
index 0000000000000000000000000000000000000000..27c30fa92f3264fd6121f4bdebb883898f3caf23
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/body_model_smplh.py
@@ -0,0 +1,98 @@
+import torch
+import torch.nn as nn
+import smplx
+
+kwargs_disable_member_var = {
+ "create_body_pose": False,
+ "create_betas": False,
+ "create_global_orient": False,
+ "create_transl": False,
+ "create_left_hand_pose": False,
+ "create_right_hand_pose": False,
+}
+
+
+class BodyModelSMPLH(nn.Module):
+ """Support Batch inference"""
+
+ def __init__(self, model_path, **kwargs):
+ super().__init__()
+ # enable flexible batchsize, handle missing variable at forward()
+ kwargs.update(kwargs_disable_member_var)
+ self.bm = smplx.create(model_path=model_path, **kwargs)
+ self.faces = self.bm.faces
+ self.is_smpl = kwargs.get("model_type", "smpl") == "smpl"
+ if not self.is_smpl:
+ self.hand_pose_dim = self.bm.num_pca_comps if self.bm.use_pca else 3 * self.bm.NUM_HAND_JOINTS
+
+ # For fast computing of skeleton under beta
+ shapedirs = self.bm.shapedirs # (V, 3, 10)
+ J_regressor = self.bm.J_regressor[:22, :] # (22, V)
+ v_template = self.bm.v_template # (V, 3)
+ J_template = J_regressor @ v_template # (22, 3)
+ J_shapedirs = torch.einsum("jv, vcd -> jcd", J_regressor, shapedirs) # (22, 3, 10)
+ self.register_buffer("J_template", J_template, False)
+ self.register_buffer("J_shapedirs", J_shapedirs, False)
+
+ def forward(
+ self,
+ betas=None,
+ global_orient=None,
+ transl=None,
+ body_pose=None,
+ left_hand_pose=None,
+ right_hand_pose=None,
+ **kwargs
+ ):
+
+ device, dtype = self.bm.shapedirs.device, self.bm.shapedirs.dtype
+
+ model_vars = [betas, global_orient, body_pose, transl, left_hand_pose, right_hand_pose]
+ batch_size = 1
+ for var in model_vars:
+ if var is None:
+ continue
+ batch_size = max(batch_size, len(var))
+
+ if global_orient is None:
+ global_orient = torch.zeros([batch_size, 3], dtype=dtype, device=device)
+ if body_pose is None:
+ body_pose = (
+ torch.zeros(3 * self.bm.NUM_BODY_JOINTS, device=device, dtype=dtype)[None]
+ .expand(batch_size, -1)
+ .contiguous()
+ )
+ if not self.is_smpl:
+ if left_hand_pose is None:
+ left_hand_pose = (
+ torch.zeros(self.hand_pose_dim, device=device, dtype=dtype)[None]
+ .expand(batch_size, -1)
+ .contiguous()
+ )
+ if right_hand_pose is None:
+ right_hand_pose = (
+ torch.zeros(self.hand_pose_dim, device=device, dtype=dtype)[None]
+ .expand(batch_size, -1)
+ .contiguous()
+ )
+ if betas is None:
+ betas = torch.zeros([batch_size, self.bm.num_betas], dtype=dtype, device=device)
+ if transl is None:
+ transl = torch.zeros([batch_size, 3], dtype=dtype, device=device)
+
+ bm_out = self.bm(
+ betas=betas,
+ global_orient=global_orient,
+ body_pose=body_pose,
+ left_hand_pose=left_hand_pose,
+ right_hand_pose=right_hand_pose,
+ transl=transl,
+ **kwargs
+ )
+
+ return bm_out
+
+ def get_skeleton(self, betas):
+ """betas: (*, 10) -> skeleton_beta: (*, 22, 3)"""
+ skeleton_beta = self.J_template + torch.einsum("...d, jcd -> ...jc", betas, self.J_shapedirs) # (22, 3)
+ return skeleton_beta
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/body_model_smplx.py b/third_party/GVHMR/hmr4d/utils/body_model/body_model_smplx.py
new file mode 100644
index 0000000000000000000000000000000000000000..ccc8aab7f9082c1008f6b1055b1051b54a359046
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/body_model_smplx.py
@@ -0,0 +1,132 @@
+import torch
+import torch.nn as nn
+import smplx
+
+kwargs_disable_member_var = {
+ "create_body_pose": False,
+ "create_betas": False,
+ "create_global_orient": False,
+ "create_transl": False,
+ "create_left_hand_pose": False,
+ "create_right_hand_pose": False,
+ "create_expression": False,
+ "create_jaw_pose": False,
+ "create_leye_pose": False,
+ "create_reye_pose": False,
+}
+
+
+class BodyModelSMPLX(nn.Module):
+ """Support Batch inference"""
+
+ def __init__(self, model_path, **kwargs):
+ super().__init__()
+ # enable flexible batchsize, handle missing variable at forward()
+ kwargs.update(kwargs_disable_member_var)
+ self.bm = smplx.create(model_path=model_path, **kwargs)
+ self.faces = self.bm.faces
+ self.hand_pose_dim = self.bm.num_pca_comps if self.bm.use_pca else 3 * self.bm.NUM_HAND_JOINTS
+
+ # For fast computing of skeleton under beta
+ shapedirs = self.bm.shapedirs # (V, 3, 10)
+ J_regressor = self.bm.J_regressor[:22, :] # (22, V)
+ v_template = self.bm.v_template # (V, 3)
+ J_template = J_regressor @ v_template # (22, 3)
+ J_shapedirs = torch.einsum("jv, vcd -> jcd", J_regressor, shapedirs) # (22, 3, 10)
+ self.register_buffer("J_template", J_template, False)
+ self.register_buffer("J_shapedirs", J_shapedirs, False)
+
+ def forward(
+ self,
+ betas=None,
+ global_orient=None,
+ transl=None,
+ body_pose=None,
+ left_hand_pose=None,
+ right_hand_pose=None,
+ expression=None,
+ jaw_pose=None,
+ leye_pose=None,
+ reye_pose=None,
+ **kwargs
+ ):
+
+ device, dtype = self.bm.shapedirs.device, self.bm.shapedirs.dtype
+
+ model_vars = [
+ betas,
+ global_orient,
+ body_pose,
+ transl,
+ expression,
+ left_hand_pose,
+ right_hand_pose,
+ jaw_pose,
+ leye_pose,
+ reye_pose,
+ ]
+ batch_size = 1
+ for var in model_vars:
+ if var is None:
+ continue
+ batch_size = max(batch_size, len(var))
+
+ if global_orient is None:
+ global_orient = torch.zeros([batch_size, 3], dtype=dtype, device=device)
+ if body_pose is None:
+ body_pose = (
+ torch.zeros(3 * self.bm.NUM_BODY_JOINTS, device=device, dtype=dtype)[None]
+ .expand(batch_size, -1)
+ .contiguous()
+ )
+ if left_hand_pose is None:
+ left_hand_pose = (
+ torch.zeros(self.hand_pose_dim, device=device, dtype=dtype)[None].expand(batch_size, -1).contiguous()
+ )
+ if right_hand_pose is None:
+ right_hand_pose = (
+ torch.zeros(self.hand_pose_dim, device=device, dtype=dtype)[None].expand(batch_size, -1).contiguous()
+ )
+ if jaw_pose is None:
+ jaw_pose = torch.zeros([batch_size, 3], dtype=dtype, device=device)
+ if leye_pose is None:
+ leye_pose = torch.zeros([batch_size, 3], dtype=dtype, device=device)
+ if reye_pose is None:
+ reye_pose = torch.zeros([batch_size, 3], dtype=dtype, device=device)
+ if expression is None:
+ expression = torch.zeros([batch_size, self.bm.num_expression_coeffs], dtype=dtype, device=device)
+ if betas is None:
+ betas = torch.zeros([batch_size, self.bm.num_betas], dtype=dtype, device=device)
+ if transl is None:
+ transl = torch.zeros([batch_size, 3], dtype=dtype, device=device)
+
+ bm_out = self.bm(
+ betas=betas,
+ global_orient=global_orient,
+ body_pose=body_pose,
+ left_hand_pose=left_hand_pose,
+ right_hand_pose=right_hand_pose,
+ transl=transl,
+ expression=expression,
+ jaw_pose=jaw_pose,
+ leye_pose=leye_pose,
+ reye_pose=reye_pose,
+ **kwargs
+ )
+
+ return bm_out
+
+ def get_skeleton(self, betas):
+ """betas: (*, 10) -> skeleton_beta: (*, 22, 3)"""
+ skeleton_beta = self.J_template + torch.einsum("...d, jcd -> ...jc", betas, self.J_shapedirs) # (22, 3)
+ return skeleton_beta
+
+ def forward_bfc(self, **kwargs):
+ """Wrap (B, F, C) to (B*F, C) and unwrap (B*F, C) to (B, F, C)"""
+ for k in kwargs:
+ assert len(kwargs[k].shape) == 3
+ B, F = kwargs["body_pose"].shape[:2]
+ smplx_out = self.forward(**{k: v.reshape(B * F, -1) for k, v in kwargs.items()})
+ smplx_out.vertices = smplx_out.vertices.reshape(B, F, -1, 3)
+ smplx_out.joints = smplx_out.joints.reshape(B, F, -1, 3)
+ return smplx_out
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/coco_aug_dict.pth b/third_party/GVHMR/hmr4d/utils/body_model/coco_aug_dict.pth
new file mode 100644
index 0000000000000000000000000000000000000000..5d9de9af1fad7096be9e20f1910e85f9d897948b
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/coco_aug_dict.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9d045cc3e507e1f7d91ef89904cdfa26e29cf18710f52f9fd305a88ae5d4e539
+size 1759
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/min_lbs.py b/third_party/GVHMR/hmr4d/utils/body_model/min_lbs.py
new file mode 100644
index 0000000000000000000000000000000000000000..8c7fbd957f597fca8e334e6083a12046003226a5
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/min_lbs.py
@@ -0,0 +1,106 @@
+import numpy as np
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from pytorch3d.transforms import axis_angle_to_matrix
+from smplx.utils import Struct, to_np, to_tensor
+from hmr4d.utils.smplx_utils import forward_kinematics_motion
+
+
+class MinimalLBS(nn.Module):
+ def __init__(self, sp_ids, bm_dir='models/smplh', num_betas=16, model_type='smplh', **kwargs):
+ super().__init__()
+ self.num_betas = num_betas
+ self.sensor_point_vid = torch.tensor(sp_ids)
+
+ # load struct data on predefined sensor-point
+ self.load_struct_on_sp(f'{bm_dir}/male/model.npz', prefix='male')
+ self.load_struct_on_sp(f'{bm_dir}/female/model.npz', prefix='female')
+
+ def load_struct_on_sp(self, bm_path, prefix='m'):
+ """
+ Load 4 weights from body-model-struct.
+ Keep the sensor points only. Use prefix to label different bm.
+ """
+ num_betas = self.num_betas
+ sp_vid = self.sensor_point_vid
+ # load data
+ data_struct = Struct(**np.load(bm_path, encoding='latin1'))
+
+ # v-template
+ v_template = to_tensor(to_np(data_struct.v_template)) # (V, 3)
+ v_template_sp = v_template[sp_vid] # (N, 3)
+ self.register_buffer(f'{prefix}_v_template_sp', v_template_sp, False)
+
+ # shapedirs
+ shapedirs = to_tensor(to_np(data_struct.shapedirs[:, :, :num_betas])) # (V, 3, NB)
+ shapedirs_sp = shapedirs[sp_vid]
+ self.register_buffer(f'{prefix}_shapedirs_sp', shapedirs_sp, False)
+
+ # posedirs
+ posedirs = to_tensor(to_np(data_struct.posedirs)) # (V, 3, 51*9)
+ posedirs_sp = posedirs[sp_vid]
+ posedirs_sp = posedirs_sp.reshape(len(sp_vid)*3, -1).T # (51*9, N*3)
+ self.register_buffer(f'{prefix}_posedirs_sp', posedirs_sp, False)
+
+ # lbs_weights
+ lbs_weights = to_tensor(to_np(data_struct.weights)) # (V, J+1)
+ lbs_weights_sp = lbs_weights[sp_vid]
+ self.register_buffer(f'{prefix}_lbs_weights_sp', lbs_weights_sp, False)
+
+ def forward(self, root_orient=None, pose_body=None, trans=None, betas=None, A=None, recompute_A=False, genders=None,
+ joints_zero=None):
+ """
+ Args:
+ root_orient, Optional: (B, T, 3)
+ pose_body: (B, T, J*3)
+ trans: (B, T, 3)
+ betas: (B, T, 16)
+ A, Optional: (B, T, J+1, 4, 4)
+ recompute_A: if True, root_orient should be given, otherwise use A
+ genders, List: ['male', 'female', ...]
+ joints_zero: (B, J+1, 3), required when recompute_A is True
+ Returns:
+ sensor_verts: (B, T, N, 3)
+ """
+ B, T = pose_body.shape[:2]
+
+ v_template = torch.stack([getattr(self, f'{g}_v_template_sp') for g in genders]) # (B, N, 3)
+ shapedirs = torch.stack([getattr(self, f'{g}_shapedirs_sp') for g in genders]) # (B, N, 3, NB)
+ posedirs = torch.stack([getattr(self, f'{g}_posedirs_sp') for g in genders]) # (B, 51*9, N*3)
+ lbs_weights = torch.stack([getattr(self, f'{g}_lbs_weights_sp') for g in genders]) # (B, N, J+1)
+
+ # ===== LBS, handle T ===== #
+ # 2. Add shape contribution
+ if betas.shape[1] == 1:
+ betas = betas.expand(-1, T, -1)
+ blend_shape = torch.einsum('btl,bmkl->btmk', [betas, shapedirs])
+ v_shaped = v_template[:, None] + blend_shape
+
+ # 3. Add pose blend shapes
+ ident = torch.eye(3).to(pose_body)
+ aa = pose_body.reshape(B, T, -1, 3)
+ R = axis_angle_to_matrix(aa)
+ pose_feature = (R - ident).view(B, T, -1)
+ dim_pf = pose_feature.shape[-1]
+ # (B, T, P) @ (B, P, N*3) -> (B, T, N, 3)
+ pose_offsets = torch.matmul(pose_feature, posedirs[:, :dim_pf]).view(B, T, -1, 3)
+ v_posed = pose_offsets + v_shaped
+
+ # 4. Compute A
+ if recompute_A:
+ _, _, A = forward_kinematics_motion(root_orient, pose_body, trans, joints_zero)
+
+ # 5. Skinning
+ W = lbs_weights
+ # (B, 1, N, J+1)) @ (B, T, J+1, 16)
+ num_joints = A.shape[-3] # 22
+ Ts = torch.matmul(W[:, None, :, :num_joints], A.view(B, T, num_joints, 16))
+ Ts = Ts.view(B, T, -1, 4, 4) # (B, T, N, 4, 4)
+ v_posed_homo = F.pad(v_posed, (0, 1), value=1) # (B, T, N, 4)
+ v_homo = torch.matmul(Ts, torch.unsqueeze(v_posed_homo, dim=-1))
+
+ # 6. translate
+ sensor_verts = v_homo[:, :, :, :3, 0] + trans[:, :, None]
+
+ return sensor_verts
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/seg_part_info.npy b/third_party/GVHMR/hmr4d/utils/body_model/seg_part_info.npy
new file mode 100644
index 0000000000000000000000000000000000000000..fa6f2583aac224ff7e4912c0bcc5e6f7ca043144
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/seg_part_info.npy
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a9a9272563da3d16811bc455fb618902d31f3adcf719da2e46197853045a2a10
+size 59151
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/smpl_3dpw14_J_regressor_sparse.pt b/third_party/GVHMR/hmr4d/utils/body_model/smpl_3dpw14_J_regressor_sparse.pt
new file mode 100644
index 0000000000000000000000000000000000000000..8d8cda5d595ef01f036ff4eb4c5c455a71d462fa
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/smpl_3dpw14_J_regressor_sparse.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f3f4cf476e206d36ac806beab9b728b886c10f1534a74318839391f76ad5ab0a
+size 2791
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/smpl_coco17_J_regressor.pt b/third_party/GVHMR/hmr4d/utils/body_model/smpl_coco17_J_regressor.pt
new file mode 100644
index 0000000000000000000000000000000000000000..2723218b38239856575c0c35d0d3b5516e0b01e3
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/smpl_coco17_J_regressor.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:bacdaf756629493994cc869f4c27d179f5e4a5d06b8797ee3dcb94571522079f
+size 469227
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/smpl_lite.py b/third_party/GVHMR/hmr4d/utils/body_model/smpl_lite.py
new file mode 100644
index 0000000000000000000000000000000000000000..4e84fc7ef1441783f3ff6de1b064d541696de43c
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/smpl_lite.py
@@ -0,0 +1,143 @@
+import numpy as np
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from pathlib import Path
+from pytorch3d.transforms import axis_angle_to_matrix
+from smplx.utils import Struct, to_np, to_tensor
+from einops import einsum, rearrange
+from time import time
+
+import pickle
+
+from .smplx_lite import batch_rigid_transform_v2
+
+
+class SmplLite(nn.Module):
+ def __init__(
+ self,
+ model_path="inputs/checkpoints/body_models/smpl",
+ gender="neutral",
+ num_betas=10,
+ ):
+ super().__init__()
+
+ # Load the model
+ model_path = Path(model_path)
+ if model_path.is_dir():
+ smpl_path = Path(model_path) / f"SMPL_{gender.upper()}.pkl"
+ else:
+ smpl_path = model_path
+ assert smpl_path.exists()
+ with open(smpl_path, "rb") as smpl_file:
+ data_struct = Struct(**pickle.load(smpl_file, encoding="latin1"))
+ self.faces = data_struct.f # (F, 3)
+
+ self.register_smpl_buffers(data_struct, num_betas)
+ self.register_fast_skeleton_computing_buffers()
+
+ def register_smpl_buffers(self, data_struct, num_betas):
+ # shapedirs, (V, 3, N_betas), V=10475 for SMPLX
+ shapedirs = to_tensor(to_np(data_struct.shapedirs[:, :, :num_betas])).float()
+ self.register_buffer("shapedirs", shapedirs, False)
+
+ # v_template, (V, 3)
+ v_template = to_tensor(to_np(data_struct.v_template)).float()
+ self.register_buffer("v_template", v_template, False)
+
+ # J_regressor, (J, V), J=55 for SMPLX
+ J_regressor = to_tensor(to_np(data_struct.J_regressor)).float()
+ self.register_buffer("J_regressor", J_regressor, False)
+
+ # posedirs, (54*9, V, 3), note that the first global_orient is not included
+ posedirs = to_tensor(to_np(data_struct.posedirs)).float() # (V, 3, 54*9)
+ posedirs = rearrange(posedirs, "v c n -> n v c")
+ self.register_buffer("posedirs", posedirs, False)
+
+ # lbs_weights, (V, J), J=55
+ lbs_weights = to_tensor(to_np(data_struct.weights)).float()
+ self.register_buffer("lbs_weights", lbs_weights, False)
+
+ # parents, (J), long
+ parents = to_tensor(to_np(data_struct.kintree_table[0])).long()
+ parents[0] = -1
+ self.register_buffer("parents", parents, False)
+
+ def register_fast_skeleton_computing_buffers(self):
+ # For fast computing of skeleton under beta
+ J_template = self.J_regressor @ self.v_template # (J, 3)
+ J_shapedirs = torch.einsum("jv, vcd -> jcd", self.J_regressor, self.shapedirs) # (J, 3, 10)
+ self.register_buffer("J_template", J_template, False)
+ self.register_buffer("J_shapedirs", J_shapedirs, False)
+
+ def get_skeleton(self, betas):
+ return self.J_template + einsum(betas, self.J_shapedirs, "... k, j c k -> ... j c")
+
+ def forward(
+ self,
+ body_pose,
+ betas,
+ global_orient,
+ transl,
+ ):
+ """
+ Args:
+ body_pose: (B, L, 63)
+ betas: (B, L, 10)
+ global_orient: (B, L, 3)
+ transl: (B, L, 3)
+ Returns:
+ vertices: (B, L, V, 3)
+ """
+ # 1. Convert [global_orient, body_pose] to rot_mats
+ full_pose = torch.cat([global_orient, body_pose], dim=-1)
+ rot_mats = axis_angle_to_matrix(full_pose.reshape(*full_pose.shape[:-1], full_pose.shape[-1] // 3, 3))
+
+ # 2. Forward Kinematics
+ J = self.get_skeleton(betas) # (*, 55, 3)
+ A = batch_rigid_transform_v2(rot_mats, J, self.parents)[1]
+
+ # 3. Canonical v_posed = v_template + shaped_offsets + pose_offsets
+ pose_feature = rot_mats[..., 1:, :, :] - rot_mats.new([[1, 0, 0], [0, 1, 0], [0, 0, 1]])
+ pose_feature = pose_feature.view(*pose_feature.shape[:-3], -1) # (*, 55*3*3)
+ v_posed = (
+ self.v_template
+ + einsum(betas, self.shapedirs, "... k, v c k -> ... v c")
+ + einsum(pose_feature, self.posedirs, "... k, k v c -> ... v c")
+ )
+ del pose_feature, rot_mats, full_pose
+
+ # 4. Skinning
+ T = einsum(self.lbs_weights, A, "v j, ... j c d -> ... v c d")
+ verts = einsum(T[..., :3, :3], v_posed, "... v c d, ... v d -> ... v c") + T[..., :3, 3]
+
+ # 5. Translation
+ verts = verts + transl[..., None, :]
+ return verts
+
+
+class SmplxLiteJ24(SmplLite):
+ def __init__(self, **kwargs):
+ super().__init__(**kwargs)
+
+ # Compute mapping
+ smpl2j24 = self.J_regressor # (24, 6890)
+
+ jids, smplx_vids = torch.where(smpl2j24 != 0)
+ interestd = torch.zeros([len(smplx_vids), 24])
+ for idx, (jid, smplx_vid) in enumerate(zip(jids, smplx_vids)):
+ interestd[idx, jid] = smpl2j24[jid, smplx_vid]
+ self.register_buffer("interestd", interestd, False) # (236, 24)
+
+ # Update to vertices of interest
+ self.v_template = self.v_template[smplx_vids].clone() # (V', 3)
+ self.shapedirs = self.shapedirs[smplx_vids].clone() # (V', 3, K)
+ self.posedirs = self.posedirs[:, smplx_vids].clone() # (K, V', 3)
+ self.lbs_weights = self.lbs_weights[smplx_vids].clone() # (V', J)
+
+ def forward(self, body_pose, betas, global_orient, transl):
+ """Returns: joints (*, J, 3). (B, L) or (B,) are both supported."""
+ # Use super class's forward to get verts
+ verts = super().forward(body_pose, betas, global_orient, transl) # (*, 236, 3)
+ joints = einsum(self.interestd, verts, "v j, ... v c -> ... j c")
+ return joints
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/smpl_neutral_J_regressor.pt b/third_party/GVHMR/hmr4d/utils/body_model/smpl_neutral_J_regressor.pt
new file mode 100644
index 0000000000000000000000000000000000000000..4428566cb5e090c1617d46bc0bc975ae3eda9bbd
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/smpl_neutral_J_regressor.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:70e3213bd30fe8d8ce37b54675282745e406f915a51511a003aeff99b6da04cf
+size 662187
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/smpl_vert_segmentation.json b/third_party/GVHMR/hmr4d/utils/body_model/smpl_vert_segmentation.json
new file mode 100644
index 0000000000000000000000000000000000000000..b3244cce450e13f1095a1c3af676f4c8fdea5633
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/smpl_vert_segmentation.json
@@ -0,0 +1,7440 @@
+{
+ "rightHand": [
+ 5442,
+ 5443,
+ 5444,
+ 5445,
+ 5446,
+ 5447,
+ 5448,
+ 5449,
+ 5450,
+ 5451,
+ 5452,
+ 5453,
+ 5454,
+ 5455,
+ 5456,
+ 5457,
+ 5458,
+ 5459,
+ 5460,
+ 5461,
+ 5462,
+ 5463,
+ 5464,
+ 5465,
+ 5466,
+ 5467,
+ 5468,
+ 5469,
+ 5470,
+ 5471,
+ 5472,
+ 5473,
+ 5474,
+ 5475,
+ 5476,
+ 5477,
+ 5478,
+ 5479,
+ 5480,
+ 5481,
+ 5482,
+ 5483,
+ 5484,
+ 5485,
+ 5486,
+ 5487,
+ 5492,
+ 5493,
+ 5494,
+ 5495,
+ 5496,
+ 5497,
+ 5502,
+ 5503,
+ 5504,
+ 5505,
+ 5506,
+ 5507,
+ 5508,
+ 5509,
+ 5510,
+ 5511,
+ 5512,
+ 5513,
+ 5514,
+ 5515,
+ 5516,
+ 5517,
+ 5518,
+ 5519,
+ 5520,
+ 5521,
+ 5522,
+ 5523,
+ 5524,
+ 5525,
+ 5526,
+ 5527,
+ 5530,
+ 5531,
+ 5532,
+ 5533,
+ 5534,
+ 5535,
+ 5536,
+ 5537,
+ 5538,
+ 5539,
+ 5540,
+ 5541,
+ 5542,
+ 5543,
+ 5544,
+ 5545,
+ 5546,
+ 5547,
+ 5548,
+ 5549,
+ 5550,
+ 5551,
+ 5552,
+ 5553,
+ 5554,
+ 5555,
+ 5556,
+ 5557,
+ 5558,
+ 5559,
+ 5560,
+ 5561,
+ 5562,
+ 5569,
+ 5571,
+ 5574,
+ 5575,
+ 5576,
+ 5577,
+ 5578,
+ 5579,
+ 5580,
+ 5581,
+ 5582,
+ 5583,
+ 5588,
+ 5589,
+ 5592,
+ 5593,
+ 5594,
+ 5595,
+ 5596,
+ 5597,
+ 5598,
+ 5599,
+ 5600,
+ 5601,
+ 5602,
+ 5603,
+ 5604,
+ 5605,
+ 5610,
+ 5611,
+ 5612,
+ 5613,
+ 5614,
+ 5621,
+ 5622,
+ 5625,
+ 5631,
+ 5632,
+ 5633,
+ 5634,
+ 5635,
+ 5636,
+ 5637,
+ 5638,
+ 5639,
+ 5640,
+ 5641,
+ 5643,
+ 5644,
+ 5645,
+ 5646,
+ 5649,
+ 5650,
+ 5652,
+ 5653,
+ 5654,
+ 5655,
+ 5656,
+ 5657,
+ 5658,
+ 5659,
+ 5660,
+ 5661,
+ 5662,
+ 5663,
+ 5664,
+ 5667,
+ 5670,
+ 5671,
+ 5672,
+ 5673,
+ 5674,
+ 5675,
+ 5682,
+ 5683,
+ 5684,
+ 5685,
+ 5686,
+ 5687,
+ 5688,
+ 5689,
+ 5690,
+ 5692,
+ 5695,
+ 5697,
+ 5698,
+ 5699,
+ 5700,
+ 5701,
+ 5707,
+ 5708,
+ 5709,
+ 5710,
+ 5711,
+ 5712,
+ 5713,
+ 5714,
+ 5715,
+ 5716,
+ 5717,
+ 5718,
+ 5719,
+ 5720,
+ 5721,
+ 5723,
+ 5724,
+ 5725,
+ 5726,
+ 5727,
+ 5728,
+ 5729,
+ 5730,
+ 5731,
+ 5732,
+ 5735,
+ 5736,
+ 5737,
+ 5738,
+ 5739,
+ 5740,
+ 5745,
+ 5746,
+ 5748,
+ 5749,
+ 5750,
+ 5751,
+ 5752,
+ 6056,
+ 6057,
+ 6066,
+ 6067,
+ 6158,
+ 6159,
+ 6160,
+ 6161,
+ 6162,
+ 6163,
+ 6164,
+ 6165,
+ 6166,
+ 6167,
+ 6168,
+ 6169,
+ 6170,
+ 6171,
+ 6172,
+ 6173,
+ 6174,
+ 6175,
+ 6176,
+ 6177,
+ 6178,
+ 6179,
+ 6180,
+ 6181,
+ 6182,
+ 6183,
+ 6184,
+ 6185,
+ 6186,
+ 6187,
+ 6188,
+ 6189,
+ 6190,
+ 6191,
+ 6192,
+ 6193,
+ 6194,
+ 6195,
+ 6196,
+ 6197,
+ 6198,
+ 6199,
+ 6200,
+ 6201,
+ 6202,
+ 6203,
+ 6204,
+ 6205,
+ 6206,
+ 6207,
+ 6208,
+ 6209,
+ 6210,
+ 6211,
+ 6212,
+ 6213,
+ 6214,
+ 6215,
+ 6216,
+ 6217,
+ 6218,
+ 6219,
+ 6220,
+ 6221,
+ 6222,
+ 6223,
+ 6224,
+ 6225,
+ 6226,
+ 6227,
+ 6228,
+ 6229,
+ 6230,
+ 6231,
+ 6232,
+ 6233,
+ 6234,
+ 6235,
+ 6236,
+ 6237,
+ 6238,
+ 6239
+ ],
+ "rightUpLeg": [
+ 4320,
+ 4321,
+ 4323,
+ 4324,
+ 4333,
+ 4334,
+ 4335,
+ 4336,
+ 4337,
+ 4338,
+ 4339,
+ 4340,
+ 4356,
+ 4357,
+ 4358,
+ 4359,
+ 4360,
+ 4361,
+ 4362,
+ 4363,
+ 4364,
+ 4365,
+ 4366,
+ 4367,
+ 4383,
+ 4384,
+ 4385,
+ 4386,
+ 4387,
+ 4388,
+ 4389,
+ 4390,
+ 4391,
+ 4392,
+ 4393,
+ 4394,
+ 4395,
+ 4396,
+ 4397,
+ 4398,
+ 4399,
+ 4400,
+ 4401,
+ 4419,
+ 4420,
+ 4421,
+ 4422,
+ 4430,
+ 4431,
+ 4432,
+ 4433,
+ 4434,
+ 4435,
+ 4436,
+ 4437,
+ 4438,
+ 4439,
+ 4440,
+ 4441,
+ 4442,
+ 4443,
+ 4444,
+ 4445,
+ 4446,
+ 4447,
+ 4448,
+ 4449,
+ 4450,
+ 4451,
+ 4452,
+ 4453,
+ 4454,
+ 4455,
+ 4456,
+ 4457,
+ 4458,
+ 4459,
+ 4460,
+ 4461,
+ 4462,
+ 4463,
+ 4464,
+ 4465,
+ 4466,
+ 4467,
+ 4468,
+ 4469,
+ 4470,
+ 4471,
+ 4472,
+ 4473,
+ 4474,
+ 4475,
+ 4476,
+ 4477,
+ 4478,
+ 4479,
+ 4480,
+ 4481,
+ 4482,
+ 4483,
+ 4484,
+ 4485,
+ 4486,
+ 4487,
+ 4488,
+ 4489,
+ 4490,
+ 4491,
+ 4492,
+ 4493,
+ 4494,
+ 4495,
+ 4496,
+ 4497,
+ 4498,
+ 4499,
+ 4500,
+ 4501,
+ 4502,
+ 4503,
+ 4504,
+ 4505,
+ 4506,
+ 4507,
+ 4508,
+ 4509,
+ 4510,
+ 4511,
+ 4512,
+ 4513,
+ 4514,
+ 4515,
+ 4516,
+ 4517,
+ 4518,
+ 4519,
+ 4520,
+ 4521,
+ 4522,
+ 4523,
+ 4524,
+ 4525,
+ 4526,
+ 4527,
+ 4528,
+ 4529,
+ 4530,
+ 4531,
+ 4532,
+ 4623,
+ 4624,
+ 4625,
+ 4626,
+ 4627,
+ 4628,
+ 4629,
+ 4630,
+ 4631,
+ 4632,
+ 4633,
+ 4634,
+ 4645,
+ 4646,
+ 4647,
+ 4648,
+ 4649,
+ 4650,
+ 4651,
+ 4652,
+ 4653,
+ 4654,
+ 4655,
+ 4656,
+ 4657,
+ 4658,
+ 4659,
+ 4660,
+ 4670,
+ 4671,
+ 4672,
+ 4673,
+ 4704,
+ 4705,
+ 4706,
+ 4707,
+ 4708,
+ 4709,
+ 4710,
+ 4711,
+ 4712,
+ 4713,
+ 4745,
+ 4746,
+ 4757,
+ 4758,
+ 4759,
+ 4760,
+ 4801,
+ 4802,
+ 4829,
+ 4834,
+ 4835,
+ 4836,
+ 4837,
+ 4838,
+ 4839,
+ 4840,
+ 4841,
+ 4924,
+ 4925,
+ 4926,
+ 4928,
+ 4929,
+ 4930,
+ 4931,
+ 4932,
+ 4933,
+ 4934,
+ 4935,
+ 4936,
+ 4948,
+ 4949,
+ 4950,
+ 4951,
+ 4952,
+ 4970,
+ 4971,
+ 4972,
+ 4973,
+ 4983,
+ 4984,
+ 4985,
+ 4986,
+ 4987,
+ 4988,
+ 4989,
+ 4990,
+ 4991,
+ 4992,
+ 4993,
+ 5004,
+ 5005,
+ 6546,
+ 6547,
+ 6548,
+ 6549,
+ 6552,
+ 6553,
+ 6554,
+ 6555,
+ 6556,
+ 6873,
+ 6877
+ ],
+ "leftArm": [
+ 626,
+ 627,
+ 628,
+ 629,
+ 634,
+ 635,
+ 680,
+ 681,
+ 716,
+ 717,
+ 718,
+ 719,
+ 769,
+ 770,
+ 771,
+ 772,
+ 773,
+ 774,
+ 775,
+ 776,
+ 777,
+ 778,
+ 779,
+ 780,
+ 784,
+ 785,
+ 786,
+ 787,
+ 788,
+ 789,
+ 790,
+ 791,
+ 792,
+ 793,
+ 1231,
+ 1232,
+ 1233,
+ 1234,
+ 1258,
+ 1259,
+ 1260,
+ 1261,
+ 1271,
+ 1281,
+ 1282,
+ 1310,
+ 1311,
+ 1314,
+ 1315,
+ 1340,
+ 1341,
+ 1342,
+ 1343,
+ 1355,
+ 1356,
+ 1357,
+ 1358,
+ 1376,
+ 1377,
+ 1378,
+ 1379,
+ 1380,
+ 1381,
+ 1382,
+ 1383,
+ 1384,
+ 1385,
+ 1386,
+ 1387,
+ 1388,
+ 1389,
+ 1390,
+ 1391,
+ 1392,
+ 1393,
+ 1394,
+ 1395,
+ 1396,
+ 1397,
+ 1398,
+ 1399,
+ 1400,
+ 1402,
+ 1403,
+ 1405,
+ 1406,
+ 1407,
+ 1408,
+ 1409,
+ 1410,
+ 1411,
+ 1412,
+ 1413,
+ 1414,
+ 1415,
+ 1416,
+ 1428,
+ 1429,
+ 1430,
+ 1431,
+ 1432,
+ 1433,
+ 1438,
+ 1439,
+ 1440,
+ 1441,
+ 1442,
+ 1443,
+ 1444,
+ 1445,
+ 1502,
+ 1505,
+ 1506,
+ 1507,
+ 1508,
+ 1509,
+ 1510,
+ 1538,
+ 1541,
+ 1542,
+ 1543,
+ 1545,
+ 1619,
+ 1620,
+ 1621,
+ 1622,
+ 1631,
+ 1632,
+ 1633,
+ 1634,
+ 1635,
+ 1636,
+ 1637,
+ 1638,
+ 1639,
+ 1640,
+ 1641,
+ 1642,
+ 1645,
+ 1646,
+ 1647,
+ 1648,
+ 1649,
+ 1650,
+ 1651,
+ 1652,
+ 1653,
+ 1654,
+ 1655,
+ 1656,
+ 1658,
+ 1659,
+ 1661,
+ 1662,
+ 1664,
+ 1666,
+ 1667,
+ 1668,
+ 1669,
+ 1670,
+ 1671,
+ 1672,
+ 1673,
+ 1674,
+ 1675,
+ 1676,
+ 1677,
+ 1678,
+ 1679,
+ 1680,
+ 1681,
+ 1682,
+ 1683,
+ 1684,
+ 1696,
+ 1697,
+ 1698,
+ 1703,
+ 1704,
+ 1705,
+ 1706,
+ 1707,
+ 1708,
+ 1709,
+ 1710,
+ 1711,
+ 1712,
+ 1713,
+ 1714,
+ 1715,
+ 1716,
+ 1717,
+ 1718,
+ 1719,
+ 1720,
+ 1725,
+ 1731,
+ 1732,
+ 1733,
+ 1734,
+ 1735,
+ 1737,
+ 1739,
+ 1740,
+ 1745,
+ 1746,
+ 1747,
+ 1748,
+ 1749,
+ 1751,
+ 1761,
+ 1830,
+ 1831,
+ 1844,
+ 1845,
+ 1846,
+ 1850,
+ 1851,
+ 1854,
+ 1855,
+ 1858,
+ 1860,
+ 1865,
+ 1866,
+ 1867,
+ 1869,
+ 1870,
+ 1871,
+ 1874,
+ 1875,
+ 1876,
+ 1877,
+ 1878,
+ 1882,
+ 1883,
+ 1888,
+ 1889,
+ 1892,
+ 1900,
+ 1901,
+ 1902,
+ 1903,
+ 1904,
+ 1909,
+ 2819,
+ 2820,
+ 2821,
+ 2822,
+ 2895,
+ 2896,
+ 2897,
+ 2898,
+ 2899,
+ 2900,
+ 2901,
+ 2902,
+ 2903,
+ 2945,
+ 2946,
+ 2974,
+ 2975,
+ 2976,
+ 2977,
+ 2978,
+ 2979,
+ 2980,
+ 2981,
+ 2982,
+ 2983,
+ 2984,
+ 2985,
+ 2986,
+ 2987,
+ 2988,
+ 2989,
+ 2990,
+ 2991,
+ 2992,
+ 2993,
+ 2994,
+ 2995,
+ 2996,
+ 3002,
+ 3013
+ ],
+ "leftLeg": [
+ 995,
+ 998,
+ 999,
+ 1002,
+ 1004,
+ 1005,
+ 1008,
+ 1010,
+ 1012,
+ 1015,
+ 1016,
+ 1018,
+ 1019,
+ 1043,
+ 1044,
+ 1047,
+ 1048,
+ 1049,
+ 1050,
+ 1051,
+ 1052,
+ 1053,
+ 1054,
+ 1055,
+ 1056,
+ 1057,
+ 1058,
+ 1059,
+ 1060,
+ 1061,
+ 1062,
+ 1063,
+ 1064,
+ 1065,
+ 1066,
+ 1067,
+ 1068,
+ 1069,
+ 1070,
+ 1071,
+ 1072,
+ 1073,
+ 1074,
+ 1075,
+ 1076,
+ 1077,
+ 1078,
+ 1079,
+ 1080,
+ 1081,
+ 1082,
+ 1083,
+ 1084,
+ 1085,
+ 1086,
+ 1087,
+ 1088,
+ 1089,
+ 1090,
+ 1091,
+ 1092,
+ 1093,
+ 1094,
+ 1095,
+ 1096,
+ 1097,
+ 1098,
+ 1099,
+ 1100,
+ 1101,
+ 1102,
+ 1103,
+ 1104,
+ 1105,
+ 1106,
+ 1107,
+ 1108,
+ 1109,
+ 1110,
+ 1111,
+ 1112,
+ 1113,
+ 1114,
+ 1115,
+ 1116,
+ 1117,
+ 1118,
+ 1119,
+ 1120,
+ 1121,
+ 1122,
+ 1123,
+ 1124,
+ 1125,
+ 1126,
+ 1127,
+ 1128,
+ 1129,
+ 1130,
+ 1131,
+ 1132,
+ 1133,
+ 1134,
+ 1135,
+ 1136,
+ 1148,
+ 1149,
+ 1150,
+ 1151,
+ 1152,
+ 1153,
+ 1154,
+ 1155,
+ 1156,
+ 1157,
+ 1158,
+ 1175,
+ 1176,
+ 1177,
+ 1178,
+ 1179,
+ 1180,
+ 1181,
+ 1182,
+ 1183,
+ 1369,
+ 1370,
+ 1371,
+ 1372,
+ 1373,
+ 1374,
+ 1375,
+ 1464,
+ 1465,
+ 1466,
+ 1467,
+ 1468,
+ 1469,
+ 1470,
+ 1471,
+ 1472,
+ 1473,
+ 1474,
+ 1522,
+ 1523,
+ 1524,
+ 1525,
+ 1526,
+ 1527,
+ 1528,
+ 1529,
+ 1530,
+ 1531,
+ 1532,
+ 3174,
+ 3175,
+ 3176,
+ 3177,
+ 3178,
+ 3179,
+ 3180,
+ 3181,
+ 3182,
+ 3183,
+ 3184,
+ 3185,
+ 3186,
+ 3187,
+ 3188,
+ 3189,
+ 3190,
+ 3191,
+ 3192,
+ 3193,
+ 3194,
+ 3195,
+ 3196,
+ 3197,
+ 3198,
+ 3199,
+ 3200,
+ 3201,
+ 3202,
+ 3203,
+ 3204,
+ 3205,
+ 3206,
+ 3207,
+ 3208,
+ 3209,
+ 3210,
+ 3319,
+ 3320,
+ 3321,
+ 3322,
+ 3323,
+ 3324,
+ 3325,
+ 3326,
+ 3327,
+ 3328,
+ 3329,
+ 3330,
+ 3331,
+ 3332,
+ 3333,
+ 3334,
+ 3335,
+ 3432,
+ 3433,
+ 3434,
+ 3435,
+ 3436,
+ 3469,
+ 3472,
+ 3473,
+ 3474
+ ],
+ "leftToeBase": [
+ 3211,
+ 3212,
+ 3213,
+ 3214,
+ 3215,
+ 3216,
+ 3217,
+ 3218,
+ 3219,
+ 3220,
+ 3221,
+ 3222,
+ 3223,
+ 3224,
+ 3225,
+ 3226,
+ 3227,
+ 3228,
+ 3229,
+ 3230,
+ 3231,
+ 3232,
+ 3233,
+ 3234,
+ 3235,
+ 3236,
+ 3237,
+ 3238,
+ 3239,
+ 3240,
+ 3241,
+ 3242,
+ 3243,
+ 3244,
+ 3245,
+ 3246,
+ 3247,
+ 3248,
+ 3249,
+ 3250,
+ 3251,
+ 3252,
+ 3253,
+ 3254,
+ 3255,
+ 3256,
+ 3257,
+ 3258,
+ 3259,
+ 3260,
+ 3261,
+ 3262,
+ 3263,
+ 3264,
+ 3265,
+ 3266,
+ 3267,
+ 3268,
+ 3269,
+ 3270,
+ 3271,
+ 3272,
+ 3273,
+ 3274,
+ 3275,
+ 3276,
+ 3277,
+ 3278,
+ 3279,
+ 3280,
+ 3281,
+ 3282,
+ 3283,
+ 3284,
+ 3285,
+ 3286,
+ 3287,
+ 3288,
+ 3289,
+ 3290,
+ 3291,
+ 3292,
+ 3293,
+ 3294,
+ 3295,
+ 3296,
+ 3297,
+ 3298,
+ 3299,
+ 3300,
+ 3301,
+ 3302,
+ 3303,
+ 3304,
+ 3305,
+ 3306,
+ 3307,
+ 3308,
+ 3309,
+ 3310,
+ 3311,
+ 3312,
+ 3313,
+ 3314,
+ 3315,
+ 3316,
+ 3317,
+ 3318,
+ 3336,
+ 3337,
+ 3340,
+ 3342,
+ 3344,
+ 3346,
+ 3348,
+ 3350,
+ 3352,
+ 3354,
+ 3357,
+ 3358,
+ 3360,
+ 3362
+ ],
+ "leftFoot": [
+ 3327,
+ 3328,
+ 3329,
+ 3330,
+ 3331,
+ 3332,
+ 3333,
+ 3334,
+ 3335,
+ 3336,
+ 3337,
+ 3338,
+ 3339,
+ 3340,
+ 3341,
+ 3342,
+ 3343,
+ 3344,
+ 3345,
+ 3346,
+ 3347,
+ 3348,
+ 3349,
+ 3350,
+ 3351,
+ 3352,
+ 3353,
+ 3354,
+ 3355,
+ 3356,
+ 3357,
+ 3358,
+ 3359,
+ 3360,
+ 3361,
+ 3362,
+ 3363,
+ 3364,
+ 3365,
+ 3366,
+ 3367,
+ 3368,
+ 3369,
+ 3370,
+ 3371,
+ 3372,
+ 3373,
+ 3374,
+ 3375,
+ 3376,
+ 3377,
+ 3378,
+ 3379,
+ 3380,
+ 3381,
+ 3382,
+ 3383,
+ 3384,
+ 3385,
+ 3386,
+ 3387,
+ 3388,
+ 3389,
+ 3390,
+ 3391,
+ 3392,
+ 3393,
+ 3394,
+ 3395,
+ 3396,
+ 3397,
+ 3398,
+ 3399,
+ 3400,
+ 3401,
+ 3402,
+ 3403,
+ 3404,
+ 3405,
+ 3406,
+ 3407,
+ 3408,
+ 3409,
+ 3410,
+ 3411,
+ 3412,
+ 3413,
+ 3414,
+ 3415,
+ 3416,
+ 3417,
+ 3418,
+ 3419,
+ 3420,
+ 3421,
+ 3422,
+ 3423,
+ 3424,
+ 3425,
+ 3426,
+ 3427,
+ 3428,
+ 3429,
+ 3430,
+ 3431,
+ 3432,
+ 3433,
+ 3434,
+ 3435,
+ 3436,
+ 3437,
+ 3438,
+ 3439,
+ 3440,
+ 3441,
+ 3442,
+ 3443,
+ 3444,
+ 3445,
+ 3446,
+ 3447,
+ 3448,
+ 3449,
+ 3450,
+ 3451,
+ 3452,
+ 3453,
+ 3454,
+ 3455,
+ 3456,
+ 3457,
+ 3458,
+ 3459,
+ 3460,
+ 3461,
+ 3462,
+ 3463,
+ 3464,
+ 3465,
+ 3466,
+ 3467,
+ 3468,
+ 3469
+ ],
+ "spine1": [
+ 598,
+ 599,
+ 600,
+ 601,
+ 610,
+ 611,
+ 612,
+ 613,
+ 614,
+ 615,
+ 616,
+ 617,
+ 618,
+ 619,
+ 620,
+ 621,
+ 642,
+ 645,
+ 646,
+ 647,
+ 652,
+ 653,
+ 658,
+ 659,
+ 660,
+ 661,
+ 668,
+ 669,
+ 670,
+ 671,
+ 684,
+ 685,
+ 686,
+ 687,
+ 688,
+ 689,
+ 690,
+ 691,
+ 692,
+ 722,
+ 723,
+ 724,
+ 725,
+ 736,
+ 750,
+ 751,
+ 761,
+ 764,
+ 766,
+ 767,
+ 794,
+ 795,
+ 891,
+ 892,
+ 893,
+ 894,
+ 925,
+ 926,
+ 927,
+ 928,
+ 929,
+ 940,
+ 941,
+ 942,
+ 943,
+ 1190,
+ 1191,
+ 1192,
+ 1193,
+ 1194,
+ 1195,
+ 1196,
+ 1197,
+ 1200,
+ 1201,
+ 1202,
+ 1212,
+ 1236,
+ 1252,
+ 1253,
+ 1254,
+ 1255,
+ 1268,
+ 1269,
+ 1270,
+ 1329,
+ 1330,
+ 1348,
+ 1349,
+ 1351,
+ 1420,
+ 1421,
+ 1423,
+ 1424,
+ 1425,
+ 1426,
+ 1436,
+ 1437,
+ 1756,
+ 1757,
+ 1758,
+ 2839,
+ 2840,
+ 2841,
+ 2842,
+ 2843,
+ 2844,
+ 2845,
+ 2846,
+ 2847,
+ 2848,
+ 2849,
+ 2850,
+ 2851,
+ 2870,
+ 2871,
+ 2883,
+ 2906,
+ 2908,
+ 3014,
+ 3017,
+ 3025,
+ 3030,
+ 3033,
+ 3034,
+ 3037,
+ 3039,
+ 3040,
+ 3041,
+ 3042,
+ 3043,
+ 3044,
+ 3076,
+ 3077,
+ 3079,
+ 3480,
+ 3505,
+ 3511,
+ 4086,
+ 4087,
+ 4088,
+ 4089,
+ 4098,
+ 4099,
+ 4100,
+ 4101,
+ 4102,
+ 4103,
+ 4104,
+ 4105,
+ 4106,
+ 4107,
+ 4108,
+ 4109,
+ 4130,
+ 4131,
+ 4134,
+ 4135,
+ 4140,
+ 4141,
+ 4146,
+ 4147,
+ 4148,
+ 4149,
+ 4156,
+ 4157,
+ 4158,
+ 4159,
+ 4172,
+ 4173,
+ 4174,
+ 4175,
+ 4176,
+ 4177,
+ 4178,
+ 4179,
+ 4180,
+ 4210,
+ 4211,
+ 4212,
+ 4213,
+ 4225,
+ 4239,
+ 4240,
+ 4249,
+ 4250,
+ 4255,
+ 4256,
+ 4282,
+ 4283,
+ 4377,
+ 4378,
+ 4379,
+ 4380,
+ 4411,
+ 4412,
+ 4413,
+ 4414,
+ 4415,
+ 4426,
+ 4427,
+ 4428,
+ 4429,
+ 4676,
+ 4677,
+ 4678,
+ 4679,
+ 4680,
+ 4681,
+ 4682,
+ 4683,
+ 4686,
+ 4687,
+ 4688,
+ 4695,
+ 4719,
+ 4735,
+ 4736,
+ 4737,
+ 4740,
+ 4751,
+ 4752,
+ 4753,
+ 4824,
+ 4825,
+ 4828,
+ 4893,
+ 4894,
+ 4895,
+ 4897,
+ 4898,
+ 4899,
+ 4908,
+ 4909,
+ 5223,
+ 5224,
+ 5225,
+ 6300,
+ 6301,
+ 6302,
+ 6303,
+ 6304,
+ 6305,
+ 6306,
+ 6307,
+ 6308,
+ 6309,
+ 6310,
+ 6311,
+ 6312,
+ 6331,
+ 6332,
+ 6342,
+ 6366,
+ 6367,
+ 6475,
+ 6477,
+ 6478,
+ 6481,
+ 6482,
+ 6485,
+ 6487,
+ 6488,
+ 6489,
+ 6490,
+ 6491,
+ 6878
+ ],
+ "spine2": [
+ 570,
+ 571,
+ 572,
+ 573,
+ 584,
+ 585,
+ 586,
+ 587,
+ 588,
+ 589,
+ 590,
+ 591,
+ 592,
+ 593,
+ 594,
+ 595,
+ 596,
+ 597,
+ 602,
+ 603,
+ 604,
+ 605,
+ 606,
+ 607,
+ 608,
+ 609,
+ 622,
+ 623,
+ 624,
+ 625,
+ 638,
+ 639,
+ 640,
+ 641,
+ 643,
+ 644,
+ 648,
+ 649,
+ 650,
+ 651,
+ 666,
+ 667,
+ 672,
+ 673,
+ 674,
+ 675,
+ 680,
+ 681,
+ 682,
+ 683,
+ 693,
+ 694,
+ 695,
+ 696,
+ 697,
+ 698,
+ 699,
+ 700,
+ 701,
+ 702,
+ 703,
+ 704,
+ 713,
+ 714,
+ 715,
+ 716,
+ 717,
+ 726,
+ 727,
+ 728,
+ 729,
+ 730,
+ 731,
+ 732,
+ 733,
+ 735,
+ 737,
+ 738,
+ 739,
+ 740,
+ 741,
+ 742,
+ 743,
+ 744,
+ 745,
+ 746,
+ 747,
+ 748,
+ 749,
+ 752,
+ 753,
+ 754,
+ 755,
+ 756,
+ 757,
+ 758,
+ 759,
+ 760,
+ 762,
+ 763,
+ 803,
+ 804,
+ 805,
+ 806,
+ 811,
+ 812,
+ 813,
+ 814,
+ 817,
+ 818,
+ 819,
+ 820,
+ 821,
+ 824,
+ 825,
+ 826,
+ 827,
+ 828,
+ 895,
+ 896,
+ 930,
+ 931,
+ 1198,
+ 1199,
+ 1213,
+ 1214,
+ 1215,
+ 1216,
+ 1217,
+ 1218,
+ 1219,
+ 1220,
+ 1235,
+ 1237,
+ 1256,
+ 1257,
+ 1271,
+ 1272,
+ 1273,
+ 1279,
+ 1280,
+ 1283,
+ 1284,
+ 1285,
+ 1286,
+ 1287,
+ 1288,
+ 1289,
+ 1290,
+ 1291,
+ 1292,
+ 1293,
+ 1294,
+ 1295,
+ 1296,
+ 1297,
+ 1298,
+ 1299,
+ 1300,
+ 1301,
+ 1302,
+ 1303,
+ 1304,
+ 1305,
+ 1306,
+ 1307,
+ 1308,
+ 1309,
+ 1312,
+ 1313,
+ 1319,
+ 1320,
+ 1346,
+ 1347,
+ 1350,
+ 1352,
+ 1401,
+ 1417,
+ 1418,
+ 1419,
+ 1422,
+ 1427,
+ 1434,
+ 1435,
+ 1503,
+ 1504,
+ 1536,
+ 1537,
+ 1544,
+ 1545,
+ 1753,
+ 1754,
+ 1755,
+ 1759,
+ 1760,
+ 1761,
+ 1762,
+ 1763,
+ 1808,
+ 1809,
+ 1810,
+ 1811,
+ 1816,
+ 1817,
+ 1818,
+ 1819,
+ 1820,
+ 1834,
+ 1835,
+ 1836,
+ 1837,
+ 1838,
+ 1839,
+ 1868,
+ 1879,
+ 1880,
+ 2812,
+ 2813,
+ 2852,
+ 2853,
+ 2854,
+ 2855,
+ 2856,
+ 2857,
+ 2858,
+ 2859,
+ 2860,
+ 2861,
+ 2862,
+ 2863,
+ 2864,
+ 2865,
+ 2866,
+ 2867,
+ 2868,
+ 2869,
+ 2872,
+ 2875,
+ 2876,
+ 2877,
+ 2878,
+ 2881,
+ 2882,
+ 2884,
+ 2885,
+ 2886,
+ 2904,
+ 2905,
+ 2907,
+ 2931,
+ 2932,
+ 2933,
+ 2934,
+ 2935,
+ 2936,
+ 2937,
+ 2941,
+ 2950,
+ 2951,
+ 2952,
+ 2953,
+ 2954,
+ 2955,
+ 2956,
+ 2957,
+ 2958,
+ 2959,
+ 2960,
+ 2961,
+ 2962,
+ 2963,
+ 2964,
+ 2965,
+ 2966,
+ 2967,
+ 2968,
+ 2969,
+ 2970,
+ 2971,
+ 2972,
+ 2973,
+ 2997,
+ 2998,
+ 3006,
+ 3007,
+ 3012,
+ 3015,
+ 3026,
+ 3027,
+ 3028,
+ 3029,
+ 3031,
+ 3032,
+ 3035,
+ 3036,
+ 3038,
+ 3059,
+ 3060,
+ 3061,
+ 3062,
+ 3063,
+ 3064,
+ 3065,
+ 3066,
+ 3067,
+ 3073,
+ 3074,
+ 3075,
+ 3078,
+ 3168,
+ 3169,
+ 3171,
+ 3470,
+ 3471,
+ 3482,
+ 3483,
+ 3495,
+ 3496,
+ 3497,
+ 3498,
+ 3506,
+ 3508,
+ 4058,
+ 4059,
+ 4060,
+ 4061,
+ 4072,
+ 4073,
+ 4074,
+ 4075,
+ 4076,
+ 4077,
+ 4078,
+ 4079,
+ 4080,
+ 4081,
+ 4082,
+ 4083,
+ 4084,
+ 4085,
+ 4090,
+ 4091,
+ 4092,
+ 4093,
+ 4094,
+ 4095,
+ 4096,
+ 4097,
+ 4110,
+ 4111,
+ 4112,
+ 4113,
+ 4126,
+ 4127,
+ 4128,
+ 4129,
+ 4132,
+ 4133,
+ 4136,
+ 4137,
+ 4138,
+ 4139,
+ 4154,
+ 4155,
+ 4160,
+ 4161,
+ 4162,
+ 4163,
+ 4168,
+ 4169,
+ 4170,
+ 4171,
+ 4181,
+ 4182,
+ 4183,
+ 4184,
+ 4185,
+ 4186,
+ 4187,
+ 4188,
+ 4189,
+ 4190,
+ 4191,
+ 4192,
+ 4201,
+ 4202,
+ 4203,
+ 4204,
+ 4207,
+ 4214,
+ 4215,
+ 4216,
+ 4217,
+ 4218,
+ 4219,
+ 4220,
+ 4221,
+ 4223,
+ 4224,
+ 4226,
+ 4227,
+ 4228,
+ 4229,
+ 4230,
+ 4231,
+ 4232,
+ 4233,
+ 4234,
+ 4235,
+ 4236,
+ 4237,
+ 4238,
+ 4241,
+ 4242,
+ 4243,
+ 4244,
+ 4245,
+ 4246,
+ 4247,
+ 4248,
+ 4251,
+ 4252,
+ 4291,
+ 4292,
+ 4293,
+ 4294,
+ 4299,
+ 4300,
+ 4301,
+ 4302,
+ 4305,
+ 4306,
+ 4307,
+ 4308,
+ 4309,
+ 4312,
+ 4313,
+ 4314,
+ 4315,
+ 4381,
+ 4382,
+ 4416,
+ 4417,
+ 4684,
+ 4685,
+ 4696,
+ 4697,
+ 4698,
+ 4699,
+ 4700,
+ 4701,
+ 4702,
+ 4703,
+ 4718,
+ 4720,
+ 4738,
+ 4739,
+ 4754,
+ 4755,
+ 4756,
+ 4761,
+ 4762,
+ 4765,
+ 4766,
+ 4767,
+ 4768,
+ 4769,
+ 4770,
+ 4771,
+ 4772,
+ 4773,
+ 4774,
+ 4775,
+ 4776,
+ 4777,
+ 4778,
+ 4779,
+ 4780,
+ 4781,
+ 4782,
+ 4783,
+ 4784,
+ 4785,
+ 4786,
+ 4787,
+ 4788,
+ 4789,
+ 4792,
+ 4793,
+ 4799,
+ 4800,
+ 4822,
+ 4823,
+ 4826,
+ 4827,
+ 4874,
+ 4890,
+ 4891,
+ 4892,
+ 4896,
+ 4900,
+ 4907,
+ 4910,
+ 4975,
+ 4976,
+ 5007,
+ 5008,
+ 5013,
+ 5014,
+ 5222,
+ 5226,
+ 5227,
+ 5228,
+ 5229,
+ 5230,
+ 5269,
+ 5270,
+ 5271,
+ 5272,
+ 5277,
+ 5278,
+ 5279,
+ 5280,
+ 5281,
+ 5295,
+ 5296,
+ 5297,
+ 5298,
+ 5299,
+ 5300,
+ 5329,
+ 5340,
+ 5341,
+ 6273,
+ 6274,
+ 6313,
+ 6314,
+ 6315,
+ 6316,
+ 6317,
+ 6318,
+ 6319,
+ 6320,
+ 6321,
+ 6322,
+ 6323,
+ 6324,
+ 6325,
+ 6326,
+ 6327,
+ 6328,
+ 6329,
+ 6330,
+ 6333,
+ 6336,
+ 6337,
+ 6340,
+ 6341,
+ 6343,
+ 6344,
+ 6345,
+ 6363,
+ 6364,
+ 6365,
+ 6390,
+ 6391,
+ 6392,
+ 6393,
+ 6394,
+ 6395,
+ 6396,
+ 6398,
+ 6409,
+ 6410,
+ 6411,
+ 6412,
+ 6413,
+ 6414,
+ 6415,
+ 6416,
+ 6417,
+ 6418,
+ 6419,
+ 6420,
+ 6421,
+ 6422,
+ 6423,
+ 6424,
+ 6425,
+ 6426,
+ 6427,
+ 6428,
+ 6429,
+ 6430,
+ 6431,
+ 6432,
+ 6456,
+ 6457,
+ 6465,
+ 6466,
+ 6476,
+ 6479,
+ 6480,
+ 6483,
+ 6484,
+ 6486,
+ 6496,
+ 6497,
+ 6498,
+ 6499,
+ 6500,
+ 6501,
+ 6502,
+ 6503,
+ 6879
+ ],
+ "leftShoulder": [
+ 591,
+ 604,
+ 605,
+ 606,
+ 609,
+ 634,
+ 635,
+ 636,
+ 637,
+ 674,
+ 706,
+ 707,
+ 708,
+ 709,
+ 710,
+ 711,
+ 712,
+ 713,
+ 715,
+ 717,
+ 730,
+ 733,
+ 734,
+ 735,
+ 781,
+ 782,
+ 783,
+ 1238,
+ 1239,
+ 1240,
+ 1241,
+ 1242,
+ 1243,
+ 1244,
+ 1245,
+ 1290,
+ 1291,
+ 1294,
+ 1316,
+ 1317,
+ 1318,
+ 1401,
+ 1402,
+ 1403,
+ 1404,
+ 1509,
+ 1535,
+ 1545,
+ 1808,
+ 1810,
+ 1811,
+ 1812,
+ 1813,
+ 1814,
+ 1815,
+ 1818,
+ 1819,
+ 1821,
+ 1822,
+ 1823,
+ 1824,
+ 1825,
+ 1826,
+ 1827,
+ 1828,
+ 1829,
+ 1830,
+ 1831,
+ 1832,
+ 1833,
+ 1837,
+ 1840,
+ 1841,
+ 1842,
+ 1843,
+ 1844,
+ 1845,
+ 1846,
+ 1847,
+ 1848,
+ 1849,
+ 1850,
+ 1851,
+ 1852,
+ 1853,
+ 1854,
+ 1855,
+ 1856,
+ 1857,
+ 1858,
+ 1859,
+ 1861,
+ 1862,
+ 1863,
+ 1864,
+ 1872,
+ 1873,
+ 1880,
+ 1881,
+ 1884,
+ 1885,
+ 1886,
+ 1887,
+ 1890,
+ 1891,
+ 1893,
+ 1894,
+ 1895,
+ 1896,
+ 1897,
+ 1898,
+ 1899,
+ 2879,
+ 2880,
+ 2881,
+ 2886,
+ 2887,
+ 2888,
+ 2889,
+ 2890,
+ 2891,
+ 2892,
+ 2893,
+ 2894,
+ 2903,
+ 2938,
+ 2939,
+ 2940,
+ 2941,
+ 2942,
+ 2943,
+ 2944,
+ 2945,
+ 2946,
+ 2947,
+ 2948,
+ 2949,
+ 2965,
+ 2967,
+ 2969,
+ 2999,
+ 3000,
+ 3001,
+ 3002,
+ 3003,
+ 3004,
+ 3005,
+ 3008,
+ 3009,
+ 3010,
+ 3011
+ ],
+ "rightShoulder": [
+ 4077,
+ 4091,
+ 4092,
+ 4094,
+ 4095,
+ 4122,
+ 4123,
+ 4124,
+ 4125,
+ 4162,
+ 4194,
+ 4195,
+ 4196,
+ 4197,
+ 4198,
+ 4199,
+ 4200,
+ 4201,
+ 4203,
+ 4207,
+ 4218,
+ 4219,
+ 4222,
+ 4223,
+ 4269,
+ 4270,
+ 4271,
+ 4721,
+ 4722,
+ 4723,
+ 4724,
+ 4725,
+ 4726,
+ 4727,
+ 4728,
+ 4773,
+ 4774,
+ 4778,
+ 4796,
+ 4797,
+ 4798,
+ 4874,
+ 4875,
+ 4876,
+ 4877,
+ 4982,
+ 5006,
+ 5014,
+ 5269,
+ 5271,
+ 5272,
+ 5273,
+ 5274,
+ 5275,
+ 5276,
+ 5279,
+ 5281,
+ 5282,
+ 5283,
+ 5284,
+ 5285,
+ 5286,
+ 5287,
+ 5288,
+ 5289,
+ 5290,
+ 5291,
+ 5292,
+ 5293,
+ 5294,
+ 5298,
+ 5301,
+ 5302,
+ 5303,
+ 5304,
+ 5305,
+ 5306,
+ 5307,
+ 5308,
+ 5309,
+ 5310,
+ 5311,
+ 5312,
+ 5313,
+ 5314,
+ 5315,
+ 5316,
+ 5317,
+ 5318,
+ 5319,
+ 5320,
+ 5322,
+ 5323,
+ 5324,
+ 5325,
+ 5333,
+ 5334,
+ 5341,
+ 5342,
+ 5345,
+ 5346,
+ 5347,
+ 5348,
+ 5351,
+ 5352,
+ 5354,
+ 5355,
+ 5356,
+ 5357,
+ 5358,
+ 5359,
+ 5360,
+ 6338,
+ 6339,
+ 6340,
+ 6345,
+ 6346,
+ 6347,
+ 6348,
+ 6349,
+ 6350,
+ 6351,
+ 6352,
+ 6353,
+ 6362,
+ 6397,
+ 6398,
+ 6399,
+ 6400,
+ 6401,
+ 6402,
+ 6403,
+ 6404,
+ 6405,
+ 6406,
+ 6407,
+ 6408,
+ 6424,
+ 6425,
+ 6428,
+ 6458,
+ 6459,
+ 6460,
+ 6461,
+ 6462,
+ 6463,
+ 6464,
+ 6467,
+ 6468,
+ 6469,
+ 6470
+ ],
+ "rightFoot": [
+ 6727,
+ 6728,
+ 6729,
+ 6730,
+ 6731,
+ 6732,
+ 6733,
+ 6734,
+ 6735,
+ 6736,
+ 6737,
+ 6738,
+ 6739,
+ 6740,
+ 6741,
+ 6742,
+ 6743,
+ 6744,
+ 6745,
+ 6746,
+ 6747,
+ 6748,
+ 6749,
+ 6750,
+ 6751,
+ 6752,
+ 6753,
+ 6754,
+ 6755,
+ 6756,
+ 6757,
+ 6758,
+ 6759,
+ 6760,
+ 6761,
+ 6762,
+ 6763,
+ 6764,
+ 6765,
+ 6766,
+ 6767,
+ 6768,
+ 6769,
+ 6770,
+ 6771,
+ 6772,
+ 6773,
+ 6774,
+ 6775,
+ 6776,
+ 6777,
+ 6778,
+ 6779,
+ 6780,
+ 6781,
+ 6782,
+ 6783,
+ 6784,
+ 6785,
+ 6786,
+ 6787,
+ 6788,
+ 6789,
+ 6790,
+ 6791,
+ 6792,
+ 6793,
+ 6794,
+ 6795,
+ 6796,
+ 6797,
+ 6798,
+ 6799,
+ 6800,
+ 6801,
+ 6802,
+ 6803,
+ 6804,
+ 6805,
+ 6806,
+ 6807,
+ 6808,
+ 6809,
+ 6810,
+ 6811,
+ 6812,
+ 6813,
+ 6814,
+ 6815,
+ 6816,
+ 6817,
+ 6818,
+ 6819,
+ 6820,
+ 6821,
+ 6822,
+ 6823,
+ 6824,
+ 6825,
+ 6826,
+ 6827,
+ 6828,
+ 6829,
+ 6830,
+ 6831,
+ 6832,
+ 6833,
+ 6834,
+ 6835,
+ 6836,
+ 6837,
+ 6838,
+ 6839,
+ 6840,
+ 6841,
+ 6842,
+ 6843,
+ 6844,
+ 6845,
+ 6846,
+ 6847,
+ 6848,
+ 6849,
+ 6850,
+ 6851,
+ 6852,
+ 6853,
+ 6854,
+ 6855,
+ 6856,
+ 6857,
+ 6858,
+ 6859,
+ 6860,
+ 6861,
+ 6862,
+ 6863,
+ 6864,
+ 6865,
+ 6866,
+ 6867,
+ 6868,
+ 6869
+ ],
+ "head": [
+ 0,
+ 1,
+ 2,
+ 3,
+ 4,
+ 5,
+ 6,
+ 7,
+ 8,
+ 9,
+ 10,
+ 11,
+ 12,
+ 13,
+ 14,
+ 15,
+ 16,
+ 17,
+ 18,
+ 19,
+ 20,
+ 21,
+ 22,
+ 23,
+ 24,
+ 25,
+ 26,
+ 27,
+ 28,
+ 29,
+ 30,
+ 31,
+ 32,
+ 33,
+ 34,
+ 35,
+ 36,
+ 37,
+ 38,
+ 39,
+ 40,
+ 41,
+ 42,
+ 43,
+ 44,
+ 45,
+ 46,
+ 47,
+ 48,
+ 49,
+ 50,
+ 51,
+ 52,
+ 53,
+ 54,
+ 55,
+ 56,
+ 57,
+ 58,
+ 59,
+ 60,
+ 61,
+ 62,
+ 63,
+ 64,
+ 65,
+ 66,
+ 67,
+ 68,
+ 69,
+ 70,
+ 71,
+ 72,
+ 73,
+ 74,
+ 75,
+ 76,
+ 77,
+ 78,
+ 79,
+ 80,
+ 81,
+ 82,
+ 83,
+ 84,
+ 85,
+ 86,
+ 87,
+ 88,
+ 89,
+ 90,
+ 91,
+ 92,
+ 93,
+ 94,
+ 95,
+ 96,
+ 97,
+ 98,
+ 99,
+ 100,
+ 101,
+ 102,
+ 103,
+ 104,
+ 105,
+ 106,
+ 107,
+ 108,
+ 109,
+ 110,
+ 111,
+ 112,
+ 113,
+ 114,
+ 115,
+ 116,
+ 117,
+ 118,
+ 119,
+ 120,
+ 121,
+ 122,
+ 123,
+ 124,
+ 125,
+ 126,
+ 127,
+ 128,
+ 129,
+ 130,
+ 131,
+ 132,
+ 133,
+ 134,
+ 135,
+ 136,
+ 137,
+ 138,
+ 139,
+ 140,
+ 141,
+ 142,
+ 143,
+ 144,
+ 145,
+ 146,
+ 147,
+ 148,
+ 149,
+ 154,
+ 155,
+ 156,
+ 157,
+ 158,
+ 159,
+ 160,
+ 161,
+ 162,
+ 163,
+ 164,
+ 165,
+ 166,
+ 167,
+ 168,
+ 169,
+ 170,
+ 171,
+ 172,
+ 173,
+ 176,
+ 177,
+ 178,
+ 179,
+ 180,
+ 181,
+ 182,
+ 183,
+ 184,
+ 185,
+ 186,
+ 187,
+ 188,
+ 189,
+ 190,
+ 191,
+ 192,
+ 193,
+ 194,
+ 195,
+ 196,
+ 197,
+ 198,
+ 199,
+ 200,
+ 201,
+ 202,
+ 203,
+ 204,
+ 205,
+ 220,
+ 221,
+ 225,
+ 226,
+ 227,
+ 228,
+ 229,
+ 230,
+ 231,
+ 232,
+ 233,
+ 234,
+ 235,
+ 236,
+ 237,
+ 238,
+ 239,
+ 240,
+ 241,
+ 242,
+ 243,
+ 244,
+ 245,
+ 246,
+ 247,
+ 248,
+ 249,
+ 250,
+ 251,
+ 252,
+ 253,
+ 254,
+ 255,
+ 258,
+ 259,
+ 260,
+ 261,
+ 262,
+ 263,
+ 264,
+ 265,
+ 266,
+ 267,
+ 268,
+ 269,
+ 270,
+ 271,
+ 272,
+ 273,
+ 274,
+ 275,
+ 276,
+ 277,
+ 278,
+ 279,
+ 280,
+ 281,
+ 282,
+ 283,
+ 286,
+ 287,
+ 288,
+ 289,
+ 290,
+ 291,
+ 292,
+ 293,
+ 294,
+ 295,
+ 303,
+ 304,
+ 306,
+ 307,
+ 310,
+ 311,
+ 312,
+ 313,
+ 314,
+ 315,
+ 316,
+ 317,
+ 318,
+ 319,
+ 320,
+ 321,
+ 322,
+ 323,
+ 324,
+ 325,
+ 326,
+ 327,
+ 328,
+ 329,
+ 330,
+ 331,
+ 332,
+ 335,
+ 336,
+ 337,
+ 338,
+ 339,
+ 340,
+ 341,
+ 342,
+ 343,
+ 344,
+ 345,
+ 346,
+ 347,
+ 348,
+ 349,
+ 350,
+ 351,
+ 352,
+ 353,
+ 354,
+ 355,
+ 356,
+ 357,
+ 358,
+ 359,
+ 360,
+ 361,
+ 362,
+ 363,
+ 364,
+ 365,
+ 366,
+ 367,
+ 368,
+ 369,
+ 370,
+ 371,
+ 372,
+ 373,
+ 374,
+ 375,
+ 376,
+ 377,
+ 378,
+ 379,
+ 380,
+ 381,
+ 382,
+ 383,
+ 384,
+ 385,
+ 386,
+ 387,
+ 388,
+ 389,
+ 390,
+ 391,
+ 392,
+ 393,
+ 394,
+ 395,
+ 396,
+ 397,
+ 398,
+ 399,
+ 400,
+ 401,
+ 402,
+ 403,
+ 404,
+ 405,
+ 406,
+ 407,
+ 408,
+ 409,
+ 410,
+ 411,
+ 412,
+ 413,
+ 414,
+ 415,
+ 416,
+ 417,
+ 418,
+ 419,
+ 420,
+ 421,
+ 422,
+ 427,
+ 428,
+ 429,
+ 430,
+ 431,
+ 432,
+ 433,
+ 434,
+ 435,
+ 436,
+ 437,
+ 438,
+ 439,
+ 442,
+ 443,
+ 444,
+ 445,
+ 446,
+ 447,
+ 448,
+ 449,
+ 450,
+ 454,
+ 455,
+ 456,
+ 457,
+ 458,
+ 459,
+ 461,
+ 462,
+ 463,
+ 464,
+ 465,
+ 466,
+ 467,
+ 468,
+ 469,
+ 470,
+ 471,
+ 472,
+ 473,
+ 474,
+ 475,
+ 476,
+ 477,
+ 478,
+ 479,
+ 480,
+ 481,
+ 482,
+ 483,
+ 484,
+ 485,
+ 486,
+ 487,
+ 488,
+ 489,
+ 490,
+ 491,
+ 492,
+ 493,
+ 494,
+ 495,
+ 496,
+ 497,
+ 498,
+ 499,
+ 500,
+ 501,
+ 502,
+ 503,
+ 504,
+ 505,
+ 506,
+ 507,
+ 508,
+ 509,
+ 510,
+ 511,
+ 512,
+ 513,
+ 514,
+ 515,
+ 516,
+ 517,
+ 518,
+ 519,
+ 520,
+ 521,
+ 522,
+ 523,
+ 524,
+ 525,
+ 526,
+ 527,
+ 528,
+ 529,
+ 530,
+ 531,
+ 532,
+ 533,
+ 534,
+ 535,
+ 536,
+ 537,
+ 538,
+ 539,
+ 540,
+ 541,
+ 542,
+ 543,
+ 544,
+ 545,
+ 546,
+ 547,
+ 548,
+ 549,
+ 550,
+ 551,
+ 552,
+ 553,
+ 554,
+ 555,
+ 556,
+ 557,
+ 558,
+ 559,
+ 560,
+ 561,
+ 562,
+ 563,
+ 564,
+ 565,
+ 566,
+ 567,
+ 568,
+ 569,
+ 574,
+ 575,
+ 576,
+ 577,
+ 578,
+ 579,
+ 580,
+ 581,
+ 582,
+ 583,
+ 1764,
+ 1765,
+ 1766,
+ 1770,
+ 1771,
+ 1772,
+ 1773,
+ 1774,
+ 1775,
+ 1776,
+ 1777,
+ 1778,
+ 1905,
+ 1906,
+ 1907,
+ 1908,
+ 2779,
+ 2780,
+ 2781,
+ 2782,
+ 2783,
+ 2784,
+ 2785,
+ 2786,
+ 2787,
+ 2788,
+ 2789,
+ 2790,
+ 2791,
+ 2792,
+ 2793,
+ 2794,
+ 2795,
+ 2796,
+ 2797,
+ 2798,
+ 2799,
+ 2800,
+ 2801,
+ 2802,
+ 2803,
+ 2804,
+ 2805,
+ 2806,
+ 2807,
+ 2808,
+ 2809,
+ 2810,
+ 2811,
+ 2814,
+ 2815,
+ 2816,
+ 2817,
+ 2818,
+ 3045,
+ 3046,
+ 3047,
+ 3048,
+ 3051,
+ 3052,
+ 3053,
+ 3054,
+ 3055,
+ 3056,
+ 3058,
+ 3069,
+ 3070,
+ 3071,
+ 3072,
+ 3161,
+ 3162,
+ 3163,
+ 3165,
+ 3166,
+ 3167,
+ 3485,
+ 3486,
+ 3487,
+ 3488,
+ 3489,
+ 3490,
+ 3491,
+ 3492,
+ 3493,
+ 3494,
+ 3499,
+ 3512,
+ 3513,
+ 3514,
+ 3515,
+ 3516,
+ 3517,
+ 3518,
+ 3519,
+ 3520,
+ 3521,
+ 3522,
+ 3523,
+ 3524,
+ 3525,
+ 3526,
+ 3527,
+ 3528,
+ 3529,
+ 3530,
+ 3531,
+ 3532,
+ 3533,
+ 3534,
+ 3535,
+ 3536,
+ 3537,
+ 3538,
+ 3539,
+ 3540,
+ 3541,
+ 3542,
+ 3543,
+ 3544,
+ 3545,
+ 3546,
+ 3547,
+ 3548,
+ 3549,
+ 3550,
+ 3551,
+ 3552,
+ 3553,
+ 3554,
+ 3555,
+ 3556,
+ 3557,
+ 3558,
+ 3559,
+ 3560,
+ 3561,
+ 3562,
+ 3563,
+ 3564,
+ 3565,
+ 3566,
+ 3567,
+ 3568,
+ 3569,
+ 3570,
+ 3571,
+ 3572,
+ 3573,
+ 3574,
+ 3575,
+ 3576,
+ 3577,
+ 3578,
+ 3579,
+ 3580,
+ 3581,
+ 3582,
+ 3583,
+ 3584,
+ 3585,
+ 3586,
+ 3587,
+ 3588,
+ 3589,
+ 3590,
+ 3591,
+ 3592,
+ 3593,
+ 3594,
+ 3595,
+ 3596,
+ 3597,
+ 3598,
+ 3599,
+ 3600,
+ 3601,
+ 3602,
+ 3603,
+ 3604,
+ 3605,
+ 3606,
+ 3607,
+ 3608,
+ 3609,
+ 3610,
+ 3611,
+ 3612,
+ 3613,
+ 3614,
+ 3615,
+ 3616,
+ 3617,
+ 3618,
+ 3619,
+ 3620,
+ 3621,
+ 3622,
+ 3623,
+ 3624,
+ 3625,
+ 3626,
+ 3627,
+ 3628,
+ 3629,
+ 3630,
+ 3631,
+ 3632,
+ 3633,
+ 3634,
+ 3635,
+ 3636,
+ 3637,
+ 3638,
+ 3639,
+ 3640,
+ 3641,
+ 3642,
+ 3643,
+ 3644,
+ 3645,
+ 3646,
+ 3647,
+ 3648,
+ 3649,
+ 3650,
+ 3651,
+ 3652,
+ 3653,
+ 3654,
+ 3655,
+ 3656,
+ 3657,
+ 3658,
+ 3659,
+ 3660,
+ 3661,
+ 3666,
+ 3667,
+ 3668,
+ 3669,
+ 3670,
+ 3671,
+ 3672,
+ 3673,
+ 3674,
+ 3675,
+ 3676,
+ 3677,
+ 3678,
+ 3679,
+ 3680,
+ 3681,
+ 3682,
+ 3683,
+ 3684,
+ 3685,
+ 3688,
+ 3689,
+ 3690,
+ 3691,
+ 3692,
+ 3693,
+ 3694,
+ 3695,
+ 3696,
+ 3697,
+ 3698,
+ 3699,
+ 3700,
+ 3701,
+ 3702,
+ 3703,
+ 3704,
+ 3705,
+ 3706,
+ 3707,
+ 3708,
+ 3709,
+ 3710,
+ 3711,
+ 3712,
+ 3713,
+ 3714,
+ 3715,
+ 3716,
+ 3717,
+ 3732,
+ 3733,
+ 3737,
+ 3738,
+ 3739,
+ 3740,
+ 3741,
+ 3742,
+ 3743,
+ 3744,
+ 3745,
+ 3746,
+ 3747,
+ 3748,
+ 3749,
+ 3750,
+ 3751,
+ 3752,
+ 3753,
+ 3754,
+ 3755,
+ 3756,
+ 3757,
+ 3758,
+ 3759,
+ 3760,
+ 3761,
+ 3762,
+ 3763,
+ 3764,
+ 3765,
+ 3766,
+ 3767,
+ 3770,
+ 3771,
+ 3772,
+ 3773,
+ 3774,
+ 3775,
+ 3776,
+ 3777,
+ 3778,
+ 3779,
+ 3780,
+ 3781,
+ 3782,
+ 3783,
+ 3784,
+ 3785,
+ 3786,
+ 3787,
+ 3788,
+ 3789,
+ 3790,
+ 3791,
+ 3792,
+ 3793,
+ 3794,
+ 3795,
+ 3798,
+ 3799,
+ 3800,
+ 3801,
+ 3802,
+ 3803,
+ 3804,
+ 3805,
+ 3806,
+ 3807,
+ 3815,
+ 3816,
+ 3819,
+ 3820,
+ 3821,
+ 3822,
+ 3823,
+ 3824,
+ 3825,
+ 3826,
+ 3827,
+ 3828,
+ 3829,
+ 3830,
+ 3831,
+ 3832,
+ 3833,
+ 3834,
+ 3835,
+ 3836,
+ 3837,
+ 3838,
+ 3841,
+ 3842,
+ 3843,
+ 3844,
+ 3845,
+ 3846,
+ 3847,
+ 3848,
+ 3849,
+ 3850,
+ 3851,
+ 3852,
+ 3853,
+ 3854,
+ 3855,
+ 3856,
+ 3857,
+ 3858,
+ 3859,
+ 3860,
+ 3861,
+ 3862,
+ 3863,
+ 3864,
+ 3865,
+ 3866,
+ 3867,
+ 3868,
+ 3869,
+ 3870,
+ 3871,
+ 3872,
+ 3873,
+ 3874,
+ 3875,
+ 3876,
+ 3877,
+ 3878,
+ 3879,
+ 3880,
+ 3881,
+ 3882,
+ 3883,
+ 3884,
+ 3885,
+ 3886,
+ 3887,
+ 3888,
+ 3889,
+ 3890,
+ 3891,
+ 3892,
+ 3893,
+ 3894,
+ 3895,
+ 3896,
+ 3897,
+ 3898,
+ 3899,
+ 3900,
+ 3901,
+ 3902,
+ 3903,
+ 3904,
+ 3905,
+ 3906,
+ 3907,
+ 3908,
+ 3909,
+ 3910,
+ 3911,
+ 3912,
+ 3913,
+ 3914,
+ 3915,
+ 3916,
+ 3917,
+ 3922,
+ 3923,
+ 3924,
+ 3925,
+ 3926,
+ 3927,
+ 3928,
+ 3929,
+ 3930,
+ 3931,
+ 3932,
+ 3933,
+ 3936,
+ 3937,
+ 3938,
+ 3939,
+ 3940,
+ 3941,
+ 3945,
+ 3946,
+ 3947,
+ 3948,
+ 3949,
+ 3950,
+ 3951,
+ 3952,
+ 3953,
+ 3954,
+ 3955,
+ 3956,
+ 3957,
+ 3958,
+ 3959,
+ 3960,
+ 3961,
+ 3962,
+ 3963,
+ 3964,
+ 3965,
+ 3966,
+ 3967,
+ 3968,
+ 3969,
+ 3970,
+ 3971,
+ 3972,
+ 3973,
+ 3974,
+ 3975,
+ 3976,
+ 3977,
+ 3978,
+ 3979,
+ 3980,
+ 3981,
+ 3982,
+ 3983,
+ 3984,
+ 3985,
+ 3986,
+ 3987,
+ 3988,
+ 3989,
+ 3990,
+ 3991,
+ 3992,
+ 3993,
+ 3994,
+ 3995,
+ 3996,
+ 3997,
+ 3998,
+ 3999,
+ 4000,
+ 4001,
+ 4002,
+ 4003,
+ 4004,
+ 4005,
+ 4006,
+ 4007,
+ 4008,
+ 4009,
+ 4010,
+ 4011,
+ 4012,
+ 4013,
+ 4014,
+ 4015,
+ 4016,
+ 4017,
+ 4018,
+ 4019,
+ 4020,
+ 4021,
+ 4022,
+ 4023,
+ 4024,
+ 4025,
+ 4026,
+ 4027,
+ 4028,
+ 4029,
+ 4030,
+ 4031,
+ 4032,
+ 4033,
+ 4034,
+ 4035,
+ 4036,
+ 4037,
+ 4038,
+ 4039,
+ 4040,
+ 4041,
+ 4042,
+ 4043,
+ 4044,
+ 4045,
+ 4046,
+ 4047,
+ 4048,
+ 4049,
+ 4050,
+ 4051,
+ 4052,
+ 4053,
+ 4054,
+ 4055,
+ 4056,
+ 4057,
+ 4062,
+ 4063,
+ 4064,
+ 4065,
+ 4066,
+ 4067,
+ 4068,
+ 4069,
+ 4070,
+ 4071,
+ 5231,
+ 5232,
+ 5233,
+ 5235,
+ 5236,
+ 5237,
+ 5238,
+ 5239,
+ 5240,
+ 5241,
+ 5242,
+ 5243,
+ 5366,
+ 5367,
+ 5368,
+ 5369,
+ 6240,
+ 6241,
+ 6242,
+ 6243,
+ 6244,
+ 6245,
+ 6246,
+ 6247,
+ 6248,
+ 6249,
+ 6250,
+ 6251,
+ 6252,
+ 6253,
+ 6254,
+ 6255,
+ 6256,
+ 6257,
+ 6258,
+ 6259,
+ 6260,
+ 6261,
+ 6262,
+ 6263,
+ 6264,
+ 6265,
+ 6266,
+ 6267,
+ 6268,
+ 6269,
+ 6270,
+ 6271,
+ 6272,
+ 6275,
+ 6276,
+ 6277,
+ 6278,
+ 6279,
+ 6492,
+ 6493,
+ 6494,
+ 6495,
+ 6880,
+ 6881,
+ 6882,
+ 6883,
+ 6884,
+ 6885,
+ 6886,
+ 6887,
+ 6888,
+ 6889
+ ],
+ "rightArm": [
+ 4114,
+ 4115,
+ 4116,
+ 4117,
+ 4122,
+ 4125,
+ 4168,
+ 4171,
+ 4204,
+ 4205,
+ 4206,
+ 4207,
+ 4257,
+ 4258,
+ 4259,
+ 4260,
+ 4261,
+ 4262,
+ 4263,
+ 4264,
+ 4265,
+ 4266,
+ 4267,
+ 4268,
+ 4272,
+ 4273,
+ 4274,
+ 4275,
+ 4276,
+ 4277,
+ 4278,
+ 4279,
+ 4280,
+ 4281,
+ 4714,
+ 4715,
+ 4716,
+ 4717,
+ 4741,
+ 4742,
+ 4743,
+ 4744,
+ 4756,
+ 4763,
+ 4764,
+ 4790,
+ 4791,
+ 4794,
+ 4795,
+ 4816,
+ 4817,
+ 4818,
+ 4819,
+ 4830,
+ 4831,
+ 4832,
+ 4833,
+ 4849,
+ 4850,
+ 4851,
+ 4852,
+ 4853,
+ 4854,
+ 4855,
+ 4856,
+ 4857,
+ 4858,
+ 4859,
+ 4860,
+ 4861,
+ 4862,
+ 4863,
+ 4864,
+ 4865,
+ 4866,
+ 4867,
+ 4868,
+ 4869,
+ 4870,
+ 4871,
+ 4872,
+ 4873,
+ 4876,
+ 4877,
+ 4878,
+ 4879,
+ 4880,
+ 4881,
+ 4882,
+ 4883,
+ 4884,
+ 4885,
+ 4886,
+ 4887,
+ 4888,
+ 4889,
+ 4901,
+ 4902,
+ 4903,
+ 4904,
+ 4905,
+ 4906,
+ 4911,
+ 4912,
+ 4913,
+ 4914,
+ 4915,
+ 4916,
+ 4917,
+ 4918,
+ 4974,
+ 4977,
+ 4978,
+ 4979,
+ 4980,
+ 4981,
+ 4982,
+ 5009,
+ 5010,
+ 5011,
+ 5012,
+ 5014,
+ 5088,
+ 5089,
+ 5090,
+ 5091,
+ 5100,
+ 5101,
+ 5102,
+ 5103,
+ 5104,
+ 5105,
+ 5106,
+ 5107,
+ 5108,
+ 5109,
+ 5110,
+ 5111,
+ 5114,
+ 5115,
+ 5116,
+ 5117,
+ 5118,
+ 5119,
+ 5120,
+ 5121,
+ 5122,
+ 5123,
+ 5124,
+ 5125,
+ 5128,
+ 5129,
+ 5130,
+ 5131,
+ 5134,
+ 5135,
+ 5136,
+ 5137,
+ 5138,
+ 5139,
+ 5140,
+ 5141,
+ 5142,
+ 5143,
+ 5144,
+ 5145,
+ 5146,
+ 5147,
+ 5148,
+ 5149,
+ 5150,
+ 5151,
+ 5152,
+ 5153,
+ 5165,
+ 5166,
+ 5167,
+ 5172,
+ 5173,
+ 5174,
+ 5175,
+ 5176,
+ 5177,
+ 5178,
+ 5179,
+ 5180,
+ 5181,
+ 5182,
+ 5183,
+ 5184,
+ 5185,
+ 5186,
+ 5187,
+ 5188,
+ 5189,
+ 5194,
+ 5200,
+ 5201,
+ 5202,
+ 5203,
+ 5204,
+ 5206,
+ 5208,
+ 5209,
+ 5214,
+ 5215,
+ 5216,
+ 5217,
+ 5218,
+ 5220,
+ 5229,
+ 5292,
+ 5293,
+ 5303,
+ 5306,
+ 5309,
+ 5311,
+ 5314,
+ 5315,
+ 5318,
+ 5319,
+ 5321,
+ 5326,
+ 5327,
+ 5328,
+ 5330,
+ 5331,
+ 5332,
+ 5335,
+ 5336,
+ 5337,
+ 5338,
+ 5339,
+ 5343,
+ 5344,
+ 5349,
+ 5350,
+ 5353,
+ 5361,
+ 5362,
+ 5363,
+ 5364,
+ 5365,
+ 5370,
+ 6280,
+ 6281,
+ 6282,
+ 6283,
+ 6354,
+ 6355,
+ 6356,
+ 6357,
+ 6358,
+ 6359,
+ 6360,
+ 6361,
+ 6362,
+ 6404,
+ 6405,
+ 6433,
+ 6434,
+ 6435,
+ 6436,
+ 6437,
+ 6438,
+ 6439,
+ 6440,
+ 6441,
+ 6442,
+ 6443,
+ 6444,
+ 6445,
+ 6446,
+ 6447,
+ 6448,
+ 6449,
+ 6450,
+ 6451,
+ 6452,
+ 6453,
+ 6454,
+ 6455,
+ 6461,
+ 6471
+ ],
+ "leftHandIndex1": [
+ 2027,
+ 2028,
+ 2029,
+ 2030,
+ 2037,
+ 2038,
+ 2039,
+ 2040,
+ 2057,
+ 2067,
+ 2068,
+ 2123,
+ 2124,
+ 2125,
+ 2126,
+ 2127,
+ 2128,
+ 2129,
+ 2130,
+ 2132,
+ 2145,
+ 2146,
+ 2152,
+ 2153,
+ 2154,
+ 2156,
+ 2157,
+ 2158,
+ 2159,
+ 2160,
+ 2161,
+ 2162,
+ 2163,
+ 2164,
+ 2165,
+ 2166,
+ 2167,
+ 2168,
+ 2169,
+ 2177,
+ 2178,
+ 2179,
+ 2181,
+ 2186,
+ 2187,
+ 2190,
+ 2191,
+ 2204,
+ 2205,
+ 2215,
+ 2216,
+ 2217,
+ 2218,
+ 2219,
+ 2220,
+ 2232,
+ 2233,
+ 2245,
+ 2246,
+ 2247,
+ 2258,
+ 2259,
+ 2261,
+ 2262,
+ 2263,
+ 2269,
+ 2270,
+ 2272,
+ 2273,
+ 2274,
+ 2276,
+ 2277,
+ 2280,
+ 2281,
+ 2282,
+ 2283,
+ 2291,
+ 2292,
+ 2293,
+ 2294,
+ 2295,
+ 2296,
+ 2297,
+ 2298,
+ 2299,
+ 2300,
+ 2301,
+ 2302,
+ 2303,
+ 2304,
+ 2305,
+ 2306,
+ 2307,
+ 2308,
+ 2309,
+ 2310,
+ 2311,
+ 2312,
+ 2313,
+ 2314,
+ 2315,
+ 2316,
+ 2317,
+ 2318,
+ 2319,
+ 2320,
+ 2321,
+ 2322,
+ 2323,
+ 2324,
+ 2325,
+ 2326,
+ 2327,
+ 2328,
+ 2329,
+ 2330,
+ 2331,
+ 2332,
+ 2333,
+ 2334,
+ 2335,
+ 2336,
+ 2337,
+ 2338,
+ 2339,
+ 2340,
+ 2341,
+ 2342,
+ 2343,
+ 2344,
+ 2345,
+ 2346,
+ 2347,
+ 2348,
+ 2349,
+ 2350,
+ 2351,
+ 2352,
+ 2353,
+ 2354,
+ 2355,
+ 2356,
+ 2357,
+ 2358,
+ 2359,
+ 2360,
+ 2361,
+ 2362,
+ 2363,
+ 2364,
+ 2365,
+ 2366,
+ 2367,
+ 2368,
+ 2369,
+ 2370,
+ 2371,
+ 2372,
+ 2373,
+ 2374,
+ 2375,
+ 2376,
+ 2377,
+ 2378,
+ 2379,
+ 2380,
+ 2381,
+ 2382,
+ 2383,
+ 2384,
+ 2385,
+ 2386,
+ 2387,
+ 2388,
+ 2389,
+ 2390,
+ 2391,
+ 2392,
+ 2393,
+ 2394,
+ 2395,
+ 2396,
+ 2397,
+ 2398,
+ 2399,
+ 2400,
+ 2401,
+ 2402,
+ 2403,
+ 2404,
+ 2405,
+ 2406,
+ 2407,
+ 2408,
+ 2409,
+ 2410,
+ 2411,
+ 2412,
+ 2413,
+ 2414,
+ 2415,
+ 2416,
+ 2417,
+ 2418,
+ 2419,
+ 2420,
+ 2421,
+ 2422,
+ 2423,
+ 2424,
+ 2425,
+ 2426,
+ 2427,
+ 2428,
+ 2429,
+ 2430,
+ 2431,
+ 2432,
+ 2433,
+ 2434,
+ 2435,
+ 2436,
+ 2437,
+ 2438,
+ 2439,
+ 2440,
+ 2441,
+ 2442,
+ 2443,
+ 2444,
+ 2445,
+ 2446,
+ 2447,
+ 2448,
+ 2449,
+ 2450,
+ 2451,
+ 2452,
+ 2453,
+ 2454,
+ 2455,
+ 2456,
+ 2457,
+ 2458,
+ 2459,
+ 2460,
+ 2461,
+ 2462,
+ 2463,
+ 2464,
+ 2465,
+ 2466,
+ 2467,
+ 2468,
+ 2469,
+ 2470,
+ 2471,
+ 2472,
+ 2473,
+ 2474,
+ 2475,
+ 2476,
+ 2477,
+ 2478,
+ 2479,
+ 2480,
+ 2481,
+ 2482,
+ 2483,
+ 2484,
+ 2485,
+ 2486,
+ 2487,
+ 2488,
+ 2489,
+ 2490,
+ 2491,
+ 2492,
+ 2493,
+ 2494,
+ 2495,
+ 2496,
+ 2497,
+ 2498,
+ 2499,
+ 2500,
+ 2501,
+ 2502,
+ 2503,
+ 2504,
+ 2505,
+ 2506,
+ 2507,
+ 2508,
+ 2509,
+ 2510,
+ 2511,
+ 2512,
+ 2513,
+ 2514,
+ 2515,
+ 2516,
+ 2517,
+ 2518,
+ 2519,
+ 2520,
+ 2521,
+ 2522,
+ 2523,
+ 2524,
+ 2525,
+ 2526,
+ 2527,
+ 2528,
+ 2529,
+ 2530,
+ 2531,
+ 2532,
+ 2533,
+ 2534,
+ 2535,
+ 2536,
+ 2537,
+ 2538,
+ 2539,
+ 2540,
+ 2541,
+ 2542,
+ 2543,
+ 2544,
+ 2545,
+ 2546,
+ 2547,
+ 2548,
+ 2549,
+ 2550,
+ 2551,
+ 2552,
+ 2553,
+ 2554,
+ 2555,
+ 2556,
+ 2557,
+ 2558,
+ 2559,
+ 2560,
+ 2561,
+ 2562,
+ 2563,
+ 2564,
+ 2565,
+ 2566,
+ 2567,
+ 2568,
+ 2569,
+ 2570,
+ 2571,
+ 2572,
+ 2573,
+ 2574,
+ 2575,
+ 2576,
+ 2577,
+ 2578,
+ 2579,
+ 2580,
+ 2581,
+ 2582,
+ 2583,
+ 2584,
+ 2585,
+ 2586,
+ 2587,
+ 2588,
+ 2589,
+ 2590,
+ 2591,
+ 2592,
+ 2593,
+ 2594,
+ 2596,
+ 2597,
+ 2599,
+ 2600,
+ 2601,
+ 2602,
+ 2603,
+ 2604,
+ 2606,
+ 2607,
+ 2609,
+ 2610,
+ 2611,
+ 2612,
+ 2613,
+ 2614,
+ 2615,
+ 2616,
+ 2617,
+ 2618,
+ 2619,
+ 2620,
+ 2621,
+ 2622,
+ 2623,
+ 2624,
+ 2625,
+ 2626,
+ 2627,
+ 2628,
+ 2629,
+ 2630,
+ 2631,
+ 2632,
+ 2633,
+ 2634,
+ 2635,
+ 2636,
+ 2637,
+ 2638,
+ 2639,
+ 2640,
+ 2641,
+ 2642,
+ 2643,
+ 2644,
+ 2645,
+ 2646,
+ 2647,
+ 2648,
+ 2649,
+ 2650,
+ 2651,
+ 2652,
+ 2653,
+ 2654,
+ 2655,
+ 2656,
+ 2657,
+ 2658,
+ 2659,
+ 2660,
+ 2661,
+ 2662,
+ 2663,
+ 2664,
+ 2665,
+ 2666,
+ 2667,
+ 2668,
+ 2669,
+ 2670,
+ 2671,
+ 2672,
+ 2673,
+ 2674,
+ 2675,
+ 2676,
+ 2677,
+ 2678,
+ 2679,
+ 2680,
+ 2681,
+ 2682,
+ 2683,
+ 2684,
+ 2685,
+ 2686,
+ 2687,
+ 2688,
+ 2689,
+ 2690,
+ 2691,
+ 2692,
+ 2693,
+ 2694,
+ 2695,
+ 2696
+ ],
+ "rightLeg": [
+ 4481,
+ 4482,
+ 4485,
+ 4486,
+ 4491,
+ 4492,
+ 4493,
+ 4495,
+ 4498,
+ 4500,
+ 4501,
+ 4505,
+ 4506,
+ 4529,
+ 4532,
+ 4533,
+ 4534,
+ 4535,
+ 4536,
+ 4537,
+ 4538,
+ 4539,
+ 4540,
+ 4541,
+ 4542,
+ 4543,
+ 4544,
+ 4545,
+ 4546,
+ 4547,
+ 4548,
+ 4549,
+ 4550,
+ 4551,
+ 4552,
+ 4553,
+ 4554,
+ 4555,
+ 4556,
+ 4557,
+ 4558,
+ 4559,
+ 4560,
+ 4561,
+ 4562,
+ 4563,
+ 4564,
+ 4565,
+ 4566,
+ 4567,
+ 4568,
+ 4569,
+ 4570,
+ 4571,
+ 4572,
+ 4573,
+ 4574,
+ 4575,
+ 4576,
+ 4577,
+ 4578,
+ 4579,
+ 4580,
+ 4581,
+ 4582,
+ 4583,
+ 4584,
+ 4585,
+ 4586,
+ 4587,
+ 4588,
+ 4589,
+ 4590,
+ 4591,
+ 4592,
+ 4593,
+ 4594,
+ 4595,
+ 4596,
+ 4597,
+ 4598,
+ 4599,
+ 4600,
+ 4601,
+ 4602,
+ 4603,
+ 4604,
+ 4605,
+ 4606,
+ 4607,
+ 4608,
+ 4609,
+ 4610,
+ 4611,
+ 4612,
+ 4613,
+ 4614,
+ 4615,
+ 4616,
+ 4617,
+ 4618,
+ 4619,
+ 4620,
+ 4621,
+ 4622,
+ 4634,
+ 4635,
+ 4636,
+ 4637,
+ 4638,
+ 4639,
+ 4640,
+ 4641,
+ 4642,
+ 4643,
+ 4644,
+ 4661,
+ 4662,
+ 4663,
+ 4664,
+ 4665,
+ 4666,
+ 4667,
+ 4668,
+ 4669,
+ 4842,
+ 4843,
+ 4844,
+ 4845,
+ 4846,
+ 4847,
+ 4848,
+ 4937,
+ 4938,
+ 4939,
+ 4940,
+ 4941,
+ 4942,
+ 4943,
+ 4944,
+ 4945,
+ 4946,
+ 4947,
+ 4993,
+ 4994,
+ 4995,
+ 4996,
+ 4997,
+ 4998,
+ 4999,
+ 5000,
+ 5001,
+ 5002,
+ 5003,
+ 6574,
+ 6575,
+ 6576,
+ 6577,
+ 6578,
+ 6579,
+ 6580,
+ 6581,
+ 6582,
+ 6583,
+ 6584,
+ 6585,
+ 6586,
+ 6587,
+ 6588,
+ 6589,
+ 6590,
+ 6591,
+ 6592,
+ 6593,
+ 6594,
+ 6595,
+ 6596,
+ 6597,
+ 6598,
+ 6599,
+ 6600,
+ 6601,
+ 6602,
+ 6603,
+ 6604,
+ 6605,
+ 6606,
+ 6607,
+ 6608,
+ 6609,
+ 6610,
+ 6719,
+ 6720,
+ 6721,
+ 6722,
+ 6723,
+ 6724,
+ 6725,
+ 6726,
+ 6727,
+ 6728,
+ 6729,
+ 6730,
+ 6731,
+ 6732,
+ 6733,
+ 6734,
+ 6735,
+ 6832,
+ 6833,
+ 6834,
+ 6835,
+ 6836,
+ 6869,
+ 6870,
+ 6871,
+ 6872
+ ],
+ "rightHandIndex1": [
+ 5488,
+ 5489,
+ 5490,
+ 5491,
+ 5498,
+ 5499,
+ 5500,
+ 5501,
+ 5518,
+ 5528,
+ 5529,
+ 5584,
+ 5585,
+ 5586,
+ 5587,
+ 5588,
+ 5589,
+ 5590,
+ 5591,
+ 5592,
+ 5606,
+ 5607,
+ 5613,
+ 5615,
+ 5616,
+ 5617,
+ 5618,
+ 5619,
+ 5620,
+ 5621,
+ 5622,
+ 5623,
+ 5624,
+ 5625,
+ 5626,
+ 5627,
+ 5628,
+ 5629,
+ 5630,
+ 5638,
+ 5639,
+ 5640,
+ 5642,
+ 5647,
+ 5648,
+ 5650,
+ 5651,
+ 5665,
+ 5666,
+ 5676,
+ 5677,
+ 5678,
+ 5679,
+ 5680,
+ 5681,
+ 5693,
+ 5694,
+ 5706,
+ 5707,
+ 5708,
+ 5719,
+ 5721,
+ 5722,
+ 5723,
+ 5724,
+ 5730,
+ 5731,
+ 5733,
+ 5734,
+ 5735,
+ 5737,
+ 5738,
+ 5741,
+ 5742,
+ 5743,
+ 5744,
+ 5752,
+ 5753,
+ 5754,
+ 5755,
+ 5756,
+ 5757,
+ 5758,
+ 5759,
+ 5760,
+ 5761,
+ 5762,
+ 5763,
+ 5764,
+ 5765,
+ 5766,
+ 5767,
+ 5768,
+ 5769,
+ 5770,
+ 5771,
+ 5772,
+ 5773,
+ 5774,
+ 5775,
+ 5776,
+ 5777,
+ 5778,
+ 5779,
+ 5780,
+ 5781,
+ 5782,
+ 5783,
+ 5784,
+ 5785,
+ 5786,
+ 5787,
+ 5788,
+ 5789,
+ 5790,
+ 5791,
+ 5792,
+ 5793,
+ 5794,
+ 5795,
+ 5796,
+ 5797,
+ 5798,
+ 5799,
+ 5800,
+ 5801,
+ 5802,
+ 5803,
+ 5804,
+ 5805,
+ 5806,
+ 5807,
+ 5808,
+ 5809,
+ 5810,
+ 5811,
+ 5812,
+ 5813,
+ 5814,
+ 5815,
+ 5816,
+ 5817,
+ 5818,
+ 5819,
+ 5820,
+ 5821,
+ 5822,
+ 5823,
+ 5824,
+ 5825,
+ 5826,
+ 5827,
+ 5828,
+ 5829,
+ 5830,
+ 5831,
+ 5832,
+ 5833,
+ 5834,
+ 5835,
+ 5836,
+ 5837,
+ 5838,
+ 5839,
+ 5840,
+ 5841,
+ 5842,
+ 5843,
+ 5844,
+ 5845,
+ 5846,
+ 5847,
+ 5848,
+ 5849,
+ 5850,
+ 5851,
+ 5852,
+ 5853,
+ 5854,
+ 5855,
+ 5856,
+ 5857,
+ 5858,
+ 5859,
+ 5860,
+ 5861,
+ 5862,
+ 5863,
+ 5864,
+ 5865,
+ 5866,
+ 5867,
+ 5868,
+ 5869,
+ 5870,
+ 5871,
+ 5872,
+ 5873,
+ 5874,
+ 5875,
+ 5876,
+ 5877,
+ 5878,
+ 5879,
+ 5880,
+ 5881,
+ 5882,
+ 5883,
+ 5884,
+ 5885,
+ 5886,
+ 5887,
+ 5888,
+ 5889,
+ 5890,
+ 5891,
+ 5892,
+ 5893,
+ 5894,
+ 5895,
+ 5896,
+ 5897,
+ 5898,
+ 5899,
+ 5900,
+ 5901,
+ 5902,
+ 5903,
+ 5904,
+ 5905,
+ 5906,
+ 5907,
+ 5908,
+ 5909,
+ 5910,
+ 5911,
+ 5912,
+ 5913,
+ 5914,
+ 5915,
+ 5916,
+ 5917,
+ 5918,
+ 5919,
+ 5920,
+ 5921,
+ 5922,
+ 5923,
+ 5924,
+ 5925,
+ 5926,
+ 5927,
+ 5928,
+ 5929,
+ 5930,
+ 5931,
+ 5932,
+ 5933,
+ 5934,
+ 5935,
+ 5936,
+ 5937,
+ 5938,
+ 5939,
+ 5940,
+ 5941,
+ 5942,
+ 5943,
+ 5944,
+ 5945,
+ 5946,
+ 5947,
+ 5948,
+ 5949,
+ 5950,
+ 5951,
+ 5952,
+ 5953,
+ 5954,
+ 5955,
+ 5956,
+ 5957,
+ 5958,
+ 5959,
+ 5960,
+ 5961,
+ 5962,
+ 5963,
+ 5964,
+ 5965,
+ 5966,
+ 5967,
+ 5968,
+ 5969,
+ 5970,
+ 5971,
+ 5972,
+ 5973,
+ 5974,
+ 5975,
+ 5976,
+ 5977,
+ 5978,
+ 5979,
+ 5980,
+ 5981,
+ 5982,
+ 5983,
+ 5984,
+ 5985,
+ 5986,
+ 5987,
+ 5988,
+ 5989,
+ 5990,
+ 5991,
+ 5992,
+ 5993,
+ 5994,
+ 5995,
+ 5996,
+ 5997,
+ 5998,
+ 5999,
+ 6000,
+ 6001,
+ 6002,
+ 6003,
+ 6004,
+ 6005,
+ 6006,
+ 6007,
+ 6008,
+ 6009,
+ 6010,
+ 6011,
+ 6012,
+ 6013,
+ 6014,
+ 6015,
+ 6016,
+ 6017,
+ 6018,
+ 6019,
+ 6020,
+ 6021,
+ 6022,
+ 6023,
+ 6024,
+ 6025,
+ 6026,
+ 6027,
+ 6028,
+ 6029,
+ 6030,
+ 6031,
+ 6032,
+ 6033,
+ 6034,
+ 6035,
+ 6036,
+ 6037,
+ 6038,
+ 6039,
+ 6040,
+ 6041,
+ 6042,
+ 6043,
+ 6044,
+ 6045,
+ 6046,
+ 6047,
+ 6048,
+ 6049,
+ 6050,
+ 6051,
+ 6052,
+ 6053,
+ 6054,
+ 6055,
+ 6058,
+ 6059,
+ 6060,
+ 6061,
+ 6062,
+ 6063,
+ 6064,
+ 6065,
+ 6068,
+ 6069,
+ 6070,
+ 6071,
+ 6072,
+ 6073,
+ 6074,
+ 6075,
+ 6076,
+ 6077,
+ 6078,
+ 6079,
+ 6080,
+ 6081,
+ 6082,
+ 6083,
+ 6084,
+ 6085,
+ 6086,
+ 6087,
+ 6088,
+ 6089,
+ 6090,
+ 6091,
+ 6092,
+ 6093,
+ 6094,
+ 6095,
+ 6096,
+ 6097,
+ 6098,
+ 6099,
+ 6100,
+ 6101,
+ 6102,
+ 6103,
+ 6104,
+ 6105,
+ 6106,
+ 6107,
+ 6108,
+ 6109,
+ 6110,
+ 6111,
+ 6112,
+ 6113,
+ 6114,
+ 6115,
+ 6116,
+ 6117,
+ 6118,
+ 6119,
+ 6120,
+ 6121,
+ 6122,
+ 6123,
+ 6124,
+ 6125,
+ 6126,
+ 6127,
+ 6128,
+ 6129,
+ 6130,
+ 6131,
+ 6132,
+ 6133,
+ 6134,
+ 6135,
+ 6136,
+ 6137,
+ 6138,
+ 6139,
+ 6140,
+ 6141,
+ 6142,
+ 6143,
+ 6144,
+ 6145,
+ 6146,
+ 6147,
+ 6148,
+ 6149,
+ 6150,
+ 6151,
+ 6152,
+ 6153,
+ 6154,
+ 6155,
+ 6156,
+ 6157
+ ],
+ "leftForeArm": [
+ 1546,
+ 1547,
+ 1548,
+ 1549,
+ 1550,
+ 1551,
+ 1552,
+ 1553,
+ 1554,
+ 1555,
+ 1556,
+ 1557,
+ 1558,
+ 1559,
+ 1560,
+ 1561,
+ 1562,
+ 1563,
+ 1564,
+ 1565,
+ 1566,
+ 1567,
+ 1568,
+ 1569,
+ 1570,
+ 1571,
+ 1572,
+ 1573,
+ 1574,
+ 1575,
+ 1576,
+ 1577,
+ 1578,
+ 1579,
+ 1580,
+ 1581,
+ 1582,
+ 1583,
+ 1584,
+ 1585,
+ 1586,
+ 1587,
+ 1588,
+ 1589,
+ 1590,
+ 1591,
+ 1592,
+ 1593,
+ 1594,
+ 1595,
+ 1596,
+ 1597,
+ 1598,
+ 1599,
+ 1600,
+ 1601,
+ 1602,
+ 1603,
+ 1604,
+ 1605,
+ 1606,
+ 1607,
+ 1608,
+ 1609,
+ 1610,
+ 1611,
+ 1612,
+ 1613,
+ 1614,
+ 1615,
+ 1616,
+ 1617,
+ 1618,
+ 1620,
+ 1621,
+ 1623,
+ 1624,
+ 1625,
+ 1626,
+ 1627,
+ 1628,
+ 1629,
+ 1630,
+ 1643,
+ 1644,
+ 1646,
+ 1647,
+ 1650,
+ 1651,
+ 1654,
+ 1655,
+ 1657,
+ 1658,
+ 1659,
+ 1660,
+ 1661,
+ 1662,
+ 1663,
+ 1664,
+ 1665,
+ 1666,
+ 1685,
+ 1686,
+ 1687,
+ 1688,
+ 1689,
+ 1690,
+ 1691,
+ 1692,
+ 1693,
+ 1694,
+ 1695,
+ 1699,
+ 1700,
+ 1701,
+ 1702,
+ 1721,
+ 1722,
+ 1723,
+ 1724,
+ 1725,
+ 1726,
+ 1727,
+ 1728,
+ 1729,
+ 1730,
+ 1732,
+ 1736,
+ 1738,
+ 1741,
+ 1742,
+ 1743,
+ 1744,
+ 1750,
+ 1752,
+ 1900,
+ 1909,
+ 1910,
+ 1911,
+ 1912,
+ 1913,
+ 1914,
+ 1915,
+ 1916,
+ 1917,
+ 1918,
+ 1919,
+ 1920,
+ 1921,
+ 1922,
+ 1923,
+ 1924,
+ 1925,
+ 1926,
+ 1927,
+ 1928,
+ 1929,
+ 1930,
+ 1931,
+ 1932,
+ 1933,
+ 1934,
+ 1935,
+ 1936,
+ 1937,
+ 1938,
+ 1939,
+ 1940,
+ 1941,
+ 1942,
+ 1943,
+ 1944,
+ 1945,
+ 1946,
+ 1947,
+ 1948,
+ 1949,
+ 1950,
+ 1951,
+ 1952,
+ 1953,
+ 1954,
+ 1955,
+ 1956,
+ 1957,
+ 1958,
+ 1959,
+ 1960,
+ 1961,
+ 1962,
+ 1963,
+ 1964,
+ 1965,
+ 1966,
+ 1967,
+ 1968,
+ 1969,
+ 1970,
+ 1971,
+ 1972,
+ 1973,
+ 1974,
+ 1975,
+ 1976,
+ 1977,
+ 1978,
+ 1979,
+ 1980,
+ 2019,
+ 2059,
+ 2060,
+ 2073,
+ 2089,
+ 2098,
+ 2099,
+ 2100,
+ 2101,
+ 2102,
+ 2103,
+ 2104,
+ 2105,
+ 2106,
+ 2107,
+ 2108,
+ 2109,
+ 2110,
+ 2111,
+ 2112,
+ 2147,
+ 2148,
+ 2206,
+ 2207,
+ 2208,
+ 2209,
+ 2228,
+ 2230,
+ 2234,
+ 2235,
+ 2241,
+ 2242,
+ 2243,
+ 2244,
+ 2279,
+ 2286,
+ 2873,
+ 2874
+ ],
+ "rightForeArm": [
+ 5015,
+ 5016,
+ 5017,
+ 5018,
+ 5019,
+ 5020,
+ 5021,
+ 5022,
+ 5023,
+ 5024,
+ 5025,
+ 5026,
+ 5027,
+ 5028,
+ 5029,
+ 5030,
+ 5031,
+ 5032,
+ 5033,
+ 5034,
+ 5035,
+ 5036,
+ 5037,
+ 5038,
+ 5039,
+ 5040,
+ 5041,
+ 5042,
+ 5043,
+ 5044,
+ 5045,
+ 5046,
+ 5047,
+ 5048,
+ 5049,
+ 5050,
+ 5051,
+ 5052,
+ 5053,
+ 5054,
+ 5055,
+ 5056,
+ 5057,
+ 5058,
+ 5059,
+ 5060,
+ 5061,
+ 5062,
+ 5063,
+ 5064,
+ 5065,
+ 5066,
+ 5067,
+ 5068,
+ 5069,
+ 5070,
+ 5071,
+ 5072,
+ 5073,
+ 5074,
+ 5075,
+ 5076,
+ 5077,
+ 5078,
+ 5079,
+ 5080,
+ 5081,
+ 5082,
+ 5083,
+ 5084,
+ 5085,
+ 5086,
+ 5087,
+ 5090,
+ 5091,
+ 5092,
+ 5093,
+ 5094,
+ 5095,
+ 5096,
+ 5097,
+ 5098,
+ 5099,
+ 5112,
+ 5113,
+ 5116,
+ 5117,
+ 5120,
+ 5121,
+ 5124,
+ 5125,
+ 5126,
+ 5127,
+ 5128,
+ 5129,
+ 5130,
+ 5131,
+ 5132,
+ 5133,
+ 5134,
+ 5135,
+ 5154,
+ 5155,
+ 5156,
+ 5157,
+ 5158,
+ 5159,
+ 5160,
+ 5161,
+ 5162,
+ 5163,
+ 5164,
+ 5168,
+ 5169,
+ 5170,
+ 5171,
+ 5190,
+ 5191,
+ 5192,
+ 5193,
+ 5194,
+ 5195,
+ 5196,
+ 5197,
+ 5198,
+ 5199,
+ 5202,
+ 5205,
+ 5207,
+ 5210,
+ 5211,
+ 5212,
+ 5213,
+ 5219,
+ 5221,
+ 5361,
+ 5370,
+ 5371,
+ 5372,
+ 5373,
+ 5374,
+ 5375,
+ 5376,
+ 5377,
+ 5378,
+ 5379,
+ 5380,
+ 5381,
+ 5382,
+ 5383,
+ 5384,
+ 5385,
+ 5386,
+ 5387,
+ 5388,
+ 5389,
+ 5390,
+ 5391,
+ 5392,
+ 5393,
+ 5394,
+ 5395,
+ 5396,
+ 5397,
+ 5398,
+ 5399,
+ 5400,
+ 5401,
+ 5402,
+ 5403,
+ 5404,
+ 5405,
+ 5406,
+ 5407,
+ 5408,
+ 5409,
+ 5410,
+ 5411,
+ 5412,
+ 5413,
+ 5414,
+ 5415,
+ 5416,
+ 5417,
+ 5418,
+ 5419,
+ 5420,
+ 5421,
+ 5422,
+ 5423,
+ 5424,
+ 5425,
+ 5426,
+ 5427,
+ 5428,
+ 5429,
+ 5430,
+ 5431,
+ 5432,
+ 5433,
+ 5434,
+ 5435,
+ 5436,
+ 5437,
+ 5438,
+ 5439,
+ 5440,
+ 5441,
+ 5480,
+ 5520,
+ 5521,
+ 5534,
+ 5550,
+ 5559,
+ 5560,
+ 5561,
+ 5562,
+ 5563,
+ 5564,
+ 5565,
+ 5566,
+ 5567,
+ 5568,
+ 5569,
+ 5570,
+ 5571,
+ 5572,
+ 5573,
+ 5608,
+ 5609,
+ 5667,
+ 5668,
+ 5669,
+ 5670,
+ 5689,
+ 5691,
+ 5695,
+ 5696,
+ 5702,
+ 5703,
+ 5704,
+ 5705,
+ 5740,
+ 5747,
+ 6334,
+ 6335
+ ],
+ "neck": [
+ 148,
+ 150,
+ 151,
+ 152,
+ 153,
+ 172,
+ 174,
+ 175,
+ 201,
+ 202,
+ 204,
+ 205,
+ 206,
+ 207,
+ 208,
+ 209,
+ 210,
+ 211,
+ 212,
+ 213,
+ 214,
+ 215,
+ 216,
+ 217,
+ 218,
+ 219,
+ 222,
+ 223,
+ 224,
+ 225,
+ 256,
+ 257,
+ 284,
+ 285,
+ 295,
+ 296,
+ 297,
+ 298,
+ 299,
+ 300,
+ 301,
+ 302,
+ 303,
+ 304,
+ 305,
+ 306,
+ 307,
+ 308,
+ 309,
+ 333,
+ 334,
+ 423,
+ 424,
+ 425,
+ 426,
+ 440,
+ 441,
+ 451,
+ 452,
+ 453,
+ 460,
+ 461,
+ 571,
+ 572,
+ 824,
+ 825,
+ 826,
+ 827,
+ 828,
+ 829,
+ 1279,
+ 1280,
+ 1312,
+ 1313,
+ 1319,
+ 1320,
+ 1331,
+ 3049,
+ 3050,
+ 3057,
+ 3058,
+ 3059,
+ 3068,
+ 3164,
+ 3661,
+ 3662,
+ 3663,
+ 3664,
+ 3665,
+ 3685,
+ 3686,
+ 3687,
+ 3714,
+ 3715,
+ 3716,
+ 3717,
+ 3718,
+ 3719,
+ 3720,
+ 3721,
+ 3722,
+ 3723,
+ 3724,
+ 3725,
+ 3726,
+ 3727,
+ 3728,
+ 3729,
+ 3730,
+ 3731,
+ 3734,
+ 3735,
+ 3736,
+ 3737,
+ 3768,
+ 3769,
+ 3796,
+ 3797,
+ 3807,
+ 3808,
+ 3809,
+ 3810,
+ 3811,
+ 3812,
+ 3813,
+ 3814,
+ 3815,
+ 3816,
+ 3817,
+ 3818,
+ 3819,
+ 3839,
+ 3840,
+ 3918,
+ 3919,
+ 3920,
+ 3921,
+ 3934,
+ 3935,
+ 3942,
+ 3943,
+ 3944,
+ 3950,
+ 4060,
+ 4061,
+ 4312,
+ 4313,
+ 4314,
+ 4315,
+ 4761,
+ 4762,
+ 4792,
+ 4793,
+ 4799,
+ 4800,
+ 4807
+ ],
+ "rightToeBase": [
+ 6611,
+ 6612,
+ 6613,
+ 6614,
+ 6615,
+ 6616,
+ 6617,
+ 6618,
+ 6619,
+ 6620,
+ 6621,
+ 6622,
+ 6623,
+ 6624,
+ 6625,
+ 6626,
+ 6627,
+ 6628,
+ 6629,
+ 6630,
+ 6631,
+ 6632,
+ 6633,
+ 6634,
+ 6635,
+ 6636,
+ 6637,
+ 6638,
+ 6639,
+ 6640,
+ 6641,
+ 6642,
+ 6643,
+ 6644,
+ 6645,
+ 6646,
+ 6647,
+ 6648,
+ 6649,
+ 6650,
+ 6651,
+ 6652,
+ 6653,
+ 6654,
+ 6655,
+ 6656,
+ 6657,
+ 6658,
+ 6659,
+ 6660,
+ 6661,
+ 6662,
+ 6663,
+ 6664,
+ 6665,
+ 6666,
+ 6667,
+ 6668,
+ 6669,
+ 6670,
+ 6671,
+ 6672,
+ 6673,
+ 6674,
+ 6675,
+ 6676,
+ 6677,
+ 6678,
+ 6679,
+ 6680,
+ 6681,
+ 6682,
+ 6683,
+ 6684,
+ 6685,
+ 6686,
+ 6687,
+ 6688,
+ 6689,
+ 6690,
+ 6691,
+ 6692,
+ 6693,
+ 6694,
+ 6695,
+ 6696,
+ 6697,
+ 6698,
+ 6699,
+ 6700,
+ 6701,
+ 6702,
+ 6703,
+ 6704,
+ 6705,
+ 6706,
+ 6707,
+ 6708,
+ 6709,
+ 6710,
+ 6711,
+ 6712,
+ 6713,
+ 6714,
+ 6715,
+ 6716,
+ 6717,
+ 6718,
+ 6736,
+ 6739,
+ 6741,
+ 6743,
+ 6745,
+ 6747,
+ 6749,
+ 6750,
+ 6752,
+ 6754,
+ 6757,
+ 6758,
+ 6760,
+ 6762
+ ],
+ "spine": [
+ 616,
+ 617,
+ 630,
+ 631,
+ 632,
+ 633,
+ 654,
+ 655,
+ 656,
+ 657,
+ 662,
+ 663,
+ 664,
+ 665,
+ 720,
+ 721,
+ 765,
+ 766,
+ 767,
+ 768,
+ 796,
+ 797,
+ 798,
+ 799,
+ 889,
+ 890,
+ 916,
+ 917,
+ 918,
+ 919,
+ 921,
+ 922,
+ 923,
+ 924,
+ 925,
+ 926,
+ 1188,
+ 1189,
+ 1211,
+ 1212,
+ 1248,
+ 1249,
+ 1250,
+ 1251,
+ 1264,
+ 1265,
+ 1266,
+ 1267,
+ 1323,
+ 1324,
+ 1325,
+ 1326,
+ 1327,
+ 1328,
+ 1332,
+ 1333,
+ 1334,
+ 1335,
+ 1336,
+ 1344,
+ 1345,
+ 1481,
+ 1482,
+ 1483,
+ 1484,
+ 1485,
+ 1486,
+ 1487,
+ 1488,
+ 1489,
+ 1490,
+ 1491,
+ 1492,
+ 1493,
+ 1494,
+ 1495,
+ 1496,
+ 1767,
+ 2823,
+ 2824,
+ 2825,
+ 2826,
+ 2827,
+ 2828,
+ 2829,
+ 2830,
+ 2831,
+ 2832,
+ 2833,
+ 2834,
+ 2835,
+ 2836,
+ 2837,
+ 2838,
+ 2839,
+ 2840,
+ 2841,
+ 2842,
+ 2843,
+ 2844,
+ 2845,
+ 2847,
+ 2848,
+ 2851,
+ 3016,
+ 3017,
+ 3018,
+ 3019,
+ 3020,
+ 3023,
+ 3024,
+ 3124,
+ 3173,
+ 3476,
+ 3477,
+ 3478,
+ 3480,
+ 3500,
+ 3501,
+ 3502,
+ 3504,
+ 3509,
+ 3511,
+ 4103,
+ 4104,
+ 4118,
+ 4119,
+ 4120,
+ 4121,
+ 4142,
+ 4143,
+ 4144,
+ 4145,
+ 4150,
+ 4151,
+ 4152,
+ 4153,
+ 4208,
+ 4209,
+ 4253,
+ 4254,
+ 4255,
+ 4256,
+ 4284,
+ 4285,
+ 4286,
+ 4287,
+ 4375,
+ 4376,
+ 4402,
+ 4403,
+ 4405,
+ 4406,
+ 4407,
+ 4408,
+ 4409,
+ 4410,
+ 4411,
+ 4412,
+ 4674,
+ 4675,
+ 4694,
+ 4695,
+ 4731,
+ 4732,
+ 4733,
+ 4734,
+ 4747,
+ 4748,
+ 4749,
+ 4750,
+ 4803,
+ 4804,
+ 4805,
+ 4806,
+ 4808,
+ 4809,
+ 4810,
+ 4811,
+ 4812,
+ 4820,
+ 4821,
+ 4953,
+ 4954,
+ 4955,
+ 4956,
+ 4957,
+ 4958,
+ 4959,
+ 4960,
+ 4961,
+ 4962,
+ 4963,
+ 4964,
+ 4965,
+ 4966,
+ 4967,
+ 4968,
+ 5234,
+ 6284,
+ 6285,
+ 6286,
+ 6287,
+ 6288,
+ 6289,
+ 6290,
+ 6291,
+ 6292,
+ 6293,
+ 6294,
+ 6295,
+ 6296,
+ 6297,
+ 6298,
+ 6299,
+ 6300,
+ 6301,
+ 6302,
+ 6303,
+ 6304,
+ 6305,
+ 6306,
+ 6308,
+ 6309,
+ 6312,
+ 6472,
+ 6473,
+ 6474,
+ 6545,
+ 6874,
+ 6875,
+ 6876,
+ 6878
+ ],
+ "leftUpLeg": [
+ 833,
+ 834,
+ 838,
+ 839,
+ 847,
+ 848,
+ 849,
+ 850,
+ 851,
+ 852,
+ 853,
+ 854,
+ 870,
+ 871,
+ 872,
+ 873,
+ 874,
+ 875,
+ 876,
+ 877,
+ 878,
+ 879,
+ 880,
+ 881,
+ 897,
+ 898,
+ 899,
+ 900,
+ 901,
+ 902,
+ 903,
+ 904,
+ 905,
+ 906,
+ 907,
+ 908,
+ 909,
+ 910,
+ 911,
+ 912,
+ 913,
+ 914,
+ 915,
+ 933,
+ 934,
+ 935,
+ 936,
+ 944,
+ 945,
+ 946,
+ 947,
+ 948,
+ 949,
+ 950,
+ 951,
+ 952,
+ 953,
+ 954,
+ 955,
+ 956,
+ 957,
+ 958,
+ 959,
+ 960,
+ 961,
+ 962,
+ 963,
+ 964,
+ 965,
+ 966,
+ 967,
+ 968,
+ 969,
+ 970,
+ 971,
+ 972,
+ 973,
+ 974,
+ 975,
+ 976,
+ 977,
+ 978,
+ 979,
+ 980,
+ 981,
+ 982,
+ 983,
+ 984,
+ 985,
+ 986,
+ 987,
+ 988,
+ 989,
+ 990,
+ 991,
+ 992,
+ 993,
+ 994,
+ 995,
+ 996,
+ 997,
+ 998,
+ 999,
+ 1000,
+ 1001,
+ 1002,
+ 1003,
+ 1004,
+ 1005,
+ 1006,
+ 1007,
+ 1008,
+ 1009,
+ 1010,
+ 1011,
+ 1012,
+ 1013,
+ 1014,
+ 1015,
+ 1016,
+ 1017,
+ 1018,
+ 1019,
+ 1020,
+ 1021,
+ 1022,
+ 1023,
+ 1024,
+ 1025,
+ 1026,
+ 1027,
+ 1028,
+ 1029,
+ 1030,
+ 1031,
+ 1032,
+ 1033,
+ 1034,
+ 1035,
+ 1036,
+ 1037,
+ 1038,
+ 1039,
+ 1040,
+ 1041,
+ 1042,
+ 1043,
+ 1044,
+ 1045,
+ 1046,
+ 1137,
+ 1138,
+ 1139,
+ 1140,
+ 1141,
+ 1142,
+ 1143,
+ 1144,
+ 1145,
+ 1146,
+ 1147,
+ 1148,
+ 1159,
+ 1160,
+ 1161,
+ 1162,
+ 1163,
+ 1164,
+ 1165,
+ 1166,
+ 1167,
+ 1168,
+ 1169,
+ 1170,
+ 1171,
+ 1172,
+ 1173,
+ 1174,
+ 1184,
+ 1185,
+ 1186,
+ 1187,
+ 1221,
+ 1222,
+ 1223,
+ 1224,
+ 1225,
+ 1226,
+ 1227,
+ 1228,
+ 1229,
+ 1230,
+ 1262,
+ 1263,
+ 1274,
+ 1275,
+ 1276,
+ 1277,
+ 1321,
+ 1322,
+ 1354,
+ 1359,
+ 1360,
+ 1361,
+ 1362,
+ 1365,
+ 1366,
+ 1367,
+ 1368,
+ 1451,
+ 1452,
+ 1453,
+ 1455,
+ 1456,
+ 1457,
+ 1458,
+ 1459,
+ 1460,
+ 1461,
+ 1462,
+ 1463,
+ 1475,
+ 1477,
+ 1478,
+ 1479,
+ 1480,
+ 1498,
+ 1499,
+ 1500,
+ 1501,
+ 1511,
+ 1512,
+ 1513,
+ 1514,
+ 1516,
+ 1517,
+ 1518,
+ 1519,
+ 1520,
+ 1521,
+ 1522,
+ 1533,
+ 1534,
+ 3125,
+ 3126,
+ 3127,
+ 3128,
+ 3131,
+ 3132,
+ 3133,
+ 3134,
+ 3135,
+ 3475,
+ 3479
+ ],
+ "leftHand": [
+ 1981,
+ 1982,
+ 1983,
+ 1984,
+ 1985,
+ 1986,
+ 1987,
+ 1988,
+ 1989,
+ 1990,
+ 1991,
+ 1992,
+ 1993,
+ 1994,
+ 1995,
+ 1996,
+ 1997,
+ 1998,
+ 1999,
+ 2000,
+ 2001,
+ 2002,
+ 2003,
+ 2004,
+ 2005,
+ 2006,
+ 2007,
+ 2008,
+ 2009,
+ 2010,
+ 2011,
+ 2012,
+ 2013,
+ 2014,
+ 2015,
+ 2016,
+ 2017,
+ 2018,
+ 2019,
+ 2020,
+ 2021,
+ 2022,
+ 2023,
+ 2024,
+ 2025,
+ 2026,
+ 2031,
+ 2032,
+ 2033,
+ 2034,
+ 2035,
+ 2036,
+ 2041,
+ 2042,
+ 2043,
+ 2044,
+ 2045,
+ 2046,
+ 2047,
+ 2048,
+ 2049,
+ 2050,
+ 2051,
+ 2052,
+ 2053,
+ 2054,
+ 2055,
+ 2056,
+ 2057,
+ 2058,
+ 2059,
+ 2060,
+ 2061,
+ 2062,
+ 2063,
+ 2064,
+ 2065,
+ 2066,
+ 2069,
+ 2070,
+ 2071,
+ 2072,
+ 2073,
+ 2074,
+ 2075,
+ 2076,
+ 2077,
+ 2078,
+ 2079,
+ 2080,
+ 2081,
+ 2082,
+ 2083,
+ 2084,
+ 2085,
+ 2086,
+ 2087,
+ 2088,
+ 2089,
+ 2090,
+ 2091,
+ 2092,
+ 2093,
+ 2094,
+ 2095,
+ 2096,
+ 2097,
+ 2098,
+ 2099,
+ 2100,
+ 2101,
+ 2107,
+ 2111,
+ 2113,
+ 2114,
+ 2115,
+ 2116,
+ 2117,
+ 2118,
+ 2119,
+ 2120,
+ 2121,
+ 2122,
+ 2127,
+ 2130,
+ 2131,
+ 2132,
+ 2133,
+ 2134,
+ 2135,
+ 2136,
+ 2137,
+ 2138,
+ 2139,
+ 2140,
+ 2141,
+ 2142,
+ 2143,
+ 2144,
+ 2149,
+ 2150,
+ 2151,
+ 2152,
+ 2155,
+ 2160,
+ 2163,
+ 2164,
+ 2170,
+ 2171,
+ 2172,
+ 2173,
+ 2174,
+ 2175,
+ 2176,
+ 2177,
+ 2178,
+ 2179,
+ 2180,
+ 2182,
+ 2183,
+ 2184,
+ 2185,
+ 2188,
+ 2189,
+ 2191,
+ 2192,
+ 2193,
+ 2194,
+ 2195,
+ 2196,
+ 2197,
+ 2198,
+ 2199,
+ 2200,
+ 2201,
+ 2202,
+ 2203,
+ 2207,
+ 2209,
+ 2210,
+ 2211,
+ 2212,
+ 2213,
+ 2214,
+ 2221,
+ 2222,
+ 2223,
+ 2224,
+ 2225,
+ 2226,
+ 2227,
+ 2228,
+ 2229,
+ 2231,
+ 2234,
+ 2236,
+ 2237,
+ 2238,
+ 2239,
+ 2240,
+ 2246,
+ 2247,
+ 2248,
+ 2249,
+ 2250,
+ 2251,
+ 2252,
+ 2253,
+ 2254,
+ 2255,
+ 2256,
+ 2257,
+ 2258,
+ 2259,
+ 2260,
+ 2262,
+ 2263,
+ 2264,
+ 2265,
+ 2266,
+ 2267,
+ 2268,
+ 2269,
+ 2270,
+ 2271,
+ 2274,
+ 2275,
+ 2276,
+ 2277,
+ 2278,
+ 2279,
+ 2284,
+ 2285,
+ 2287,
+ 2288,
+ 2289,
+ 2290,
+ 2293,
+ 2595,
+ 2598,
+ 2605,
+ 2608,
+ 2697,
+ 2698,
+ 2699,
+ 2700,
+ 2701,
+ 2702,
+ 2703,
+ 2704,
+ 2705,
+ 2706,
+ 2707,
+ 2708,
+ 2709,
+ 2710,
+ 2711,
+ 2712,
+ 2713,
+ 2714,
+ 2715,
+ 2716,
+ 2717,
+ 2718,
+ 2719,
+ 2720,
+ 2721,
+ 2722,
+ 2723,
+ 2724,
+ 2725,
+ 2726,
+ 2727,
+ 2728,
+ 2729,
+ 2730,
+ 2731,
+ 2732,
+ 2733,
+ 2734,
+ 2735,
+ 2736,
+ 2737,
+ 2738,
+ 2739,
+ 2740,
+ 2741,
+ 2742,
+ 2743,
+ 2744,
+ 2745,
+ 2746,
+ 2747,
+ 2748,
+ 2749,
+ 2750,
+ 2751,
+ 2752,
+ 2753,
+ 2754,
+ 2755,
+ 2756,
+ 2757,
+ 2758,
+ 2759,
+ 2760,
+ 2761,
+ 2762,
+ 2763,
+ 2764,
+ 2765,
+ 2766,
+ 2767,
+ 2768,
+ 2769,
+ 2770,
+ 2771,
+ 2772,
+ 2773,
+ 2774,
+ 2775,
+ 2776,
+ 2777,
+ 2778
+ ],
+ "hips": [
+ 631,
+ 632,
+ 654,
+ 657,
+ 662,
+ 665,
+ 676,
+ 677,
+ 678,
+ 679,
+ 705,
+ 720,
+ 796,
+ 799,
+ 800,
+ 801,
+ 802,
+ 807,
+ 808,
+ 809,
+ 810,
+ 815,
+ 816,
+ 822,
+ 823,
+ 830,
+ 831,
+ 832,
+ 833,
+ 834,
+ 835,
+ 836,
+ 837,
+ 838,
+ 839,
+ 840,
+ 841,
+ 842,
+ 843,
+ 844,
+ 845,
+ 846,
+ 855,
+ 856,
+ 857,
+ 858,
+ 859,
+ 860,
+ 861,
+ 862,
+ 863,
+ 864,
+ 865,
+ 866,
+ 867,
+ 868,
+ 869,
+ 871,
+ 878,
+ 881,
+ 882,
+ 883,
+ 884,
+ 885,
+ 886,
+ 887,
+ 888,
+ 889,
+ 890,
+ 912,
+ 915,
+ 916,
+ 917,
+ 918,
+ 919,
+ 920,
+ 932,
+ 937,
+ 938,
+ 939,
+ 1163,
+ 1166,
+ 1203,
+ 1204,
+ 1205,
+ 1206,
+ 1207,
+ 1208,
+ 1209,
+ 1210,
+ 1246,
+ 1247,
+ 1262,
+ 1263,
+ 1276,
+ 1277,
+ 1278,
+ 1321,
+ 1336,
+ 1337,
+ 1338,
+ 1339,
+ 1353,
+ 1354,
+ 1361,
+ 1362,
+ 1363,
+ 1364,
+ 1446,
+ 1447,
+ 1448,
+ 1449,
+ 1450,
+ 1454,
+ 1476,
+ 1497,
+ 1511,
+ 1513,
+ 1514,
+ 1515,
+ 1533,
+ 1534,
+ 1539,
+ 1540,
+ 1768,
+ 1769,
+ 1779,
+ 1780,
+ 1781,
+ 1782,
+ 1783,
+ 1784,
+ 1785,
+ 1786,
+ 1787,
+ 1788,
+ 1789,
+ 1790,
+ 1791,
+ 1792,
+ 1793,
+ 1794,
+ 1795,
+ 1796,
+ 1797,
+ 1798,
+ 1799,
+ 1800,
+ 1801,
+ 1802,
+ 1803,
+ 1804,
+ 1805,
+ 1806,
+ 1807,
+ 2909,
+ 2910,
+ 2911,
+ 2912,
+ 2913,
+ 2914,
+ 2915,
+ 2916,
+ 2917,
+ 2918,
+ 2919,
+ 2920,
+ 2921,
+ 2922,
+ 2923,
+ 2924,
+ 2925,
+ 2926,
+ 2927,
+ 2928,
+ 2929,
+ 2930,
+ 3018,
+ 3019,
+ 3021,
+ 3022,
+ 3080,
+ 3081,
+ 3082,
+ 3083,
+ 3084,
+ 3085,
+ 3086,
+ 3087,
+ 3088,
+ 3089,
+ 3090,
+ 3091,
+ 3092,
+ 3093,
+ 3094,
+ 3095,
+ 3096,
+ 3097,
+ 3098,
+ 3099,
+ 3100,
+ 3101,
+ 3102,
+ 3103,
+ 3104,
+ 3105,
+ 3106,
+ 3107,
+ 3108,
+ 3109,
+ 3110,
+ 3111,
+ 3112,
+ 3113,
+ 3114,
+ 3115,
+ 3116,
+ 3117,
+ 3118,
+ 3119,
+ 3120,
+ 3121,
+ 3122,
+ 3123,
+ 3124,
+ 3128,
+ 3129,
+ 3130,
+ 3136,
+ 3137,
+ 3138,
+ 3139,
+ 3140,
+ 3141,
+ 3142,
+ 3143,
+ 3144,
+ 3145,
+ 3146,
+ 3147,
+ 3148,
+ 3149,
+ 3150,
+ 3151,
+ 3152,
+ 3153,
+ 3154,
+ 3155,
+ 3156,
+ 3157,
+ 3158,
+ 3159,
+ 3160,
+ 3170,
+ 3172,
+ 3481,
+ 3484,
+ 3500,
+ 3502,
+ 3503,
+ 3507,
+ 3510,
+ 4120,
+ 4121,
+ 4142,
+ 4143,
+ 4150,
+ 4151,
+ 4164,
+ 4165,
+ 4166,
+ 4167,
+ 4193,
+ 4208,
+ 4284,
+ 4285,
+ 4288,
+ 4289,
+ 4290,
+ 4295,
+ 4296,
+ 4297,
+ 4298,
+ 4303,
+ 4304,
+ 4310,
+ 4311,
+ 4316,
+ 4317,
+ 4318,
+ 4319,
+ 4320,
+ 4321,
+ 4322,
+ 4323,
+ 4324,
+ 4325,
+ 4326,
+ 4327,
+ 4328,
+ 4329,
+ 4330,
+ 4331,
+ 4332,
+ 4341,
+ 4342,
+ 4343,
+ 4344,
+ 4345,
+ 4346,
+ 4347,
+ 4348,
+ 4349,
+ 4350,
+ 4351,
+ 4352,
+ 4353,
+ 4354,
+ 4355,
+ 4356,
+ 4364,
+ 4365,
+ 4368,
+ 4369,
+ 4370,
+ 4371,
+ 4372,
+ 4373,
+ 4374,
+ 4375,
+ 4376,
+ 4398,
+ 4399,
+ 4402,
+ 4403,
+ 4404,
+ 4405,
+ 4406,
+ 4418,
+ 4423,
+ 4424,
+ 4425,
+ 4649,
+ 4650,
+ 4689,
+ 4690,
+ 4691,
+ 4692,
+ 4693,
+ 4729,
+ 4730,
+ 4745,
+ 4746,
+ 4759,
+ 4760,
+ 4801,
+ 4812,
+ 4813,
+ 4814,
+ 4815,
+ 4829,
+ 4836,
+ 4837,
+ 4919,
+ 4920,
+ 4921,
+ 4922,
+ 4923,
+ 4927,
+ 4969,
+ 4983,
+ 4984,
+ 4986,
+ 5004,
+ 5005,
+ 5244,
+ 5245,
+ 5246,
+ 5247,
+ 5248,
+ 5249,
+ 5250,
+ 5251,
+ 5252,
+ 5253,
+ 5254,
+ 5255,
+ 5256,
+ 5257,
+ 5258,
+ 5259,
+ 5260,
+ 5261,
+ 5262,
+ 5263,
+ 5264,
+ 5265,
+ 5266,
+ 5267,
+ 5268,
+ 6368,
+ 6369,
+ 6370,
+ 6371,
+ 6372,
+ 6373,
+ 6374,
+ 6375,
+ 6376,
+ 6377,
+ 6378,
+ 6379,
+ 6380,
+ 6381,
+ 6382,
+ 6383,
+ 6384,
+ 6385,
+ 6386,
+ 6387,
+ 6388,
+ 6389,
+ 6473,
+ 6474,
+ 6504,
+ 6505,
+ 6506,
+ 6507,
+ 6508,
+ 6509,
+ 6510,
+ 6511,
+ 6512,
+ 6513,
+ 6514,
+ 6515,
+ 6516,
+ 6517,
+ 6518,
+ 6519,
+ 6520,
+ 6521,
+ 6522,
+ 6523,
+ 6524,
+ 6525,
+ 6526,
+ 6527,
+ 6528,
+ 6529,
+ 6530,
+ 6531,
+ 6532,
+ 6533,
+ 6534,
+ 6535,
+ 6536,
+ 6537,
+ 6538,
+ 6539,
+ 6540,
+ 6541,
+ 6542,
+ 6543,
+ 6544,
+ 6545,
+ 6549,
+ 6550,
+ 6551,
+ 6557,
+ 6558,
+ 6559,
+ 6560,
+ 6561,
+ 6562,
+ 6563,
+ 6564,
+ 6565,
+ 6566,
+ 6567,
+ 6568,
+ 6569,
+ 6570,
+ 6571,
+ 6572,
+ 6573
+ ]
+}
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/smplx2smpl_sparse.pt b/third_party/GVHMR/hmr4d/utils/body_model/smplx2smpl_sparse.pt
new file mode 100644
index 0000000000000000000000000000000000000000..0595089d31dec584e97b169fa936a4a8dfbd36bb
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/smplx2smpl_sparse.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0fc821a9e79ec3e76d6a9796b96d5bef8cd67055e18497bbe370d8aed9e07e06
+size 208935
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/smplx_lite.py b/third_party/GVHMR/hmr4d/utils/body_model/smplx_lite.py
new file mode 100644
index 0000000000000000000000000000000000000000..013a26874b62e18b606384d5052147099a936851
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/smplx_lite.py
@@ -0,0 +1,302 @@
+import numpy as np
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from pathlib import Path
+from pytorch3d.transforms import axis_angle_to_matrix, rotation_6d_to_matrix
+from smplx.utils import Struct, to_np, to_tensor
+from einops import einsum, rearrange
+from time import time
+
+from hmr4d import PROJ_ROOT
+
+
+class SmplxLite(nn.Module):
+ def __init__(
+ self,
+ model_path=PROJ_ROOT / "inputs/checkpoints/body_models/smplx",
+ gender="neutral",
+ num_betas=10,
+ ):
+ super().__init__()
+
+ # Load the model
+ model_path = Path(model_path)
+ if model_path.is_dir():
+ smplx_path = Path(model_path) / f"SMPLX_{gender.upper()}.npz"
+ else:
+ smplx_path = model_path
+ assert smplx_path.exists()
+ model_data = np.load(smplx_path, allow_pickle=True)
+
+ data_struct = Struct(**model_data)
+ self.faces = data_struct.f # (F, 3)
+
+ self.register_smpl_buffers(data_struct, num_betas)
+ # self.register_smplh_buffers(data_struct, num_pca_comps, flat_hand_mean)
+ # self.register_smplx_buffers(data_struct)
+ self.register_fast_skeleton_computing_buffers()
+
+ # default_pose (99,) for torch.cat([global_orient, body_pose, default_pose])
+ other_default_pose = torch.cat(
+ [
+ torch.zeros(9),
+ to_tensor(data_struct.hands_meanl).float(),
+ to_tensor(data_struct.hands_meanr).float(),
+ ]
+ )
+ self.register_buffer("other_default_pose", other_default_pose, False)
+
+ def register_smpl_buffers(self, data_struct, num_betas):
+ # shapedirs, (V, 3, N_betas), V=10475 for SMPLX
+ shapedirs = to_tensor(to_np(data_struct.shapedirs[:, :, :num_betas])).float()
+ self.register_buffer("shapedirs", shapedirs, False)
+
+ # v_template, (V, 3)
+ v_template = to_tensor(to_np(data_struct.v_template)).float()
+ self.register_buffer("v_template", v_template, False)
+
+ # J_regressor, (J, V), J=55 for SMPLX
+ J_regressor = to_tensor(to_np(data_struct.J_regressor)).float()
+ self.register_buffer("J_regressor", J_regressor, False)
+
+ # posedirs, (54*9, V, 3), note that the first global_orient is not included
+ posedirs = to_tensor(to_np(data_struct.posedirs)).float() # (V, 3, 54*9)
+ posedirs = rearrange(posedirs, "v c n -> n v c")
+ self.register_buffer("posedirs", posedirs, False)
+
+ # lbs_weights, (V, J), J=55
+ lbs_weights = to_tensor(to_np(data_struct.weights)).float()
+ self.register_buffer("lbs_weights", lbs_weights, False)
+
+ # parents, (J), long
+ parents = to_tensor(to_np(data_struct.kintree_table[0])).long()
+ parents[0] = -1
+ self.register_buffer("parents", parents, False)
+
+ def register_smplh_buffers(self, data_struct, num_pca_comps, flat_hand_mean):
+ # hand_pca, (N_pca, 45)
+ left_hand_components = to_tensor(data_struct.hands_componentsl[:num_pca_comps]).float()
+ right_hand_components = to_tensor(data_struct.hands_componentsr[:num_pca_comps]).float()
+ self.register_buffer("left_hand_components", left_hand_components, False)
+ self.register_buffer("right_hand_components", right_hand_components, False)
+
+ # hand_mean, (45,)
+ left_hand_mean = to_tensor(data_struct.hands_meanl).float()
+ right_hand_mean = to_tensor(data_struct.hands_meanr).float()
+ if not flat_hand_mean:
+ left_hand_mean = torch.zeros_like(left_hand_mean)
+ right_hand_mean = torch.zeros_like(right_hand_mean)
+ self.register_buffer("left_hand_mean", left_hand_mean, False)
+ self.register_buffer("right_hand_mean", right_hand_mean, False)
+
+ def register_smplx_buffers(self, data_struct):
+ # expr_dirs, (V, 3, N_expr)
+ expr_dirs = to_tensor(to_np(data_struct.shapedirs[:, :, 300:310])).float()
+ self.register_buffer("expr_dirs", expr_dirs, False)
+
+ def register_fast_skeleton_computing_buffers(self):
+ # For fast computing of skeleton under beta
+ J_template = self.J_regressor @ self.v_template # (J, 3)
+ J_shapedirs = torch.einsum("jv, vcd -> jcd", self.J_regressor, self.shapedirs) # (J, 3, 10)
+ self.register_buffer("J_template", J_template, False)
+ self.register_buffer("J_shapedirs", J_shapedirs, False)
+
+ def get_skeleton(self, betas):
+ return self.J_template + einsum(betas, self.J_shapedirs, "... k, j c k -> ... j c")
+
+ def forward(
+ self,
+ body_pose,
+ betas,
+ global_orient,
+ transl=None,
+ rotation_type="aa",
+ ):
+ """
+ Args:
+ body_pose: (B, L, 63)
+ betas: (B, L, 10)
+ global_orient: (B, L, 3)
+ transl: (B, L, 3)
+ Returns:
+ vertices: (B, L, V, 3)
+ """
+ # 1. Convert [global_orient, body_pose, other_default_pose] to rot_mats
+ other_default_pose = self.other_default_pose # (99,)
+ if rotation_type == "aa":
+ other_default_pose = other_default_pose.expand(*body_pose.shape[:-1], -1)
+ full_pose = torch.cat([global_orient, body_pose, other_default_pose], dim=-1)
+ rot_mats = axis_angle_to_matrix(full_pose.reshape(*full_pose.shape[:-1], 55, 3))
+ del full_pose, other_default_pose
+ else:
+ assert rotation_type == "r6d" # useful when doing smplify
+ other_default_pose = axis_angle_to_matrix(other_default_pose.view(33, 3))
+ part_full_pose = torch.cat([global_orient, body_pose], dim=-1)
+ rot_mats = rotation_6d_to_matrix(part_full_pose.view(*part_full_pose.shape[:-1], 22, 6))
+ other_default_pose = other_default_pose.expand(*rot_mats.shape[:-3], -1, -1, -1)
+ rot_mats = torch.cat([rot_mats, other_default_pose], dim=-3)
+ del part_full_pose, other_default_pose
+
+ # 2. Forward Kinematics
+ J = self.get_skeleton(betas) # (*, 55, 3)
+ A = batch_rigid_transform_v2(rot_mats, J, self.parents)[1]
+
+ # 3. Canonical v_posed = v_template + shaped_offsets + pose_offsets
+ pose_feature = rot_mats[..., 1:, :, :] - rot_mats.new([[1, 0, 0], [0, 1, 0], [0, 0, 1]])
+ pose_feature = pose_feature.view(*pose_feature.shape[:-3], -1) # (*, 55*3*3)
+ v_posed = (
+ self.v_template
+ + einsum(betas, self.shapedirs, "... k, v c k -> ... v c")
+ + einsum(pose_feature, self.posedirs, "... k, k v c -> ... v c")
+ )
+ del pose_feature, rot_mats
+
+ # 4. Skinning
+ T = einsum(self.lbs_weights, A, "v j, ... j c d -> ... v c d")
+ verts = einsum(T[..., :3, :3], v_posed, "... v c d, ... v d -> ... v c") + T[..., :3, 3]
+
+ # 5. Translation
+ if transl is not None:
+ verts = verts + transl[..., None, :]
+ return verts
+
+
+class SmplxLiteCoco17(SmplxLite):
+ """Output COCO17 joints (Faster, but cannot output vertices)"""
+
+ def __init__(self, **kwargs):
+ super().__init__(**kwargs)
+
+ # Compute mapping
+ smplx2smpl = torch.load(PROJ_ROOT / "hmr4d/utils/body_model/smplx2smpl_sparse.pt")
+ COCO17_regressor = torch.load(PROJ_ROOT / "hmr4d/utils/body_model/smpl_coco17_J_regressor.pt")
+ smplx2coco17 = torch.matmul(COCO17_regressor, smplx2smpl.to_dense())
+
+ jids, smplx_vids = torch.where(smplx2coco17 != 0)
+ smplx2coco17_interestd = torch.zeros([len(smplx_vids), 17])
+ for idx, (jid, smplx_vid) in enumerate(zip(jids, smplx_vids)):
+ smplx2coco17_interestd[idx, jid] = smplx2coco17[jid, smplx_vid]
+ self.register_buffer("smplx2coco17_interestd", smplx2coco17_interestd, False) # (132, 17)
+
+ # Update to vertices of interest
+ self.v_template = self.v_template[smplx_vids].clone() # (V', 3)
+ self.shapedirs = self.shapedirs[smplx_vids].clone() # (V', 3, K)
+ self.posedirs = self.posedirs[:, smplx_vids].clone() # (K, V', 3)
+ self.lbs_weights = self.lbs_weights[smplx_vids].clone() # (V', J)
+
+ def forward(self, body_pose, betas, global_orient, transl):
+ """Returns: joints (*, 17, 3). (B, L) or (B,) are both supported."""
+ # Use super class's forward to get verts
+ verts = super().forward(body_pose, betas, global_orient, transl) # (*, 132, 3)
+ joints = einsum(self.smplx2coco17_interestd, verts, "v j, ... v c -> ... j c")
+ return joints
+
+
+class SmplxLiteV437Coco17(SmplxLite):
+ def __init__(self, **kwargs):
+ super().__init__(**kwargs)
+
+ # Compute mapping (COCO17)
+ smplx2smpl = torch.load(PROJ_ROOT / "hmr4d/utils/body_model/smplx2smpl_sparse.pt")
+ COCO17_regressor = torch.load(PROJ_ROOT / "hmr4d/utils/body_model/smpl_coco17_J_regressor.pt")
+ smplx2coco17 = torch.matmul(COCO17_regressor, smplx2smpl.to_dense())
+
+ jids, smplx_vids = torch.where(smplx2coco17 != 0)
+ smplx2coco17_interestd = torch.zeros([len(smplx_vids), 17])
+ for idx, (jid, smplx_vid) in enumerate(zip(jids, smplx_vids)):
+ smplx2coco17_interestd[idx, jid] = smplx2coco17[jid, smplx_vid]
+ self.register_buffer("smplx2coco17_interestd", smplx2coco17_interestd, False) # (132, 17)
+ assert len(smplx_vids) == 132
+
+ # Verts437
+ smplx_vids2 = torch.load(PROJ_ROOT / "hmr4d/utils/body_model/smplx_verts437.pt")
+ smplx_vids = torch.cat([smplx_vids, smplx_vids2])
+
+ # Update to vertices of interest
+ self.v_template = self.v_template[smplx_vids].clone() # (V', 3)
+ self.shapedirs = self.shapedirs[smplx_vids].clone() # (V', 3, K)
+ self.posedirs = self.posedirs[:, smplx_vids].clone() # (K, V', 3)
+ self.lbs_weights = self.lbs_weights[smplx_vids].clone() # (V', J)
+
+ def forward(self, body_pose, betas, global_orient, transl):
+ """
+ Returns:
+ verts_437: (*, 437, 3)
+ joints (*, 17, 3). (B, L) or (B,) are both supported.
+ """
+ # Use super class's forward to get verts
+ verts = super().forward(body_pose, betas, global_orient, transl) # (*, 132+437, 3)
+
+ verts_437 = verts[..., 132:, :].clone()
+ joints = einsum(self.smplx2coco17_interestd, verts[..., :132, :], "v j, ... v c -> ... j c")
+ return verts_437, joints
+
+
+class SmplxLiteSmplN24(SmplxLite):
+ """Output SMPL(not smplx)-Neutral 24 joints (Faster, but cannot output vertices)"""
+
+ def __init__(self, **kwargs):
+ super().__init__(**kwargs)
+
+ # Compute mapping
+ smplx2smpl = torch.load(PROJ_ROOT / "hmr4d/utils/body_model/smplx2smpl_sparse.pt")
+ smpl2joints = torch.load(PROJ_ROOT / "hmr4d/utils/body_model/smpl_neutral_J_regressor.pt")
+ smplx2joints = torch.matmul(smpl2joints, smplx2smpl.to_dense())
+
+ jids, smplx_vids = torch.where(smplx2joints != 0)
+ smplx2joints_interested = torch.zeros([len(smplx_vids), smplx2joints.size(0)])
+ for idx, (jid, smplx_vid) in enumerate(zip(jids, smplx_vids)):
+ smplx2joints_interested[idx, jid] = smplx2joints[jid, smplx_vid]
+ self.register_buffer("smplx2joints_interested", smplx2joints_interested, False) # (V', J)
+
+ # Update to vertices of interest
+ self.v_template = self.v_template[smplx_vids].clone() # (V', 3)
+ self.shapedirs = self.shapedirs[smplx_vids].clone() # (V', 3, K)
+ self.posedirs = self.posedirs[:, smplx_vids].clone() # (K, V', 3)
+ self.lbs_weights = self.lbs_weights[smplx_vids].clone() # (V', J)
+
+ def forward(self, body_pose, betas, global_orient, transl):
+ """Returns: joints (*, J, 3). (B, L) or (B,) are both supported."""
+ # Use super class's forward to get verts
+ verts = super().forward(body_pose, betas, global_orient, transl) # (*, V', 3)
+ joints = einsum(self.smplx2joints_interested, verts, "v j, ... v c -> ... j c")
+ return joints
+
+
+def batch_rigid_transform_v2(rot_mats, joints, parents):
+ """
+ Args:
+ rot_mats: (*, J, 3, 3)
+ joints: (*, J, 3)
+ """
+ # check shape, since sometimes beta has shape=1
+ rot_mats_shape_prefix = rot_mats.shape[:-3]
+ if rot_mats_shape_prefix != joints.shape[:-2]:
+ joints = joints.expand(*rot_mats_shape_prefix, -1, -1)
+
+ rel_joints = joints.clone()
+ rel_joints[..., 1:, :] -= joints[..., parents[1:], :]
+ transforms_mat = torch.cat([rot_mats, rel_joints[..., :, None]], dim=-1) # (*, J, 3, 4)
+ transforms_mat = F.pad(transforms_mat, [0, 0, 0, 1], value=0.0)
+ transforms_mat[..., 3, 3] = 1.0 # (*, J, 4, 4)
+
+ transform_chain = [transforms_mat[..., 0, :, :]]
+ for i in range(1, parents.shape[0]):
+ # Subtract the joint location at the rest pose
+ # No need for rotation, since it's identity when at rest
+ curr_res = torch.matmul(transform_chain[parents[i]], transforms_mat[..., i, :, :])
+ transform_chain.append(curr_res)
+
+ transforms = torch.stack(transform_chain, dim=-3) # (*, J, 4, 4)
+
+ # The last column of the transformations contains the posed joints
+ posed_joints = transforms[..., :3, 3].clone()
+ rel_transforms = transforms.clone()
+ rel_transforms[..., :3, 3] -= einsum(transforms[..., :3, :3], joints, "... j c d, ... j d -> ... j c")
+ return posed_joints, rel_transforms
+
+
+def sync_time():
+ torch.cuda.synchronize()
+ return time()
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/smplx_verts437.pt b/third_party/GVHMR/hmr4d/utils/body_model/smplx_verts437.pt
new file mode 100644
index 0000000000000000000000000000000000000000..9434221e3687fa1f1adc4b0b4bffbd19b479a7f8
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/smplx_verts437.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ef0ea64c470a1fea80adb5e4b866c78a792a04f5f98744c9367b15e57bfb4a4d
+size 4203
diff --git a/third_party/GVHMR/hmr4d/utils/body_model/utils.py b/third_party/GVHMR/hmr4d/utils/body_model/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..17856e6851c58a2a9587b8ad6ae15e2ece70b39a
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/body_model/utils.py
@@ -0,0 +1,229 @@
+import os
+import numpy as np
+import torch
+
+SMPLH_JOINT_NAMES = [
+ 'pelvis',
+ 'left_hip',
+ 'right_hip',
+ 'spine1',
+ 'left_knee',
+ 'right_knee',
+ 'spine2',
+ 'left_ankle',
+ 'right_ankle',
+ 'spine3',
+ 'left_foot',
+ 'right_foot',
+ 'neck',
+ 'left_collar',
+ 'right_collar',
+ 'head',
+ 'left_shoulder',
+ 'right_shoulder',
+ 'left_elbow',
+ 'right_elbow',
+ 'left_wrist',
+ 'right_wrist',
+ 'left_index1',
+ 'left_index2',
+ 'left_index3',
+ 'left_middle1',
+ 'left_middle2',
+ 'left_middle3',
+ 'left_pinky1',
+ 'left_pinky2',
+ 'left_pinky3',
+ 'left_ring1',
+ 'left_ring2',
+ 'left_ring3',
+ 'left_thumb1',
+ 'left_thumb2',
+ 'left_thumb3',
+ 'right_index1',
+ 'right_index2',
+ 'right_index3',
+ 'right_middle1',
+ 'right_middle2',
+ 'right_middle3',
+ 'right_pinky1',
+ 'right_pinky2',
+ 'right_pinky3',
+ 'right_ring1',
+ 'right_ring2',
+ 'right_ring3',
+ 'right_thumb1',
+ 'right_thumb2',
+ 'right_thumb3',
+ 'nose',
+ 'right_eye',
+ 'left_eye',
+ 'right_ear',
+ 'left_ear',
+ 'left_big_toe',
+ 'left_small_toe',
+ 'left_heel',
+ 'right_big_toe',
+ 'right_small_toe',
+ 'right_heel',
+ 'left_thumb',
+ 'left_index',
+ 'left_middle',
+ 'left_ring',
+ 'left_pinky',
+ 'right_thumb',
+ 'right_index',
+ 'right_middle',
+ 'right_ring',
+ 'right_pinky',
+]
+
+SMPLH_LEFT_LEG = ['left_hip', 'left_knee', 'left_ankle', 'left_foot']
+SMPLH_RIGHT_LEG = ['right_hip', 'right_knee', 'right_ankle', 'right_foot']
+SMPLH_LEFT_ARM = ['left_collar', 'left_shoulder', 'left_elbow', 'left_wrist']
+SMPLH_RIGHT_ARM = ['right_collar', 'right_shoulder', 'right_elbow', 'right_wrist']
+SMPLH_HEAD = ['neck', 'head']
+SMPLH_SPINE = ['spine1', 'spine2', 'spine3']
+
+# name to 21 index (without pelvis, hand, and extra)
+_name_2_idx = {j: i for i, j in enumerate(SMPLH_JOINT_NAMES[1:22])}
+SMPLH_PART_IDX = {
+ 'left_leg': [_name_2_idx[x] for x in SMPLH_LEFT_LEG],
+ 'right_leg': [_name_2_idx[x] for x in SMPLH_RIGHT_LEG],
+ 'left_arm': [_name_2_idx[x] for x in SMPLH_LEFT_ARM],
+ 'right_arm': [_name_2_idx[x] for x in SMPLH_RIGHT_ARM],
+ 'two_legs': [_name_2_idx[x] for x in SMPLH_LEFT_LEG + SMPLH_RIGHT_LEG],
+ 'left_arm_and_leg': [_name_2_idx[x] for x in SMPLH_LEFT_ARM + SMPLH_LEFT_LEG],
+ 'right_arm_and_leg': [_name_2_idx[x] for x in SMPLH_RIGHT_ARM + SMPLH_RIGHT_LEG],
+}
+
+# name to full index
+_name_2_idx_full = {j: i for i, j in enumerate(SMPLH_JOINT_NAMES)}
+SMPLH_PART_IDX_FULL = {
+ 'lower_body': [_name_2_idx_full[x] for x in ['pelvis'] + SMPLH_LEFT_LEG + SMPLH_RIGHT_LEG]
+}
+
+# ===== ⬇️ Fitting optimizer ⬇️ ===== #
+SMPL_JOINTS = {'hips': 0, 'leftUpLeg': 1, 'rightUpLeg': 2, 'spine': 3, 'leftLeg': 4, 'rightLeg': 5,
+ 'spine1': 6, 'leftFoot': 7, 'rightFoot': 8, 'spine2': 9, 'leftToeBase': 10, 'rightToeBase': 11,
+ 'neck': 12, 'leftShoulder': 13, 'rightShoulder': 14, 'head': 15, 'leftArm': 16, 'rightArm': 17,
+ 'leftForeArm': 18, 'rightForeArm': 19, 'leftHand': 20, 'rightHand': 21}
+
+# chosen virtual mocap markers that are "keypoints" to work with
+KEYPT_VERTS = [4404, 920, 3076, 3169, 823, 4310, 1010, 1085, 4495, 4569, 6615, 3217, 3313, 6713,
+ 6785, 3383, 6607, 3207, 1241, 1508, 4797, 4122, 1618, 1569, 5135, 5040, 5691, 5636,
+ 5404, 2230, 2173, 2108, 134, 3645, 6543, 3123, 3024, 4194, 1306, 182, 3694, 4294, 744]
+
+
+# From https://github.com/vchoutas/smplify-x/blob/master/smplifyx/utils.py
+# Please see license for usage restrictions.
+def smpl_to_openpose(model_type='smplx', use_hands=True, use_face=True,
+ use_face_contour=False, openpose_format='coco25'):
+ ''' Returns the indices of the permutation that maps SMPL to OpenPose
+
+ Parameters
+ ----------
+ model_type: str, optional
+ The type of SMPL-like model that is used. The default mapping
+ returned is for the SMPLX model
+ use_hands: bool, optional
+ Flag for adding to the returned permutation the mapping for the
+ hand keypoints. Defaults to True
+ use_face: bool, optional
+ Flag for adding to the returned permutation the mapping for the
+ face keypoints. Defaults to True
+ use_face_contour: bool, optional
+ Flag for appending the facial contour keypoints. Defaults to False
+ openpose_format: bool, optional
+ The output format of OpenPose. For now only COCO-25 and COCO-19 is
+ supported. Defaults to 'coco25'
+
+ '''
+ if openpose_format.lower() == 'coco25':
+ if model_type == 'smpl':
+ return np.array([24, 12, 17, 19, 21, 16, 18, 20, 0, 2, 5, 8, 1, 4,
+ 7, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34],
+ dtype=np.int32)
+ elif model_type == 'smplh':
+ body_mapping = np.array([52, 12, 17, 19, 21, 16, 18, 20, 0, 2, 5,
+ 8, 1, 4, 7, 53, 54, 55, 56, 57, 58, 59,
+ 60, 61, 62], dtype=np.int32)
+ mapping = [body_mapping]
+ if use_hands:
+ lhand_mapping = np.array([20, 34, 35, 36, 63, 22, 23, 24, 64,
+ 25, 26, 27, 65, 31, 32, 33, 66, 28,
+ 29, 30, 67], dtype=np.int32)
+ rhand_mapping = np.array([21, 49, 50, 51, 68, 37, 38, 39, 69,
+ 40, 41, 42, 70, 46, 47, 48, 71, 43,
+ 44, 45, 72], dtype=np.int32)
+ mapping += [lhand_mapping, rhand_mapping]
+ return np.concatenate(mapping)
+ # SMPLX
+ elif model_type == 'smplx':
+ body_mapping = np.array([55, 12, 17, 19, 21, 16, 18, 20, 0, 2, 5,
+ 8, 1, 4, 7, 56, 57, 58, 59, 60, 61, 62,
+ 63, 64, 65], dtype=np.int32)
+ mapping = [body_mapping]
+ if use_hands:
+ lhand_mapping = np.array([20, 37, 38, 39, 66, 25, 26, 27,
+ 67, 28, 29, 30, 68, 34, 35, 36, 69,
+ 31, 32, 33, 70], dtype=np.int32)
+ rhand_mapping = np.array([21, 52, 53, 54, 71, 40, 41, 42, 72,
+ 43, 44, 45, 73, 49, 50, 51, 74, 46,
+ 47, 48, 75], dtype=np.int32)
+
+ mapping += [lhand_mapping, rhand_mapping]
+ if use_face:
+ # end_idx = 127 + 17 * use_face_contour
+ face_mapping = np.arange(76, 127 + 17 * use_face_contour,
+ dtype=np.int32)
+ mapping += [face_mapping]
+
+ return np.concatenate(mapping)
+ else:
+ raise ValueError('Unknown model type: {}'.format(model_type))
+ elif openpose_format == 'coco19':
+ if model_type == 'smpl':
+ return np.array([24, 12, 17, 19, 21, 16, 18, 20, 0, 2, 5, 8,
+ 1, 4, 7, 25, 26, 27, 28],
+ dtype=np.int32)
+ elif model_type == 'smplh':
+ body_mapping = np.array([52, 12, 17, 19, 21, 16, 18, 20, 0, 2, 5,
+ 8, 1, 4, 7, 53, 54, 55, 56],
+ dtype=np.int32)
+ mapping = [body_mapping]
+ if use_hands:
+ lhand_mapping = np.array([20, 34, 35, 36, 57, 22, 23, 24, 58,
+ 25, 26, 27, 59, 31, 32, 33, 60, 28,
+ 29, 30, 61], dtype=np.int32)
+ rhand_mapping = np.array([21, 49, 50, 51, 62, 37, 38, 39, 63,
+ 40, 41, 42, 64, 46, 47, 48, 65, 43,
+ 44, 45, 66], dtype=np.int32)
+ mapping += [lhand_mapping, rhand_mapping]
+ return np.concatenate(mapping)
+ # SMPLX
+ elif model_type == 'smplx':
+ body_mapping = np.array([55, 12, 17, 19, 21, 16, 18, 20, 0, 2, 5,
+ 8, 1, 4, 7, 56, 57, 58, 59],
+ dtype=np.int32)
+ mapping = [body_mapping]
+ if use_hands:
+ lhand_mapping = np.array([20, 37, 38, 39, 60, 25, 26, 27,
+ 61, 28, 29, 30, 62, 34, 35, 36, 63,
+ 31, 32, 33, 64], dtype=np.int32)
+ rhand_mapping = np.array([21, 52, 53, 54, 65, 40, 41, 42, 66,
+ 43, 44, 45, 67, 49, 50, 51, 68, 46,
+ 47, 48, 69], dtype=np.int32)
+
+ mapping += [lhand_mapping, rhand_mapping]
+ if use_face:
+ face_mapping = np.arange(70, 70 + 51 +
+ 17 * use_face_contour,
+ dtype=np.int32)
+ mapping += [face_mapping]
+
+ return np.concatenate(mapping)
+ else:
+ raise ValueError('Unknown model type: {}'.format(model_type))
+ else:
+ raise ValueError('Unknown joint format: {}'.format(openpose_format))
diff --git a/third_party/GVHMR/hmr4d/utils/callbacks/lr_monitor.py b/third_party/GVHMR/hmr4d/utils/callbacks/lr_monitor.py
new file mode 100644
index 0000000000000000000000000000000000000000..fd26ec1470ab68b073c6c30959f2469c636eaacc
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/callbacks/lr_monitor.py
@@ -0,0 +1,5 @@
+from pytorch_lightning.callbacks import LearningRateMonitor
+from hmr4d.configs import builds, MainStore
+
+
+MainStore.store(name="pl", node=builds(LearningRateMonitor), group="callbacks/lr_monitor")
diff --git a/third_party/GVHMR/hmr4d/utils/callbacks/prog_bar.py b/third_party/GVHMR/hmr4d/utils/callbacks/prog_bar.py
new file mode 100644
index 0000000000000000000000000000000000000000..97798fb638fe7ca9dacc7befccc222a9a70440a4
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/callbacks/prog_bar.py
@@ -0,0 +1,436 @@
+from collections import OrderedDict
+from numbers import Number
+from datetime import datetime, timedelta
+from typing import Any, Dict, Union
+from pytorch_lightning.utilities.types import STEP_OUTPUT
+import torch
+from pytorch_lightning.callbacks.progress.tqdm_progress import TQDMProgressBar, Tqdm, convert_inf
+from pytorch_lightning.callbacks.progress import ProgressBar
+from pytorch_lightning.utilities import rank_zero_only
+import pytorch_lightning as pl
+
+from hmr4d.utils.pylogger import Log
+from time import time
+from collections import deque
+import sys
+from hmr4d.configs import MainStore, builds
+
+# ========== Helper functions ========== #
+
+
+def format_num(n):
+ f = "{0:.3g}".format(n).replace("+0", "+").replace("-0", "-")
+ n = str(n)
+ return f if len(f) < len(n) else n
+
+
+def convert_kwargs_to_str(**kwargs):
+ # Sort in alphabetical order to be more deterministic
+ postfix = OrderedDict([])
+ for key in sorted(kwargs.keys()):
+ new_key = key.split("/")[-1]
+ postfix[new_key] = kwargs[key]
+ # Preprocess stats according to datatype
+ for key in postfix.keys():
+ # Number: limit the length of the string
+ if isinstance(postfix[key], Number):
+ postfix[key] = format_num(postfix[key])
+ # Else for any other type, try to get the string conversion
+ elif not isinstance(postfix[key], str):
+ postfix[key] = str(postfix[key])
+ # Else if it's a string, don't need to preprocess anything
+ # Stitch together to get the final postfix
+ postfix = ", ".join(key + "=" + postfix[key].strip() for key in postfix.keys())
+ return postfix
+
+
+def convert_t_to_str(t):
+ """Convert time in second to string in format hour:minute:second.
+ If hour is 0, don't show it. Always show minute and second.
+ """
+ t_str = timedelta(seconds=t) # e.g. 0:00:00.704186
+ t_str = str(t_str).split(".")[0] # e.g. 0:00:00
+ if t_str[:2] == "0:":
+ t_str = t_str[2:]
+ return t_str
+
+
+class MyTQDMProgressBar(TQDMProgressBar, pl.Callback):
+ def init_train_tqdm(self):
+ bar = Tqdm(
+ desc="Training", # this will be overwritten anyway
+ bar_format="{desc}{percentage:3.0f}%[{bar:10}][{n_fmt}/{total_fmt}, {elapsed}→{remaining},{rate_fmt}]{postfix}",
+ position=(2 * self.process_position),
+ disable=self.is_disabled,
+ leave=False,
+ smoothing=0,
+ dynamic_ncols=False,
+ )
+ return bar
+
+ @rank_zero_only
+ def on_train_batch_end(self, trainer, pl_module, outputs, batch, batch_idx):
+ # this function also updates the main progress bar
+ super().on_train_batch_end(trainer, pl_module, outputs, batch, batch_idx)
+ # in this function, we only set the postfix of the main progress bar
+ n = batch_idx + 1
+ if self._should_update(n, self.train_progress_bar.total):
+ # Set post-fix string
+ # 1. maximum GPU usage
+ max_mem = torch.cuda.max_memory_allocated() / 1024.0 / 1024.0 / 1024.0
+ post_fix_str = f"maxGPU={max_mem:.1f}GB"
+
+ # 2. training metrics
+ training_metrics = self.get_metrics(trainer, pl_module)
+ training_metrics.pop("v_num", None)
+ post_fix_str += ", " + convert_kwargs_to_str(**training_metrics)
+
+ # extra message if applicable
+ if "message" in outputs:
+ post_fix_str += ", " + outputs["message"]
+
+ self.train_progress_bar.set_postfix_str(post_fix_str)
+
+
+class ProgressReporter(ProgressBar, pl.Callback):
+ def __init__(
+ self,
+ log_every_percent: float = 0.1, # report interval
+ exp_name=None, # if None, use pl_module.exp_name or "Unnamed Experiment"
+ data_name=None, # if None, use pl_module.exp_name or "Unknown Data"
+ **kwargs,
+ ):
+ super().__init__()
+ self.enable = True
+ # 1. Store experiment meta data.
+ self.log_every_percent = log_every_percent
+ self.exp_name = exp_name
+ self.data_name = data_name
+ self.batch_time_queue = deque(maxlen=5)
+ self.start_prompt = "🚀"
+ self.finish_prompt = "✅"
+ # 2. Utils for evaluation
+ self.n_finished = 0
+
+ def disable(self):
+ self.enable = False
+
+ def setup(self, trainer: pl.Trainer, pl_module: pl.LightningModule, stage: str) -> None:
+ # Connect to the trainer object.
+ super().setup(trainer, pl_module, stage)
+ self.stage = stage
+ self.time_exp_start = time()
+ self.epoch_exp_start = trainer.current_epoch
+
+ if self.exp_name is None:
+ if hasattr(pl_module, "exp_name"):
+ self.exp_name = pl_module.exp_name
+ else:
+ self.exp_name = "Unnamed Experiment"
+ if self.data_name is None:
+ if hasattr(pl_module, "data_name"):
+ self.data_name = pl_module.data_name
+ else:
+ self.data_name = "Unknown Data"
+
+ def print(self, *args: Any, **kwargs: Any) -> None:
+ print(*args)
+
+ def get_metrics(self, trainer: pl.Trainer, pl_module: pl.LightningModule) -> Dict[str, Union[str, float]]:
+ """Get metrics from trainer for progress bar."""
+ items = super().get_metrics(trainer, pl_module)
+ items.pop("v_num", None)
+ return items
+
+ def _should_update(self, n_finished: int, total: int) -> bool:
+ """
+ Rule: Log every `log_every_percent` percent, or the last batch.
+ """
+ log_interval = max(int(total * self.log_every_percent), 1)
+ able = n_finished % log_interval == 0 or n_finished == total
+ if log_interval > 10:
+ able = able or n_finished in [5, 10] # always log
+ able = able and self.enable
+ return able
+
+ @rank_zero_only
+ def on_train_epoch_start(self, trainer: "pl.Trainer", *_: Any) -> None:
+ self.print("=" * 80)
+ Log.info(
+ f"{self.start_prompt}[FIT][Epoch {trainer.current_epoch}] Data: {self.data_name} Experiment: {self.exp_name}"
+ )
+ self.time_train_epoch_start = time()
+
+ @rank_zero_only
+ def on_train_batch_end(self, trainer, pl_module, outputs, batch, batch_idx):
+ super().on_train_batch_end(trainer, pl_module, outputs, batch, batch_idx) # don't forget this :)
+ total = self.total_train_batches
+
+ # Speed
+ n_finished = batch_idx + 1
+ percent = 100 * n_finished / total
+ time_current = time()
+ self.batch_time_queue.append(time_current)
+ time_elapsed = time_current - self.time_train_epoch_start # second
+ time_remaining = time_elapsed * (total - n_finished) / n_finished # second
+ if len(self.batch_time_queue) == 1: # cannot compute speed
+ speed = 1 / time_elapsed
+ else:
+ speed = (len(self.batch_time_queue) - 1) / (self.batch_time_queue[-1] - self.batch_time_queue[0])
+
+ # Skip if not update
+ if not self._should_update(n_finished, total):
+ return
+
+ # ===== Set Prefix string ===== #
+ # General
+ desc = f"[Train]"
+
+ # Speed: Get elapsed time and estimated remaining time
+ time_elapsed_str = convert_t_to_str(time_elapsed)
+ time_remaining_str = convert_t_to_str(time_remaining)
+ speed_str = f"{speed:.2f}it/s" if speed > 1 else f"{1/speed:.1f}s/it"
+ n_digit = len(str(total))
+ desc_speed = (
+ f"[{n_finished:{n_digit}d}/{total}={percent:3.0f}%, {time_elapsed_str} → {time_remaining_str}, {speed_str}]"
+ )
+
+ # ===== Set postfix string ===== #
+ # 1. maximum GPU usage
+ max_mem = torch.cuda.max_memory_allocated() / 1024.0 / 1024.0 / 1024.0
+ post_fix_str = f"maxGPU={max_mem:.1f}GB"
+
+ # 2. training step metrics
+ train_metrics = self.get_metrics(trainer, pl_module)
+ train_metrics = {k: v for k, v in train_metrics.items() if ("train" in k and "epoch" not in k)}
+ post_fix_str += ", " + convert_kwargs_to_str(**train_metrics)
+
+ # extra message if applicable
+ if "message" in outputs:
+ post_fix_str += ", " + outputs["message"]
+ post_fix_str = f"[{post_fix_str}]"
+
+ # ===== Output ===== #
+ bar_output = f"{desc}{desc_speed}{post_fix_str}"
+ self.print(bar_output)
+
+ @rank_zero_only
+ def on_train_epoch_end(self, trainer: pl.Trainer, pl_module: pl.LightningModule) -> None:
+ super().on_train_epoch_end(trainer, pl_module)
+
+ # Clear
+ self.batch_time_queue.clear()
+
+ # Estimate Epoch time
+ n_finished = trainer.current_epoch + 1 - self.epoch_exp_start
+ n_to_finish = trainer.max_epochs - trainer.current_epoch - 1
+ time_current = time()
+ time_elapsed = time_current - self.time_exp_start
+ time_remaining = time_elapsed * n_to_finish / n_finished
+ time_elapsed_str = convert_t_to_str(time_elapsed)
+ time_remaining_str = convert_t_to_str(time_remaining)
+
+ # Metrics
+ # training epoch metrics
+ train_metrics = self.get_metrics(trainer, pl_module)
+ train_metrics = {k: v for k, v in train_metrics.items() if ("train" in k and "epoch" in k)}
+ train_metrics_str = convert_kwargs_to_str(**train_metrics)
+
+ Log.info(
+ f"{self.finish_prompt}[FIT][Epoch {trainer.current_epoch}] finished! {time_elapsed_str}→{time_remaining_str} | {train_metrics_str}"
+ )
+
+ # ===== Validation/Test/Prediction ===== #
+ @rank_zero_only
+ def on_validation_epoch_start(self, trainer, pl_module):
+ self.time_val_epoch_start = time()
+
+ @rank_zero_only
+ def on_validation_batch_end(self, trainer, pl_module, outputs, batch, batch_idx, dataloader_idx=0):
+ self.n_finished += 1
+ n_finished = self.n_finished
+ total = self.total_val_batches
+ if not self._should_update(n_finished, total):
+ return
+
+ # General
+ desc = f"[Val]"
+
+ # Speed
+ percent = 100 * n_finished / total
+ time_current = time()
+ time_elapsed = time_current - self.time_val_epoch_start # second
+ time_remaining = time_elapsed * (total - n_finished) / n_finished # second
+ time_elapsed_str = convert_t_to_str(time_elapsed)
+ time_remaining_str = convert_t_to_str(time_remaining)
+ desc_speed = f"[{n_finished}/{total} ={percent:3.0f}%, {time_elapsed_str}→{time_remaining_str}]"
+
+ # Output
+ bar_output = f"{desc} {desc_speed}"
+ self.print(bar_output)
+
+ def on_validation_epoch_end(self, trainer: pl.Trainer, pl_module: pl.LightningModule) -> None:
+ # Reset
+ self.n_finished = 0
+
+
+class EmojiProgressReporter(ProgressBar, pl.Callback):
+ def __init__(
+ self,
+ refresh_rate_batch: Union[int, None] = 1, # report interval of batch, set None to disable it
+ refresh_rate_epoch: int = 1, # report interval of epoch
+ **kwargs,
+ ):
+ super().__init__()
+ self.enable = True
+ # Store experiment meta data.
+ self.refresh_rate_batch = refresh_rate_batch
+ self.refresh_rate_epoch = refresh_rate_epoch
+
+ # Style of the progress bar.
+ self.title_prompt = "📝"
+ self.prog_prompt = "🚀"
+ self.timer_prompt = "⌛️"
+ self.metric_prompt = "📌"
+ self.finish_prompt = "✅"
+
+ def disable(self):
+ self.enable = False
+
+ def setup(self, trainer: pl.Trainer, pl_module: pl.LightningModule, stage: str):
+ # Connect to the trainer object.
+ super().setup(trainer, pl_module, stage)
+ self.stage = stage
+ self.time_start_batch = None
+ self.time_start_epoch = None
+ if hasattr(pl_module, "exp_name"):
+ self.exp_name = pl_module.exp_name
+ else:
+ self.exp_name = "Unnamed Experiment"
+ Log.warn("Experiment name not found, please set it to `pl_module.exp_name`!")
+
+ def print(self, *args: Any, **kwargs: Any):
+ print(*args)
+
+ def get_metrics(self, trainer: pl.Trainer, pl_module: pl.LightningModule) -> Dict[str, Union[str, float]]:
+ """Get metrics from trainer for progress bar."""
+ items = super().get_metrics(trainer, pl_module)
+ items.pop("v_num", None)
+ return dict(sorted(items.items()))
+
+ def _should_log_batch(self, n: int) -> bool:
+ # Disable batch log.
+ if self.refresh_rate_batch is None:
+ return False
+ # Log at the first & last batch, and every `self.refresh_rate_batch` batches.
+ able = n % self.refresh_rate_batch == 0 or n == self.total_train_batches - 1
+ able = able and self.enable
+ return able
+
+ def _should_log_epoch(self, n: int) -> bool:
+ # Log at the first & last epoch, and every `self.refresh_rate_epoch` epochs.
+ able = n % self.refresh_rate_epoch == 0 or n == self.trainer.max_epochs - 1
+ able = able and self.enable
+ return able
+
+ def timestamp_delta_to_str(self, timestamp_delta: float):
+ """Convert delta timestamp to string."""
+ time_rest = timedelta(seconds=timestamp_delta)
+ hours, remainder = divmod(time_rest.seconds, 3600)
+ minutes, seconds = divmod(remainder, 60)
+ time_str = ""
+
+ # Check if the time is valid. Note that, if `hours` is visible, then `minutes` must be visible.
+ if hours <= 0:
+ hours = None
+ if minutes <= 0:
+ minutes = None
+ if seconds <= 0:
+ seconds = None
+
+ time_str += f"{hours}h " if hours is not None else ""
+ time_str += f"{minutes}m " if minutes is not None else ""
+ time_str += f"{seconds}s" if seconds is not None else ""
+ return time_str
+
+ @rank_zero_only
+ def on_train_batch_start(self, trainer: pl.Trainer, pl_module: pl.LightningModule, batch: Any, batch_idx: int):
+ super().on_train_batch_start(trainer, pl_module, batch, batch_idx)
+ # Initialize some meta data.
+ if self.time_start_batch is None:
+ self.time_start_batch = datetime.now().timestamp()
+
+ @rank_zero_only
+ def on_train_batch_end(self, trainer, pl_module, outputs, batch, batch_idx):
+ super().on_train_batch_end(trainer, pl_module, outputs, batch, batch_idx) # don't forget this :)
+ # Get some meta data.
+ epoch_idx = trainer.current_epoch
+ percent = 100 * (batch_idx + 1) / (self.total_train_batches + 1)
+ metrics = self.get_metrics(trainer, pl_module)
+
+ # Current time.
+ time_cur_stamp = datetime.now().timestamp()
+ time_cur_str = datetime.fromtimestamp(time_cur_stamp).strftime("%m-%d %H:%M:%S")
+ # Rest time.
+ time_rest_stamp = (time_cur_stamp - self.time_start_batch) * (100 - percent) / percent
+ time_rest_str = self.timestamp_delta_to_str(time_rest_stamp)
+
+ if not self._should_log_batch(batch_idx):
+ return
+
+ # Print the logs.
+ self.print(f"{self.title_prompt} [{self.stage.upper()}] Exp: {self.exp_name}...")
+ self.print(
+ f"{self.prog_prompt} Ep {epoch_idx}: {int(percent):02d}% <= [{batch_idx}/{self.total_train_batches}]"
+ )
+ self.print(f"{self.timer_prompt} Time: {time_cur_str} | Ep Rest: {time_rest_str}")
+ for k, v in metrics.items():
+ self.print(f"{self.metric_prompt} {k}: {v}")
+ self.print("") # Add a blank line.
+
+ def on_train_epoch_start(self, trainer: pl.Trainer, pl_module: pl.LightningModule):
+ super().on_train_epoch_start(trainer, pl_module)
+ # Initialize some meta data.
+ self.time_start_batch = None
+ if self.time_start_epoch is None:
+ self.time_start_epoch = datetime.now().timestamp()
+
+ @rank_zero_only
+ def on_train_epoch_end(self, trainer: pl.Trainer, pl_module: pl.LightningModule):
+ super().on_train_epoch_end(trainer, pl_module)
+ # Get some meta data.
+ epoch_idx = trainer.current_epoch
+ percent = 100 * (epoch_idx + 1) / (self.trainer.max_epochs + 1)
+ metrics = self.get_metrics(trainer, pl_module)
+
+ # Current time.
+ time_cur = datetime.now().timestamp()
+ time_str = datetime.fromtimestamp(time_cur).strftime("%m-%d %H: %M:%S")
+ # Rest time.
+ time_rest_stamp = (time_cur - self.time_start_epoch) * (100 - percent) / percent
+ time_rest_str = self.timestamp_delta_to_str(time_rest_stamp)
+
+ if not self._should_log_batch(epoch_idx):
+ return
+
+ # Print the logs.
+ self.print(f">> >> >> >>")
+ self.print(f"{self.title_prompt} [{self.stage.upper()}] Exp: {self.exp_name}")
+ self.print(f"{self.finish_prompt} Ep {epoch_idx} finished!")
+ self.print(f"{self.timer_prompt} Time: {time_str} | Rest: {time_rest_str}")
+ for k, v in metrics.items():
+ self.print(f"{self.metric_prompt} {k}: {v}")
+ self.print(f"<< << << <<")
+ self.print("") # Add a blank line.
+
+
+group_name = "callbacks/prog_bar"
+prog_reporter_base = builds(
+ ProgressReporter,
+ log_every_percent=0.1,
+ exp_name="${exp_name}",
+ data_name="${data_name}",
+ populate_full_signature=True,
+)
+MainStore.store(name="prog_reporter_every0.1", node=prog_reporter_base, group=group_name)
+MainStore.store(name="prog_reporter_every0.2", node=prog_reporter_base(log_every_percent=0.2), group=group_name)
diff --git a/third_party/GVHMR/hmr4d/utils/callbacks/simple_ckpt_saver.py b/third_party/GVHMR/hmr4d/utils/callbacks/simple_ckpt_saver.py
new file mode 100644
index 0000000000000000000000000000000000000000..5565339df71bfff03d77eda01504b62ed8ead472
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/callbacks/simple_ckpt_saver.py
@@ -0,0 +1,93 @@
+from pathlib import Path
+import torch
+import pytorch_lightning as pl
+from pytorch_lightning.callbacks.checkpoint import Checkpoint
+from pytorch_lightning.utilities import rank_zero_only
+
+from hmr4d.utils.pylogger import Log
+from hmr4d.configs import MainStore, builds
+
+
+class SimpleCkptSaver(Checkpoint):
+ """
+ This callback runs at the end of each training epoch.
+ Check {every_n_epochs} and save at most {save_top_k} model if it is time.
+ """
+
+ def __init__(
+ self,
+ output_dir,
+ filename="e{epoch:03d}-s{step:06d}.ckpt",
+ save_top_k=1,
+ every_n_epochs=1,
+ save_last=None,
+ save_weights_only=True,
+ ):
+ super().__init__()
+ self.output_dir = Path(output_dir)
+ self.filename = filename
+ self.save_top_k = save_top_k
+ self.every_n_epochs = every_n_epochs
+ self.save_last = save_last
+ self.save_weights_only = save_weights_only
+
+ # Setup output dir
+ if rank_zero_only.rank == 0:
+ self.output_dir.mkdir(parents=True, exist_ok=True)
+ Log.info(f"[Simple Ckpt Saver]: Save to `{self.output_dir}'")
+
+ @rank_zero_only
+ def on_train_epoch_end(self, trainer, pl_module):
+ """Save a checkpoint at the end of the training epoch."""
+ if self.every_n_epochs >= 1 and (trainer.current_epoch + 1) % self.every_n_epochs == 0:
+ if self.save_top_k == 0:
+ return
+
+ # Current saved ckpts in the output_dir
+ model_paths = []
+ for p in sorted(list(self.output_dir.glob("*.ckpt"))):
+ model_paths.append(p)
+ model_to_remove = model_paths[0] if len(model_paths) >= self.save_top_k else None
+
+ # Save cureent checkpoint
+ filepath = self.output_dir / self.filename.format(epoch=trainer.current_epoch, step=trainer.global_step)
+ checkpoint = {
+ "epoch": trainer.current_epoch,
+ "global_step": trainer.global_step,
+ "pytorch-lightning_version": pl.__version__,
+ "state_dict": pl_module.state_dict(),
+ }
+ pl_module.on_save_checkpoint(checkpoint)
+
+ if not self.save_weights_only:
+ # optimizer
+ optimizer_states = []
+ for i, optimizer in enumerate(trainer.optimizers):
+ # Rely on accelerator to dump optimizer state
+ optimizer_state = trainer.strategy.optimizer_state(optimizer)
+ optimizer_states.append(optimizer_state)
+ checkpoint["optimizer_states"] = optimizer_states
+
+ # lr_scheduler
+ lr_schedulers = []
+ for config in trainer.lr_scheduler_configs:
+ lr_schedulers.append(config.scheduler.state_dict())
+ checkpoint["lr_schedulers"] = lr_schedulers
+
+ # trainer.strategy.checkpoint_io.save_checkpoint(checkpoint, filepath)
+ torch.save(checkpoint, filepath)
+
+ # Remove the earliest checkpoint
+ if model_to_remove:
+ trainer.strategy.remove_checkpoint(model_paths[0])
+
+
+group_name = "callbacks/simple_ckpt_saver"
+base = builds(SimpleCkptSaver, output_dir="${output_dir}/checkpoints/", populate_full_signature=True)
+MainStore.store(name="base", node=base, group=group_name)
+MainStore.store(name="every1e", node=base, group=group_name)
+MainStore.store(name="every2e", node=base(every_n_epochs=2), group=group_name)
+MainStore.store(name="every5e", node=base(every_n_epochs=5), group=group_name)
+MainStore.store(name="every5e_top100", node=base(every_n_epochs=5, save_top_k=100), group=group_name)
+MainStore.store(name="every10e", node=base(every_n_epochs=10), group=group_name)
+MainStore.store(name="every10e_top100", node=base(every_n_epochs=10, save_top_k=100), group=group_name)
diff --git a/third_party/GVHMR/hmr4d/utils/callbacks/train_speed_timer.py b/third_party/GVHMR/hmr4d/utils/callbacks/train_speed_timer.py
new file mode 100644
index 0000000000000000000000000000000000000000..6c8c038e2f8539c587c5ceb38cd70123a368adb9
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/callbacks/train_speed_timer.py
@@ -0,0 +1,71 @@
+import pytorch_lightning as pl
+from pytorch_lightning.utilities import rank_zero_only
+from time import time
+from collections import deque
+
+from hmr4d.configs import MainStore, builds
+
+
+class TrainSpeedTimer(pl.Callback):
+ def __init__(self, N_avg=5):
+ """
+ This callback times the training speed (averge over recent 5 iterations)
+ 1. Data waiting time: this should be small, otherwise the data loading should be improved
+ 2. Single batch time: this is the time for one batch of training (excluding data waiting)
+ """
+ super().__init__()
+ self.last_batch_end = None
+ self.this_batch_start = None
+
+ # time queues for averaging
+ self.data_waiting_time_queue = deque(maxlen=N_avg)
+ self.single_batch_time_queue = deque(maxlen=N_avg)
+
+ @rank_zero_only
+ def on_train_batch_start(self, trainer, pl_module, batch, batch_idx):
+ """Count the time of data waiting"""
+ if self.last_batch_end is not None:
+ # This should be small, otherwise the data loading should be improved
+ data_waiting = time() - self.last_batch_end
+
+ # Average the time
+ self.data_waiting_time_queue.append(data_waiting)
+ average_time = sum(self.data_waiting_time_queue) / len(self.data_waiting_time_queue)
+
+ # Log to prog-bar
+ pl_module.log(
+ "train_timer/data_waiting", average_time, on_step=True, on_epoch=False, prog_bar=True, logger=True
+ )
+
+ self.this_batch_start = time()
+
+ @rank_zero_only
+ def on_train_batch_end(self, trainer, pl_module, outputs, batch, batch_idx):
+ # Effective training time elapsed (excluding data waiting)
+ single_batch = time() - self.this_batch_start
+
+ # Average the time
+ self.single_batch_time_queue.append(single_batch)
+ average_time = sum(self.single_batch_time_queue) / len(self.single_batch_time_queue)
+
+ # Log iter time
+ pl_module.log(
+ "train_timer/single_batch", average_time, on_step=True, on_epoch=False, prog_bar=False, logger=True
+ )
+
+ # Set timer for counting data waiting
+ self.last_batch_end = time()
+
+ @rank_zero_only
+ def on_train_epoch_end(self, trainer, pl_module):
+ # Reset the timer
+ self.last_batch_end = None
+ self.this_batch_start = None
+ # Clear the queue
+ self.data_waiting_time_queue.clear()
+ self.single_batch_time_queue.clear()
+
+
+group_name = "callbacks/train_speed_timer"
+base = builds(TrainSpeedTimer, populate_full_signature=True)
+MainStore.store(name="base", node=base, group=group_name)
diff --git a/third_party/GVHMR/hmr4d/utils/comm/gather.py b/third_party/GVHMR/hmr4d/utils/comm/gather.py
new file mode 100644
index 0000000000000000000000000000000000000000..741a6cfbac1d8f030bd77c131fda221184e42355
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/comm/gather.py
@@ -0,0 +1,257 @@
+# Copyright (c) Facebook, Inc. and its affiliates. All Rights Reserved
+"""
+[Copied from detectron2]
+This file contains primitives for multi-gpu communication.
+This is useful when doing distributed training.
+"""
+
+import functools
+import logging
+import numpy as np
+import pickle
+import torch
+import torch.distributed as dist
+
+_LOCAL_PROCESS_GROUP = None
+"""
+A torch process group which only includes processes that on the same machine as the current process.
+This variable is set when processes are spawned by `launch()` in "engine/launch.py".
+"""
+
+
+def get_world_size() -> int:
+ if not dist.is_available():
+ return 1
+ if not dist.is_initialized():
+ return 1
+ return dist.get_world_size()
+
+
+def get_rank() -> int:
+ if not dist.is_available():
+ return 0
+ if not dist.is_initialized():
+ return 0
+ return dist.get_rank()
+
+
+def get_local_rank() -> int:
+ """
+ Returns:
+ The rank of the current process within the local (per-machine) process group.
+ """
+ if not dist.is_available():
+ return 0
+ if not dist.is_initialized():
+ return 0
+ assert _LOCAL_PROCESS_GROUP is not None
+ return dist.get_rank(group=_LOCAL_PROCESS_GROUP)
+
+
+def get_local_size() -> int:
+ """
+ Returns:
+ The size of the per-machine process group,
+ i.e. the number of processes per machine.
+ """
+ if not dist.is_available():
+ return 1
+ if not dist.is_initialized():
+ return 1
+ return dist.get_world_size(group=_LOCAL_PROCESS_GROUP)
+
+
+def is_main_process() -> bool:
+ return get_rank() == 0
+
+
+def synchronize():
+ """
+ Helper function to synchronize (barrier) among all processes when
+ using distributed training
+ """
+ if not dist.is_available():
+ return
+ if not dist.is_initialized():
+ return
+ world_size = dist.get_world_size()
+ if world_size == 1:
+ return
+ dist.barrier()
+
+
+@functools.lru_cache()
+def _get_global_gloo_group():
+ """
+ Return a process group based on gloo backend, containing all the ranks
+ The result is cached.
+ """
+ if dist.get_backend() == "nccl":
+ return dist.new_group(backend="gloo")
+ else:
+ return dist.group.WORLD
+
+
+def _serialize_to_tensor(data, group):
+ backend = dist.get_backend(group)
+ assert backend in ["gloo", "nccl"]
+ device = torch.device("cpu" if backend == "gloo" else "cuda")
+
+ buffer = pickle.dumps(data)
+ if len(buffer) > 1024**3:
+ logger = logging.getLogger(__name__)
+ logger.warning(
+ "Rank {} trying to all-gather {:.2f} GB of data on device {}".format(
+ get_rank(), len(buffer) / (1024**3), device
+ )
+ )
+ storage = torch.ByteStorage.from_buffer(buffer)
+ tensor = torch.ByteTensor(storage).to(device=device)
+ return tensor
+
+
+def _pad_to_largest_tensor(tensor, group):
+ """
+ Returns:
+ list[int]: size of the tensor, on each rank
+ Tensor: padded tensor that has the max size
+ """
+ world_size = dist.get_world_size(group=group)
+ assert world_size >= 1, "comm.gather/all_gather must be called from ranks within the given group!"
+ local_size = torch.tensor([tensor.numel()], dtype=torch.int64, device=tensor.device)
+ size_list = [torch.zeros([1], dtype=torch.int64, device=tensor.device) for _ in range(world_size)]
+ dist.all_gather(size_list, local_size, group=group)
+
+ size_list = [int(size.item()) for size in size_list]
+
+ max_size = max(size_list)
+
+ # we pad the tensor because torch all_gather does not support
+ # gathering tensors of different shapes
+ if local_size != max_size:
+ padding = torch.zeros((max_size - local_size,), dtype=torch.uint8, device=tensor.device)
+ tensor = torch.cat((tensor, padding), dim=0)
+ return size_list, tensor
+
+
+def all_gather(data, group=None):
+ """
+ Run all_gather on arbitrary picklable data (not necessarily tensors).
+
+ Args:
+ data: any picklable object
+ group: a torch process group. By default, will use a group which
+ contains all ranks on gloo backend.
+
+ Returns:
+ list[data]: list of data gathered from each rank
+ """
+ if get_world_size() == 1:
+ return [data]
+ if group is None:
+ group = _get_global_gloo_group()
+ if dist.get_world_size(group) == 1:
+ return [data]
+
+ tensor = _serialize_to_tensor(data, group)
+
+ size_list, tensor = _pad_to_largest_tensor(tensor, group)
+ max_size = max(size_list)
+
+ # receiving Tensor from all ranks
+ tensor_list = [torch.empty((max_size,), dtype=torch.uint8, device=tensor.device) for _ in size_list]
+ dist.all_gather(tensor_list, tensor, group=group)
+
+ data_list = []
+ for size, tensor in zip(size_list, tensor_list):
+ buffer = tensor.cpu().numpy().tobytes()[:size]
+ data_list.append(pickle.loads(buffer))
+
+ return data_list
+
+
+def gather(data, dst=0, group=None):
+ """
+ Run gather on arbitrary picklable data (not necessarily tensors).
+
+ Args:
+ data: any picklable object
+ dst (int): destination rank
+ group: a torch process group. By default, will use a group which
+ contains all ranks on gloo backend.
+
+ Returns:
+ list[data]: on dst, a list of data gathered from each rank. Otherwise,
+ an empty list.
+ """
+ if get_world_size() == 1:
+ return [data]
+ if group is None:
+ group = _get_global_gloo_group()
+ if dist.get_world_size(group=group) == 1:
+ return [data]
+ rank = dist.get_rank(group=group)
+
+ tensor = _serialize_to_tensor(data, group)
+ size_list, tensor = _pad_to_largest_tensor(tensor, group)
+
+ # receiving Tensor from all ranks
+ if rank == dst:
+ max_size = max(size_list)
+ tensor_list = [torch.empty((max_size,), dtype=torch.uint8, device=tensor.device) for _ in size_list]
+ dist.gather(tensor, tensor_list, dst=dst, group=group)
+
+ data_list = []
+ for size, tensor in zip(size_list, tensor_list):
+ buffer = tensor.cpu().numpy().tobytes()[:size]
+ data_list.append(pickle.loads(buffer))
+ return data_list
+ else:
+ dist.gather(tensor, [], dst=dst, group=group)
+ return []
+
+
+def shared_random_seed():
+ """
+ Returns:
+ int: a random number that is the same across all workers.
+ If workers need a shared RNG, they can use this shared seed to
+ create one.
+
+ All workers must call this function, otherwise it will deadlock.
+ """
+ ints = np.random.randint(2**31)
+ all_ints = all_gather(ints)
+ return all_ints[0]
+
+
+def reduce_dict(input_dict, average=True):
+ """
+ Reduce the values in the dictionary from all processes so that process with rank
+ 0 has the reduced results.
+
+ Args:
+ input_dict (dict): inputs to be reduced. All the values must be scalar CUDA Tensor.
+ average (bool): whether to do average or sum
+
+ Returns:
+ a dict with the same keys as input_dict, after reduction.
+ """
+ world_size = get_world_size()
+ if world_size < 2:
+ return input_dict
+ with torch.no_grad():
+ names = []
+ values = []
+ # sort the keys so that they are consistent across processes
+ for k in sorted(input_dict.keys()):
+ names.append(k)
+ values.append(input_dict[k])
+ values = torch.stack(values, dim=0)
+ dist.reduce(values, dst=0)
+ if dist.get_rank() == 0 and average:
+ # only main process gets accumulated, so only divide by
+ # world_size in this case
+ values /= world_size
+ reduced_dict = {k: v for k, v in zip(names, values)}
+ return reduced_dict
diff --git a/third_party/GVHMR/hmr4d/utils/eval/eval_utils.py b/third_party/GVHMR/hmr4d/utils/eval/eval_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..6f488988cf550ab5d8f30e24d29c9ff43c0cdc76
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/eval/eval_utils.py
@@ -0,0 +1,457 @@
+import torch
+import numpy as np
+
+
+@torch.no_grad()
+def compute_camcoord_metrics(batch, pelvis_idxs=[1, 2], fps=30, mask=None):
+ """
+ Args:
+ batch (dict): {
+ "pred_j3d": (..., J, 3) tensor
+ "target_j3d":
+ "pred_verts":
+ "target_verts":
+ }
+ Returns:
+ cam_coord_metrics (dict): {
+ "pa_mpjpe": (..., ) numpy array
+ "mpjpe":
+ "pve":
+ "accel":
+ }
+ """
+ # All data is in camera coordinates
+ pred_j3d = batch["pred_j3d"].cpu() # (..., J, 3)
+ target_j3d = batch["target_j3d"].cpu()
+ pred_verts = batch["pred_verts"].cpu()
+ target_verts = batch["target_verts"].cpu()
+
+ if mask is not None:
+ mask = mask.cpu()
+ pred_j3d = pred_j3d[mask].clone()
+ target_j3d = target_j3d[mask].clone()
+ pred_verts = pred_verts[mask].clone()
+ target_verts = target_verts[mask].clone()
+ assert "mask" not in batch
+
+ # Align by pelvis
+ pred_j3d, target_j3d, pred_verts, target_verts = batch_align_by_pelvis(
+ [pred_j3d, target_j3d, pred_verts, target_verts], pelvis_idxs=pelvis_idxs
+ )
+
+ # Metrics
+ m2mm = 1000
+ S1_hat = batch_compute_similarity_transform_torch(pred_j3d, target_j3d)
+ pa_mpjpe = compute_jpe(S1_hat, target_j3d) * m2mm
+ mpjpe = compute_jpe(pred_j3d, target_j3d) * m2mm
+ pve = compute_jpe(pred_verts, target_verts) * m2mm
+ accel = compute_error_accel(joints_pred=pred_j3d, joints_gt=target_j3d, fps=fps)
+
+ camcoord_metrics = {
+ "pa_mpjpe": pa_mpjpe,
+ "mpjpe": mpjpe,
+ "pve": pve,
+ "accel": accel,
+ }
+ return camcoord_metrics
+
+
+@torch.no_grad()
+def compute_global_metrics(batch, mask=None):
+ """Follow WHAM, the input has skipped invalid frames
+ Args:
+ batch (dict): {
+ "pred_j3d_glob": (F, J, 3) tensor
+ "target_j3d_glob":
+ "pred_verts_glob":
+ "target_verts_glob":
+ }
+ Returns:
+ global_metrics (dict): {
+ "wa2_mpjpe": (F, ) numpy array
+ "waa_mpjpe":
+ "rte":
+ "jitter":
+ "fs":
+ }
+ """
+ # All data is in global coordinates
+ pred_j3d_glob = batch["pred_j3d_glob"].cpu() # (..., J, 3)
+ target_j3d_glob = batch["target_j3d_glob"].cpu()
+ pred_verts_glob = batch["pred_verts_glob"].cpu()
+ target_verts_glob = batch["target_verts_glob"].cpu()
+ if mask is not None:
+ mask = mask.cpu()
+ pred_j3d_glob = pred_j3d_glob[mask].clone()
+ target_j3d_glob = target_j3d_glob[mask].clone()
+ pred_verts_glob = pred_verts_glob[mask].clone()
+ target_verts_glob = target_verts_glob[mask].clone()
+ assert "mask" not in batch
+
+ seq_length = pred_j3d_glob.shape[0]
+
+ # Use chunk to compare
+ chunk_length = 100
+ wa2_mpjpe, waa_mpjpe = [], []
+ for start in range(0, seq_length, chunk_length):
+ end = min(seq_length, start + chunk_length)
+
+ target_j3d = target_j3d_glob[start:end].clone().cpu()
+ pred_j3d = pred_j3d_glob[start:end].clone().cpu()
+
+ w_j3d = first_align_joints(target_j3d, pred_j3d)
+ wa_j3d = global_align_joints(target_j3d, pred_j3d)
+
+ if False:
+ from hmr4d.utils.wis3d_utils import make_wis3d, add_motion_as_lines
+
+ wis3d = make_wis3d(name="debug-metric_utils")
+ add_motion_as_lines(target_j3d, wis3d, name="target_j3d")
+ add_motion_as_lines(pred_j3d, wis3d, name="pred_j3d")
+ add_motion_as_lines(w_j3d, wis3d, name="pred_w2_j3d")
+ add_motion_as_lines(wa_j3d, wis3d, name="pred_wa_j3d")
+
+ wa2_mpjpe.append(compute_jpe(target_j3d, w_j3d))
+ waa_mpjpe.append(compute_jpe(target_j3d, wa_j3d))
+
+ # Metrics
+ m2mm = 1000
+ wa2_mpjpe = np.concatenate(wa2_mpjpe) * m2mm
+ waa_mpjpe = np.concatenate(waa_mpjpe) * m2mm
+
+ # Additional Metrics
+ rte = compute_rte(target_j3d_glob[:, 0].cpu(), pred_j3d_glob[:, 0].cpu()) * 1e2
+ jitter = compute_jitter(pred_j3d_glob, fps=30)
+ foot_sliding = compute_foot_sliding(target_verts_glob, pred_verts_glob) * m2mm
+
+ global_metrics = {
+ "wa2_mpjpe": wa2_mpjpe,
+ "waa_mpjpe": waa_mpjpe,
+ "rte": rte,
+ "jitter": jitter,
+ "fs": foot_sliding,
+ }
+ return global_metrics
+
+
+@torch.no_grad()
+def compute_camcoord_perjoint_metrics(batch, pelvis_idxs=[1, 2]):
+ """
+ Args:
+ batch (dict): {
+ "pred_j3d": (..., J, 3) tensor
+ "target_j3d":
+ }
+ Returns:
+ cam_coord_metrics (dict): {
+ "pa_mpjpe": (..., ) numpy array
+ "mpjpe":
+ "pve":
+ "accel":
+ }
+ """
+ # All data is in camera coordinates
+ pred_j3d = batch["pred_j3d"].cpu() # (..., J, 3)
+ target_j3d = batch["target_j3d"].cpu()
+ pred_verts = batch["pred_verts"].cpu()
+ target_verts = batch["target_verts"].cpu()
+
+ # Align by pelvis
+ pred_j3d, target_j3d, pred_verts, target_verts = batch_align_by_pelvis(
+ [pred_j3d, target_j3d, pred_verts, target_verts], pelvis_idxs=pelvis_idxs
+ )
+ # Metrics
+ m2mm = 1000
+ perjoint_mpjpe = compute_perjoint_jpe(pred_j3d, target_j3d) * m2mm
+
+ camcoord_perjoint_metrics = {
+ "mpjpe": perjoint_mpjpe,
+ }
+ return camcoord_perjoint_metrics
+
+
+# ===== Utilities =====
+
+
+def compute_jpe(S1, S2):
+ return torch.sqrt(((S1 - S2) ** 2).sum(dim=-1)).mean(dim=-1).numpy()
+
+
+def compute_perjoint_jpe(S1, S2):
+ return torch.sqrt(((S1 - S2) ** 2).sum(dim=-1)).numpy()
+
+
+def batch_align_by_pelvis(data_list, pelvis_idxs=[1, 2]):
+ """
+ Assumes data is given as [pred_j3d, target_j3d, pred_verts, target_verts].
+ Each data is in shape of (frames, num_points, 3)
+ Pelvis is notated as one / two joints indices.
+ Align all data to the corresponding pelvis location.
+ """
+
+ pred_j3d, target_j3d, pred_verts, target_verts = data_list
+
+ pred_pelvis = pred_j3d[:, pelvis_idxs].mean(dim=1, keepdims=True).clone()
+ target_pelvis = target_j3d[:, pelvis_idxs].mean(dim=1, keepdims=True).clone()
+
+ # Align to the pelvis
+ pred_j3d = pred_j3d - pred_pelvis
+ target_j3d = target_j3d - target_pelvis
+ pred_verts = pred_verts - pred_pelvis
+ target_verts = target_verts - target_pelvis
+
+ return (pred_j3d, target_j3d, pred_verts, target_verts)
+
+
+def batch_compute_similarity_transform_torch(S1, S2):
+ """
+ Computes a similarity transform (sR, t) that takes
+ a set of 3D points S1 (3 x N) closest to a set of 3D points S2,
+ where R is an 3x3 rotation matrix, t 3x1 translation, s scale.
+ i.e. solves the orthogonal Procrutes problem.
+ """
+ transposed = False
+ if S1.shape[0] != 3 and S1.shape[0] != 2:
+ S1 = S1.permute(0, 2, 1)
+ S2 = S2.permute(0, 2, 1)
+ transposed = True
+ assert S2.shape[1] == S1.shape[1]
+
+ # 1. Remove mean.
+ mu1 = S1.mean(axis=-1, keepdims=True)
+ mu2 = S2.mean(axis=-1, keepdims=True)
+
+ X1 = S1 - mu1
+ X2 = S2 - mu2
+
+ # 2. Compute variance of X1 used for scale.
+ var1 = torch.sum(X1**2, dim=1).sum(dim=1)
+
+ # 3. The outer product of X1 and X2.
+ K = X1.bmm(X2.permute(0, 2, 1))
+
+ # 4. Solution that Maximizes trace(R'K) is R=U*V', where U, V are
+ # singular vectors of K.
+ U, s, V = torch.svd(K)
+
+ # Construct Z that fixes the orientation of R to get det(R)=1.
+ Z = torch.eye(U.shape[1], device=S1.device).unsqueeze(0)
+ Z = Z.repeat(U.shape[0], 1, 1)
+ Z[:, -1, -1] *= torch.sign(torch.det(U.bmm(V.permute(0, 2, 1))))
+
+ # Construct R.
+ R = V.bmm(Z.bmm(U.permute(0, 2, 1)))
+
+ # 5. Recover scale.
+ scale = torch.cat([torch.trace(x).unsqueeze(0) for x in R.bmm(K)]) / var1
+
+ # 6. Recover translation.
+ t = mu2 - (scale.unsqueeze(-1).unsqueeze(-1) * (R.bmm(mu1)))
+
+ # 7. Error:
+ S1_hat = scale.unsqueeze(-1).unsqueeze(-1) * R.bmm(S1) + t
+
+ if transposed:
+ S1_hat = S1_hat.permute(0, 2, 1)
+
+ return S1_hat
+
+
+def compute_error_accel(joints_gt, joints_pred, valid_mask=None, fps=None):
+ """
+ Use [i-1, i, i+1] to compute acc at frame_i. The acceleration error:
+ 1/(n-2) \sum_{i=1}^{n-1} X_{i-1} - 2X_i + X_{i+1}
+ Note that for each frame that is not visible, three entries(-1, 0, +1) in the
+ acceleration error will be zero'd out.
+ Args:
+ joints_gt : (F, J, 3)
+ joints_pred : (F, J, 3)
+ valid_mask : (F)
+ Returns:
+ error_accel (F-2) when valid_mask is None, else (F'), F' <= F-2
+ """
+ # (F, J, 3) -> (F-2) per-joint
+ accel_gt = joints_gt[:-2] - 2 * joints_gt[1:-1] + joints_gt[2:]
+ accel_pred = joints_pred[:-2] - 2 * joints_pred[1:-1] + joints_pred[2:]
+ normed = np.linalg.norm(accel_pred - accel_gt, axis=-1).mean(axis=-1)
+ if fps is not None:
+ normed = normed * fps**2
+
+ if valid_mask is None:
+ new_vis = np.ones(len(normed), dtype=bool)
+ else:
+ invis = np.logical_not(valid_mask)
+ invis1 = np.roll(invis, -1)
+ invis2 = np.roll(invis, -2)
+ new_invis = np.logical_or(invis, np.logical_or(invis1, invis2))[:-2]
+ new_vis = np.logical_not(new_invis)
+ if new_vis.sum() == 0:
+ print("Warning!!! no valid acceleration error to compute.")
+
+ return normed[new_vis]
+
+
+def compute_rte(target_trans, pred_trans):
+ # Compute the global alignment
+ _, rot, trans = align_pcl(target_trans[None, :], pred_trans[None, :], fixed_scale=True)
+ pred_trans_hat = (torch.einsum("tij,tnj->tni", rot, pred_trans[None, :]) + trans[None, :])[0]
+
+ # Compute the entire displacement of ground truth trajectory
+ disps, disp = [], 0
+ for p1, p2 in zip(target_trans, target_trans[1:]):
+ delta = (p2 - p1).norm(2, dim=-1)
+ disp += delta
+ disps.append(disp)
+
+ # Compute absolute root-translation-error (RTE)
+ rte = torch.norm(target_trans - pred_trans_hat, 2, dim=-1)
+
+ # Normalize it to the displacement
+ return (rte / disp).numpy()
+
+
+def compute_jitter(joints, fps=30):
+ """compute jitter of the motion
+ Args:
+ joints (N, J, 3).
+ fps (float).
+ Returns:
+ jitter (N-3).
+ """
+ pred_jitter = torch.norm(
+ (joints[3:] - 3 * joints[2:-1] + 3 * joints[1:-2] - joints[:-3]) * (fps**3),
+ dim=2,
+ ).mean(dim=-1)
+
+ return pred_jitter.cpu().numpy() / 10.0
+
+
+def compute_foot_sliding(target_verts, pred_verts, thr=1e-2):
+ """compute foot sliding error
+ The foot ground contact label is computed by the threshold of 1 cm/frame
+ Args:
+ target_verts (N, 6890, 3).
+ pred_verts (N, 6890, 3).
+ Returns:
+ error (N frames in contact).
+ """
+ assert target_verts.shape == pred_verts.shape
+ assert target_verts.shape[-2] == 6890
+
+ # Foot vertices idxs
+ foot_idxs = [3216, 3387, 6617, 6787]
+
+ # Compute contact label
+ foot_loc = target_verts[:, foot_idxs]
+ foot_disp = (foot_loc[1:] - foot_loc[:-1]).norm(2, dim=-1)
+ contact = foot_disp[:] < thr
+
+ pred_feet_loc = pred_verts[:, foot_idxs]
+ pred_disp = (pred_feet_loc[1:] - pred_feet_loc[:-1]).norm(2, dim=-1)
+
+ error = pred_disp[contact]
+
+ return error.cpu().numpy()
+
+
+def convert_joints22_to_24(joints22, ratio2220=0.3438, ratio2321=0.3345):
+ joints24 = torch.zeros(*joints22.shape[:-2], 24, 3).to(joints22.device)
+ joints24[..., :22, :] = joints22
+ joints24[..., 22, :] = joints22[..., 20, :] + ratio2220 * (joints22[..., 20, :] - joints22[..., 18, :])
+ joints24[..., 23, :] = joints22[..., 21, :] + ratio2321 * (joints22[..., 21, :] - joints22[..., 19, :])
+ return joints24
+
+
+def align_pcl(Y, X, weight=None, fixed_scale=False):
+ """align similarity transform to align X with Y using umeyama method
+ X' = s * R * X + t is aligned with Y
+ :param Y (*, N, 3) first trajectory
+ :param X (*, N, 3) second trajectory
+ :param weight (*, N, 1) optional weight of valid correspondences
+ :returns s (*, 1), R (*, 3, 3), t (*, 3)
+ """
+ *dims, N, _ = Y.shape
+ N = torch.ones(*dims, 1, 1) * N
+
+ if weight is not None:
+ Y = Y * weight
+ X = X * weight
+ N = weight.sum(dim=-2, keepdim=True) # (*, 1, 1)
+
+ # subtract mean
+ my = Y.sum(dim=-2) / N[..., 0] # (*, 3)
+ mx = X.sum(dim=-2) / N[..., 0]
+ y0 = Y - my[..., None, :] # (*, N, 3)
+ x0 = X - mx[..., None, :]
+
+ if weight is not None:
+ y0 = y0 * weight
+ x0 = x0 * weight
+
+ # correlation
+ C = torch.matmul(y0.transpose(-1, -2), x0) / N # (*, 3, 3)
+ U, D, Vh = torch.linalg.svd(C) # (*, 3, 3), (*, 3), (*, 3, 3)
+
+ S = torch.eye(3).reshape(*(1,) * (len(dims)), 3, 3).repeat(*dims, 1, 1)
+ neg = torch.det(U) * torch.det(Vh.transpose(-1, -2)) < 0
+ S[neg, 2, 2] = -1
+
+ R = torch.matmul(U, torch.matmul(S, Vh)) # (*, 3, 3)
+
+ D = torch.diag_embed(D) # (*, 3, 3)
+ if fixed_scale:
+ s = torch.ones(*dims, 1, device=Y.device, dtype=torch.float32)
+ else:
+ var = torch.sum(torch.square(x0), dim=(-1, -2), keepdim=True) / N # (*, 1, 1)
+ s = torch.diagonal(torch.matmul(D, S), dim1=-2, dim2=-1).sum(dim=-1, keepdim=True) / var[..., 0] # (*, 1)
+
+ t = my - s * torch.matmul(R, mx[..., None])[..., 0] # (*, 3)
+
+ return s, R, t
+
+
+def global_align_joints(gt_joints, pred_joints):
+ """
+ :param gt_joints (T, J, 3)
+ :param pred_joints (T, J, 3)
+ """
+ s_glob, R_glob, t_glob = align_pcl(gt_joints.reshape(-1, 3), pred_joints.reshape(-1, 3))
+ pred_glob = s_glob * torch.einsum("ij,tnj->tni", R_glob, pred_joints) + t_glob[None, None]
+ return pred_glob
+
+
+def first_align_joints(gt_joints, pred_joints):
+ """
+ align the first two frames
+ :param gt_joints (T, J, 3)
+ :param pred_joints (T, J, 3)
+ """
+ # (1, 1), (1, 3, 3), (1, 3)
+ s_first, R_first, t_first = align_pcl(gt_joints[:2].reshape(1, -1, 3), pred_joints[:2].reshape(1, -1, 3))
+ pred_first = s_first * torch.einsum("tij,tnj->tni", R_first, pred_joints) + t_first[:, None]
+ return pred_first
+
+
+def rearrange_by_mask(x, mask):
+ """
+ x (L, *)
+ mask (M,), M >= L
+ """
+ M = mask.size(0)
+ L = x.size(0)
+ if M == L:
+ return x
+ assert M > L
+ assert mask.sum() == L
+ x_rearranged = torch.zeros((M, *x.size()[1:]), dtype=x.dtype, device=x.device)
+ x_rearranged[mask] = x
+ return x_rearranged
+
+
+def as_np_array(d):
+ if isinstance(d, torch.Tensor):
+ return d.cpu().numpy()
+ elif isinstance(d, np.ndarray):
+ return d
+ else:
+ return np.array(d)
diff --git a/third_party/GVHMR/hmr4d/utils/geo/augment_noisy_pose.py b/third_party/GVHMR/hmr4d/utils/geo/augment_noisy_pose.py
new file mode 100644
index 0000000000000000000000000000000000000000..3c8b9e5e96f5ac9281aded966efb145ba9869f9b
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/geo/augment_noisy_pose.py
@@ -0,0 +1,207 @@
+import torch
+from pytorch3d.transforms import axis_angle_to_matrix, matrix_to_axis_angle, matrix_to_rotation_6d
+import hmr4d.utils.matrix as matrix
+from hmr4d import PROJ_ROOT
+
+COCO17_AUG = {k: v.flatten() for k, v in torch.load(PROJ_ROOT / "hmr4d/utils/body_model/coco_aug_dict.pth").items()}
+COCO17_AUG_CUDA = {}
+COCO17_TREE = [[5, 6], 0, 0, 1, 2, -1, -1, 5, 6, 7, 8, -1, -1, 11, 12, 13, 14, 15, 15, 15, 16, 16, 16]
+
+
+def gaussian_augment(body_pose, std_angle=10.0, to_R=True):
+ """
+ Args:
+ body_pose torch.Tensor: (..., J, 3) axis-angle if to_R is True, else rotmat (..., J, 3, 3)
+ std_angle: scalar or list, in degree
+ """
+
+ body_pose = body_pose.clone()
+
+ if to_R:
+ body_pose_R = axis_angle_to_matrix(body_pose) # (B, L, J, 3, 3)
+ else:
+ body_pose_R = body_pose
+ shape = body_pose_R.shape[:-2]
+ device = body_pose.device
+
+ # 1. Simulate noise
+ # angle:
+ std_angle = torch.tensor(std_angle).to(device).reshape(-1) # allow scalar or list
+ noise_angle = torch.randn(shape, device=device) * std_angle * torch.pi / 180
+
+ # axis: avoid zero vector
+ noise_axis = torch.rand((*shape, 3), device=device)
+ mask_ = torch.norm(noise_axis, dim=-1) < 1e-6
+ noise_axis[mask_] = 1
+
+ noise_axis = noise_axis / torch.norm(noise_axis, dim=-1, keepdim=True)
+ noise_aa = noise_angle[..., None] * noise_axis # (B, L, J, 3)
+ noise_R = axis_angle_to_matrix(noise_aa) # (B, L, J, 3, 3)
+
+ # 2. Add noise to body pose
+ new_body_pose_R = matrix.get_mat_BfromA(body_pose_R, noise_R) # (B, L, J, 3, 3)
+ # new_body_pose_R = torch.matmul(noise_R, body_pose_R)
+ new_body_pose_r6d = matrix_to_rotation_6d(new_body_pose_R) # (B, L, J, 6)
+ new_body_pose_aa = matrix_to_axis_angle(new_body_pose_R) # (B, L, J, 3)
+
+ return new_body_pose_R, new_body_pose_r6d, new_body_pose_aa
+
+
+# ========= Augment Joint 3D ======== #
+
+
+def get_jitter(shape=(8, 120), s_jittering=5e-2):
+ """Guassian jitter modeling."""
+ jittering_noise = (
+ torch.normal(
+ mean=torch.zeros((*shape, 17, 3)),
+ std=COCO17_AUG["jittering"].reshape(1, 1, 17, 1).expand(*shape, -1, 3),
+ )
+ * s_jittering
+ )
+ return jittering_noise
+
+
+def get_jitter_cuda(shape=(8, 120), s_jittering=5e-2):
+ if "jittering" not in COCO17_AUG_CUDA:
+ COCO17_AUG_CUDA["jittering"] = COCO17_AUG["jittering"].cuda().reshape(1, 1, 17, 1)
+ jittering = COCO17_AUG_CUDA["jittering"]
+ jittering_noise = torch.randn((*shape, 17, 3), device="cuda") * jittering * s_jittering
+ return jittering_noise
+
+
+def get_lfhp(shape=(8, 120), s_peak=3e-1, s_peak_mask=5e-3):
+ """Low-frequency high-peak noise modeling."""
+
+ def get_peak_noise_mask():
+ peak_noise_mask = torch.rand(*shape, 17) * COCO17_AUG["pmask"]
+ peak_noise_mask = peak_noise_mask < s_peak_mask
+ return peak_noise_mask
+
+ peak_noise_mask = get_peak_noise_mask() # (B, L, 17)
+ peak_noise = peak_noise_mask.float().unsqueeze(-1).repeat(1, 1, 1, 3)
+ peak_noise = peak_noise * torch.randn(3) * COCO17_AUG["peak"].reshape(17, 1) * s_peak
+ return peak_noise
+
+
+def get_lfhp_cuda(shape=(8, 120), s_peak=3e-1, s_peak_mask=5e-3):
+ if "peak" not in COCO17_AUG_CUDA:
+ COCO17_AUG_CUDA["pmask"] = COCO17_AUG["pmask"].cuda()
+ COCO17_AUG_CUDA["peak"] = COCO17_AUG["peak"].cuda().reshape(17, 1)
+
+ pmask = COCO17_AUG_CUDA["pmask"]
+ peak = COCO17_AUG_CUDA["peak"]
+ peak_noise_mask = torch.rand(*shape, 17, device="cuda") * pmask < s_peak_mask
+ peak_noise = (
+ peak_noise_mask.float().unsqueeze(-1).expand(-1, -1, -1, 3) * torch.randn(3, device="cuda") * peak * s_peak
+ )
+ return peak_noise
+
+
+def get_bias(shape=(8, 120), s_bias=1e-1):
+ """Bias noise modeling."""
+ b, l = shape
+ bias_noise = torch.normal(mean=torch.zeros((b, 17, 3)), std=COCO17_AUG["bias"].reshape(1, 17, 1)) * s_bias
+ bias_noise = bias_noise[:, None].expand(-1, l, -1, -1) # (B, L, J, 3), the whole sequence is moved by the same bias
+ return bias_noise
+
+
+def get_bias_cuda(shape=(8, 120), s_bias=1e-1):
+ if "bias" not in COCO17_AUG_CUDA:
+ COCO17_AUG_CUDA["bias"] = COCO17_AUG["bias"].cuda().reshape(1, 17, 1)
+
+ bias = COCO17_AUG_CUDA["bias"]
+ bias_noise = torch.randn((shape[0], 17, 3), device="cuda") * bias * s_bias
+ bias_noise = bias_noise[:, None].expand(-1, shape[1], -1, -1)
+ return bias_noise
+
+
+def get_wham_aug_kp3d(shape=(8, 120)):
+ # aug = get_bias(shape).cuda() + get_lfhp(shape).cuda() + get_jitter(shape).cuda()
+ aug = get_bias_cuda(shape) + get_lfhp_cuda(shape) + get_jitter_cuda(shape)
+ return aug
+
+
+def get_visible_mask(shape=(8, 120), s_mask=0.03):
+ """Mask modeling."""
+ # Per-frame and joint
+ mask = torch.rand(*shape, 17) < s_mask
+ visible = (~mask).clone() # (B, L, 17)
+
+ visible = visible.reshape(-1, 17) # (BL, 17)
+ for child in range(17):
+ parent = COCO17_TREE[child]
+ if parent == -1:
+ continue
+ if isinstance(parent, list):
+ visible[:, child] *= visible[:, parent[0]] * visible[:, parent[1]]
+ else:
+ visible[:, child] *= visible[:, parent]
+ visible = visible.reshape(*shape, 17).clone() # (B, L, J)
+ return visible
+
+
+def get_invisible_legs_mask(shape, s_mask=0.03):
+ """
+ Both legs are invisible for a random duration.
+ """
+ B, L = shape
+ starts = torch.randint(0, L - 90, (B,))
+ ends = starts + torch.randint(30, 90, (B,))
+ mask_range = torch.arange(L).unsqueeze(0).expand(B, -1)
+ mask_to_apply = (mask_range >= starts.unsqueeze(1)) & (mask_range < ends.unsqueeze(1))
+ mask_to_apply = mask_to_apply.unsqueeze(2).expand(-1, -1, 17).clone()
+ mask_to_apply[:, :, :11] = False # only both legs are invisible
+ mask_to_apply = mask_to_apply & (torch.rand(B, 1, 1) < s_mask)
+ return mask_to_apply
+
+
+def randomly_occlude_lower_half(i_x2d, s_mask=0.03):
+ """
+ Randomly occlude the lower half of the image.
+ """
+ raise NotImplementedError
+ B, L, N, _ = i_x2d.shape
+ i_x2d = i_x2d.clone()
+
+ # a period of time when the lower half of the image is invisible
+ starts = torch.randint(0, L - 90, (B,))
+ ends = starts + torch.randint(30, 90, (B,))
+ mask_range = torch.arange(L).unsqueeze(0).expand(B, -1)
+ mask_to_apply = (mask_range >= starts.unsqueeze(1)) & (mask_range < ends.unsqueeze(1))
+ mask_to_apply = mask_to_apply.unsqueeze(2).expand(-1, -1, N) # (B, L, N)
+
+ # only the lower half of the image is invisible
+ i_x2d
+ i_x2d[..., 1] / 2
+
+ mask_to_apply = mask_to_apply & (torch.rand(B, 1, 1) < s_mask)
+ return mask_to_apply
+
+
+def randomly_modify_hands_legs(j3d):
+ hands = [9, 10]
+ legs = [15, 16]
+
+ B, L, J, _ = j3d.shape
+ p_switch_hand = 0.001
+ p_switch_leg = 0.001
+ p_wrong_hand0 = 0.001
+ p_wrong_hand1 = 0.001
+ p_wrong_leg0 = 0.001
+ p_wrong_leg1 = 0.001
+
+ mask = torch.rand(B, L) < p_switch_hand
+ j3d[mask][:, hands] = j3d[mask][:, hands[::-1]]
+ mask = torch.rand(B, L) < p_switch_leg
+ j3d[mask][:, legs] = j3d[mask][:, legs[::-1]]
+ mask = torch.rand(B, L) < p_wrong_hand0
+ j3d[mask][:, 9] = j3d[mask][:, 10]
+ mask = torch.rand(B, L) < p_wrong_hand1
+ j3d[mask][:, 10] = j3d[mask][:, 9]
+ mask = torch.rand(B, L) < p_wrong_leg0
+ j3d[mask][:, 15] = j3d[mask][:, 16]
+ mask = torch.rand(B, L) < p_wrong_leg1
+ j3d[mask][:, 16] = j3d[mask][:, 15]
+
+ return j3d
diff --git a/third_party/GVHMR/hmr4d/utils/geo/flip_utils.py b/third_party/GVHMR/hmr4d/utils/geo/flip_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..81fc81585edf477323b715d11b43cb21f7ffd04c
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/geo/flip_utils.py
@@ -0,0 +1,86 @@
+import torch
+from pytorch3d.transforms import axis_angle_to_matrix, matrix_to_axis_angle
+
+
+def flip_heatmap_coco17(output_flipped):
+ assert output_flipped.ndim == 4, "output_flipped should be [B, J, H, W]"
+ shape_ori = output_flipped.shape
+ channels = 1
+ output_flipped = output_flipped.reshape(shape_ori[0], -1, channels, shape_ori[2], shape_ori[3])
+ output_flipped_back = output_flipped.clone()
+
+ # Swap left-right parts
+ for left, right in [[1, 2], [3, 4], [5, 6], [7, 8], [9, 10], [11, 12], [13, 14], [15, 16]]:
+ output_flipped_back[:, left, ...] = output_flipped[:, right, ...]
+ output_flipped_back[:, right, ...] = output_flipped[:, left, ...]
+ output_flipped_back = output_flipped_back.reshape(shape_ori)
+ # Flip horizontally
+ output_flipped_back = output_flipped_back.flip(3)
+ return output_flipped_back
+
+
+def flip_bbx_xys(bbx_xys, w):
+ """
+ bbx_xys: (F, 3)
+ """
+ bbx_xys_flip = bbx_xys.clone()
+ bbx_xys_flip[:, 0] = w - bbx_xys_flip[:, 0]
+ return bbx_xys_flip
+
+
+def flip_kp2d_coco17(kp2d, w):
+ """Flip keypoints."""
+ kp2d = kp2d.clone()
+ flipped_parts = [0, 2, 1, 4, 3, 6, 5, 8, 7, 10, 9, 12, 11, 14, 13, 16, 15]
+ kp2d = kp2d[..., flipped_parts, :]
+ kp2d[..., 0] = w - kp2d[..., 0]
+ return kp2d
+
+
+def flip_smplx_params(smplx_params):
+ """Flip pose.
+ The flipping is based on SMPLX parameters.
+ """
+ rotation = torch.cat([smplx_params["global_orient"], smplx_params["body_pose"]], dim=1)
+
+ BN = rotation.shape[0]
+ pose = rotation.reshape(BN, -1).transpose(0, 1)
+
+ SMPL_JOINTS_FLIP_PERM = [0, 2, 1, 3, 5, 4, 6, 8, 7, 9, 11, 10, 12, 14, 13, 15, 17, 16, 19, 18, 21, 20] # , 23, 22]
+ SMPL_POSE_FLIP_PERM = []
+ for i in SMPL_JOINTS_FLIP_PERM:
+ SMPL_POSE_FLIP_PERM.append(3 * i)
+ SMPL_POSE_FLIP_PERM.append(3 * i + 1)
+ SMPL_POSE_FLIP_PERM.append(3 * i + 2)
+
+ pose = pose[SMPL_POSE_FLIP_PERM]
+
+ # we also negate the second and the third dimension of the axis-angle
+ pose[1::3] = -pose[1::3]
+ pose[2::3] = -pose[2::3]
+ pose = pose.transpose(0, 1).reshape(BN, -1, 3)
+
+ smplx_params_flipped = smplx_params.copy()
+ smplx_params_flipped["global_orient"] = pose[:, :1]
+ smplx_params_flipped["body_pose"] = pose[:, 1:]
+ return smplx_params_flipped
+
+
+def avg_smplx_aa(aa1, aa2):
+ def avg_rot(rot):
+ # input [B,...,3,3] --> output [...,3,3]
+ rot = rot.mean(dim=0)
+ U, _, V = torch.svd(rot)
+ rot = U @ V.transpose(-1, -2)
+ return rot
+
+ B, J3 = aa1.shape
+ aa1 = aa1.reshape(B, -1, 3)
+ aa2 = aa2.reshape(B, -1, 3)
+
+ R1 = axis_angle_to_matrix(aa1)
+ R2 = axis_angle_to_matrix(aa2)
+ R_avg = avg_rot(torch.stack([R1, R2]))
+ aa_avg = matrix_to_axis_angle(R_avg).reshape(B, -1)
+
+ return aa_avg
diff --git a/third_party/GVHMR/hmr4d/utils/geo/hmr_cam.py b/third_party/GVHMR/hmr4d/utils/geo/hmr_cam.py
new file mode 100644
index 0000000000000000000000000000000000000000..48a7365b74d4f0c57a10accf12aa7d13bcf8ce40
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/geo/hmr_cam.py
@@ -0,0 +1,398 @@
+import torch
+import numpy as np
+from hmr4d.utils.geo_transform import project_p2d, convert_bbx_xys_to_lurb, cvt_to_bi01_p2d
+
+
+def estimate_focal_length(img_w, img_h):
+ return (img_w**2 + img_h**2) ** 0.5 # Diagonal FOV = 2*arctan(0.5) * 180/pi = 53
+
+
+def estimate_K(img_w, img_h):
+ focal_length = estimate_focal_length(img_w, img_h)
+ K = torch.eye(3).float()
+ K[0, 0] = focal_length
+ K[1, 1] = focal_length
+ K[0, 2] = img_w / 2.0
+ K[1, 2] = img_h / 2.0
+ return K
+
+
+def convert_K_to_K4(K):
+ K4 = torch.stack([K[0, 0], K[1, 1], K[0, 2], K[1, 2]]).float()
+ return K4
+
+
+def convert_f_to_K(focal_length, img_w, img_h):
+ K = torch.eye(3).float()
+ K[0, 0] = focal_length
+ K[1, 1] = focal_length
+ K[0, 2] = img_w / 2.0
+ K[1, 2] = img_h / 2.0
+ return K
+
+
+def resize_K(K, f=0.5):
+ K = K.clone() * f
+ K[..., 2, 2] = 1.0
+ return K
+
+
+def create_camera_sensor(width=None, height=None, f_fullframe=None):
+ if width is None or height is None:
+ # The 4:3 aspect ratio is widely adopted by image sensors in mobile phones.
+ if np.random.rand() < 0.5:
+ width, height = 1200, 1600
+ else:
+ width, height = 1600, 1200
+
+ # Sample FOV from common options:
+ # 1. wide-angle lenses are common in mobile phones,
+ # 2. telephoto lenses has less perspective effect, which should makes it easy to learn
+ if f_fullframe is None:
+ f_fullframe_options = [24, 26, 28, 30, 35, 40, 50, 60, 70]
+ f_fullframe = np.random.choice(f_fullframe_options)
+
+ # We use diag to map focal-length: https://www.nikonians.org/reviews/fov-tables
+ diag_fullframe = (24**2 + 36**2) ** 0.5
+ diag_img = (width**2 + height**2) ** 0.5
+ focal_length = diag_img / diag_fullframe * f_fullframe
+
+ K_fullimg = torch.eye(3)
+ K_fullimg[0, 0] = focal_length
+ K_fullimg[1, 1] = focal_length
+ K_fullimg[0, 2] = width / 2
+ K_fullimg[1, 2] = height / 2
+
+ return width, height, K_fullimg
+
+
+# ====== Compute cliffcam ===== #
+
+
+def convert_xys_to_cliff_cam_wham(xys, res):
+ """
+ Args:
+ xys: (N, 3) in pixel. Note s should not be touched by 200
+ res: (2), e.g. [4112., 3008.] (w,h)
+ Returns:
+ cliff_cam: (N, 3), normalized representation
+ """
+
+ def normalize_keypoints_to_image(x, res):
+ """
+ Args:
+ x: (N, 2), centers
+ res: (2), e.g. [4112., 3008.]
+ Returns:
+ x_normalized: (N, 2)
+ """
+ res = res.to(x.device)
+ scale = res.max(-1)[0].reshape(-1)
+ mean = torch.stack([res[..., 0] / scale, res[..., 1] / scale], dim=-1).to(x.device)
+ x = 2 * x / scale.reshape(*[1 for i in range(len(x.shape[1:]))]) - mean.reshape(
+ *[1 for i in range(len(x.shape[1:-1]))], -1
+ )
+ return x
+
+ centers = normalize_keypoints_to_image(xys[:, :2], res) # (N, 2)
+ scale = xys[:, 2:] / res.max()
+ location = torch.cat((centers, scale), dim=-1)
+ return location
+
+
+def compute_bbox_info_bedlam(bbx_xys, K_fullimg):
+ """impl as in BEDLAM
+ Args:
+ bbx_xys: ((B), N, 3), in pixel space described by K_fullimg
+ K_fullimg: ((B), (N), 3, 3)
+ Returns:
+ bbox_info: ((B), N, 3)
+ """
+ fl = K_fullimg[..., 0, 0].unsqueeze(-1)
+ icx = K_fullimg[..., 0, 2]
+ icy = K_fullimg[..., 1, 2]
+
+ cx, cy, b = bbx_xys[..., 0], bbx_xys[..., 1], bbx_xys[..., 2]
+ bbox_info = torch.stack([cx - icx, cy - icy, b], dim=-1)
+ bbox_info = bbox_info / fl
+ return bbox_info
+
+
+# ====== Convert Prediction to Cam-t ===== #
+
+
+def compute_transl_full_cam(pred_cam, bbx_xys, K_fullimg):
+ s, tx, ty = pred_cam[..., 0], pred_cam[..., 1], pred_cam[..., 2]
+ focal_length = K_fullimg[..., 0, 0]
+
+ icx = K_fullimg[..., 0, 2]
+ icy = K_fullimg[..., 1, 2]
+ sb = s * bbx_xys[..., 2]
+ cx = 2 * (bbx_xys[..., 0] - icx) / (sb + 1e-9)
+ cy = 2 * (bbx_xys[..., 1] - icy) / (sb + 1e-9)
+ tz = 2 * focal_length / (sb + 1e-9)
+
+ cam_t = torch.stack([tx + cx, ty + cy, tz], dim=-1)
+ return cam_t
+
+
+def get_a_pred_cam(transl, bbx_xys, K_fullimg):
+ """Inverse operation of compute_transl_full_cam"""
+ assert transl.ndim == bbx_xys.ndim # (*, L, 3)
+ assert K_fullimg.ndim == (bbx_xys.ndim + 1) # (*, L, 3, 3)
+ f = K_fullimg[..., 0, 0]
+ cx = K_fullimg[..., 0, 2]
+ cy = K_fullimg[..., 1, 2]
+ gt_s = 2 * f / (transl[..., 2] * bbx_xys[..., 2]) # (B, L)
+ gt_x = transl[..., 0] - transl[..., 2] / f * (bbx_xys[..., 0] - cx)
+ gt_y = transl[..., 1] - transl[..., 2] / f * (bbx_xys[..., 1] - cy)
+ gt_pred_cam = torch.stack([gt_s, gt_x, gt_y], dim=-1)
+ return gt_pred_cam
+
+
+# ====== 3D to 2D ===== #
+
+
+def project_to_bi01(points, bbx_xys, K_fullimg):
+ """
+ points: (B, L, J, 3)
+ bbx_xys: (B, L, 3)
+ K_fullimg: (B, L, 3, 3)
+ """
+ # p2d = project_p2d(points, K_fullimg)
+ p2d = perspective_projection(points, K_fullimg)
+ bbx_lurb = convert_bbx_xys_to_lurb(bbx_xys)
+ p2d_bi01 = cvt_to_bi01_p2d(p2d, bbx_lurb)
+ return p2d_bi01
+
+
+def perspective_projection(points, K):
+ # points: (B, L, J, 3)
+ # K: (B, L, 3, 3)
+ projected_points = points / points[..., -1].unsqueeze(-1)
+ projected_points = torch.einsum("...ij,...kj->...ki", K, projected_points.float())
+ return projected_points[..., :-1]
+
+
+# ====== 2D (bbx from j2d) ===== #
+
+
+def normalize_kp2d(obs_kp2d, bbx_xys, clamp_scale_min=False):
+ """
+ Args:
+ obs_kp2d: (B, L, J, 3) [x, y, c]
+ bbx_xys: (B, L, 3)
+ Returns:
+ obs: (B, L, J, 3) [x, y, c]
+ """
+ obs_xy = obs_kp2d[..., :2] # (B, L, J, 2)
+ obs_conf = obs_kp2d[..., 2] # (B, L, J)
+ center = bbx_xys[..., :2]
+ scale = bbx_xys[..., [2]]
+
+ # Mark keypoints outside the bounding box as invisible
+ xy_max = center + scale / 2
+ xy_min = center - scale / 2
+ invisible_mask = (
+ (obs_xy[..., 0] < xy_min[..., None, 0])
+ + (obs_xy[..., 0] > xy_max[..., None, 0])
+ + (obs_xy[..., 1] < xy_min[..., None, 1])
+ + (obs_xy[..., 1] > xy_max[..., None, 1])
+ )
+ obs_conf = obs_conf * ~invisible_mask
+ if clamp_scale_min:
+ scale = scale.clamp(min=1e-5)
+ normalized_obs_xy = 2 * (obs_xy - center.unsqueeze(-2)) / scale.unsqueeze(-2)
+
+ return torch.cat([normalized_obs_xy, obs_conf[..., None]], dim=-1)
+
+
+def get_bbx_xys(i_j2d, bbx_ratio=[192, 256], do_augment=False, base_enlarge=1.2):
+ """Args: (B, L, J, 3) [x,y,c] -> Returns: (B, L, 3)"""
+ # Center
+ min_x = i_j2d[..., 0].min(-1)[0]
+ max_x = i_j2d[..., 0].max(-1)[0]
+ min_y = i_j2d[..., 1].min(-1)[0]
+ max_y = i_j2d[..., 1].max(-1)[0]
+ center_x = (min_x + max_x) / 2
+ center_y = (min_y + max_y) / 2
+
+ # Size
+ h = max_y - min_y # (B, L)
+ w = max_x - min_x # (B, L)
+
+ if True: # fit w and h into aspect-ratio
+ aspect_ratio = bbx_ratio[0] / bbx_ratio[1]
+ mask1 = w > aspect_ratio * h
+ h[mask1] = w[mask1] / aspect_ratio
+ mask2 = w < aspect_ratio * h
+ w[mask2] = h[mask2] * aspect_ratio
+
+ # apply a common factor to enlarge the bounding box
+ bbx_size = torch.max(h, w) * base_enlarge
+
+ if do_augment:
+ B, L = bbx_size.shape[:2]
+ device = bbx_size.device
+ if True:
+ scaleFactor = torch.rand((B, L), device=device) * 0.3 + 1.05 # 1.05~1.35
+ txFactor = torch.rand((B, L), device=device) * 1.6 - 0.8 # -0.8~0.8
+ tyFactor = torch.rand((B, L), device=device) * 1.6 - 0.8 # -0.8~0.8
+ else:
+ scaleFactor = torch.rand((B, 1), device=device) * 0.3 + 1.05 # 1.05~1.35
+ txFactor = torch.rand((B, 1), device=device) * 1.6 - 0.8 # -0.8~0.8
+ tyFactor = torch.rand((B, 1), device=device) * 1.6 - 0.8 # -0.8~0.8
+
+ raw_bbx_size = bbx_size / base_enlarge
+ bbx_size = raw_bbx_size * scaleFactor
+ center_x += raw_bbx_size / 2 * ((scaleFactor - 1) * txFactor)
+ center_y += raw_bbx_size / 2 * ((scaleFactor - 1) * tyFactor)
+
+ return torch.stack([center_x, center_y, bbx_size], dim=-1)
+
+
+def safely_render_x3d_K(x3d, K_fullimg, thr):
+ """
+ Args:
+ x3d: (B, L, V, 3), should as least have a safe points (not examined here)
+ K_fullimg: (B, L, 3, 3)
+ Returns:
+ bbx_xys: (B, L, 3)
+ i_x2d: (B, L, V, 2)
+ """
+ # For each frame, update unsafe z ( 0:
+ x3d[..., 2][x3d_unsafe_mask] = thr
+ if False:
+ from hmr4d.utils.wis3d_utils import make_wis3d
+
+ wis3d = make_wis3d(name="debug-update-z")
+ bs, ls, vs = torch.where(x3d_unsafe_mask)
+ bs = torch.unique(bs)
+ for b in bs:
+ for f in range(x3d.size(1)):
+ wis3d.set_scene_id(f)
+ wis3d.add_point_cloud(x3d[b, f], name="unsafe")
+ pass
+
+ # renfer
+ i_x2d = perspective_projection(x3d, K_fullimg) # (B, L, V, 2)
+ return i_x2d
+
+
+def get_bbx_xys_from_xyxy(bbx_xyxy, base_enlarge=1.2):
+ """
+ Args:
+ bbx_xyxy: (N, 4) [x1, y1, x2, y2]
+ Returns:
+ bbx_xys: (N, 3) [center_x, center_y, size]
+ """
+
+ i_p2d = torch.stack([bbx_xyxy[:, [0, 1]], bbx_xyxy[:, [2, 3]]], dim=1) # (L, 2, 2)
+ bbx_xys = get_bbx_xys(i_p2d[None], base_enlarge=base_enlarge)[0]
+ return bbx_xys
+
+
+def bbx_xyxy_from_x(p2d):
+ """
+ Args:
+ p2d: (*, V, 2) - Tensor containing 2D points.
+
+ Returns:
+ bbx_xyxy: (*, 4) - Bounding box coordinates in the format (xmin, ymin, xmax, ymax).
+ """
+ # Compute the minimum and maximum coordinates for the bounding box
+ xy_min = p2d.min(dim=-2).values # (*, 2)
+ xy_max = p2d.max(dim=-2).values # (*, 2)
+
+ # Concatenate min and max coordinates to form the bounding box
+ bbx_xyxy = torch.cat([xy_min, xy_max], dim=-1) # (*, 4)
+
+ return bbx_xyxy
+
+
+def bbx_xyxy_from_masked_x(p2d, mask):
+ """
+ Args:
+ p2d: (*, V, 2) - Tensor containing 2D points.
+ mask: (*, V) - Boolean tensor indicating valid points.
+
+ Returns:
+ bbx_xyxy: (*, 4) - Bounding box coordinates in the format (xmin, ymin, xmax, ymax).
+ """
+ # Ensure the shapes of p2d and mask are compatible
+ assert p2d.shape[:-1] == mask.shape, "The shape of p2d and mask are not compatible."
+
+ # Flatten the input tensors for batch processing
+ p2d_flat = p2d.view(-1, p2d.shape[-2], p2d.shape[-1])
+ mask_flat = mask.view(-1, mask.shape[-1])
+
+ # Set masked out values to a large positive and negative value respectively
+ p2d_min = torch.where(mask_flat.unsqueeze(-1), p2d_flat, torch.tensor(float("inf")).to(p2d_flat))
+ p2d_max = torch.where(mask_flat.unsqueeze(-1), p2d_flat, torch.tensor(float("-inf")).to(p2d_flat))
+
+ # Compute the minimum and maximum coordinates for the bounding box
+ xy_min = p2d_min.min(dim=1).values # (BL, 2)
+ xy_max = p2d_max.max(dim=1).values # (BL, 2)
+
+ # Concatenate min and max coordinates to form the bounding box
+ bbx_xyxy = torch.cat([xy_min, xy_max], dim=-1) # (BL, 4)
+
+ # Reshape back to the original shape prefix
+ bbx_xyxy = bbx_xyxy.view(*p2d.shape[:-2], 4)
+
+ return bbx_xyxy
+
+
+def bbx_xyxy_ratio(xyxy1, xyxy2):
+ """Designed for fov/unbounded
+ Args:
+ xyxy1: (*, 4)
+ xyxy2: (*, 4)
+ Return:
+ ratio: (*), squared_area(xyxy1) / squared_area(xyxy2)
+ """
+ area1 = (xyxy1[..., 2] - xyxy1[..., 0]) * (xyxy1[..., 3] - xyxy1[..., 1])
+ area2 = (xyxy2[..., 2] - xyxy2[..., 0]) * (xyxy2[..., 3] - xyxy2[..., 1])
+ # Check
+ area1[~torch.isfinite(area1)] = 0 # replace inf in area1 with 0
+ assert (area2 > 0).all(), "area2 should be positive"
+ return area1 / area2
+
+
+def get_mesh_in_fov_category(mask):
+ """mask: (L, V)
+ The definition:
+ 1. FullyVisible: The mesh in every frame is entirely within the field of view (FOV).
+ 2. PartiallyVisible: In some frames, parts of the mesh are outside the FOV, while other parts are within the FOV.
+ 3. PartiallyOut: In some frames, the mesh is completely outside the FOV, while in others, it is visible.
+ 4. FullyOut: The mesh is completely outside the FOV in every frame.
+ """
+ mask = mask.clone().cpu()
+ is_class1 = mask.all() # FullyVisible
+ is_class2 = mask.any(1).all() * ~is_class1 # PartiallyVisible
+ is_class4 = ~(mask.any()) # PartiallyOut
+ is_class3 = ~is_class1 * ~is_class2 * ~is_class4 # FullyOut
+
+ mask_frame_any_verts = mask.any(1)
+ assert is_class1.int() + is_class2.int() + is_class3.int() + is_class4.int() == 1
+ class_type = is_class1.int() + 2 * is_class2.int() + 3 * is_class3.int() + 4 * is_class4.int()
+ return class_type.item(), mask_frame_any_verts
+
+
+def get_infov_mask(p2d, w_real, h_real):
+ """
+ Args:
+ p2d: (B, L, V, 2)
+ w_real, h_real: (B, L) or int
+ Returns:
+ mask: (B, L, V)
+ """
+ x, y = p2d[..., 0], p2d[..., 1]
+ if isinstance(w_real, int):
+ mask = (x >= 0) * (x < w_real) * (y >= 0) * (y < h_real)
+ else:
+ mask = (x >= 0) * (x < w_real[..., None]) * (y >= 0) * (y < h_real[..., None])
+ return mask
diff --git a/third_party/GVHMR/hmr4d/utils/geo/hmr_global.py b/third_party/GVHMR/hmr4d/utils/geo/hmr_global.py
new file mode 100644
index 0000000000000000000000000000000000000000..5483370d8dd07e56c1b7318f2ac049323aa4594f
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/geo/hmr_global.py
@@ -0,0 +1,345 @@
+import torch
+from pytorch3d.transforms import axis_angle_to_matrix, matrix_to_axis_angle, matrix_to_quaternion, quaternion_to_matrix
+import hmr4d.utils.matrix as matrix
+from hmr4d.utils.net_utils import gaussian_smooth
+
+
+def get_R_c2gv(R_w2c, axis_gravity_in_w=[0, 0, -1]):
+ """
+ Args:
+ R_w2c: (*, 3, 3)
+ Returns:
+ R_c2gv: (*, 3, 3)
+ """
+ if isinstance(axis_gravity_in_w, list):
+ axis_gravity_in_w = torch.tensor(axis_gravity_in_w).float() # gravity direction in world coord
+ axis_z_in_c = torch.tensor([0, 0, 1]).float()
+
+ # get gv-coord axes in in c-coord
+ axis_y_of_gv = R_w2c @ axis_gravity_in_w # (*, 3)
+ axis_x_of_gv = axis_y_of_gv.cross(axis_z_in_c.expand_as(axis_y_of_gv), dim=-1)
+ # normalize
+ axis_x_of_gv_norm = axis_x_of_gv.norm(dim=-1, keepdim=True)
+ axis_x_of_gv = axis_x_of_gv / (axis_x_of_gv_norm + 1e-5)
+ axis_x_of_gv[axis_x_of_gv_norm.squeeze(-1) < 1e-5] = torch.tensor([1.0, 0.0, 0.0]) # use cam x-axis as axis_x_of_gv
+ axis_z_of_gv = axis_x_of_gv.cross(axis_y_of_gv, dim=-1)
+
+ R_gv2c = torch.stack([axis_x_of_gv, axis_y_of_gv, axis_z_of_gv], dim=-1) # (*, 3, 3)
+ R_c2gv = R_gv2c.transpose(-1, -2) # (*, 3, 3)
+ return R_c2gv
+
+
+tsf_axisangle = {
+ "ay->ay": [0, 0, 0],
+ "any->ay": [0, 0, torch.pi],
+ "az->ay": [-torch.pi / 2, 0, 0],
+ "ay->any": [0, 0, torch.pi],
+}
+
+
+def get_tgtcoord_rootparam(global_orient, transl, gravity_vec=None, tgt_gravity_vec=None, tsf="ay->ay"):
+ """Rotate around the origin center, to match the new gravity direction
+ Args:
+ global_orient: torch.tensor, (*, 3)
+ transl: torch.tensor, (*, 3)
+ gravity_vec: torch.tensor, (3,)
+ tgt_gravity_vec: torch.tensor, (3,)
+ Returns:
+ tgt_global_orient: torch.tensor, (*, 3)
+ tgt_transl: torch.tensor, (*, 3)
+ R_g2tg: (3, 3)
+ """
+ # get rotation matrix
+ device = global_orient.device
+ if gravity_vec is None and tgt_gravity_vec is None:
+ aa = torch.tensor(tsf_axisangle[tsf]).to(device)
+ R_g2tg = axis_angle_to_matrix(aa) # (3, 3)
+ else:
+ raise NotImplementedError
+ # TODO: Impl this function
+ gravity_vec = torch.tensor(gravity_vec).float().to(device)
+ gravity_vec = gravity_vec / gravity_vec.norm()
+ tgt_gravity_vec = torch.tensor(tgt_gravity_vec).float().to(device)
+ tgt_gravity_vec = tgt_gravity_vec / tgt_gravity_vec.norm()
+ # pick one identity axis
+ axis_identity = torch.tensor([0, 0, 0]).float().to(device)
+ for i in (gravity_vec == 0) & (tgt_global_orient == 0):
+ if i:
+ axis_identity[i] = 1
+ break
+
+ # rotate
+ global_orient_R = axis_angle_to_matrix(global_orient) # (*, 3, 3)
+ tgt_global_orient = matrix_to_axis_angle(R_g2tg @ global_orient_R) # (*, 3, 3)
+ tgt_transl = torch.einsum("...ij,...j->...i", R_g2tg, transl)
+
+ return tgt_global_orient, tgt_transl, R_g2tg
+
+
+def get_c_rootparam(global_orient, transl, T_w2c, offset):
+ """
+ Args:
+ global_orient: torch.tensor, (F, 3)
+ transl: torch.tensor, (F, 3)
+ T_w2c: torch.tensor, (*, 4, 4)
+ offset: torch.tensor, (3,)
+ Returns:
+ R_c: torch.tensor, (F, 3)
+ t_c: torch.tensor, (F, 3)
+ """
+ assert global_orient.shape == transl.shape and len(global_orient.shape) == 2
+ R_w = axis_angle_to_matrix(global_orient) # (F, 3, 3)
+ t_w = transl # (F, 3)
+
+ R_w2c = T_w2c[..., :3, :3] # (*, 3, 3)
+ t_w2c = T_w2c[..., :3, 3] # (*, 3)
+ if len(R_w2c.shape) == 2:
+ R_w2c = R_w2c[None].expand(R_w.size(0), -1, -1) # (F, 3, 3)
+ t_w2c = t_w2c[None].expand(t_w.size(0), -1)
+
+ R_c = matrix_to_axis_angle(R_w2c @ R_w) # (F, 3)
+ t_c = torch.einsum("fij,fj->fi", R_w2c, t_w + offset) + t_w2c - offset # (F, 3)
+ return R_c, t_c
+
+
+def get_T_w2c_from_wcparams(global_orient_w, transl_w, global_orient_c, transl_c, offset):
+ """
+ Args:
+ global_orient_w: torch.tensor, (F, 3)
+ transl_w: torch.tensor, (F, 3)
+ global_orient_c: torch.tensor, (F, 3)
+ transl_c: torch.tensor, (F, 3)
+ offset: torch.tensor, (*, 3)
+ Returns:
+ T_w2c: torch.tensor, (F, 4, 4)
+ """
+ assert global_orient_w.shape == transl_w.shape and len(global_orient_w.shape) == 2
+ assert global_orient_c.shape == transl_c.shape and len(global_orient_c.shape) == 2
+
+ R_w = axis_angle_to_matrix(global_orient_w) # (F, 3, 3)
+ t_w = transl_w # (F, 3)
+ R_c = axis_angle_to_matrix(global_orient_c) # (F, 3, 3)
+ t_c = transl_c # (F, 3)
+
+ R_w2c = R_c @ R_w.transpose(-1, -2) # (F, 3, 3)
+ t_w2c = t_c + offset - torch.einsum("fij,fj->fi", R_w2c, t_w + offset) # (F, 3)
+ T_w2c = torch.eye(4, device=global_orient_w.device).repeat(R_w.size(0), 1, 1) # (F, 4, 4)
+ T_w2c[..., :3, :3] = R_w2c # (F, 3, 3)
+ T_w2c[..., :3, 3] = t_w2c # (F, 3)
+ return T_w2c
+
+
+def get_local_transl_vel(transl, global_orient):
+ """
+ transl velocity is in local coordinate (or, SMPL-coord)
+ Args:
+ transl: (*, L, 3)
+ global_orient: (*, L, 3)
+ Returns:
+ transl_vel: (*, L, 3)
+ """
+ assert len(transl.shape) == len(global_orient.shape)
+ global_orient_R = axis_angle_to_matrix(global_orient) # (B, L, 3, 3)
+ transl_vel = transl[..., 1:, :] - transl[..., :-1, :] # (B, L-1, 3)
+ transl_vel = torch.cat([transl_vel, transl_vel[..., [-1], :]], dim=-2) # (B, L, 3) last-padding
+
+ # v_local = R^T @ v_global
+ local_transl_vel = torch.einsum("...lij,...li->...lj", global_orient_R, transl_vel)
+ return local_transl_vel
+
+
+def rollout_local_transl_vel(local_transl_vel, global_orient, transl_0=None):
+ """
+ transl velocity is in local coordinate (or, SMPL-coord)
+ Args:
+ local_transl_vel: (*, L, 3)
+ global_orient: (*, L, 3)
+ transl_0: (*, 1, 3), if not provided, the start point is 0
+ Returns:
+ transl: (*, L, 3)
+ """
+ global_orient_R = axis_angle_to_matrix(global_orient)
+ transl_vel = torch.einsum("...lij,...lj->...li", global_orient_R, local_transl_vel)
+
+ # set start point
+ if transl_0 is None:
+ transl_0 = transl_vel[..., :1, :].clone().detach().zero_()
+ transl_ = torch.cat([transl_0, transl_vel[..., :-1, :]], dim=-2)
+
+ # rollout from start point
+ transl = torch.cumsum(transl_, dim=-2)
+ return transl
+
+
+def get_local_transl_vel_alignhead(transl, global_orient):
+ # assume global_orient is ay
+ global_orient_rot = axis_angle_to_matrix(global_orient) # (*, 3, 3)
+ global_orient_quat = matrix_to_quaternion(global_orient_rot) # (*, 4)
+
+ global_orient_quat_xyzw = matrix.quat_wxyz2xyzw(global_orient_quat) # (*, 4)
+ head_quat_xyzw = matrix.calc_heading_quat(global_orient_quat_xyzw, head_ind=2, gravity_axis="y") # (*, 4)
+ head_quat = matrix.quat_xyzw2wxyz(head_quat_xyzw) # (*, 4)
+ head_rot = quaternion_to_matrix(head_quat)
+ head_aa = matrix_to_axis_angle(head_rot)
+
+ local_transl_vel_alignhead = get_local_transl_vel(transl, head_aa)
+ return local_transl_vel_alignhead
+
+
+def rollout_local_transl_vel_alignhead(local_transl_vel_alignhead, global_orient, transl_0=None):
+ # assume global_orient is ay
+ global_orient_rot = axis_angle_to_matrix(global_orient) # (*, 3, 3)
+ global_orient_quat = matrix_to_quaternion(global_orient_rot) # (*, 4)
+
+ global_orient_quat_xyzw = matrix.quat_wxyz2xyzw(global_orient_quat) # (*, 4)
+ head_quat_xyzw = matrix.calc_heading_quat(global_orient_quat_xyzw, head_ind=2, gravity_axis="y") # (*, 4)
+ head_quat = matrix.quat_xyzw2wxyz(head_quat_xyzw) # (*, 4)
+ head_rot = quaternion_to_matrix(head_quat)
+ head_aa = matrix_to_axis_angle(head_rot)
+
+ transl = rollout_local_transl_vel(local_transl_vel_alignhead, head_aa, transl_0)
+ return transl
+
+
+def get_local_transl_vel_alignhead_absy(transl, global_orient):
+ # assume global_orient is ay
+ global_orient_rot = axis_angle_to_matrix(global_orient) # (*, 3, 3)
+ global_orient_quat = matrix_to_quaternion(global_orient_rot) # (*, 4)
+
+ global_orient_quat_xyzw = matrix.quat_wxyz2xyzw(global_orient_quat) # (*, 4)
+ head_quat_xyzw = matrix.calc_heading_quat(global_orient_quat_xyzw, head_ind=2, gravity_axis="y") # (*, 4)
+ head_quat = matrix.quat_xyzw2wxyz(head_quat_xyzw) # (*, 4)
+ head_rot = quaternion_to_matrix(head_quat)
+ head_aa = matrix_to_axis_angle(head_rot)
+
+ local_transl_vel_alignhead = get_local_transl_vel(transl, head_aa)
+ abs_y = torch.cumsum(local_transl_vel_alignhead[..., [1]], dim=-2) # (*, L, 1)
+ local_transl_vel_alignhead_absy = torch.cat(
+ [local_transl_vel_alignhead[..., [0]], abs_y, local_transl_vel_alignhead[..., [2]]], dim=-1
+ )
+
+ return local_transl_vel_alignhead_absy
+
+
+def rollout_local_transl_vel_alignhead_absy(local_transl_vel_alignhead_absy, global_orient, transl_0=None):
+ # assume global_orient is ay
+ global_orient_rot = axis_angle_to_matrix(global_orient) # (*, 3, 3)
+ global_orient_quat = matrix_to_quaternion(global_orient_rot) # (*, 4)
+
+ global_orient_quat_xyzw = matrix.quat_wxyz2xyzw(global_orient_quat) # (*, 4)
+ head_quat_xyzw = matrix.calc_heading_quat(global_orient_quat_xyzw, head_ind=2, gravity_axis="y") # (*, 4)
+ head_quat = matrix.quat_xyzw2wxyz(head_quat_xyzw) # (*, 4)
+ head_rot = quaternion_to_matrix(head_quat)
+ head_aa = matrix_to_axis_angle(head_rot)
+
+ local_transl_vel_alignhead_y = (
+ local_transl_vel_alignhead_absy[..., 1:, [1]] - local_transl_vel_alignhead_absy[..., :-1, [1]]
+ )
+ local_transl_vel_alignhead_y = torch.cat(
+ [local_transl_vel_alignhead_absy[..., :1, [1]], local_transl_vel_alignhead_y], dim=-2
+ )
+ local_transl_vel_alignhead = torch.cat(
+ [
+ local_transl_vel_alignhead_absy[..., [0]],
+ local_transl_vel_alignhead_y,
+ local_transl_vel_alignhead_absy[..., [2]],
+ ],
+ dim=-1,
+ )
+
+ transl = rollout_local_transl_vel(local_transl_vel_alignhead, head_aa, transl_0)
+ return transl
+
+
+def get_local_transl_vel_alignhead_absgy(transl, global_orient):
+ # assume global_orient is ay
+ global_orient_rot = axis_angle_to_matrix(global_orient) # (*, 3, 3)
+ global_orient_quat = matrix_to_quaternion(global_orient_rot) # (*, 4)
+
+ global_orient_quat_xyzw = matrix.quat_wxyz2xyzw(global_orient_quat) # (*, 4)
+ head_quat_xyzw = matrix.calc_heading_quat(global_orient_quat_xyzw, head_ind=2, gravity_axis="y") # (*, 4)
+ head_quat = matrix.quat_xyzw2wxyz(head_quat_xyzw) # (*, 4)
+ head_rot = quaternion_to_matrix(head_quat)
+ head_aa = matrix_to_axis_angle(head_rot)
+
+ local_transl_vel_alignhead = get_local_transl_vel(transl, head_aa)
+ abs_y = transl[..., [1]] # (*, L, 1)
+ local_transl_vel_alignhead_absy = torch.cat(
+ [local_transl_vel_alignhead[..., [0]], abs_y, local_transl_vel_alignhead[..., [2]]], dim=-1
+ )
+
+ return local_transl_vel_alignhead_absy
+
+
+def rollout_local_transl_vel_alignhead_absgy(local_transl_vel_alignhead_absgy, global_orient, transl_0=None):
+ # assume global_orient is ay
+ global_orient_rot = axis_angle_to_matrix(global_orient) # (*, 3, 3)
+ global_orient_quat = matrix_to_quaternion(global_orient_rot) # (*, 4)
+
+ global_orient_quat_xyzw = matrix.quat_wxyz2xyzw(global_orient_quat) # (*, 4)
+ head_quat_xyzw = matrix.calc_heading_quat(global_orient_quat_xyzw, head_ind=2, gravity_axis="y") # (*, 4)
+ head_quat = matrix.quat_xyzw2wxyz(head_quat_xyzw) # (*, 4)
+ head_rot = quaternion_to_matrix(head_quat)
+ head_aa = matrix_to_axis_angle(head_rot)
+
+ local_transl_vel_alignhead_y = (
+ local_transl_vel_alignhead_absgy[..., 1:, [1]] - local_transl_vel_alignhead_absgy[..., :-1, [1]]
+ )
+ local_transl_vel_alignhead_y = torch.cat(
+ [local_transl_vel_alignhead_y, local_transl_vel_alignhead_y[..., -1:, :]], dim=-2
+ )
+ if transl_0 is not None:
+ transl_0 = transl_0.clone()
+ transl_0[..., 1] = local_transl_vel_alignhead_absgy[..., :1, 1]
+ else:
+ transl_0 = local_transl_vel_alignhead_absgy.clone()[..., :1, :] # (*, 1, 3)
+ transl_0[..., :1, 0] = 0.0
+ transl_0[..., :1, 2] = 0.0
+
+ local_transl_vel_alignhead = torch.cat(
+ [
+ local_transl_vel_alignhead_absgy[..., [0]],
+ local_transl_vel_alignhead_y,
+ local_transl_vel_alignhead_absgy[..., [2]],
+ ],
+ dim=-1,
+ )
+
+ transl = rollout_local_transl_vel(local_transl_vel_alignhead, head_aa, transl_0)
+ return transl
+
+
+def rollout_vel(vel, transl_0=None):
+ """
+ Args:
+ vel: (*, L, 3)
+ transl_0: (*, 1, 3), if not provided, the start point is 0
+ Returns:
+ transl: (*, L, 3)
+ """
+ # set start point
+ if transl_0 is None:
+ assert len(vel.shape) == len(transl_0.shape)
+ transl_0 = vel[..., :1, :].clone().detach().zero_()
+ transl_ = torch.cat([transl_0, vel[..., :-1, :]], dim=-2)
+
+ # rollout from start point
+ transl = torch.cumsum(transl_, dim=-2)
+ return transl
+
+
+def get_static_joint_mask(w_j3d, vel_thr=0.25, smooth=False, repeat_last=False):
+ """
+ w_j3d: (*, L, J, 3)
+ vel_thr: HuMoR uses 0.15m/s
+ """
+ joint_v_ = (w_j3d[..., 1:, :, :] - w_j3d[..., :-1, :, :]).pow(2).sum(-1).sqrt() / 0.033 # (*, L-1, J)
+ if smooth:
+ joint_v_ = gaussian_smooth(joint_v_, 3, -2)
+
+ static_joint_mask = joint_v_ < vel_thr # 1 as stable, 0 as moving
+
+ if repeat_last: # repeat the last frame, this makes the shape same as w_j3d
+ static_joint_mask = torch.cat([static_joint_mask, static_joint_mask[..., [-1], :]], dim=-2)
+
+ return static_joint_mask
diff --git a/third_party/GVHMR/hmr4d/utils/geo/quaternion.py b/third_party/GVHMR/hmr4d/utils/geo/quaternion.py
new file mode 100644
index 0000000000000000000000000000000000000000..2aaff90bf878310dee45361e5872944dc7916f15
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/geo/quaternion.py
@@ -0,0 +1,440 @@
+# Copyright (c) 2018-present, Facebook, Inc.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+#
+
+import torch
+import numpy as np
+
+_EPS4 = np.finfo(float).eps * 4.0
+
+try:
+ _FLOAT_EPS = np.finfo(np.float).eps
+except:
+ _FLOAT_EPS = np.finfo(float).eps
+
+
+# PyTorch-backed implementations
+def qinv(q):
+ assert q.shape[-1] == 4, "q must be a tensor of shape (*, 4)"
+ mask = torch.ones_like(q)
+ mask[..., 1:] = -mask[..., 1:]
+ return q * mask
+
+
+def qinv_np(q):
+ assert q.shape[-1] == 4, "q must be a tensor of shape (*, 4)"
+ return qinv(torch.from_numpy(q).float()).numpy()
+
+
+def qnormalize(q):
+ assert q.shape[-1] == 4, "q must be a tensor of shape (*, 4)"
+ return q / torch.clamp(torch.norm(q, dim=-1, keepdim=True), min=1e-8)
+
+
+def qmul(q, r):
+ """
+ Multiply quaternion(s) q with quaternion(s) r.
+ Expects two equally-sized tensors of shape (*, 4), where * denotes any number of dimensions.
+ Returns q*r as a tensor of shape (*, 4).
+ """
+ assert q.shape[-1] == 4
+ assert r.shape[-1] == 4
+
+ original_shape = q.shape
+
+ # Compute outer product
+ terms = torch.bmm(r.reshape(-1, 4, 1), q.reshape(-1, 1, 4))
+
+ w = terms[:, 0, 0] - terms[:, 1, 1] - terms[:, 2, 2] - terms[:, 3, 3]
+ x = terms[:, 0, 1] + terms[:, 1, 0] - terms[:, 2, 3] + terms[:, 3, 2]
+ y = terms[:, 0, 2] + terms[:, 1, 3] + terms[:, 2, 0] - terms[:, 3, 1]
+ z = terms[:, 0, 3] - terms[:, 1, 2] + terms[:, 2, 1] + terms[:, 3, 0]
+ return torch.stack((w, x, y, z), dim=1).view(original_shape)
+
+
+def qrot(q, v):
+ """
+ Rotate vector(s) v about the rotation described by quaternion(s) q.
+ Expects a tensor of shape (*, 4) for q and a tensor of shape (*, 3) for v,
+ where * denotes any number of dimensions.
+ Returns a tensor of shape (*, 3).
+ """
+ assert q.shape[-1] == 4
+ assert v.shape[-1] == 3
+ assert q.shape[:-1] == v.shape[:-1]
+
+ original_shape = list(v.shape)
+ # print(q.shape)
+ q = q.contiguous().view(-1, 4)
+ v = v.contiguous().view(-1, 3)
+
+ qvec = q[:, 1:]
+ uv = torch.cross(qvec, v, dim=1)
+ uuv = torch.cross(qvec, uv, dim=1)
+ return (v + 2 * (q[:, :1] * uv + uuv)).view(original_shape)
+
+
+def qeuler(q, order, epsilon=0, deg=True):
+ """
+ Convert quaternion(s) q to Euler angles.
+ Expects a tensor of shape (*, 4), where * denotes any number of dimensions.
+ Returns a tensor of shape (*, 3).
+ """
+ assert q.shape[-1] == 4
+
+ original_shape = list(q.shape)
+ original_shape[-1] = 3
+ q = q.view(-1, 4)
+
+ q0 = q[:, 0]
+ q1 = q[:, 1]
+ q2 = q[:, 2]
+ q3 = q[:, 3]
+
+ if order == "xyz":
+ x = torch.atan2(2 * (q0 * q1 - q2 * q3), 1 - 2 * (q1 * q1 + q2 * q2))
+ y = torch.asin(torch.clamp(2 * (q1 * q3 + q0 * q2), -1 + epsilon, 1 - epsilon))
+ z = torch.atan2(2 * (q0 * q3 - q1 * q2), 1 - 2 * (q2 * q2 + q3 * q3))
+ elif order == "yzx":
+ x = torch.atan2(2 * (q0 * q1 - q2 * q3), 1 - 2 * (q1 * q1 + q3 * q3))
+ y = torch.atan2(2 * (q0 * q2 - q1 * q3), 1 - 2 * (q2 * q2 + q3 * q3))
+ z = torch.asin(torch.clamp(2 * (q1 * q2 + q0 * q3), -1 + epsilon, 1 - epsilon))
+ elif order == "zxy":
+ x = torch.asin(torch.clamp(2 * (q0 * q1 + q2 * q3), -1 + epsilon, 1 - epsilon))
+ y = torch.atan2(2 * (q0 * q2 - q1 * q3), 1 - 2 * (q1 * q1 + q2 * q2))
+ z = torch.atan2(2 * (q0 * q3 - q1 * q2), 1 - 2 * (q1 * q1 + q3 * q3))
+ elif order == "xzy":
+ x = torch.atan2(2 * (q0 * q1 + q2 * q3), 1 - 2 * (q1 * q1 + q3 * q3))
+ y = torch.atan2(2 * (q0 * q2 + q1 * q3), 1 - 2 * (q2 * q2 + q3 * q3))
+ z = torch.asin(torch.clamp(2 * (q0 * q3 - q1 * q2), -1 + epsilon, 1 - epsilon))
+ elif order == "yxz":
+ x = torch.asin(torch.clamp(2 * (q0 * q1 - q2 * q3), -1 + epsilon, 1 - epsilon))
+ y = torch.atan2(2 * (q1 * q3 + q0 * q2), 1 - 2 * (q1 * q1 + q2 * q2))
+ z = torch.atan2(2 * (q1 * q2 + q0 * q3), 1 - 2 * (q1 * q1 + q3 * q3))
+ elif order == "zyx":
+ x = torch.atan2(2 * (q0 * q1 + q2 * q3), 1 - 2 * (q1 * q1 + q2 * q2))
+ y = torch.asin(torch.clamp(2 * (q0 * q2 - q1 * q3), -1 + epsilon, 1 - epsilon))
+ z = torch.atan2(2 * (q0 * q3 + q1 * q2), 1 - 2 * (q2 * q2 + q3 * q3))
+ else:
+ raise
+
+ if deg:
+ return torch.stack((x, y, z), dim=1).view(original_shape) * 180 / np.pi
+ else:
+ return torch.stack((x, y, z), dim=1).view(original_shape)
+
+
+# Numpy-backed implementations
+
+
+def qmul_np(q, r):
+ q = torch.from_numpy(q).contiguous().float()
+ r = torch.from_numpy(r).contiguous().float()
+ return qmul(q, r).numpy()
+
+
+def qrot_np(q, v):
+ q = torch.from_numpy(q).contiguous().float()
+ v = torch.from_numpy(v).contiguous().float()
+ return qrot(q, v).numpy()
+
+
+def qeuler_np(q, order, epsilon=0, use_gpu=False):
+ if use_gpu:
+ q = torch.from_numpy(q).cuda().float()
+ return qeuler(q, order, epsilon).cpu().numpy()
+ else:
+ q = torch.from_numpy(q).contiguous().float()
+ return qeuler(q, order, epsilon).numpy()
+
+
+def qfix(q):
+ """
+ Enforce quaternion continuity across the time dimension by selecting
+ the representation (q or -q) with minimal distance (or, equivalently, maximal dot product)
+ between two consecutive frames.
+
+ Expects a tensor of shape (L, J, 4), where L is the sequence length and J is the number of joints.
+ Returns a tensor of the same shape.
+ """
+ assert len(q.shape) == 3
+ assert q.shape[-1] == 4
+
+ result = q.copy()
+ dot_products = np.sum(q[1:] * q[:-1], axis=2)
+ mask = dot_products < 0
+ mask = (np.cumsum(mask, axis=0) % 2).astype(bool)
+ result[1:][mask] *= -1
+ return result
+
+
+def euler2quat(e, order, deg=True):
+ """
+ Convert Euler angles to quaternions.
+ """
+ assert e.shape[-1] == 3
+
+ original_shape = list(e.shape)
+ original_shape[-1] = 4
+
+ e = e.view(-1, 3)
+
+ ## if euler angles in degrees
+ if deg:
+ e = e * np.pi / 180.0
+
+ x = e[:, 0]
+ y = e[:, 1]
+ z = e[:, 2]
+
+ rx = torch.stack((torch.cos(x / 2), torch.sin(x / 2), torch.zeros_like(x), torch.zeros_like(x)), dim=1)
+ ry = torch.stack((torch.cos(y / 2), torch.zeros_like(y), torch.sin(y / 2), torch.zeros_like(y)), dim=1)
+ rz = torch.stack((torch.cos(z / 2), torch.zeros_like(z), torch.zeros_like(z), torch.sin(z / 2)), dim=1)
+
+ result = None
+ for coord in order:
+ if coord == "x":
+ r = rx
+ elif coord == "y":
+ r = ry
+ elif coord == "z":
+ r = rz
+ else:
+ raise
+ if result is None:
+ result = r
+ else:
+ result = qmul(result, r)
+
+ # Reverse antipodal representation to have a non-negative "w"
+ if order in ["xyz", "yzx", "zxy"]:
+ result *= -1
+
+ return result.view(original_shape)
+
+
+def expmap_to_quaternion(e):
+ """
+ Convert axis-angle rotations (aka exponential maps) to quaternions.
+ Stable formula from "Practical Parameterization of Rotations Using the Exponential Map".
+ Expects a tensor of shape (*, 3), where * denotes any number of dimensions.
+ Returns a tensor of shape (*, 4).
+ """
+ assert e.shape[-1] == 3
+
+ original_shape = list(e.shape)
+ original_shape[-1] = 4
+ e = e.reshape(-1, 3)
+
+ theta = np.linalg.norm(e, axis=1).reshape(-1, 1)
+ w = np.cos(0.5 * theta).reshape(-1, 1)
+ xyz = 0.5 * np.sinc(0.5 * theta / np.pi) * e
+ return np.concatenate((w, xyz), axis=1).reshape(original_shape)
+
+
+def euler_to_quaternion(e, order):
+ """
+ Convert Euler angles to quaternions.
+ """
+ assert e.shape[-1] == 3
+
+ original_shape = list(e.shape)
+ original_shape[-1] = 4
+
+ e = e.reshape(-1, 3)
+
+ x = e[:, 0]
+ y = e[:, 1]
+ z = e[:, 2]
+
+ rx = np.stack((np.cos(x / 2), np.sin(x / 2), np.zeros_like(x), np.zeros_like(x)), axis=1)
+ ry = np.stack((np.cos(y / 2), np.zeros_like(y), np.sin(y / 2), np.zeros_like(y)), axis=1)
+ rz = np.stack((np.cos(z / 2), np.zeros_like(z), np.zeros_like(z), np.sin(z / 2)), axis=1)
+
+ result = None
+ for coord in order:
+ if coord == "x":
+ r = rx
+ elif coord == "y":
+ r = ry
+ elif coord == "z":
+ r = rz
+ else:
+ raise
+ if result is None:
+ result = r
+ else:
+ result = qmul_np(result, r)
+
+ # Reverse antipodal representation to have a non-negative "w"
+ if order in ["xyz", "yzx", "zxy"]:
+ result *= -1
+
+ return result.reshape(original_shape)
+
+
+def quaternion_to_matrix(quaternions):
+ """
+ Convert rotations given as quaternions to rotation matrices.
+ Args:
+ quaternions: quaternions with real part first,
+ as tensor of shape (..., 4).
+ Returns:
+ Rotation matrices as tensor of shape (..., 3, 3).
+ """
+ r, i, j, k = torch.unbind(quaternions, -1)
+ two_s = 2.0 / (quaternions * quaternions).sum(-1)
+
+ o = torch.stack(
+ (
+ 1 - two_s * (j * j + k * k),
+ two_s * (i * j - k * r),
+ two_s * (i * k + j * r),
+ two_s * (i * j + k * r),
+ 1 - two_s * (i * i + k * k),
+ two_s * (j * k - i * r),
+ two_s * (i * k - j * r),
+ two_s * (j * k + i * r),
+ 1 - two_s * (i * i + j * j),
+ ),
+ -1,
+ )
+ return o.reshape(quaternions.shape[:-1] + (3, 3))
+
+
+def quaternion_to_matrix_np(quaternions):
+ q = torch.from_numpy(quaternions).contiguous().float()
+ return quaternion_to_matrix(q).numpy()
+
+
+def quaternion_to_cont6d_np(quaternions):
+ rotation_mat = quaternion_to_matrix_np(quaternions)
+ cont_6d = np.concatenate([rotation_mat[..., 0], rotation_mat[..., 1]], axis=-1)
+ return cont_6d
+
+
+def quaternion_to_cont6d(quaternions):
+ rotation_mat = quaternion_to_matrix(quaternions)
+ cont_6d = torch.cat([rotation_mat[..., 0], rotation_mat[..., 1]], dim=-1)
+ return cont_6d
+
+
+def cont6d_to_matrix(cont6d):
+ assert cont6d.shape[-1] == 6, "The last dimension must be 6"
+ x_raw = cont6d[..., 0:3]
+ y_raw = cont6d[..., 3:6]
+
+ x = x_raw / torch.norm(x_raw, dim=-1, keepdim=True)
+ z = torch.cross(x, y_raw, dim=-1)
+ z = z / torch.norm(z, dim=-1, keepdim=True)
+
+ y = torch.cross(z, x, dim=-1)
+
+ x = x[..., None]
+ y = y[..., None]
+ z = z[..., None]
+
+ mat = torch.cat([x, y, z], dim=-1)
+ return mat
+
+
+def cont6d_to_matrix_np(cont6d):
+ q = torch.from_numpy(cont6d).contiguous().float()
+ return cont6d_to_matrix(q).numpy()
+
+
+def qpow(q0, t, dtype=torch.float):
+ """q0 : tensor of quaternions
+ t: tensor of powers
+ """
+ q0 = qnormalize(q0)
+ theta0 = torch.acos(q0[..., :1])
+
+ ## if theta0 is close to zero, add epsilon to avoid NaNs
+ mask = (theta0 <= 10e-10) * (theta0 >= -10e-10)
+ mask = mask.float()
+ theta0 = (1 - mask) * theta0 + mask * 10e-10
+ v0 = q0[..., 1:] / torch.sin(theta0)
+
+ if isinstance(t, torch.Tensor):
+ # Do not check here
+ q = torch.zeros(t.shape + q0.shape, device=q0.device)
+ theta = t.view(-1, 1) * theta0.view(1, -1)
+ else: ## if t is a number
+ q = torch.zeros(q0.shape, device=q0.device)
+ theta = t * theta0
+
+ q[..., :1] = torch.cos(theta)
+ q[..., 1:] = v0 * torch.sin(theta)
+
+ return q.to(dtype)
+
+
+def qslerp(q0, q1, t):
+ """
+ q0: starting quaternion
+ q1: ending quaternion
+ t: array of points along the way
+
+ Returns:
+ Tensor of Slerps: t.shape + q0.shape
+ """
+
+ q0 = qnormalize(q0)
+ q1 = qnormalize(q1)
+ q_ = qpow(qmul(q1, qinv(q0)), t)
+
+ return qmul(q_, q0)
+
+
+def qbetween(v0, v1):
+ """
+ find the quaternion used to rotate v0 to v1
+ """
+ assert v0.shape[-1] == 3, "v0 must be of the shape (*, 3)"
+ assert v1.shape[-1] == 3, "v1 must be of the shape (*, 3)"
+
+ v = torch.cross(v0, v1, dim=-1)
+
+ w = torch.sqrt((v0**2).sum(dim=-1, keepdim=True) * (v1**2).sum(dim=-1, keepdim=True)) + (v0 * v1).sum(
+ dim=-1, keepdim=True
+ )
+ y_vec = torch.zeros_like(v)
+ y_vec[..., 1] = 1.0
+ # if v0 is (0, 0, -1), v1 is (0, 0, 1), v will be 0 and w will also be 0 -> this makes below situation comes v=1 w = 2
+ mask = v.norm(dim=-1) == 0
+ # if v0 is (0, 0, 1), v1 is (0, 0, 1), v will be 0 and w will be 2 -> do nothing
+ mask2 = w.sum(dim=-1).abs() <= 1e-4
+ mask = torch.logical_and(mask, mask2)
+ v[mask] = y_vec[mask]
+
+ return qnormalize(torch.cat([w, v], dim=-1))
+
+
+def qbetween_np(v0, v1):
+ """
+ find the quaternion used to rotate v0 to v1
+ """
+ assert v0.shape[-1] == 3, "v0 must be of the shape (*, 3)"
+ assert v1.shape[-1] == 3, "v1 must be of the shape (*, 3)"
+
+ v0 = torch.from_numpy(v0).float()
+ v1 = torch.from_numpy(v1).float()
+ return qbetween(v0, v1).numpy()
+
+
+def lerp(p0, p1, t):
+ if not isinstance(t, torch.Tensor):
+ t = torch.Tensor([t])
+
+ new_shape = t.shape + p0.shape
+ new_view_t = t.shape + torch.Size([1] * len(p0.shape))
+ new_view_p = torch.Size([1] * len(t.shape)) + p0.shape
+ p0 = p0.view(new_view_p).expand(new_shape)
+ p1 = p1.view(new_view_p).expand(new_shape)
+ t = t.view(new_view_t).expand(new_shape)
+
+ return p0 + t * (p1 - p0)
diff --git a/third_party/GVHMR/hmr4d/utils/geo/transforms.py b/third_party/GVHMR/hmr4d/utils/geo/transforms.py
new file mode 100644
index 0000000000000000000000000000000000000000..3f6ad21068feaa929256542776d12dc5232239f5
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/geo/transforms.py
@@ -0,0 +1,25 @@
+import torch
+
+
+def axis_rotate_to_matrix(angle, axis="x"):
+ """Get rotation matrix for rotating around one axis
+ Args:
+ angle: (N, 1)
+ Returns:
+ R: (N, 3, 3)
+ """
+ if isinstance(angle, float):
+ angle = torch.tensor([angle], dtype=torch.float)
+
+ c = torch.cos(angle)
+ s = torch.sin(angle)
+ z = torch.zeros_like(angle)
+ o = torch.ones_like(angle)
+ if axis == "x":
+ R = torch.stack([o, z, z, z, c, -s, z, s, c], dim=1).view(-1, 3, 3)
+ elif axis == "y":
+ R = torch.stack([c, z, s, z, o, z, -s, z, c], dim=1).view(-1, 3, 3)
+ else:
+ assert axis == "z"
+ R = torch.stack([c, -s, z, s, c, z, z, z, o], dim=1).view(-1, 3, 3)
+ return R
diff --git a/third_party/GVHMR/hmr4d/utils/geo_transform.py b/third_party/GVHMR/hmr4d/utils/geo_transform.py
new file mode 100644
index 0000000000000000000000000000000000000000..028cc3568af5e89c2ad7f0badf3cae42027ced41
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/geo_transform.py
@@ -0,0 +1,673 @@
+import numpy as np
+import cv2
+import torch
+import torch.nn.functional as F
+from pytorch3d.transforms import so3_exp_map, so3_log_map
+from pytorch3d.transforms import matrix_to_quaternion, quaternion_to_axis_angle, matrix_to_rotation_6d
+import pytorch3d.ops.knn as knn
+from hmr4d.utils.pylogger import Log
+from pytorch3d.transforms import euler_angles_to_matrix
+import hmr4d.utils.matrix as matrix
+from einops import einsum, rearrange, repeat
+from hmr4d.utils.geo.quaternion import qbetween
+
+
+def homo_points(points):
+ """
+ Args:
+ points: (..., C)
+ Returns: (..., C+1), with 1 padded
+ """
+ return F.pad(points, [0, 1], value=1.0)
+
+
+def apply_Ts_on_seq_points(points, Ts):
+ """
+ perform translation matrix on related point
+ Args:
+ points: (..., N, 3)
+ Ts: (..., N, 4, 4)
+ Returns: (..., N, 3)
+ """
+ points = torch.torch.einsum("...ki,...i->...k", Ts[..., :3, :3], points) + Ts[..., :3, 3]
+ return points
+
+
+def apply_T_on_points(points, T):
+ """
+ Args:
+ points: (..., N, 3)
+ T: (..., 4, 4)
+ Returns: (..., N, 3)
+ """
+ points_T = torch.einsum("...ki,...ji->...jk", T[..., :3, :3], points) + T[..., None, :3, 3]
+ return points_T
+
+
+def T_transforms_points(T, points, pattern):
+ """manual mode of apply_T_on_points
+ T: (..., 4, 4)
+ points: (..., 3)
+ pattern: "... c d, ... d -> ... c"
+ """
+ return einsum(T, homo_points(points), pattern)[..., :3]
+
+
+def project_p2d(points, K=None, is_pinhole=True):
+ """
+ Args:
+ points: (..., (N), 3)
+ K: (..., 3, 3)
+ Returns: shape is similar to points but without z
+ """
+ points = points.clone()
+ if is_pinhole:
+ z = points[..., [-1]]
+ z.masked_fill_(z.abs() < 1e-6, 1e-6)
+ points_proj = points / z
+ else: # orthogonal
+ points_proj = F.pad(points[..., :2], (0, 1), value=1)
+
+ if K is not None:
+ # Handle N
+ if len(points_proj.shape) == len(K.shape):
+ p2d_h = torch.einsum("...ki,...ji->...jk", K, points_proj)
+ else:
+ p2d_h = torch.einsum("...ki,...i->...k", K, points_proj)
+ else:
+ p2d_h = points_proj[..., :2]
+
+ return p2d_h[..., :2]
+
+
+def gen_uv_from_HW(H, W, device="cpu"):
+ """Returns: (H, W, 2), as float. Note: uv not ij"""
+ grid_v, grid_u = torch.meshgrid(torch.arange(H), torch.arange(W))
+ return (
+ torch.stack(
+ [grid_u, grid_v],
+ dim=-1,
+ )
+ .float()
+ .to(device)
+ ) # (H, W, 2)
+
+
+def unproject_p2d(uv, z, K):
+ """we assume a pinhole camera for unprojection
+ uv: (B, N, 2)
+ z: (B, N, 1)
+ K: (B, 3, 3)
+ Returns: (B, N, 3)
+ """
+ xy_atz1 = (uv - K[:, None, :2, 2]) / K[:, None, [0, 1], [0, 1]] # (B, N, 2)
+ xyz = torch.cat([xy_atz1 * z, z], dim=-1)
+ return xyz
+
+
+def cvt_p2d_from_i_to_c(uv, K):
+ """
+ Args:
+ uv: (..., 2) or (..., N, 2)
+ K: (..., 3, 3)
+ Returns: the same shape as input uv
+ """
+ if len(uv.shape) == len(K.shape):
+ xy = (uv - K[..., None, :2, 2]) / K[..., None, [0, 1], [0, 1]]
+ else: # without N
+ xy = (uv - K[..., :2, 2]) / K[..., [0, 1], [0, 1]]
+ return xy
+
+
+def cvt_to_bi01_p2d(p2d, bbx_lurb):
+ """
+ p2d: (..., (N), 2)
+ bbx_lurb: (..., 4)
+ """
+ if len(p2d.shape) == len(bbx_lurb.shape) + 1:
+ bbx_lurb = bbx_lurb[..., None, :]
+
+ bbx_wh = bbx_lurb[..., 2:] - bbx_lurb[..., :2]
+ bi01_p2d = (p2d - bbx_lurb[..., :2]) / bbx_wh
+ return bi01_p2d
+
+
+def cvt_from_bi01_p2d(bi01_p2d, bbx_lurb):
+ """Use bbx_lurb to resize bi01_p2d to p2d (image-coordinates)
+ Args:
+ p2d: (..., 2) or (..., N, 2)
+ bbx_lurb: (..., 4)
+ Returns:
+ p2d: shape is the same as input
+ """
+ bbx_wh = bbx_lurb[..., 2:] - bbx_lurb[..., :2] # (..., 2)
+ if len(bi01_p2d.shape) == len(bbx_wh.shape) + 1:
+ p2d = (bi01_p2d * bbx_wh.unsqueeze(-2)) + bbx_lurb[..., None, :2]
+ else:
+ p2d = (bi01_p2d * bbx_wh) + bbx_lurb[..., :2]
+ return p2d
+
+
+def cvt_p2d_from_bi01_to_c(bi01, bbxs_lurb, Ks):
+ """
+ Args:
+ bi01: (..., (N), 2), value in range (0,1), the point in the bbx image
+ bbxs_lurb: (..., 4)
+ Ks: (..., 3, 3)
+ Returns:
+ c: (..., (N), 2)
+ """
+ i = cvt_from_bi01_p2d(bi01, bbxs_lurb)
+ c = cvt_p2d_from_i_to_c(i, Ks)
+ return c
+
+
+def cvt_p2d_from_pm1_to_i(p2d_pm1, bbx_xys):
+ """
+ Args:
+ p2d_pm1: (..., (N), 2), value in range (-1,1), the point in the bbx image
+ bbx_xys: (..., 3)
+ Returns:
+ p2d: (..., (N), 2)
+ """
+ return bbx_xys[..., :2] + p2d_pm1 * bbx_xys[..., [2]] / 2
+
+
+def uv2l_index(uv, W):
+ return uv[..., 0] + uv[..., 1] * W
+
+
+def l2uv_index(l, W):
+ v = torch.div(l, W, rounding_mode="floor")
+ u = l % W
+ return torch.stack([u, v], dim=-1)
+
+
+def transform_mat(R, t):
+ """
+ Args:
+ R: Bx3x3 array of a batch of rotation matrices
+ t: Bx3x(1) array of a batch of translation vectors
+ Returns:
+ T: Bx4x4 Transformation matrix
+ """
+ # No padding left or right, only add an extra row
+ if len(R.shape) > len(t.shape):
+ t = t[..., None]
+ return torch.cat([F.pad(R, [0, 0, 0, 1]), F.pad(t, [0, 0, 0, 1], value=1)], dim=-1)
+
+
+def axis_angle_to_matrix_exp_map(aa):
+ """use pytorch3d so3_exp_map
+ Args:
+ aa: (*, 3)
+ Returns:
+ R: (*, 3, 3)
+ """
+ print("Use pytorch3d.transforms.axis_angle_to_matrix instead!!!")
+ ori_shape = aa.shape[:-1]
+ return so3_exp_map(aa.reshape(-1, 3)).reshape(*ori_shape, 3, 3)
+
+
+def matrix_to_axis_angle_log_map(R):
+ """use pytorch3d so3_log_map
+ Args:
+ aa: (*, 3, 3)
+ Returns:
+ R: (*, 3)
+ """
+ print("WARINING! I met singularity problem with this function, use matrix_to_axis_angle instead!")
+ ori_shape = R.shape[:-2]
+ return so3_log_map(R.reshape(-1, 3, 3)).reshape(*ori_shape, 3)
+
+
+def matrix_to_axis_angle(R):
+ """use pytorch3d so3_log_map
+ Args:
+ aa: (*, 3, 3)
+ Returns:
+ R: (*, 3)
+ """
+ return quaternion_to_axis_angle(matrix_to_quaternion(R))
+
+
+def ransac_PnP(K, pts_2d, pts_3d, err_thr=10):
+ """solve pnp"""
+ dist_coeffs = np.zeros(shape=[8, 1], dtype="float64")
+
+ pts_2d = np.ascontiguousarray(pts_2d.astype(np.float64))
+ pts_3d = np.ascontiguousarray(pts_3d.astype(np.float64))
+ K = K.astype(np.float64)
+
+ try:
+ _, rvec, tvec, inliers = cv2.solvePnPRansac(
+ pts_3d, pts_2d, K, dist_coeffs, reprojectionError=err_thr, iterationsCount=10000, flags=cv2.SOLVEPNP_EPNP
+ )
+
+ rotation = cv2.Rodrigues(rvec)[0]
+
+ pose = np.concatenate([rotation, tvec], axis=-1)
+ pose_homo = np.concatenate([pose, np.array([[0, 0, 0, 1]])], axis=0)
+
+ inliers = [] if inliers is None else inliers
+
+ return pose, pose_homo, inliers
+ except cv2.error:
+ print("CV ERROR")
+ return np.eye(4)[:3], np.eye(4), []
+
+
+def ransac_PnP_batch(K_raw, pts_2d, pts_3d, err_thr=10):
+ fit_R, fit_t = [], []
+ for b in range(K_raw.shape[0]):
+ pose, _, inliers = ransac_PnP(K_raw[b], pts_2d[b], pts_3d[b], err_thr=err_thr)
+ fit_R.append(pose[:3, :3])
+ fit_t.append(pose[:3, 3])
+ fit_R = np.stack(fit_R, axis=0)
+ fit_t = np.stack(fit_t, axis=0)
+ return fit_R, fit_t
+
+
+def triangulate_point(Ts_w2c, c_p2d, **kwargs):
+ from hmr4d.utils.geo.triangulation import triangulate_persp
+
+ print("Deprecated, please import from hmr4d.utils.geo.triangulation")
+ return triangulate_persp(Ts_w2c, c_p2d, **kwargs)
+
+
+def triangulate_point_ortho(Ts_w2c, c_p2d, **kwargs):
+ from hmr4d.utils.geo.triangulation import triangulate_ortho
+
+ print("Deprecated, please import from hmr4d.utils.geo.triangulation")
+ return triangulate_ortho(Ts_w2c, c_p2d, **kwargs)
+
+
+def get_nearby_points(points, query_verts, padding=0.0, p=1):
+ """
+ points: (S, 3)
+ query_verts: (V, 3)
+ """
+ if p == 1:
+ max_xyz = query_verts.max(0)[0] + padding
+ min_xyz = query_verts.min(0)[0] - padding
+ idx = (((points - min_xyz) > 0).all(dim=-1) * ((points - max_xyz) < 0).all(dim=-1)).nonzero().squeeze(-1)
+ nearby_points = points[idx]
+ elif p == 2:
+ squared_dist, _, _ = knn.knn_points(points[None], query_verts[None], K=1, return_nn=False)
+ mask = squared_dist[0, :, 0] < padding**2 # (S,)
+ nearby_points = points[mask]
+
+ return nearby_points
+
+
+def unproj_bbx_to_fst(bbx_lurb, K, near_z=0.5, far_z=12.5):
+ B = bbx_lurb.size(0)
+ uv = bbx_lurb[:, [[0, 1], [2, 1], [2, 3], [0, 3], [0, 1], [2, 1], [2, 3], [0, 3]]]
+ if isinstance(near_z, float):
+ z = uv.new([near_z] * 4 + [far_z] * 4).reshape(1, 8, 1).repeat(B, 1, 1)
+ else:
+ z = torch.cat([near_z[:, None, None].repeat(1, 4, 1), far_z[:, None, None].repeat(1, 4, 1)], dim=1)
+ c_frustum_points = unproject_p2d(uv, z, K) # (B, 8, 3)
+ return c_frustum_points
+
+
+def convert_bbx_xys_to_lurb(bbx_xys):
+ """
+ Args: bbx_xys (..., 3) -> bbx_lurb (..., 4)
+ """
+ size = bbx_xys[..., 2:]
+ center = bbx_xys[..., :2]
+ lurb = torch.cat([center - size / 2, center + size / 2], dim=-1)
+ return lurb
+
+
+def convert_lurb_to_bbx_xys(bbx_lurb):
+ """
+ Args: bbx_lurb (..., 4) -> bbx_xys (..., 3) be aware that it is squared
+ """
+ size = (bbx_lurb[..., 2:] - bbx_lurb[..., :2]).max(-1, keepdim=True)[0]
+ center = (bbx_lurb[..., :2] + bbx_lurb[..., 2:]) / 2
+ return torch.cat([center, size], dim=-1)
+
+
+# ================== AZ/AY Transformations ================== #
+
+
+def compute_T_ayf2az(joints, inverse=False):
+ """
+ Args:
+ joints: (B, J, 3), in the start-frame, az-coordinate
+ Returns:
+ if inverse == False:
+ T_af2az: (B, 4, 4)
+ else :
+ T_az2af: (B, 4, 4)
+ """
+
+ t_ayf2az = joints[:, 0, :].detach().clone()
+ t_ayf2az[:, 2] = 0 # do not modify z
+
+ RL_xy_h = joints[:, 1, [0, 1]] - joints[:, 2, [0, 1]] # (B, 2), hip point to left side
+ RL_xy_s = joints[:, 16, [0, 1]] - joints[:, 17, [0, 1]] # (B, 2), shoulder point to left side
+ RL_xy = RL_xy_h + RL_xy_s
+ I_mask = RL_xy.pow(2).sum(-1) < 1e-4 # do not rotate, when can't decided the face direction
+ if I_mask.sum() > 0:
+ Log.warn("{} samples can't decide the face direction".format(I_mask.sum()))
+ x_dir = F.pad(F.normalize(RL_xy, 2, -1), (0, 1), value=0) # (B, 3)
+ y_dir = torch.zeros_like(x_dir)
+ y_dir[..., 2] = 1
+ z_dir = torch.cross(x_dir, y_dir, dim=-1)
+ R_ayf2az = torch.stack([x_dir, y_dir, z_dir], dim=-1) # (B, 3, 3)
+ R_ayf2az[I_mask] = torch.eye(3).to(R_ayf2az)
+
+ if inverse:
+ R_az2ayf = R_ayf2az.transpose(1, 2) # (B, 3, 3)
+ t_az2ayf = -einsum(R_ayf2az, t_ayf2az, "b i j , b i -> b j") # (B, 3)
+ return transform_mat(R_az2ayf, t_az2ayf)
+ else:
+ return transform_mat(R_ayf2az, t_ayf2az)
+
+
+def compute_T_ayfz2ay(joints, inverse=False):
+ """
+ Args:
+ joints: (B, J, 3), in the start-frame, ay-coordinate
+ Returns:
+ if inverse == False:
+ T_ayfz2ay: (B, 4, 4)
+ else :
+ T_ay2ayfz: (B, 4, 4)
+ """
+ t_ayfz2ay = joints[:, 0, :].detach().clone()
+ t_ayfz2ay[:, 1] = 0 # do not modify y
+
+ RL_xz_h = joints[:, 1, [0, 2]] - joints[:, 2, [0, 2]] # (B, 2), hip point to left side
+ RL_xz_s = joints[:, 16, [0, 2]] - joints[:, 17, [0, 2]] # (B, 2), shoulder point to left side
+ RL_xz = RL_xz_h + RL_xz_s
+ I_mask = RL_xz.pow(2).sum(-1) < 1e-4 # do not rotate, when can't decided the face direction
+ if I_mask.sum() > 0:
+ Log.warn("{} samples can't decide the face direction".format(I_mask.sum()))
+
+ x_dir = torch.zeros_like(t_ayfz2ay) # (B, 3)
+ x_dir[:, [0, 2]] = F.normalize(RL_xz, 2, -1)
+ y_dir = torch.zeros_like(x_dir)
+ y_dir[..., 1] = 1 # (B, 3)
+ z_dir = torch.cross(x_dir, y_dir, dim=-1)
+ R_ayfz2ay = torch.stack([x_dir, y_dir, z_dir], dim=-1) # (B, 3, 3)
+ R_ayfz2ay[I_mask] = torch.eye(3).to(R_ayfz2ay)
+
+ if inverse:
+ R_ay2ayfz = R_ayfz2ay.transpose(1, 2)
+ t_ay2ayfz = -einsum(R_ayfz2ay, t_ayfz2ay, "b i j , b i -> b j")
+ return transform_mat(R_ay2ayfz, t_ay2ayfz)
+ else:
+ return transform_mat(R_ayfz2ay, t_ayfz2ay)
+
+
+def compute_T_ay2ayrot(joints):
+ """
+ Args:
+ joints: (B, J, 3), in the start-frame, ay-coordinate
+ Returns:
+ T_ay2ayrot: (B, 4, 4)
+ """
+ t_ayrot2ay = joints[:, 0, :].detach().clone()
+ t_ayrot2ay[:, 1] = 0 # do not modify y
+
+ B = joints.shape[0]
+ euler_angle = torch.zeros((B, 3), device=joints.device)
+ yrot_angle = torch.rand((B,), device=joints.device) * 2 * torch.pi
+ euler_angle[:, 0] = yrot_angle
+ R_ay2ayrot = euler_angles_to_matrix(euler_angle, "YXZ") # (B, 3, 3)
+
+ R_ayrot2ay = R_ay2ayrot.transpose(1, 2)
+ t_ay2ayrot = -einsum(R_ayrot2ay, t_ayrot2ay, "b i j , b i -> b j")
+ return transform_mat(R_ay2ayrot, t_ay2ayrot)
+
+
+def compute_root_quaternion_ay(joints):
+ """
+ Args:
+ joints: (B, J, 3), in the start-frame, ay-coordinate
+ Returns:
+ root_quat: (B, 4) from z-axis to fz
+ """
+ joints_shape = joints.shape
+ joints = joints.reshape((-1,) + joints_shape[-2:])
+ t_ayfz2ay = joints[:, 0, :].detach().clone()
+ t_ayfz2ay[:, 1] = 0 # do not modify y
+
+ RL_xz_h = joints[:, 1, [0, 2]] - joints[:, 2, [0, 2]] # (B, 2), hip point to left side
+ RL_xz_s = joints[:, 16, [0, 2]] - joints[:, 17, [0, 2]] # (B, 2), shoulder point to left side
+ RL_xz = RL_xz_h + RL_xz_s
+ I_mask = RL_xz.pow(2).sum(-1) < 1e-4 # do not rotate, when can't decided the face direction
+ if I_mask.sum() > 0:
+ Log.warn("{} samples can't decide the face direction".format(I_mask.sum()))
+
+ x_dir = torch.zeros_like(t_ayfz2ay) # (B, 3)
+ x_dir[:, [0, 2]] = F.normalize(RL_xz, 2, -1)
+ y_dir = torch.zeros_like(x_dir)
+ y_dir[..., 1] = 1 # (B, 3)
+ z_dir = torch.cross(x_dir, y_dir, dim=-1)
+
+ z_dir[..., 2] += 1e-9
+ pos_z_vec = torch.tensor([0, 0, 1]).to(joints.device).float() # (3,)
+ root_quat = qbetween(pos_z_vec[None], z_dir) # (B, 4)
+ root_quat = root_quat.reshape(joints_shape[:-2] + (4,))
+ return root_quat
+
+
+# ================== Transformations between two sets of features ================== #
+
+
+def similarity_transform_batch(S1, S2):
+ """
+ Computes a similarity transform (sR, t) that solves the orthogonal Procrutes problem.
+ Args:
+ S1, S2: (*, L, 3)
+ """
+ assert S1.shape == S2.shape
+ S_shape = S1.shape
+ S1 = S1.reshape(-1, *S_shape[-2:])
+ S2 = S2.reshape(-1, *S_shape[-2:])
+
+ S1 = S1.transpose(-2, -1)
+ S2 = S2.transpose(-2, -1)
+
+ # --- The code is borrowed from WHAM ---
+ # 1. Remove mean.
+ mu1 = S1.mean(axis=-1, keepdims=True) # axis is along N, S1(B, 3, N)
+ mu2 = S2.mean(axis=-1, keepdims=True)
+
+ X1 = S1 - mu1
+ X2 = S2 - mu2
+
+ # 2. Compute variance of X1 used for scale.
+ var1 = torch.sum(X1**2, dim=1).sum(dim=1)
+
+ # 3. The outer product of X1 and X2.
+ K = X1.bmm(X2.permute(0, 2, 1))
+
+ # 4. Solution that Maximizes trace(R'K) is R=U*V', where U, V are
+ # singular vectors of K.
+ U, s, V = torch.svd(K)
+
+ # Construct Z that fixes the orientation of R to get det(R)=1.
+ Z = torch.eye(U.shape[1], device=S1.device).unsqueeze(0)
+ Z = Z.repeat(U.shape[0], 1, 1)
+ Z[:, -1, -1] *= torch.sign(torch.det(U.bmm(V.permute(0, 2, 1))))
+
+ # Construct R.
+ R = V.bmm(Z.bmm(U.permute(0, 2, 1)))
+
+ # 5. Recover scale.
+ scale = torch.cat([torch.trace(x).unsqueeze(0) for x in R.bmm(K)]) / var1
+
+ # 6. Recover translation.
+ t = mu2 - (scale.unsqueeze(-1).unsqueeze(-1) * (R.bmm(mu1)))
+
+ # -------
+ # reshape back
+ # sR = scale[:, None, None] * R
+ # sR = sR.reshape(*S_shape[:-2], 3, 3)
+ scale = scale.reshape(*S_shape[:-2], 1, 1)
+ R = R.reshape(*S_shape[:-2], 3, 3)
+ t = t.reshape(*S_shape[:-2], 3, 1)
+
+ return (scale, R), t
+
+
+def kabsch_algorithm_batch(X1, X2):
+ """
+ Computes a rigid transform (R, t)
+ Args:
+ X1, X2: (*, L, 3)
+ """
+ assert X1.shape == X2.shape
+ X_shape = X1.shape
+ X1 = X1.reshape(-1, *X_shape[-2:])
+ X2 = X2.reshape(-1, *X_shape[-2:])
+
+ # 1. 计算质心
+ centroid_X1 = torch.mean(X1, dim=-2, keepdim=True)
+ centroid_X2 = torch.mean(X2, dim=-2, keepdim=True)
+
+ # 2. 去中心化
+ X1_centered = X1 - centroid_X1
+ X2_centered = X2 - centroid_X2
+
+ # 3. 计算协方差矩阵
+ H = torch.matmul(X1_centered.transpose(-2, -1), X2_centered)
+
+ # 4. 奇异值分解
+ U, S, Vt = torch.linalg.svd(H)
+
+ # 5. 计算旋转矩阵
+ R = torch.matmul(Vt.transpose(-2, -1), U.transpose(-2, -1))
+
+ # 修正反射矩阵
+ d = (torch.det(R) < 0).unsqueeze(-1).unsqueeze(-1)
+ Vt = torch.where(d, -Vt, Vt)
+ R = torch.matmul(Vt.transpose(-2, -1), U.transpose(-2, -1))
+
+ # 6. 计算平移向量
+ t = centroid_X2.transpose(-2, -1) - torch.matmul(R, centroid_X1.transpose(-2, -1))
+
+ # -------
+ # reshape back
+ R = R.reshape(*X_shape[:-2], 3, 3)
+ t = t.reshape(*X_shape[:-2], 3, 1)
+
+ return R, t
+
+
+# ===== WHAM cam_angvel ===== #
+
+
+def compute_cam_angvel(R_w2c, padding_last=True):
+ """
+ R_w2c : (F, 3, 3)
+ """
+ # R @ R0 = R1, so R = R1 @ R0^T
+ cam_angvel = matrix_to_rotation_6d(R_w2c[1:] @ R_w2c[:-1].transpose(-1, -2)) # (F-1, 6)
+ # cam_angvel = (cam_angvel - torch.tensor([[1, 0, 0, 0, 1, 0]])) * FPS
+ assert padding_last
+ cam_angvel = torch.cat([cam_angvel, cam_angvel[-1:]], dim=0) # (F, 6)
+ return cam_angvel.float()
+
+
+def ransac_gravity_vec(xyz, num_iterations=100, threshold=0.05, verbose=False):
+ # xyz: (L, 3)
+ N = xyz.shape[0]
+ max_inliers = []
+ best_model = None
+ norms = xyz.norm(dim=-1) # (L,)
+
+ for _ in range(num_iterations):
+ # 随机选择一个样本
+ sample_index = np.random.randint(N)
+ sample = xyz[sample_index] # (3,)
+
+ # 计算所有点与样本点的角度差
+ dot_product = (xyz * sample).sum(dim=-1) # (L,)
+ angles = dot_product / norms * norms[sample_index] # (L,)
+ angles = torch.clamp(angles, -1, 1) # 防止数值误差导致的异常
+ angles = torch.acos(angles)
+
+ # 确定内点
+ inliers = xyz[angles < threshold]
+
+ if len(inliers) > len(max_inliers):
+ max_inliers = inliers
+ best_model = sample
+ if len(max_inliers) == N:
+ break
+ if verbose:
+ print(f"Inliers: {len(max_inliers)} / {N}")
+ result = max_inliers.mean(dim=0)
+
+ return result, max_inliers
+
+
+def sequence_best_cammat(w_j3d, c_j3d, cam_rot):
+ # get best camera estimation along the sequence, requires static camera
+ # w_j3d: (L, J, 3)
+ # c_j3d: (L, J, 3)
+ # cam_rot: (L, 3, 3)
+
+ L, J, _ = w_j3d.shape
+
+ root_in_w = w_j3d[:, 0] # (L, 3)
+ root_in_c = c_j3d[:, 0] # (L, 3)
+ cam_mat = matrix.get_TRS(cam_rot, root_in_w) # (L, 4, 4)
+ cam_pos = matrix.get_position_from(-root_in_c[:, None], cam_mat)[:, 0] # (L, 3)
+ cam_mat = matrix.set_position(cam_mat, cam_pos) # (L, 4, 4)
+
+ w_j3d_expand = w_j3d[None].expand(L, -1, -1, -1) # (L, L, J, 3)
+ w_j3d_expand = w_j3d_expand.reshape(L, -1, 3) # (L, L*J, 3)
+
+ # get reproject error
+ w_j3d_expand_in_c = matrix.get_relative_position_to(w_j3d_expand, cam_mat) # (L, L*J, 3)
+ w_j2d_expand_in_c = project_p2d(w_j3d_expand_in_c) # (L, L*J, 2)
+ w_j2d_expand_in_c = w_j2d_expand_in_c.reshape(L, L, J, 2) # (L, L, J, 2)
+ c_j2d = project_p2d(c_j3d) # (L, J, 2)
+ error = w_j2d_expand_in_c - c_j2d[None] # (L, L, J, 2)
+ error = error.norm(dim=-1).mean(dim=-1) # (L, L)
+ error = error.mean(dim=-1) # (L,)
+ ind = error.argmin()
+ return cam_mat[ind], ind
+
+
+def get_sequence_cammat(w_j3d, c_j3d, cam_rot):
+ # w_j3d: (L, J, 3)
+ # c_j3d: (L, J, 3)
+ # cam_rot: (L, 3, 3)
+
+ L, J, _ = w_j3d.shape
+
+ root_in_w = w_j3d[:, 0] # (L, 3)
+ root_in_c = c_j3d[:, 0] # (L, 3)
+ cam_mat = matrix.get_TRS(cam_rot, root_in_w) # (L, 4, 4)
+ cam_pos = matrix.get_position_from(-root_in_c[:, None], cam_mat)[:, 0] # (L, 3)
+ cam_mat = matrix.set_position(cam_mat, cam_pos) # (L, 4, 4)
+ return cam_mat
+
+
+def ransac_vec(vel, min_multiply=20, verbose=False):
+ # xyz: (L, 3)
+ # remove outlier velocity
+ N = vel.shape[0]
+ vel_1 = vel[None].expand(N, -1, -1) # (L, L, 3)
+ vel_2 = vel[:, None].expand(-1, N, -1) # (L, L, 3)
+ dist_mat = (vel_1 - vel_2).norm(dim=-1) # (L, L)
+ big_identity = torch.eye(N, device=vel.device) * 1e6
+ dist_mat_ = dist_mat + big_identity
+ threshold = dist_mat_.min() * min_multiply
+ inner_mask = dist_mat < threshold # (L, L)
+ inner_num = inner_mask.sum(dim=-1) # (L, )
+ ind = inner_num.argmax()
+ result = vel[inner_mask[ind]].mean(dim=0) # (3,)
+ if verbose:
+ print(inner_mask[ind].sum().item())
+
+ return result, inner_mask[ind]
diff --git a/third_party/GVHMR/hmr4d/utils/ik/ccd_ik.py b/third_party/GVHMR/hmr4d/utils/ik/ccd_ik.py
new file mode 100644
index 0000000000000000000000000000000000000000..9bc606c3a8c6ae53c50a938ca295d19c8fa5712c
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/ik/ccd_ik.py
@@ -0,0 +1,149 @@
+# Sebastian IK
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from einops import einsum, rearrange, repeat
+
+from pytorch3d.transforms import (
+ matrix_to_rotation_6d,
+ rotation_6d_to_matrix,
+ axis_angle_to_matrix,
+ matrix_to_axis_angle,
+ quaternion_to_matrix,
+ matrix_to_quaternion,
+)
+import hmr4d.utils.matrix as matrix
+from hmr4d.utils.geo.quaternion import qbetween, qslerp, qinv, qmul, qrot
+
+
+class CCD_IK:
+ def __init__(
+ self,
+ local_mat,
+ parent,
+ target_ind,
+ target_pos=None,
+ target_rot=None,
+ kinematic_chain=None,
+ max_iter=2, # sebas sets 25 but with converged flag, 2 is enough
+ threshold=0.001,
+ pos_weight=1.0,
+ rot_weight=0.0, # this makes optimization unstable, although sebas uses 1.0
+ ):
+ if kinematic_chain is None:
+ kinematic_chain = range(local_mat.shape[-3])
+ global_mat = matrix.forward_kinematics(local_mat, parent)
+
+ # get kinematic chain only local mat and assign root mat (do not modify root during IK)
+ local_mat = local_mat.clone()
+ local_mat = local_mat[..., kinematic_chain, :, :]
+ local_mat[..., 0, :, :] = global_mat[..., kinematic_chain[0], :, :]
+
+ parent = [i - 1 for i in range(len(kinematic_chain))]
+ self.local_mat = local_mat
+ self.global_mat = matrix.forward_kinematics(local_mat, parent) # (*, J, 4, 4)
+ self.parent = parent
+
+ self.target_ind = target_ind
+ if target_pos is not None:
+ self.target_pos = target_pos # (*, O, 3)
+ else:
+ self.target_pos = None
+ if target_rot is not None:
+ self.target_q = matrix_to_quaternion(target_rot) # (*, O, 4)
+ else:
+ self.target_q = None
+
+ self.threshold = threshold
+ self.J_N = self.local_mat.shape[-3]
+ self.target_N = len(target_ind)
+ self.max_iter = max_iter
+ self.pos_weight = pos_weight
+ self.rot_weight = rot_weight
+
+ def is_converged(self):
+ end_pos = matrix.get_position(self.global_mat)[..., self.target_ind, :] # (*, OJ, 3)
+ converged_mask = (self.target_pos - end_pos).norm(dim=-1) < self.threshold
+ self.converged_mask = converged_mask
+ if self.converged_mask.sum() > 0:
+ return False
+ return True
+
+ def solve(self):
+ for _ in range(self.max_iter):
+ # if self.is_converged():
+ # return self.local_mat
+ # do not optimize root, so start from 1
+ self.optimize(1)
+ return self.local_mat
+
+ def optimize(self, i):
+ # i: joint_i
+ if i == self.J_N - 1:
+ return
+ pos = matrix.get_position(self.global_mat)[..., i, :] # (*, 3)
+ rot = matrix.get_rotation(self.global_mat)[..., i, :, :] # (*, 3, 3)
+ quat = matrix_to_quaternion(rot) # (*, 4)
+ x_vec = torch.zeros((quat.shape[:-1] + (3,)), device=quat.device)
+ x_vec[..., 0] = 1.0
+ x_vec_sum = torch.zeros_like(x_vec)
+ y_vec = torch.zeros((quat.shape[:-1] + (3,)), device=quat.device)
+ y_vec[..., 1] = 1.0
+ y_vec_sum = torch.zeros_like(y_vec)
+
+ count = 0
+
+ for target_i, j in enumerate(self.target_ind):
+ if i >= j:
+ # do not optimise same joint or child joint of targets
+ continue
+ end_pos = matrix.get_position(self.global_mat)[..., j, :] # (*, 3)
+ end_rot = matrix.get_rotation(self.global_mat)[..., j, :, :] # (*, 3, 3)
+ end_quat = matrix_to_quaternion(end_rot) # (*, 4)
+
+ if self.target_pos is not None:
+ target_pos = self.target_pos[..., target_i, :] # (*, 3)
+ # Solve objective position
+ solved_pos_target_quat = qslerp(
+ quat,
+ qmul(qbetween(end_pos - pos, target_pos - pos), quat),
+ self.get_weight(i),
+ )
+
+ x_vec_sum += qrot(solved_pos_target_quat, x_vec)
+ y_vec_sum += qrot(solved_pos_target_quat, y_vec)
+ if self.pos_weight > 0:
+ count += 1
+
+ if self.target_q is not None:
+ if target_i < self.target_N - 1:
+ # multiple rot target makes more unstable, only keep the last one
+ continue
+ # optimize rotation target is not stable
+ target_q = self.target_q[..., target_i, :] # (*, 4)
+ # Solve objective rotation
+ solved_q_target_quat = qslerp(
+ quat,
+ qmul(qmul(target_q, qinv(end_quat)), quat),
+ self.get_weight(i),
+ )
+ x_vec_sum += qrot(solved_q_target_quat, x_vec) * self.rot_weight
+ y_vec_sum += qrot(solved_q_target_quat, y_vec) * self.rot_weight
+ if self.rot_weight > 0:
+ count += 1
+
+ if count > 0:
+ x_vec_avg = matrix.normalize(x_vec_sum / count)
+ y_vec_avg = matrix.normalize(y_vec_sum / count)
+ z_vec_avg = torch.cross(x_vec_avg, y_vec_avg, dim=-1)
+ solved_rot = torch.stack([x_vec_avg, y_vec_avg, z_vec_avg], dim=-1) # column
+
+ parent_rot = matrix.get_rotation(self.global_mat)[..., self.parent[i], :, :]
+ solved_local_rot = matrix.get_mat_BtoA(parent_rot, solved_rot)
+ self.local_mat[..., i, :-1, :-1] = solved_local_rot
+ self.global_mat = matrix.forward_kinematics(self.local_mat, self.parent)
+ self.optimize(i + 1)
+
+ def get_weight(self, i):
+ weight = (i + 1) / self.J_N
+ return weight
diff --git a/third_party/GVHMR/hmr4d/utils/kpts/kp2d_utils.py b/third_party/GVHMR/hmr4d/utils/kpts/kp2d_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..b08be88f266df10b3a16f5843696bd182aea7ed8
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/kpts/kp2d_utils.py
@@ -0,0 +1,372 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import warnings
+import cv2
+import numpy as np
+
+# expose _taylor to outside
+__all__ = ["keypoints_from_heatmaps"]
+
+
+def _taylor(heatmap, coord):
+ """Distribution aware coordinate decoding method.
+
+ Note:
+ - heatmap height: H
+ - heatmap width: W
+
+ Args:
+ heatmap (np.ndarray[H, W]): Heatmap of a particular joint type.
+ coord (np.ndarray[2,]): Coordinates of the predicted keypoints.
+
+ Returns:
+ np.ndarray[2,]: Updated coordinates.
+ """
+ H, W = heatmap.shape[:2]
+ px, py = int(coord[0]), int(coord[1])
+ if 1 < px < W - 2 and 1 < py < H - 2:
+ dx = 0.5 * (heatmap[py][px + 1] - heatmap[py][px - 1])
+ dy = 0.5 * (heatmap[py + 1][px] - heatmap[py - 1][px])
+ dxx = 0.25 * (heatmap[py][px + 2] - 2 * heatmap[py][px] + heatmap[py][px - 2])
+ dxy = 0.25 * (
+ heatmap[py + 1][px + 1] - heatmap[py - 1][px + 1] - heatmap[py + 1][px - 1] + heatmap[py - 1][px - 1]
+ )
+ dyy = 0.25 * (heatmap[py + 2 * 1][px] - 2 * heatmap[py][px] + heatmap[py - 2 * 1][px])
+ derivative = np.array([[dx], [dy]])
+ hessian = np.array([[dxx, dxy], [dxy, dyy]])
+ if dxx * dyy - dxy**2 != 0:
+ hessianinv = np.linalg.inv(hessian)
+ offset = -hessianinv @ derivative
+ offset = np.squeeze(np.array(offset.T), axis=0)
+ coord += offset
+ return coord
+
+
+def _get_max_preds(heatmaps):
+ """Get keypoint predictions from score maps.
+
+ Note:
+ batch_size: N
+ num_keypoints: K
+ heatmap height: H
+ heatmap width: W
+
+ Args:
+ heatmaps (np.ndarray[N, K, H, W]): model predicted heatmaps.
+
+ Returns:
+ tuple: A tuple containing aggregated results.
+
+ - preds (np.ndarray[N, K, 2]): Predicted keypoint location.
+ - maxvals (np.ndarray[N, K, 1]): Scores (confidence) of the keypoints.
+ """
+ assert isinstance(heatmaps, np.ndarray), "heatmaps should be numpy.ndarray"
+ assert heatmaps.ndim == 4, "batch_images should be 4-ndim"
+
+ N, K, _, W = heatmaps.shape
+ heatmaps_reshaped = heatmaps.reshape((N, K, -1))
+ idx = np.argmax(heatmaps_reshaped, 2).reshape((N, K, 1))
+ maxvals = np.amax(heatmaps_reshaped, 2).reshape((N, K, 1))
+
+ preds = np.tile(idx, (1, 1, 2)).astype(np.float32)
+ preds[:, :, 0] = preds[:, :, 0] % W
+ preds[:, :, 1] = preds[:, :, 1] // W
+
+ preds = np.where(np.tile(maxvals, (1, 1, 2)) > 0.0, preds, -1)
+ return preds, maxvals
+
+
+def post_dark_udp(coords, batch_heatmaps, kernel=3):
+ """DARK post-pocessing. Implemented by udp. Paper ref: Huang et al. The
+ Devil is in the Details: Delving into Unbiased Data Processing for Human
+ Pose Estimation (CVPR 2020). Zhang et al. Distribution-Aware Coordinate
+ Representation for Human Pose Estimation (CVPR 2020).
+
+ Note:
+ - batch size: B
+ - num keypoints: K
+ - num persons: N
+ - height of heatmaps: H
+ - width of heatmaps: W
+
+ B=1 for bottom_up paradigm where all persons share the same heatmap.
+ B=N for top_down paradigm where each person has its own heatmaps.
+
+ Args:
+ coords (np.ndarray[N, K, 2]): Initial coordinates of human pose.
+ batch_heatmaps (np.ndarray[B, K, H, W]): batch_heatmaps
+ kernel (int): Gaussian kernel size (K) for modulation.
+
+ Returns:
+ np.ndarray([N, K, 2]): Refined coordinates.
+ """
+ if not isinstance(batch_heatmaps, np.ndarray):
+ batch_heatmaps = batch_heatmaps.cpu().numpy()
+ B, K, H, W = batch_heatmaps.shape
+ N = coords.shape[0]
+ assert B == 1 or B == N
+ for heatmaps in batch_heatmaps:
+ for heatmap in heatmaps:
+ cv2.GaussianBlur(heatmap, (kernel, kernel), 0, heatmap)
+ np.clip(batch_heatmaps, 0.001, 50, batch_heatmaps)
+ np.log(batch_heatmaps, batch_heatmaps)
+
+ batch_heatmaps_pad = np.pad(batch_heatmaps, ((0, 0), (0, 0), (1, 1), (1, 1)), mode="edge").flatten()
+
+ index = coords[..., 0] + 1 + (coords[..., 1] + 1) * (W + 2)
+ index += (W + 2) * (H + 2) * np.arange(0, B * K).reshape(-1, K)
+ index = index.astype(int).reshape(-1, 1)
+ i_ = batch_heatmaps_pad[index]
+ ix1 = batch_heatmaps_pad[index + 1]
+ iy1 = batch_heatmaps_pad[index + W + 2]
+ ix1y1 = batch_heatmaps_pad[index + W + 3]
+ ix1_y1_ = batch_heatmaps_pad[index - W - 3]
+ ix1_ = batch_heatmaps_pad[index - 1]
+ iy1_ = batch_heatmaps_pad[index - 2 - W]
+
+ dx = 0.5 * (ix1 - ix1_)
+ dy = 0.5 * (iy1 - iy1_)
+ derivative = np.concatenate([dx, dy], axis=1)
+ derivative = derivative.reshape(N, K, 2, 1)
+ dxx = ix1 - 2 * i_ + ix1_
+ dyy = iy1 - 2 * i_ + iy1_
+ dxy = 0.5 * (ix1y1 - ix1 - iy1 + i_ + i_ - ix1_ - iy1_ + ix1_y1_)
+ hessian = np.concatenate([dxx, dxy, dxy, dyy], axis=1)
+ hessian = hessian.reshape(N, K, 2, 2)
+ hessian = np.linalg.inv(hessian + np.finfo(np.float32).eps * np.eye(2))
+ coords -= np.einsum("ijmn,ijnk->ijmk", hessian, derivative).squeeze()
+ return coords
+
+
+def _gaussian_blur(heatmaps, kernel=11):
+ """Modulate heatmap distribution with Gaussian.
+ sigma = 0.3*((kernel_size-1)*0.5-1)+0.8
+ sigma~=3 if k=17
+ sigma=2 if k=11;
+ sigma~=1.5 if k=7;
+ sigma~=1 if k=3;
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+ - heatmap height: H
+ - heatmap width: W
+
+ Args:
+ heatmaps (np.ndarray[N, K, H, W]): model predicted heatmaps.
+ kernel (int): Gaussian kernel size (K) for modulation, which should
+ match the heatmap gaussian sigma when training.
+ K=17 for sigma=3 and k=11 for sigma=2.
+
+ Returns:
+ np.ndarray ([N, K, H, W]): Modulated heatmap distribution.
+ """
+ assert kernel % 2 == 1
+
+ border = (kernel - 1) // 2
+ batch_size = heatmaps.shape[0]
+ num_joints = heatmaps.shape[1]
+ height = heatmaps.shape[2]
+ width = heatmaps.shape[3]
+ for i in range(batch_size):
+ for j in range(num_joints):
+ origin_max = np.max(heatmaps[i, j])
+ dr = np.zeros((height + 2 * border, width + 2 * border), dtype=np.float32)
+ dr[border:-border, border:-border] = heatmaps[i, j].copy()
+ dr = cv2.GaussianBlur(dr, (kernel, kernel), 0)
+ heatmaps[i, j] = dr[border:-border, border:-border].copy()
+ heatmaps[i, j] *= origin_max / np.max(heatmaps[i, j])
+ return heatmaps
+
+
+def keypoints_from_heatmaps(
+ heatmaps,
+ center,
+ scale,
+ unbiased=False,
+ post_process="default",
+ kernel=11,
+ valid_radius_factor=0.0546875,
+ use_udp=False,
+ target_type="GaussianHeatmap",
+):
+ """Get final keypoint predictions from heatmaps and transform them back to
+ the image.
+
+ Note:
+ - batch size: N
+ - num keypoints: K
+ - heatmap height: H
+ - heatmap width: W
+
+ Args:
+ heatmaps (np.ndarray[N, K, H, W]): model predicted heatmaps.
+ center (np.ndarray[N, 2]): Center of the bounding box (x, y).
+ scale (np.ndarray[N, 2]): Scale of the bounding box
+ wrt height/width.
+ post_process (str/None): Choice of methods to post-process
+ heatmaps. Currently supported: None, 'default', 'unbiased',
+ 'megvii'.
+ unbiased (bool): Option to use unbiased decoding. Mutually
+ exclusive with megvii.
+ Note: this arg is deprecated and unbiased=True can be replaced
+ by post_process='unbiased'
+ Paper ref: Zhang et al. Distribution-Aware Coordinate
+ Representation for Human Pose Estimation (CVPR 2020).
+ kernel (int): Gaussian kernel size (K) for modulation, which should
+ match the heatmap gaussian sigma when training.
+ K=17 for sigma=3 and k=11 for sigma=2.
+ valid_radius_factor (float): The radius factor of the positive area
+ in classification heatmap for UDP.
+ use_udp (bool): Use unbiased data processing.
+ target_type (str): 'GaussianHeatmap' or 'CombinedTarget'.
+ GaussianHeatmap: Classification target with gaussian distribution.
+ CombinedTarget: The combination of classification target
+ (response map) and regression target (offset map).
+ Paper ref: Huang et al. The Devil is in the Details: Delving into
+ Unbiased Data Processing for Human Pose Estimation (CVPR 2020).
+
+ Returns:
+ tuple: A tuple containing keypoint predictions and scores.
+
+ - preds (np.ndarray[N, K, 2]): Predicted keypoint location in images.
+ - maxvals (np.ndarray[N, K, 1]): Scores (confidence) of the keypoints.
+ """
+ # Avoid being affected
+ heatmaps = heatmaps.copy()
+
+ # detect conflicts
+ if unbiased:
+ assert post_process not in [False, None, "megvii"]
+ if post_process in ["megvii", "unbiased"]:
+ assert kernel > 0
+ if use_udp:
+ assert not post_process == "megvii"
+
+ # normalize configs
+ if post_process is False:
+ warnings.warn("post_process=False is deprecated, " "please use post_process=None instead", DeprecationWarning)
+ post_process = None
+ elif post_process is True:
+ if unbiased is True:
+ warnings.warn(
+ "post_process=True, unbiased=True is deprecated," " please use post_process='unbiased' instead",
+ DeprecationWarning,
+ )
+ post_process = "unbiased"
+ else:
+ warnings.warn(
+ "post_process=True, unbiased=False is deprecated, " "please use post_process='default' instead",
+ DeprecationWarning,
+ )
+ post_process = "default"
+ elif post_process == "default":
+ if unbiased is True:
+ warnings.warn(
+ "unbiased=True is deprecated, please use " "post_process='unbiased' instead", DeprecationWarning
+ )
+ post_process = "unbiased"
+
+ # start processing
+ if post_process == "megvii":
+ heatmaps = _gaussian_blur(heatmaps, kernel=kernel)
+
+ N, K, H, W = heatmaps.shape
+ if use_udp:
+ if target_type.lower() == "GaussianHeatMap".lower():
+ preds, maxvals = _get_max_preds(heatmaps)
+ preds = post_dark_udp(preds, heatmaps, kernel=kernel)
+ elif target_type.lower() == "CombinedTarget".lower():
+ for person_heatmaps in heatmaps:
+ for i, heatmap in enumerate(person_heatmaps):
+ kt = 2 * kernel + 1 if i % 3 == 0 else kernel
+ cv2.GaussianBlur(heatmap, (kt, kt), 0, heatmap)
+ # valid radius is in direct proportion to the height of heatmap.
+ valid_radius = valid_radius_factor * H
+ offset_x = heatmaps[:, 1::3, :].flatten() * valid_radius
+ offset_y = heatmaps[:, 2::3, :].flatten() * valid_radius
+ heatmaps = heatmaps[:, ::3, :]
+ preds, maxvals = _get_max_preds(heatmaps)
+ index = preds[..., 0] + preds[..., 1] * W
+ index += W * H * np.arange(0, N * K / 3)
+ index = index.astype(int).reshape(N, K // 3, 1)
+ preds += np.concatenate((offset_x[index], offset_y[index]), axis=2)
+ else:
+ raise ValueError("target_type should be either " "'GaussianHeatmap' or 'CombinedTarget'")
+ else:
+ preds, maxvals = _get_max_preds(heatmaps)
+ if post_process == "unbiased": # alleviate biased coordinate
+ # apply Gaussian distribution modulation.
+ heatmaps = np.log(np.maximum(_gaussian_blur(heatmaps, kernel), 1e-10))
+ for n in range(N):
+ for k in range(K):
+ preds[n][k] = _taylor(heatmaps[n][k], preds[n][k])
+ elif post_process is not None:
+ # add +/-0.25 shift to the predicted locations for higher acc.
+ for n in range(N):
+ for k in range(K):
+ heatmap = heatmaps[n][k]
+ px = int(preds[n][k][0])
+ py = int(preds[n][k][1])
+ if 1 < px < W - 1 and 1 < py < H - 1:
+ diff = np.array(
+ [heatmap[py][px + 1] - heatmap[py][px - 1], heatmap[py + 1][px] - heatmap[py - 1][px]]
+ )
+ preds[n][k] += np.sign(diff) * 0.25
+ if post_process == "megvii":
+ preds[n][k] += 0.5
+
+ # Transform back to the image
+ for i in range(N):
+ preds[i] = transform_preds(preds[i], center[i], scale[i], [W, H], use_udp=use_udp)
+
+ if post_process == "megvii":
+ maxvals = maxvals / 255.0 + 0.5
+
+ return preds, maxvals
+
+
+def transform_preds(coords, center, scale, output_size, use_udp=False):
+ """Get final keypoint predictions from heatmaps and apply scaling and
+ translation to map them back to the image.
+
+ Note:
+ num_keypoints: K
+
+ Args:
+ coords (np.ndarray[K, ndims]):
+
+ * If ndims=2, corrds are predicted keypoint location.
+ * If ndims=4, corrds are composed of (x, y, scores, tags)
+ * If ndims=5, corrds are composed of (x, y, scores, tags,
+ flipped_tags)
+
+ center (np.ndarray[2, ]): Center of the bounding box (x, y).
+ scale (np.ndarray[2, ]): Scale of the bounding box
+ wrt [width, height].
+ output_size (np.ndarray[2, ] | list(2,)): Size of the
+ destination heatmaps.
+ use_udp (bool): Use unbiased data processing
+
+ Returns:
+ np.ndarray: Predicted coordinates in the images.
+ """
+ assert coords.shape[1] in (2, 4, 5)
+ assert len(center) == 2
+ assert len(scale) == 2
+ assert len(output_size) == 2
+
+ # Recover the scale which is normalized by a factor of 200.
+ scale = scale * 200.0
+
+ if use_udp:
+ scale_x = scale[0] / (output_size[0] - 1.0)
+ scale_y = scale[1] / (output_size[1] - 1.0)
+ else:
+ scale_x = scale[0] / output_size[0]
+ scale_y = scale[1] / output_size[1]
+
+ target_coords = np.ones_like(coords)
+ target_coords[:, 0] = coords[:, 0] * scale_x + center[0] - scale[0] * 0.5
+ target_coords[:, 1] = coords[:, 1] * scale_y + center[1] - scale[1] * 0.5
+
+ return target_coords
diff --git a/third_party/GVHMR/hmr4d/utils/matrix.py b/third_party/GVHMR/hmr4d/utils/matrix.py
new file mode 100644
index 0000000000000000000000000000000000000000..ce9b1e848a87252e5ef17ea2b18add76ef3f85ef
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/matrix.py
@@ -0,0 +1,1677 @@
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import copy
+from typing import List, Optional
+
+import numpy as np
+
+import math
+
+
+def identity_mat(x=None, device="cpu", is_numpy=False):
+ if x is not None:
+ if isinstance(x, torch.Tensor):
+ mat = torch.eye(4, device=device)
+ mat = mat.repeat(x.shape[:-2] + (1, 1))
+ elif isinstance(x, np.ndarray):
+ mat = np.eye(4, dtype=np.float32)
+ if x is not None:
+ for _ in range(len(x.shape) - 2):
+ mat = mat[None]
+ mat = np.tile(mat, x.shape[:-2] + (1, 1))
+ else:
+ raise ValueError
+ else:
+ # (4, 4)
+ if is_numpy:
+ mat = np.eye(4, dtype=np.float32)
+ else:
+ mat = torch.eye(4, device=device)
+
+ return mat
+
+
+def vec2mat(vec):
+ """_summary_
+
+ Args:
+ vec (tensor): [12], pos, forward, up and right
+
+ Returns:
+ mat_world(tensor): [4, 4]
+ """
+ # Assume bs = 1
+ v = np.tile(np.array([[0, 0, 0, 1]]), (1, 1))
+ if isinstance(vec, torch.Tensor):
+ v = torch.tensor(
+ v,
+ device=vec.device,
+ dtype=vec.dtype,
+ )
+ pos = vec[:3]
+ forward = vec[3:6]
+ up = vec[6:9]
+ right = vec[9:12]
+
+ if isinstance(vec, torch.Tensor):
+ mat_world = torch.stack([right, up, forward, pos], dim=-1)
+ mat_world = torch.cat([mat_world, v], dim=-2)
+ elif isinstance(vec, np.ndarray):
+ mat_world = np.stack([right, up, forward, pos], axis=-1)
+ mat_world = np.concatenate([mat_world, v], axis=-2)
+ else:
+ raise ValueError
+ mat_world = normalized_matrix(mat_world)
+ return mat_world
+
+
+def mat2vec(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [4, 4]
+
+ Returns:
+ vec (tensor): [12], pos, forward, up and right
+ """
+ # Assume bs = 1
+ pos = mat[:-1, 3]
+ forward = normalized(mat[:-1, 2])
+ up = normalized(mat[:-1, 1])
+ right = normalized(mat[:-1, 0])
+ if isinstance(mat, torch.Tensor):
+ vec = torch.cat((pos, forward, up, right))
+ elif isinstance(mat, np.ndarray):
+ vec = np.concatenate((pos, forward, up, right))
+ else:
+ raise ValueError
+
+ return vec
+
+
+def vec2mat_batch(vec):
+ """_summary_
+
+ Args:
+ vec (tensor): [B, 12], pos, forward, up and right
+
+ Returns:
+ mat_world(tensor): [B, 4, 4]
+ """
+ # Assume bs = 1
+
+ v = np.tile(np.array([[0, 0, 0, 1]], dtype=np.float32), (vec.shape[0], 1, 1))
+ if isinstance(vec, torch.Tensor):
+ v = torch.tensor(
+ v,
+ device=vec.device,
+ dtype=vec.dtype,
+ )
+ pos = vec[..., :3]
+ forward = vec[..., 3:6]
+ up = vec[..., 6:9]
+ right = vec[..., 9:12]
+ if isinstance(vec, torch.Tensor):
+ mat_world = torch.stack([right, up, forward, pos], dim=-1)
+ mat_world = torch.cat([mat_world, v], dim=-2)
+ elif isinstance(vec, np.ndarray):
+ mat_world = np.stack([right, up, forward, pos], axis=-1)
+ mat_world = np.concatenate([mat_world, v], axis=-2)
+ else:
+ raise ValueError
+
+ mat_world = normalized_matrix(mat_world)
+ return mat_world
+
+
+def rotmat2tan_norm(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 3, 3]
+
+ Returns:
+ vec (tensor): [B, 6], tan norm
+ """
+ if isinstance(mat, np.ndarray):
+ tan = np.zeros_like(mat[..., 2])
+ norm = np.zeros_like(mat[..., 0])
+ elif isinstance(mat, torch.Tensor):
+ tan = torch.zeros_like(mat[..., 2])
+ norm = torch.zeros_like(mat[..., 0])
+ else:
+ raise ValueError
+ tan[...] = mat[..., 2, ::-1]
+ tan[..., -1] *= -1
+ norm[...] = mat[..., 0, ::-1]
+ norm[..., -1] *= -1
+ if isinstance(mat, np.ndarray):
+ tan_norm = np.concatenate((tan, norm), axis=-1)
+ elif isinstance(mat, torch.Tensor):
+ tan_norm = torch.cat((tan, norm), dim=-1)
+ else:
+ raise ValueError
+ return tan_norm
+
+
+def mat2tan_norm(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 4, 4]
+
+ Returns:
+ vec (tensor): [B, 6], tan norm
+ """
+ rot_mat = mat[..., :-1, :-1]
+ return rotmat2tan_norm(rot_mat)
+
+
+def rotmat2tan_norm(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 3, 3]
+
+ Returns:
+ vec (tensor): [B, 6], tan norm
+ """
+ if isinstance(mat, np.ndarray):
+ tan = np.zeros_like(mat[..., 2])
+ norm = np.zeros_like(mat[..., 0])
+ tan[...] = mat[..., 2, ::-1]
+ norm[...] = mat[..., 0, ::-1]
+ elif isinstance(mat, torch.Tensor):
+ tan = torch.zeros_like(mat[..., 2])
+ norm = torch.zeros_like(mat[..., 0])
+ tan[...] = torch.flip(mat[..., 2], dims=[-1])
+ norm[...] = torch.flip(mat[..., 0], dims=[-1])
+ else:
+ raise ValueError
+ tan[..., -1] *= -1
+ norm[..., -1] *= -1
+ if isinstance(mat, np.ndarray):
+ tan_norm = np.concatenate((tan, norm), axis=-1)
+ elif isinstance(mat, torch.Tensor):
+ tan_norm = torch.cat((tan, norm), dim=-1)
+ else:
+ raise ValueError
+ return tan_norm
+
+
+def tan_norm2rotmat(tan_norm):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 6]
+
+ Returns:
+ vec (tensor): [B, 3]
+ """
+ tan = copy.deepcopy(tan_norm[..., :3])
+ norm = copy.deepcopy(tan_norm[..., 3:])
+ tan[..., -1] *= -1
+ norm[..., -1] *= -1
+ if isinstance(tan_norm, np.ndarray):
+ rotmat = np.zeros(tan_norm.shape[:-1] + (3, 3))
+ tan = tan[..., ::-1]
+ norm = norm[..., ::-1]
+ other = np.cross(tan, norm)
+ elif isinstance(tan_norm, torch.Tensor):
+ rotmat = torch.zeros(tan_norm.shape[:-1] + (3, 3), device=tan_norm.device)
+ tan = torch.flip(tan, dims=[-1])
+ norm = torch.flip(norm, dims=[-1])
+ other = torch.cross(tan, norm)
+ else:
+ raise ValueError
+ rotmat[..., 2, :] = tan
+ rotmat[..., 0, :] = norm
+ rotmat[..., 1, :] = other
+ return rotmat
+
+
+def rotmat332vec_batch(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 3, 3]
+
+ Returns:
+ vec (tensor): [B, 6], forward, up, right
+ """
+ # Assume bs = 1
+ mat = normalized_matrix(mat)
+ forward = mat[..., :, 2]
+ up = mat[..., :, 1]
+ right = mat[..., :, 0]
+ if isinstance(mat, torch.Tensor):
+ vec = torch.cat((forward, up, right), dim=-1)
+ elif isinstance(mat, np.ndarray):
+ vec = np.concatenate((forward, up, right), axis=-1)
+ else:
+ raise ValueError
+ return vec
+
+
+def rotmat2vec_batch(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 4, 4]
+
+ Returns:
+ vec (tensor): [B, 9], forward, up, right
+ """
+ # Assume bs = 1
+ mat = normalized_matrix(mat)
+ forward = mat[..., :-1, 2]
+ up = mat[..., :-1, 1]
+ right = mat[..., :-1, 0]
+ if isinstance(mat, torch.Tensor):
+ vec = torch.cat((forward, up, right), dim=-1)
+ elif isinstance(mat, np.ndarray):
+ vec = np.concatenate((forward, up, right), axis=-1)
+ else:
+ raise ValueError
+ return vec
+
+
+def mat2vec_batch(mat):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 4, 4]
+
+ Returns:
+ vec (tensor): [B, 12], pos, forward, up and right
+ """
+ # Assume bs = 1
+ mat = normalized_matrix(mat)
+ pos = mat[..., :-1, 3]
+ forward = mat[..., :-1, 2]
+ up = mat[..., :-1, 1]
+ right = mat[..., :-1, 0]
+ if isinstance(mat, torch.Tensor):
+ vec = torch.cat((pos, forward, up, right), dim=-1)
+ elif isinstance(mat, np.ndarray):
+ vec = np.concatenate((pos, forward, up, right), axis=-1)
+ else:
+ raise ValueError
+ return vec
+
+
+def mat2pose_batch(mat, returnvel=True):
+ """_summary_
+
+ Args:
+ mat(tensor): [B, 4, 4]
+
+ Returns:
+ vec (tensor): [B, 12], pos, forward, up, zeros
+ """
+ # Assume bs = 1
+ mat = normalized_matrix(mat)
+ pos = mat[..., :-1, 3]
+ forward = mat[..., :-1, 2]
+ up = mat[..., :-1, 1]
+ if isinstance(mat, torch.Tensor):
+ if returnvel:
+ vel = torch.zeros_like(up)
+ vec = torch.cat((pos, forward, up, vel), dim=-1)
+ else:
+ vec = torch.cat((pos, forward, up), dim=-1)
+ elif isinstance(mat, np.ndarray):
+ if returnvel:
+ vel = np.zeros_like(up)
+ vec = np.concatenate((pos, forward, up, vel), axis=-1)
+ else:
+ vec = np.concatenate((pos, forward, up), axis=-1)
+ else:
+ raise ValueError
+ return vec
+
+
+def get_mat_BinA(matCtoA, matCtoB):
+ """
+ given matrix of the same object in two coordinate A and B,
+ return matrix B in the coordinate of A
+
+ Args:
+ matCtoA (tensor): [4, 4] world matrix
+ matCtoB (tensor): [4, 4] world matrix
+ """
+ if isinstance(matCtoA, torch.Tensor):
+ matCtoB_inv = torch.inverse(matCtoB)
+ elif isinstance(matCtoA, np.ndarray):
+ matCtoB_inv = np.linalg.inv(matCtoB)
+ else:
+ raise ValueError
+ matCtoB_inv = normalized_matrix(matCtoB_inv)
+ if isinstance(matCtoA, torch.Tensor):
+ mat_BtoA = torch.matmul(matCtoA, matCtoB_inv)
+ elif isinstance(matCtoA, np.ndarray):
+ mat_BtoA = np.matmul(matCtoA, matCtoB_inv)
+ mat_BtoA = normalized_matrix(mat_BtoA)
+ return mat_BtoA
+
+
+def get_mat_BtoA(matA, matB):
+ """
+ return matrix B in the coordinate of A
+
+ Args:
+ matA (tensor): [4, 4] world matrix
+ matB (tensor): [4, 4] world matrix
+ """
+ if isinstance(matA, torch.Tensor):
+ matA_inv = torch.inverse(matA)
+ elif isinstance(matA, np.ndarray):
+ matA_inv = np.linalg.inv(matA)
+ else:
+ raise ValueError
+ matA_inv = normalized_matrix(matA_inv)
+ if isinstance(matA, torch.Tensor):
+ mat_BtoA = torch.matmul(matA_inv, matB)
+ elif isinstance(matA, np.ndarray):
+ mat_BtoA = np.matmul(matA_inv, matB)
+ mat_BtoA = normalized_matrix(mat_BtoA)
+ return mat_BtoA
+
+
+def get_mat_BfromA(matA, matBtoA):
+ """
+ return world matrix B given matrix A and mat B realtive to A
+
+ Args:
+ matA (_type_): [4, 4] world matrix
+ matBtoA (_type_): [4, 4] matrix B relative to A
+ """
+ if isinstance(matA, torch.Tensor):
+ matB = torch.matmul(matA, matBtoA)
+ if isinstance(matA, np.ndarray):
+ matB = np.matmul(matA, matBtoA)
+ matB = normalized_matrix(matB)
+ return matB
+
+
+def get_relative_position_to(pos, mat):
+ """_summary_
+
+ Args:
+ pos (_type_): [N, M, 3] or [N, 3]
+ mat (_type_): [N, 4, 4] or [4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ if isinstance(mat, torch.Tensor):
+ mat_inv = torch.inverse(mat)
+ elif isinstance(mat, np.ndarray):
+ mat_inv = np.linalg.inv(mat)
+ else:
+ raise ValueError
+ mat_inv = normalized_matrix(mat_inv)
+ if isinstance(mat, torch.Tensor):
+ rot_pos = torch.matmul(mat_inv[..., :-1, :-1], pos.transpose(-1, -2)).transpose(-1, -2)
+ elif isinstance(mat, np.ndarray):
+ rot_pos = np.matmul(mat_inv[..., :-1, :-1], pos.swapaxes(-1, -2)).swapaxes(-1, -2)
+ world_pos = rot_pos + mat_inv[..., None, :-1, 3]
+ return world_pos
+
+
+def get_rotation(mat):
+ """_summary_
+
+ Args:
+ mat (_type_): [..., 4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ return mat[..., :-1, :-1]
+
+
+def set_rotation(mat, rotmat):
+ """_summary_
+
+ Args:
+ mat (_type_): [..., 4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ mat[..., :-1, :-1] = rotmat
+ return mat
+
+
+def set_position(mat, pos):
+ """_summary_
+
+ Args:
+ mat (_type_): [..., 4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ mat[..., :-1, 3] = pos
+ return mat
+
+
+def get_position(mat):
+ """_summary_
+
+ Args:
+ mat (_type_): [..., 4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ return mat[..., :-1, 3]
+
+
+def get_position_from(pos, mat):
+ """_summary_
+
+ Args:
+ pos (_type_): [N, M, 3] or [N, 3]
+ mat (_type_): [N, 4, 4] or [4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ if isinstance(mat, torch.Tensor):
+ rot_pos = torch.matmul(mat[..., :-1, :-1], pos.transpose(-1, -2)).transpose(-1, -2)
+ elif isinstance(mat, np.ndarray):
+ rot_pos = np.matmul(mat[..., :-1, :-1], pos.swapaxes(-1, -2)).swapaxes(-1, -2)
+ else:
+ raise ValueError
+
+ world_pos = rot_pos + mat[..., None, :-1, 3]
+ return world_pos
+
+
+def get_position_from_rotmat(pos, mat):
+ """_summary_
+
+ Args:
+ pos (_type_): [N, M, 3] or [N, 3]
+ mat (_type_): [N, 4, 4] or [4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ if isinstance(mat, torch.Tensor):
+ rot_pos = torch.matmul(mat, pos.transpose(-1, -2)).transpose(-1, -2)
+ elif isinstance(mat, np.ndarray):
+ rot_pos = np.matmul(mat, pos.swapaxes(-1, -2)).swapaxes(-1, -2)
+ else:
+ raise ValueError
+ return rot_pos
+
+
+def get_relative_direction_to(dir, mat):
+ """_summary_
+
+ Args:
+ dir (_type_): [N, M, 3] or [N, 3]
+ mat (_type_): [N, 4, 4] or [4, 4]
+
+ Returns:
+ _type_: _description_
+ """
+ if isinstance(mat, torch.Tensor):
+ mat_inv = torch.inverse(mat)
+ elif isinstance(mat, np.ndarray):
+ mat_inv = np.linalg.inv(mat)
+ else:
+ raise ValueError
+ mat_inv = normalized_matrix(mat_inv)
+ rot_mat_inv = mat_inv[..., :3, :3]
+ if isinstance(mat, torch.Tensor):
+ rel_dir = torch.matmul(rot_mat_inv, dir.transpose(-1, -2))
+ return rel_dir.transpose(-1, -2)
+ elif isinstance(mat, np.ndarray):
+ rel_dir = np.matmul(rot_mat_inv, dir.swapaxes(-1, -2))
+ return rel_dir.swapaxes(-1, -2)
+ else:
+ raise ValueError
+ return
+
+
+def get_direction_from(dir, mat):
+ """_summary_
+
+ Args:
+ dir (_type_): [N, M, 3] or [N, 3]
+ mat (_type_): [N, 4, 4] or [4, 4]
+
+ Returns:
+ tensor: [N, M, 3] or [N, 3]
+ """
+ rot_mat = mat[..., :3, :3]
+ if isinstance(mat, torch.Tensor):
+ world_dir = torch.matmul(rot_mat, dir.transpose(-1, -2))
+ return world_dir.transpose(-1, -2)
+ elif isinstance(mat, np.ndarray):
+ world_dir = np.matmul(rot_mat, dir.swapaxes(-1, -2))
+ return world_dir.swapaxes(-1, -2)
+ else:
+ raise ValueError
+ return
+
+
+def get_coord_vis(pos, rot_mat, scale=1.0):
+ forward = rot_mat[..., :, 2]
+ up = rot_mat[..., :, 1]
+ right = rot_mat[..., :, 0]
+ return pos + right * scale, pos + up * scale, pos + forward * scale
+
+
+def project_vec(vec):
+ """_summary_
+
+ Args:
+ vec (tensor): [*, 12], pos, forward, up and right
+
+ Returns:
+ proj_vec (tensor): [*, 4], posx, posz, forwardx, forwardz
+ """
+ posx = vec[..., 0:1]
+ posz = vec[..., 2:3]
+ forwardx = vec[..., 3:4]
+ forwardz = vec[..., 5:6]
+ if isinstance(vec, torch.Tensor):
+ proj_vec = torch.cat((posx, posz, forwardx, forwardz), dim=-1)
+ elif isinstance(vec, np.ndarray):
+ proj_vec = np.concatenate((posx, posz, forwardx, forwardz), axis=-1)
+ else:
+ raise ValueError
+
+ return proj_vec
+
+
+def xz2xyz(vec):
+ x = vec[..., 0:1]
+ z = vec[..., 1:2]
+ if isinstance(vec, torch.Tensor):
+ y = torch.zeros(vec.shape[:-1] + (1,), device=vec.device)
+ xyz_vec = torch.cat((x, y, z), dim=-1)
+ elif isinstance(vec, np.ndarray):
+ y = np.zeros(vec.shape[:-1] + (1,))
+ xyz_vec = np.concatenate((x, y, z), axis=-1)
+ else:
+ raise ValueError
+
+ return xyz_vec
+
+
+def normalized(vec):
+ if isinstance(vec, torch.Tensor):
+ norm_vec = vec / (vec.norm(2, dim=-1, keepdim=True) + 1e-9)
+ elif isinstance(vec, np.ndarray):
+ norm_vec = vec / (np.linalg.norm(vec, ord=2, axis=-1, keepdims=True) + 1e-9)
+ else:
+ raise ValueError
+
+ return norm_vec
+
+
+def normalized_matrix(mat):
+ if mat.shape[-1] == 4:
+ rot_mat = mat[..., :-1, :-1]
+ else:
+ rot_mat = mat
+ if isinstance(mat, torch.Tensor):
+ rot_mat_norm = rot_mat / (rot_mat.norm(2, dim=-2, keepdim=True) + 1e-9)
+ norm_mat = torch.zeros_like(mat)
+ elif isinstance(mat, np.ndarray):
+ rot_mat_norm = rot_mat / (np.linalg.norm(rot_mat, ord=2, axis=-2, keepdims=True) + 1e-9)
+ norm_mat = np.zeros_like(mat)
+ else:
+ raise ValueError
+ if mat.shape[-1] == 4:
+ norm_mat[..., :-1, :-1] = rot_mat_norm
+ norm_mat[..., :-1, -1] = mat[..., :-1, -1]
+ norm_mat[..., -1, -1] = 1.0
+ else:
+ norm_mat = rot_mat_norm
+ return norm_mat
+
+
+def get_rot_mat_from_forward(forward):
+ """_summary_
+
+ Args:
+ forward (tensor): [N, M, 3]
+
+ Returns:
+ mat (tensor): [N, M, 3, 3]
+ """
+ if isinstance(forward, torch.Tensor):
+ mat = torch.eye(3, device=forward.device).repeat(forward.shape[:-1] + (1, 1))
+ right = torch.zeros_like(forward)
+ elif isinstance(forward, np.ndarray):
+ mat = np.eye(3, dtype=np.float32)
+ for _ in range(len(forward.shape) - 1):
+ mat = mat[None]
+ mat = np.tile(mat, forward.shape[:-1] + (1, 1))
+ right = np.zeros_like(forward)
+ else:
+ raise ValueError
+
+ right[..., 0] = forward[..., 2]
+ right[..., 1] = 0.0
+ right[..., 2] = -forward[..., 0]
+ # right = torch.cross(mat[..., 1], forward) # cannot backward
+
+ mat[..., 2] = normalized(forward)
+ right = normalized(right)
+ mat[..., 0] = right
+ return mat
+
+
+def get_rot_mat_from_forward_up(forward, up):
+ """_summary_
+
+ Args:
+ forward (tensor): [N, M, 3]
+ up (tensor): [N, M, 3]
+
+ Returns:
+ mat (tensor): [N, M, 3, 3]
+ """
+ if isinstance(forward, torch.Tensor):
+ mat = torch.eye(3, device=forward.device).repeat(forward.shape[:-1] + (1, 1))
+ right = torch.cross(up, forward)
+ elif isinstance(forward, np.ndarray):
+ mat = np.eye(3, dtype=np.float32)
+ for _ in range(len(forward.shape) - 1):
+ mat = mat[None]
+ mat = np.tile(mat, forward.shape[:-1] + (1, 1))
+ right = np.cross(up, forward)
+ else:
+ raise ValueError
+
+ right = normalized(right)
+ mat[..., 2] = normalized(forward)
+ mat[..., 1] = normalized(up)
+ mat[..., 0] = right
+ return mat
+
+
+def get_rot_mat_from_pose_vec(vec):
+ """_summary_
+
+ Args:
+ vec (tensor): [N, M, 6]
+
+ Returns:
+ mat (tensor): [N, M, 3, 3]
+ """
+ forward = vec[..., :3]
+ up = vec[..., 3:6]
+ return get_rot_mat_from_forward_up(forward, up)
+
+
+def get_TRS(rot_mat, pos):
+ """_summary_
+
+ Args:
+ rot_mat (tensor): [N, 3, 3]
+ pos (tensor): [N, 3]
+
+ Returns:
+ mat (tensor): [N, 4, 4]
+ """
+ if isinstance(rot_mat, torch.Tensor):
+ mat = torch.eye(4, device=pos.device).repeat(pos.shape[:-1] + (1, 1))
+ elif isinstance(rot_mat, np.ndarray):
+ mat = np.eye(4, dtype=np.float32)
+ for _ in range(len(pos.shape) - 1):
+ mat = mat[None]
+ mat = np.tile(mat, pos.shape[:-1] + (1, 1))
+ else:
+ raise ValueError
+ mat[..., :3, :3] = rot_mat
+ mat[..., :3, 3] = pos
+ mat = normalized_matrix(mat)
+ return mat
+
+
+def xzvec2mat(vec):
+ """_summary_
+
+ Args:
+ vec (tensor): [N, 4]
+
+ Returns:
+ mat (tensor): [N, 4, 4]
+ """
+ vec_shape = vec.shape[:-1]
+ if isinstance(vec, torch.Tensor):
+ pos = torch.zeros(vec_shape + (3,))
+ forward = torch.zeros(vec_shape + (3,))
+ elif isinstance(vec, np.ndarray):
+ pos = np.zeros(vec_shape + (3,))
+ forward = np.zeros(vec_shape + (3,))
+ else:
+ raise ValueError
+
+ pos[..., 0] = vec[..., 0]
+ pos[..., 2] = vec[..., 1]
+ forward[..., 0] = vec[..., 2]
+ forward[..., 2] = vec[..., 3]
+ rot_mat = get_rot_mat_from_forward(forward)
+ mat = get_TRS(rot_mat, pos)
+ return mat
+
+
+def distance(vec1, vec2):
+ return ((vec1 - vec2) ** 2).sum() ** 0.5
+
+
+def get_relative_pose_from_vec(pose, root, N):
+ root_p_mat = xzvec2mat(root)
+ pose = pose.reshape(-1, N, 12)
+ pose[..., :3] = get_position_from(pose[..., :3], root_p_mat)
+ pose[..., 3:6] = get_direction_from(pose[..., 3:6], root_p_mat)
+ pose[..., 6:9] = get_direction_from(pose[..., 6:9], root_p_mat)
+ pose[..., 9:] = get_direction_from(pose[..., 9:], root_p_mat)
+ pos = pose[..., 0, :3]
+ rot = pose[..., 3:9].reshape(-1, N * 6)
+ pose = np.concatenate((pos, rot), axis=-1)
+ return pose
+
+
+def get_forward_from_pos(pos):
+ """_summary_
+
+ Args:
+ pos (N, J, 3): joints positions of each frame
+
+ Returns:
+ _type_: _description_
+ """
+
+ pos_y_vec = torch.tensor([0, 1, 0], dtype=torch.float32).to(pos.device)
+ face_joint_indx = [2, 1, 17, 16]
+ r_hip, l_hip, r_sdr, l_sdr = face_joint_indx # use hip and shoulder to get the cross vector
+ cross_hip = pos[..., 0, r_hip, :] - pos[..., 0, l_hip, :]
+ cross_sdr = pos[..., 0, r_sdr, :] - pos[..., 0, l_sdr, :]
+ cross_vec = cross_hip + cross_sdr # (3, )
+ forward_vec = torch.cross(pos_y_vec, cross_vec, dim=-1)
+ forward_vec = normalized(forward_vec)
+ return forward_vec
+
+
+def project_point_along_ray(p, ray, keepnorm=False):
+ """_summary_
+
+ Args:
+ p (*, 3): point positions
+ ray (*, 3): ray direction
+ keepnorm: False -> project point on the ray,
+ True -> project point on the ray and keep the point length
+
+ Returns:
+ _type_: _description_
+ """
+ ray = normalized(ray)
+ if keepnorm:
+ new_p = ray * p.norm(dim=-1, keepdim=True)
+ else:
+ dot_product = torch.sum(p * ray, dim=-1, keepdim=True)
+ new_p = dot_product * ray
+ return new_p
+
+
+def solve_point_along_ray_with_constraint(c, ray, p, constraint="x"):
+ """_summary_
+
+ Args:
+ c (*,): constraint value
+ ray (*, 3): ray direction
+ p (*, 3): start point of the ray
+
+ Returns:
+ _type_: _description_
+ """
+ ray = normalized(ray)
+ if constraint == "x":
+ ind = 0
+ elif constraint == "y":
+ ind = 1
+ elif constraint == "z":
+ ind = 2
+ else:
+ raise ValueError
+ t = (c - p[..., ind]) / ray[..., ind]
+ out_p = ray * t[..., None] + p
+
+ return out_p
+
+
+def calc_cosine(vec1, vec2, return_angle=False):
+ """_summary_
+
+ Args:
+ vec1 (*, 3): vector
+ vec2 (*, 3): vector
+ return_angle: True -> return angle, False -> return cosine
+
+ Returns:
+ _type_: _description_
+ """
+ vec1 = normalized(vec1)
+ vec2 = normalized(vec2)
+ cosine = torch.sum(vec1 * vec2, dim=-1)
+ if return_angle:
+ return torch.acos(cosine)
+ return cosine
+
+
+############################################
+#
+# quaternion assumes xyzw
+#
+############################################
+
+
+def quat_xyzw2wxyz(quat):
+ new_quat = torch.cat([quat[..., 3:4], quat[..., :3]], dim=-1)
+ return new_quat
+
+
+def quat_wxyz2xyzw(quat):
+ new_quat = torch.cat([quat[..., 1:4], quat[..., :1]], dim=-1)
+ return new_quat
+
+
+def quat_mul(a, b):
+ """
+ quaternion multiplication
+ """
+ x1, y1, z1, w1 = a[..., 0], a[..., 1], a[..., 2], a[..., 3]
+ x2, y2, z2, w2 = b[..., 0], b[..., 1], b[..., 2], b[..., 3]
+
+ w = w1 * w2 - x1 * x2 - y1 * y2 - z1 * z2
+ x = w1 * x2 + x1 * w2 + y1 * z2 - z1 * y2
+ y = w1 * y2 + y1 * w2 + z1 * x2 - x1 * z2
+ z = w1 * z2 + z1 * w2 + x1 * y2 - y1 * x2
+
+ return torch.stack([x, y, z, w], dim=-1)
+
+
+def quat_pos(x):
+ """
+ make all the real part of the quaternion positive
+ """
+ q = x
+ z = (q[..., 3:] < 0).float()
+ q = (1 - 2 * z) * q
+ return q
+
+
+def quat_abs(x):
+ """
+ quaternion norm (unit quaternion represents a 3D rotation, which has norm of 1)
+ """
+ x = x.norm(p=2, dim=-1)
+ return x
+
+
+def quat_unit(x):
+ """
+ normalized quaternion with norm of 1
+ """
+ norm = quat_abs(x).unsqueeze(-1)
+ return x / (norm.clamp(min=1e-4))
+
+
+def quat_conjugate(x):
+ """
+ quaternion with its imaginary part negated
+ """
+ return torch.cat([-x[..., :3], x[..., 3:]], dim=-1)
+
+
+def quat_real(x):
+ """
+ real component of the quaternion
+ """
+ return x[..., 3]
+
+
+def quat_imaginary(x):
+ """
+ imaginary components of the quaternion
+ """
+ return x[..., :3]
+
+
+def quat_norm_check(x):
+ """
+ verify that a quaternion has norm 1
+ """
+ assert bool((abs(x.norm(p=2, dim=-1) - 1) < 1e-3).all()), "the quaternion is has non-1 norm: {}".format(
+ abs(x.norm(p=2, dim=-1) - 1)
+ )
+ assert bool((x[..., 3] >= 0).all()), "the quaternion has negative real part"
+
+
+def quat_normalize(q):
+ """
+ Construct 3D rotation from quaternion (the quaternion needs not to be normalized).
+ """
+ q = quat_unit(quat_pos(q)) # normalized to positive and unit quaternion
+ return q
+
+
+def quat_from_xyz(xyz):
+ """
+ Construct 3D rotation from the imaginary component
+ """
+ w = (1.0 - xyz.norm()).unsqueeze(-1)
+ assert bool((w >= 0).all()), "xyz has its norm greater than 1"
+ return torch.cat([xyz, w], dim=-1)
+
+
+def quat_identity(shape: List[int]):
+ """
+ Construct 3D identity rotation given shape
+ """
+ w = torch.ones(shape + (1,))
+ xyz = torch.zeros(shape + (3,))
+ q = torch.cat([xyz, w], dim=-1)
+ return quat_normalize(q)
+
+
+def tgm_quat_from_angle_axis(angle, axis, degree: bool = False):
+ """Create a 3D rotation from angle and axis of rotation. The rotation is counter-clockwise
+ along the axis.
+
+ The rotation can be interpreted as a_R_b where frame "b" is the new frame that
+ gets rotated counter-clockwise along the axis from frame "a"
+
+ :param angle: angle of rotation
+ :type angle: Tensor
+ :param axis: axis of rotation
+ :type axis: Tensor
+ :param degree: put True here if the angle is given by degree
+ :type degree: bool, optional, default=False
+ """
+ if degree:
+ angle = angle / 180.0 * math.pi
+ theta = (angle / 2).unsqueeze(-1)
+ axis = axis / (axis.norm(p=2, dim=-1, keepdim=True).clamp(min=1e-4))
+ xyz = axis * theta.sin()
+ w = theta.cos()
+ return quat_normalize(torch.cat([w, xyz], dim=-1))
+
+
+def quat_from_rotation_matrix(m):
+ """
+ Construct a 3D rotation from a valid 3x3 rotation matrices.
+ Reference can be found here:
+ http://www.cg.info.hiroshima-cu.ac.jp/~miyazaki/knowledge/teche52.html
+
+ :param m: 3x3 orthogonal rotation matrices.
+ :type m: Tensor
+
+ :rtype: Tensor
+ """
+ m = m.unsqueeze(0)
+ diag0 = m[..., 0, 0]
+ diag1 = m[..., 1, 1]
+ diag2 = m[..., 2, 2]
+
+ # Math stuff.
+ w = (((diag0 + diag1 + diag2 + 1.0) / 4.0).clamp(0.0, None)) ** 0.5
+ x = (((diag0 - diag1 - diag2 + 1.0) / 4.0).clamp(0.0, None)) ** 0.5
+ y = (((-diag0 + diag1 - diag2 + 1.0) / 4.0).clamp(0.0, None)) ** 0.5
+ z = (((-diag0 - diag1 + diag2 + 1.0) / 4.0).clamp(0.0, None)) ** 0.5
+
+ # Only modify quaternions where w > x, y, z.
+ c0 = (w >= x) & (w >= y) & (w >= z)
+ x[c0] *= (m[..., 2, 1][c0] - m[..., 1, 2][c0]).sign()
+ y[c0] *= (m[..., 0, 2][c0] - m[..., 2, 0][c0]).sign()
+ z[c0] *= (m[..., 1, 0][c0] - m[..., 0, 1][c0]).sign()
+
+ # Only modify quaternions where x > w, y, z
+ c1 = (x >= w) & (x >= y) & (x >= z)
+ w[c1] *= (m[..., 2, 1][c1] - m[..., 1, 2][c1]).sign()
+ y[c1] *= (m[..., 1, 0][c1] + m[..., 0, 1][c1]).sign()
+ z[c1] *= (m[..., 0, 2][c1] + m[..., 2, 0][c1]).sign()
+
+ # Only modify quaternions where y > w, x, z.
+ c2 = (y >= w) & (y >= x) & (y >= z)
+ w[c2] *= (m[..., 0, 2][c2] - m[..., 2, 0][c2]).sign()
+ x[c2] *= (m[..., 1, 0][c2] + m[..., 0, 1][c2]).sign()
+ z[c2] *= (m[..., 2, 1][c2] + m[..., 1, 2][c2]).sign()
+
+ # Only modify quaternions where z > w, x, y.
+ c3 = (z >= w) & (z >= x) & (z >= y)
+ w[c3] *= (m[..., 1, 0][c3] - m[..., 0, 1][c3]).sign()
+ x[c3] *= (m[..., 2, 0][c3] + m[..., 0, 2][c3]).sign()
+ y[c3] *= (m[..., 2, 1][c3] + m[..., 1, 2][c3]).sign()
+
+ return quat_normalize(torch.stack([x, y, z, w], dim=-1)).squeeze(0)
+
+
+def quat_mul_norm(x, y):
+ """
+ Combine two set of 3D rotations together using \**\* operator. The shape needs to be
+ broadcastable
+ """
+ return quat_normalize(quat_mul(x, y))
+
+
+def quat_rotate(rot, vec):
+ """
+ Rotate a 3D vector with the 3D rotation
+ """
+ other_q = torch.cat([vec, torch.zeros_like(vec[..., :1])], dim=-1)
+ return quat_imaginary(quat_mul(quat_mul(rot, other_q), quat_conjugate(rot)))
+
+
+def quat_inverse(x):
+ """
+ The inverse of the rotation
+ """
+ return quat_conjugate(x)
+
+
+def quat_identity_like(x):
+ """
+ Construct identity 3D rotation with the same shape
+ """
+ return quat_identity(x.shape[:-1])
+
+
+def quat_angle_axis(x):
+ """
+ The (angle, axis) representation of the rotation. The axis is normalized to unit length.
+ The angle is guaranteed to be between [0, pi].
+ """
+ s = 2 * (x[..., 3] ** 2) - 1
+ angle = s.clamp(-1, 1).arccos() # just to be safe
+ axis = x[..., :3]
+ axis /= axis.norm(p=2, dim=-1, keepdim=True).clamp(min=1e-4)
+ return angle, axis
+
+
+def quat_yaw_rotation(x, z_up: bool = True):
+ """
+ Yaw rotation (rotation along z-axis)
+ """
+ q = x
+ if z_up:
+ q = torch.cat([torch.zeros_like(q[..., 0:2]), q[..., 2:3], q[..., 3:]], dim=-1)
+ else:
+ q = torch.cat(
+ [
+ torch.zeros_like(q[..., 0:1]),
+ q[..., 1:2],
+ torch.zeros_like(q[..., 2:3]),
+ q[..., 3:4],
+ ],
+ dim=-1,
+ )
+ return quat_normalize(q)
+
+
+def transform_from_rotation_translation(r: Optional[torch.Tensor] = None, t: Optional[torch.Tensor] = None):
+ """
+ Construct a transform from a quaternion and 3D translation. Only one of them can be None.
+ """
+ assert r is not None or t is not None, "rotation and translation can't be all None"
+ if r is None:
+ assert t is not None
+ r = quat_identity(list(t.shape))
+ if t is None:
+ t = torch.zeros(list(r.shape) + [3])
+ return torch.cat([r, t], dim=-1)
+
+
+def transform_identity(shape: List[int]):
+ """
+ Identity transformation with given shape
+ """
+ r = quat_identity(shape)
+ t = torch.zeros(shape + [3])
+ return transform_from_rotation_translation(r, t)
+
+
+def transform_rotation(x):
+ """Get rotation from transform"""
+ return x[..., :4]
+
+
+def transform_translation(x):
+ """Get translation from transform"""
+ return x[..., 4:]
+
+
+def transform_inverse(x):
+ """
+ Inverse transformation
+ """
+ inv_so3 = quat_inverse(transform_rotation(x))
+ return transform_from_rotation_translation(r=inv_so3, t=quat_rotate(inv_so3, -transform_translation(x)))
+
+
+def transform_identity_like(x):
+ """
+ identity transformation with the same shape
+ """
+ return transform_identity(x.shape)
+
+
+def transform_mul(x, y):
+ """
+ Combine two transformation together
+ """
+ z = transform_from_rotation_translation(
+ r=quat_mul_norm(transform_rotation(x), transform_rotation(y)),
+ t=quat_rotate(transform_rotation(x), transform_translation(y)) + transform_translation(x),
+ )
+ return z
+
+
+def transform_apply(rot, vec):
+ """
+ Transform a 3D vector
+ """
+ assert isinstance(vec, torch.Tensor)
+ return quat_rotate(transform_rotation(rot), vec) + transform_translation(rot)
+
+
+def rot_matrix_det(x):
+ """
+ Return the determinant of the 3x3 matrix. The shape of the tensor will be as same as the
+ shape of the matrix
+ """
+ a, b, c = x[..., 0, 0], x[..., 0, 1], x[..., 0, 2]
+ d, e, f = x[..., 1, 0], x[..., 1, 1], x[..., 1, 2]
+ g, h, i = x[..., 2, 0], x[..., 2, 1], x[..., 2, 2]
+ t1 = a * (e * i - f * h)
+ t2 = b * (d * i - f * g)
+ t3 = c * (d * h - e * g)
+ return t1 - t2 + t3
+
+
+def rot_matrix_integrity_check(x):
+ """
+ Verify that a rotation matrix has a determinant of one and is orthogonal
+ """
+ det = rot_matrix_det(x)
+ assert bool((abs(det - 1) < 1e-3).all()), "the matrix has non-one determinant"
+ rtr = x @ x.permute(torch.arange(x.dim() - 2), -1, -2)
+ rtr_gt = rtr.zeros_like()
+ rtr_gt[..., 0, 0] = 1
+ rtr_gt[..., 1, 1] = 1
+ rtr_gt[..., 2, 2] = 1
+ assert bool(((rtr - rtr_gt) < 1e-3).all()), "the matrix is not orthogonal"
+
+
+def rot_matrix_from_quaternion(q):
+ """
+ Construct rotation matrix from quaternion
+ """
+ # Shortcuts for individual elements (using wikipedia's convention)
+ qi, qj, qk, qr = q[..., 0], q[..., 1], q[..., 2], q[..., 3]
+
+ # Set individual elements
+ R00 = 1.0 - 2.0 * (qj**2 + qk**2)
+ R01 = 2 * (qi * qj - qk * qr)
+ R02 = 2 * (qi * qk + qj * qr)
+ R10 = 2 * (qi * qj + qk * qr)
+ R11 = 1.0 - 2.0 * (qi**2 + qk**2)
+ R12 = 2 * (qj * qk - qi * qr)
+ R20 = 2 * (qi * qk - qj * qr)
+ R21 = 2 * (qj * qk + qi * qr)
+ R22 = 1.0 - 2.0 * (qi**2 + qj**2)
+
+ R0 = torch.stack([R00, R01, R02], dim=-1)
+ R1 = torch.stack([R10, R11, R12], dim=-1)
+ R2 = torch.stack([R20, R21, R22], dim=-1)
+
+ R = torch.stack([R0, R1, R2], dim=-2)
+
+ return R
+
+
+def euclidean_to_rotation_matrix(x):
+ """
+ Get the rotation matrix on the top-left corner of a Euclidean transformation matrix
+ """
+ return x[..., :3, :3]
+
+
+def euclidean_integrity_check(x):
+ euclidean_to_rotation_matrix(x) # check 3d-rotation matrix
+ assert bool((x[..., 3, :3] == 0).all()), "the last row is illegal"
+ assert bool((x[..., 3, 3] == 1).all()), "the last row is illegal"
+
+
+def euclidean_translation(x):
+ """
+ Get the translation vector located at the last column of the matrix
+ """
+ return x[..., :3, 3]
+
+
+def euclidean_inverse(x):
+ """
+ Compute the matrix that represents the inverse rotation
+ """
+ s = x.zeros_like()
+ irot = quat_inverse(quat_from_rotation_matrix(x))
+ s[..., :3, :3] = irot
+ s[..., :3, 4] = quat_rotate(irot, -euclidean_translation(x))
+ return s
+
+
+def euclidean_to_transform(transformation_matrix):
+ """
+ Construct a transform from a Euclidean transformation matrix
+ """
+ return transform_from_rotation_translation(
+ r=quat_from_rotation_matrix(m=euclidean_to_rotation_matrix(transformation_matrix)),
+ t=euclidean_translation(transformation_matrix),
+ )
+
+
+def to_torch(x, dtype=torch.float, device="cuda:0", requires_grad=False):
+ return torch.tensor(x, dtype=dtype, device=device, requires_grad=requires_grad)
+
+
+def quat_mul(a, b):
+ assert a.shape == b.shape
+ shape = a.shape
+ a = a.reshape(-1, 4)
+ b = b.reshape(-1, 4)
+
+ x1, y1, z1, w1 = a[:, 0], a[:, 1], a[:, 2], a[:, 3]
+ x2, y2, z2, w2 = b[:, 0], b[:, 1], b[:, 2], b[:, 3]
+ ww = (z1 + x1) * (x2 + y2)
+ yy = (w1 - y1) * (w2 + z2)
+ zz = (w1 + y1) * (w2 - z2)
+ xx = ww + yy + zz
+ qq = 0.5 * (xx + (z1 - x1) * (x2 - y2))
+ w = qq - ww + (z1 - y1) * (y2 - z2)
+ x = qq - xx + (x1 + w1) * (x2 + w2)
+ y = qq - yy + (w1 - x1) * (y2 + z2)
+ z = qq - zz + (z1 + y1) * (w2 - x2)
+
+ quat = torch.stack([x, y, z, w], dim=-1).view(shape)
+
+ return quat
+
+
+def normalize(x, eps: float = 1e-9):
+ return x / x.norm(p=2, dim=-1).clamp(min=eps, max=None).unsqueeze(-1)
+
+
+def quat_apply(a, b):
+ shape = b.shape
+ a = a.reshape(-1, 4)
+ b = b.reshape(-1, 3)
+ xyz = a[:, :3]
+ t = xyz.cross(b, dim=-1) * 2
+ return (b + a[:, 3:] * t + xyz.cross(t, dim=-1)).view(shape)
+
+
+def quat_rotate(q, v):
+ shape = q.shape
+ q_w = q[:, -1]
+ q_vec = q[:, :3]
+ a = v * (2.0 * q_w**2 - 1.0).unsqueeze(-1)
+ b = torch.cross(q_vec, v, dim=-1) * q_w.unsqueeze(-1) * 2.0
+ c = q_vec * torch.bmm(q_vec.view(shape[0], 1, 3), v.view(shape[0], 3, 1)).squeeze(-1) * 2.0
+ return a + b + c
+
+
+def quat_rotate_inverse(q, v):
+ shape = q.shape
+ q_w = q[:, -1]
+ q_vec = q[:, :3]
+ a = v * (2.0 * q_w**2 - 1.0).unsqueeze(-1)
+ b = torch.cross(q_vec, v, dim=-1) * q_w.unsqueeze(-1) * 2.0
+ c = q_vec * torch.bmm(q_vec.view(shape[0], 1, 3), v.view(shape[0], 3, 1)).squeeze(-1) * 2.0
+ return a - b + c
+
+
+def quat_conjugate(a):
+ shape = a.shape
+ a = a.reshape(-1, 4)
+ return torch.cat((-a[:, :3], a[:, -1:]), dim=-1).view(shape)
+
+
+def quat_unit(a):
+ return normalize(a)
+
+
+def quat_from_angle_axis(angle, axis):
+ theta = (angle / 2).unsqueeze(-1)
+ xyz = normalize(axis) * torch.sin(theta.clone())
+ w = torch.cos(theta.clone())
+ return quat_unit(torch.cat([xyz, w], dim=-1))
+
+
+def normalize_angle(x):
+ return torch.atan2(torch.sin(x.clone()), torch.cos(x.clone()))
+
+
+def tf_inverse(q, t):
+ q_inv = quat_conjugate(q)
+ return q_inv, -quat_apply(q_inv, t)
+
+
+def tf_apply(q, t, v):
+ return quat_apply(q, v) + t
+
+
+def tf_vector(q, v):
+ return quat_apply(q, v)
+
+
+def tf_combine(q1, t1, q2, t2):
+ return quat_mul(q1, q2), quat_apply(q1, t2) + t1
+
+
+def get_basis_vector(q, v):
+ return quat_rotate(q, v)
+
+
+def get_axis_params(value, axis_idx, x_value=0.0, dtype=float, n_dims=3):
+ """construct arguments to `Vec` according to axis index."""
+ zs = np.zeros((n_dims,))
+ assert axis_idx < n_dims, "the axis dim should be within the vector dimensions"
+ zs[axis_idx] = 1.0
+ params = np.where(zs == 1.0, value, zs)
+ params[0] = x_value
+ return list(params.astype(dtype))
+
+
+def copysign(a, b):
+ # type: (float, Tensor) -> Tensor
+ a = torch.tensor(a, device=b.device, dtype=torch.float).repeat(b.shape[0])
+ return torch.abs(a) * torch.sign(b)
+
+
+def get_euler_xyz(q):
+ qx, qy, qz, qw = 0, 1, 2, 3
+ # roll (x-axis rotation)
+ sinr_cosp = 2.0 * (q[:, qw] * q[:, qx] + q[:, qy] * q[:, qz])
+ cosr_cosp = q[:, qw] * q[:, qw] - q[:, qx] * q[:, qx] - q[:, qy] * q[:, qy] + q[:, qz] * q[:, qz]
+ roll = torch.atan2(sinr_cosp, cosr_cosp)
+
+ # pitch (y-axis rotation)
+ sinp = 2.0 * (q[:, qw] * q[:, qy] - q[:, qz] * q[:, qx])
+ pitch = torch.where(torch.abs(sinp) >= 1, copysign(np.pi / 2.0, sinp), torch.asin(sinp))
+
+ # yaw (z-axis rotation)
+ siny_cosp = 2.0 * (q[:, qw] * q[:, qz] + q[:, qx] * q[:, qy])
+ cosy_cosp = q[:, qw] * q[:, qw] + q[:, qx] * q[:, qx] - q[:, qy] * q[:, qy] - q[:, qz] * q[:, qz]
+ yaw = torch.atan2(siny_cosp, cosy_cosp)
+
+ return roll % (2 * np.pi), pitch % (2 * np.pi), yaw % (2 * np.pi)
+
+
+def quat_from_euler_xyz(roll, pitch, yaw):
+ cy = torch.cos(yaw * 0.5)
+ sy = torch.sin(yaw * 0.5)
+ cr = torch.cos(roll * 0.5)
+ sr = torch.sin(roll * 0.5)
+ cp = torch.cos(pitch * 0.5)
+ sp = torch.sin(pitch * 0.5)
+
+ qw = cy * cr * cp + sy * sr * sp
+ qx = cy * sr * cp - sy * cr * sp
+ qy = cy * cr * sp + sy * sr * cp
+ qz = sy * cr * cp - cy * sr * sp
+
+ return torch.stack([qx, qy, qz, qw], dim=-1)
+
+
+def torch_rand_float(lower, upper, shape, device):
+ # type: (float, float, Tuple[int, int], str) -> Tensor
+ return (upper - lower) * torch.rand(*shape, device=device) + lower
+
+
+def torch_random_dir_2(shape, device):
+ # type: (Tuple[int, int], str) -> Tensor
+ angle = torch_rand_float(-np.pi, np.pi, shape, device).squeeze(-1)
+ return torch.stack([torch.cos(angle), torch.sin(angle)], dim=-1)
+
+
+def tensor_clamp(t, min_t, max_t):
+ return torch.max(torch.min(t, max_t), min_t)
+
+
+def scale(x, lower, upper):
+ return 0.5 * (x + 1.0) * (upper - lower) + lower
+
+
+def unscale(x, lower, upper):
+ return (2.0 * x - upper - lower) / (upper - lower)
+
+
+def unscale_np(x, lower, upper):
+ return (2.0 * x - upper - lower) / (upper - lower)
+
+
+def quat_to_angle_axis(q):
+ # type: (Tensor) -> Tuple[Tensor, Tensor]
+ # computes axis-angle representation from quaternion q
+ # q must be normalized
+ min_theta = 1e-5
+ qx, qy, qz, qw = 0, 1, 2, 3
+
+ sin_theta = torch.sqrt(1 - q[..., qw] * q[..., qw])
+ angle = 2 * torch.acos(q[..., qw])
+ angle = normalize_angle(angle)
+ sin_theta_expand = sin_theta.unsqueeze(-1)
+ axis = q[..., qx:qw] / sin_theta_expand
+
+ mask = torch.abs(sin_theta) > min_theta
+ default_axis = torch.zeros_like(axis)
+ default_axis[..., -1] = 1
+
+ angle = torch.where(mask, angle, torch.zeros_like(angle))
+ mask_expand = mask.unsqueeze(-1)
+ axis = torch.where(mask_expand, axis, default_axis)
+ return angle, axis
+
+
+def angle_axis_to_exp_map(angle, axis):
+ # type: (Tensor, Tensor) -> Tensor
+ # compute exponential map from axis-angle
+ angle_expand = angle.unsqueeze(-1)
+ exp_map = angle_expand * axis
+ return exp_map
+
+
+def quat_to_exp_map(q):
+ # type: (Tensor) -> Tensor
+ # compute exponential map from quaternion
+ # q must be normalized
+ angle, axis = quat_to_angle_axis(q)
+ exp_map = angle_axis_to_exp_map(angle, axis)
+ return exp_map
+
+
+def quat_to_tan_norm(q):
+ # type: (Tensor) -> Tensor
+ # represents a rotation using the tangent and normal vectors
+ ref_tan = torch.zeros_like(q[..., 0:3])
+ ref_tan[..., 0] = 1
+ tan = quat_rotate(q, ref_tan)
+
+ ref_norm = torch.zeros_like(q[..., 0:3])
+ ref_norm[..., -1] = 1
+ norm = quat_rotate(q, ref_norm)
+
+ norm_tan = torch.cat([tan, norm], dim=len(tan.shape) - 1)
+ return norm_tan
+
+
+def euler_xyz_to_exp_map(roll, pitch, yaw):
+ # type: (Tensor, Tensor, Tensor) -> Tensor
+ q = quat_from_euler_xyz(roll, pitch, yaw)
+ exp_map = quat_to_exp_map(q)
+ return exp_map
+
+
+def exp_map_to_angle_axis(exp_map):
+ min_theta = 1e-5
+
+ angle = torch.norm(exp_map.clone(), dim=-1) + 1e-6
+ angle_exp = torch.unsqueeze(angle, dim=-1)
+ axis = exp_map.clone() / angle_exp.clone()
+ angle = normalize_angle(angle)
+
+ default_axis = torch.zeros_like(exp_map)
+ default_axis[..., -1] = 1
+
+ mask = torch.abs(angle) > min_theta
+ angle = torch.where(mask, angle, torch.zeros_like(angle))
+ mask_expand = mask.unsqueeze(-1)
+ axis = torch.where(mask_expand, axis, default_axis)
+
+ return angle, axis
+
+
+def exp_map_to_quat(exp_map):
+ angle, axis = exp_map_to_angle_axis(exp_map)
+ q = quat_from_angle_axis(angle, axis)
+ return q
+
+
+def slerp(q0, q1, t):
+ # type: (Tensor, Tensor, Tensor) -> Tensor
+ cos_half_theta = torch.sum(q0 * q1, dim=-1)
+
+ neg_mask = cos_half_theta < 0
+ q1 = q1.clone()
+ q1[neg_mask] = -q1[neg_mask]
+ cos_half_theta = torch.abs(cos_half_theta)
+ cos_half_theta = torch.unsqueeze(cos_half_theta, dim=-1)
+
+ half_theta = torch.acos(cos_half_theta)
+ sin_half_theta = torch.sqrt(1.0 - cos_half_theta * cos_half_theta)
+
+ ratioA = torch.sin((1 - t) * half_theta) / sin_half_theta
+ ratioB = torch.sin(t * half_theta) / sin_half_theta
+
+ new_q = ratioA * q0 + ratioB * q1
+
+ new_q = torch.where(torch.abs(sin_half_theta) < 0.001, 0.5 * q0 + 0.5 * q1, new_q)
+ new_q = torch.where(torch.abs(cos_half_theta) >= 1, q0, new_q)
+
+ return new_q
+
+
+def calc_heading_vec(q, head_ind=0):
+ # type: (Tensor, int) -> Tensor
+ # calculate heading direction from quaternion
+ # the heading is the direction vector
+ # q must be normalized
+ ref_dir = torch.zeros_like(q[..., 0:3])
+ ref_dir[..., head_ind] = 1
+ rot_dir = quat_rotate(q, ref_dir)
+
+ return rot_dir
+
+
+def calc_heading(q, head_ind=0, gravity_axis="z"):
+ # type: (Tensor, int, str) -> Tensor
+ # calculate heading direction from quaternion
+ # the heading is the direction on the xy plane
+ # q must be normalized
+ ref_dir = torch.zeros_like(q[..., 0:3])
+ ref_dir[..., head_ind] = 1
+ # ref_dir[..., 0] = 1
+ shape = ref_dir.shape[:-1]
+ q = q.reshape((-1, 4))
+ ref_dir = ref_dir.reshape(-1, 3)
+ rot_dir = quat_rotate(q, ref_dir)
+ rot_dir = rot_dir.reshape(shape + (3,))
+ if gravity_axis == "z":
+ heading = torch.atan2(rot_dir[..., 1], rot_dir[..., 0])
+ elif gravity_axis == "y":
+ heading = torch.atan2(rot_dir[..., 0], rot_dir[..., 2])
+ elif gravity_axis == "x":
+ heading = torch.atan2(rot_dir[..., 2], rot_dir[..., 1])
+ return heading
+
+
+def calc_heading_quat(q, head_ind=0, gravity_axis="z"):
+ # type: (Tensor, int, str) -> Tensor
+ # calculate heading rotation from quaternion
+ # the heading is the direction on the xy plane
+ # q must be normalized
+ heading = calc_heading(q, head_ind, gravity_axis=gravity_axis)
+ axis = torch.zeros_like(q[..., 0:3])
+ if gravity_axis == "z":
+ g_axis = 2
+ elif gravity_axis == "y":
+ g_axis = 1
+ elif gravity_axis == "x":
+ g_axis = 0
+ axis[..., g_axis] = 1
+
+ heading_q = quat_from_angle_axis(heading, axis)
+ return heading_q
+
+
+def calc_heading_quat_inv(q, head_ind=0):
+ # type: (Tensor, int) -> Tensor
+ # calculate heading rotation from quaternion
+ # the heading is the direction on the xy plane
+ # q must be normalized
+ heading = calc_heading(q, head_ind)
+ axis = torch.zeros_like(q[..., 0:3])
+ axis[..., 2] = 1
+
+ heading_q = quat_from_angle_axis(-heading, axis)
+ return heading_q
+
+
+def forward_kinematics(mat, parent):
+ """_summary_
+
+ Args:
+ mat ([..., N, 3, 3]): _description_
+ parent (): _description_
+ """
+ if isinstance(mat, torch.Tensor):
+ rotations = torch.eye(mat.shape[-1], device=mat.device)
+ rotations = rotations.repeat(mat.shape[:-2] + (1, 1))
+ else:
+ rotations = np.eye(mat.shape[-1], dtype=np.float32)
+ rotations = np.tile(rotations, mat.shape[:-2] + (1, 1))
+ for i in range(mat.shape[-3]):
+ if parent[i] != -1:
+ if isinstance(mat, torch.Tensor):
+ # this way make gradient flow
+ new_mat = get_mat_BfromA(rotations[..., parent[i], :, :], mat[..., i, :, :])
+ rotations = torch.cat(
+ (
+ rotations[..., :i, :, :],
+ new_mat[..., None, :, :],
+ rotations[..., i + 1 :, :, :],
+ ),
+ dim=-3,
+ )
+ else:
+ rotations[..., i, :, :] = get_mat_BfromA(rotations[..., parent[i], :, :], mat[..., i, :, :])
+ else:
+ if isinstance(mat, torch.Tensor):
+ # this way make gradient flow
+ rotations = torch.cat((mat[..., : i + 1, :, :], rotations[..., i + 1 :, :, :]), dim=-3)
+ else:
+ rotations[..., i, :, :] = mat[..., i, :, :]
+ return rotations
diff --git a/third_party/GVHMR/hmr4d/utils/net_utils.py b/third_party/GVHMR/hmr4d/utils/net_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..698d61b8934230f55ff960a37ad9002a90600aea
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/net_utils.py
@@ -0,0 +1,185 @@
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from pathlib import Path
+from hmr4d.utils.pylogger import Log
+from pytorch_lightning.utilities.memory import recursive_detach
+from einops import repeat, rearrange
+from scipy.ndimage._filters import _gaussian_kernel1d
+
+
+def load_pretrained_model(model, ckpt_path):
+ """
+ Load ckpt to model with strategy
+ """
+ assert Path(ckpt_path).exists()
+ # use model's own load_pretrained_model method
+ if hasattr(model, "load_pretrained_model"):
+ model.load_pretrained_model(ckpt_path)
+ else:
+ Log.info(f"Loading ckpt: {ckpt_path}")
+ ckpt = torch.load(ckpt_path, "cpu")
+ model.load_state_dict(ckpt, strict=True)
+
+
+def find_last_ckpt_path(dirpath):
+ """
+ Assume ckpt is named as e{}* or last*, following the convention of pytorch-lightning.
+ """
+ assert dirpath is not None
+ dirpath = Path(dirpath)
+ assert dirpath.exists()
+ # Priority 1: last.ckpt
+ auto_last_ckpt_path = dirpath / "last.ckpt"
+ if auto_last_ckpt_path.exists():
+ return auto_last_ckpt_path
+
+ # Priority 2
+ model_paths = []
+ for p in sorted(list(dirpath.glob("*.ckpt"))):
+ if "last" in p.name:
+ continue
+ model_paths.append(p)
+ if len(model_paths) > 0:
+ return model_paths[-1]
+ else:
+ Log.info("No checkpoint found, set model_path to None")
+ return None
+
+
+def get_resume_ckpt_path(resume_mode, ckpt_dir=None):
+ if Path(resume_mode).exists(): # This is a path
+ return resume_mode
+ assert resume_mode == "last"
+ return find_last_ckpt_path(ckpt_dir)
+
+
+def select_state_dict_by_prefix(state_dict, prefix, new_prefix=""):
+ """
+ For each weight that start with {old_prefix}, remove the {old_prefic} and form a new state_dict.
+ Args:
+ state_dict: dict
+ prefix: str
+ new_prefix: str, if exists, the new key will be {new_prefix} + {old_key[len(prefix):]}
+ Returns:
+ state_dict_new: dict
+ """
+ state_dict_new = {}
+ for k in list(state_dict.keys()):
+ if k.startswith(prefix):
+ new_key = new_prefix + k[len(prefix) :]
+ state_dict_new[new_key] = state_dict[k]
+ return state_dict_new
+
+
+def detach_to_cpu(in_dict):
+ return recursive_detach(in_dict, to_cpu=True)
+
+
+def to_cuda(data):
+ """Move data in the batch to cuda(), carefully handle data that is not tensor"""
+ if isinstance(data, torch.Tensor):
+ return data.cuda()
+ elif isinstance(data, dict):
+ return {k: to_cuda(v) for k, v in data.items()}
+ elif isinstance(data, list):
+ return [to_cuda(v) for v in data]
+ else:
+ return data
+
+
+def get_valid_mask(max_len, valid_len, device="cpu"):
+ mask = torch.zeros(max_len, dtype=torch.bool).to(device)
+ mask[:valid_len] = True
+ return mask
+
+
+def length_to_mask(lengths, max_len):
+ """
+ Returns: (B, max_len)
+ """
+ mask = torch.arange(max_len, device=lengths.device).expand(len(lengths), max_len) < lengths.unsqueeze(1)
+ return mask
+
+
+def repeat_to_max_len(x, max_len, dim=0):
+ """Repeat last frame to max_len along dim"""
+ assert isinstance(x, torch.Tensor)
+ if x.shape[dim] == max_len:
+ return x
+ elif x.shape[dim] < max_len:
+ x = x.clone()
+ x = x.transpose(0, dim)
+ x = torch.cat([x, repeat(x[-1:], "b ... -> (b r) ...", r=max_len - x.shape[0])])
+ x = x.transpose(0, dim)
+ return x
+ else:
+ raise ValueError(f"Unexpected length v.s. max_len: {x.shape[0]} v.s. {max_len}")
+
+
+def repeat_to_max_len_dict(x_dict, max_len, dim=0):
+ for k, v in x_dict.items():
+ x_dict[k] = repeat_to_max_len(v, max_len, dim=dim)
+ return x_dict
+
+
+class Transpose(nn.Module):
+ def __init__(self, dim1, dim2):
+ super(Transpose, self).__init__()
+ self.dim1 = dim1
+ self.dim2 = dim2
+
+ def forward(self, x):
+ return x.transpose(self.dim1, self.dim2)
+
+
+class GaussianSmooth(nn.Module):
+ def __init__(self, sigma=3, dim=-1):
+ super(GaussianSmooth, self).__init__()
+ kernel_smooth = _gaussian_kernel1d(sigma=sigma, order=0, radius=int(4 * sigma + 0.5))
+ kernel_smooth = torch.from_numpy(kernel_smooth).float()[None, None] # (1, 1, K)
+ self.register_buffer("kernel_smooth", kernel_smooth, persistent=False)
+ self.dim = dim
+
+ def forward(self, x):
+ """x (..., f, ...) f at dim"""
+ rad = self.kernel_smooth.size(-1) // 2
+
+ x = x.transpose(self.dim, -1)
+ x_shape = x.shape[:-1]
+ x = rearrange(x, "... f -> (...) 1 f") # (NB, 1, f)
+ x = F.pad(x[None], (rad, rad, 0, 0), mode="replicate")[0]
+ x = F.conv1d(x, self.kernel_smooth)
+ x = x.squeeze(1).reshape(*x_shape, -1) # (..., f)
+ x = x.transpose(-1, self.dim)
+ return x
+
+
+def gaussian_smooth(x, sigma=3, dim=-1):
+ kernel_smooth = _gaussian_kernel1d(sigma=sigma, order=0, radius=int(4 * sigma + 0.5))
+ kernel_smooth = torch.from_numpy(kernel_smooth).float()[None, None].to(x) # (1, 1, K)
+ rad = kernel_smooth.size(-1) // 2
+
+ x = x.transpose(dim, -1)
+ x_shape = x.shape[:-1]
+ x = rearrange(x, "... f -> (...) 1 f") # (NB, 1, f)
+ x = F.pad(x[None], (rad, rad, 0, 0), mode="replicate")[0]
+ x = F.conv1d(x, kernel_smooth)
+ x = x.squeeze(1).reshape(*x_shape, -1) # (..., f)
+ x = x.transpose(-1, dim)
+ return x
+
+
+def moving_average_smooth(x, window_size=5, dim=-1):
+ kernel_smooth = torch.ones(window_size).float() / window_size
+ kernel_smooth = kernel_smooth[None, None].to(x) # (1, 1, window_size)
+ rad = kernel_smooth.size(-1) // 2
+
+ x = x.transpose(dim, -1)
+ x_shape = x.shape[:-1]
+ x = rearrange(x, "... f -> (...) 1 f") # (NB, 1, f)
+ x = F.pad(x[None], (rad, rad, 0, 0), mode="replicate")[0]
+ x = F.conv1d(x, kernel_smooth)
+ x = x.squeeze(1).reshape(*x_shape, -1) # (..., f)
+ x = x.transpose(-1, dim)
+ return x
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/__init__.py b/third_party/GVHMR/hmr4d/utils/preproc/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..d0813ee7c47e6512ac0057e75db7b023ce6113b1
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/__init__.py
@@ -0,0 +1,47 @@
+"""
+Preprocessing utilities.
+
+This package historically imported several optional/heavy dependencies at import
+time (e.g. tracker/SAM2, SLAM). That makes importing a specific submodule such
+as `hmr4d.utils.preproc.vitfeat_extractor` fail if *any* optional dependency is
+missing, because Python executes this `__init__.py` first.
+
+We keep backwards-compatible symbols (`Tracker`, `Extractor`, ...) via lazy
+imports, without importing them eagerly.
+"""
+
+from __future__ import annotations
+
+from typing import Any
+
+__all__ = [
+ "Extractor",
+ "SimpleVO",
+ "SLAMModel",
+ "Tracker",
+ "VitPoseExtractor",
+]
+
+
+def __getattr__(name: str) -> Any:
+ if name == "Extractor":
+ from hmr4d.utils.preproc.vitfeat_extractor import Extractor
+
+ return Extractor
+ if name == "VitPoseExtractor":
+ from hmr4d.utils.preproc.vitpose import VitPoseExtractor
+
+ return VitPoseExtractor
+ if name == "SimpleVO":
+ from hmr4d.utils.preproc.relpose.simple_vo import SimpleVO
+
+ return SimpleVO
+ if name == "SLAMModel":
+ from hmr4d.utils.preproc.slam import SLAMModel
+
+ return SLAMModel
+ if name == "Tracker":
+ from hmr4d.utils.preproc.tracker import Tracker
+
+ return Tracker
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/hand_extractor.py b/third_party/GVHMR/hmr4d/utils/preproc/hand_extractor.py
new file mode 100644
index 0000000000000000000000000000000000000000..1ab065ffafbaafa690367c3f91c9f553a7752575
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/hand_extractor.py
@@ -0,0 +1,150 @@
+"""
+Hand keypoint and bbox extractor using RTMPose-x Wholebody.
+Used to detect hand bounding boxes for HaMeR.
+"""
+import torch
+import numpy as np
+import cv2
+import mmcv
+from tqdm import tqdm
+from mmpose.apis import init_model, inference_topdown
+
+
+class HandExtractor:
+ """
+ Extract hand bounding boxes using RTMPose-x Wholebody (133 keypoints).
+
+ COCO-WholeBody keypoint indices:
+ - 91-111: Left hand (21 keypoints)
+ - 112-132: Right hand (21 keypoints)
+ """
+
+ # Keypoint indices
+ LEFT_HAND_INDICES = list(range(91, 112)) # 21 left hand keypoints
+ RIGHT_HAND_INDICES = list(range(112, 133)) # 21 right hand keypoints
+
+ # Wrist indices for reference
+ LEFT_WRIST_IDX = 9 # COCO body left wrist
+ RIGHT_WRIST_IDX = 10 # COCO body right wrist
+
+ def __init__(self, device='cuda:0', tqdm_leave=True):
+ # RTMPose-x wholebody (133 keypoints)
+ self.config_file = "./third_party/GVHMR/inputs/checkpoints/rtmpose-x/rtmpose-x_8xb32-270e_coco-wholebody-384x288.py"
+ self.ckpt_path = "./third_party/GVHMR/inputs/checkpoints/rtmpose-x/rtmpose-x_simcc-coco-wholebody_pt-body7_270e-384x288-401dfc90_20230629.pth"
+
+ self.device = device
+ self.pose = init_model(self.config_file, self.ckpt_path, device=device)
+ self.pose.eval()
+ self.tqdm_leave = tqdm_leave
+
+ def _keypoints_to_bbox(self, keypoints, conf_thresh=0.3, padding=1.2):
+ """
+ Convert hand keypoints to bounding box.
+
+ Args:
+ keypoints: (21, 3) hand keypoints [x, y, conf]
+ conf_thresh: Minimum confidence to include keypoint
+ padding: Bbox padding factor
+
+ Returns:
+ bbox: [x1, y1, x2, y2] or None if not enough valid keypoints
+ """
+ valid_mask = keypoints[:, 2] > conf_thresh
+ if valid_mask.sum() < 4: # Need at least 4 keypoints
+ return None
+
+ valid_kpts = keypoints[valid_mask, :2]
+ x1, y1 = valid_kpts.min(axis=0)
+ x2, y2 = valid_kpts.max(axis=0)
+
+ # Add padding
+ w, h = x2 - x1, y2 - y1
+ cx, cy = (x1 + x2) / 2, (y1 + y2) / 2
+ size = max(w, h) * padding
+
+ x1 = cx - size / 2
+ y1 = cy - size / 2
+ x2 = cx + size / 2
+ y2 = cy + size / 2
+
+ return np.array([x1, y1, x2, y2])
+
+ @torch.no_grad()
+ def extract(self, video_path, bbx_xys, masks=None):
+ """
+ Extract hand bounding boxes from video.
+
+ Args:
+ video_path: Path to video file
+ bbx_xys: (L, 3) person bounding boxes [cx, cy, size]
+ masks: Optional list of SAM masks
+
+ Returns:
+ left_hand_bboxes: List of (L,) with bbox arrays [x1,y1,x2,y2] or None
+ right_hand_bboxes: List of (L,) with bbox arrays [x1,y1,x2,y2] or None
+ left_hand_kpts: (L, 21, 3) left hand keypoints
+ right_hand_kpts: (L, 21, 3) right hand keypoints
+ """
+ if isinstance(video_path, str):
+ video = mmcv.VideoReader(video_path)
+ elif isinstance(video_path, torch.Tensor):
+ video = video_path.permute(0, 2, 3, 1).cpu().numpy()
+ if video.max() <= 1.0:
+ video = (video * 255).astype(np.uint8)
+ else:
+ video = video_path
+
+ L = len(bbx_xys)
+ left_hand_bboxes = []
+ right_hand_bboxes = []
+ left_hand_kpts = []
+ right_hand_kpts = []
+
+ for i in tqdm(range(L), desc="RTMPose Hands", leave=self.tqdm_leave):
+ frame = video[i]
+
+ # Apply mask if available
+ if masks is not None and i < len(masks) and masks[i] is not None:
+ mask = masks[i]
+ if isinstance(mask, torch.Tensor):
+ mask = mask.numpy()
+ frame_h, frame_w = frame.shape[:2]
+ if mask.shape[0] != frame_h or mask.shape[1] != frame_w:
+ mask = cv2.resize(mask.astype(np.uint8), (frame_w, frame_h), interpolation=cv2.INTER_NEAREST)
+ gray_bg = np.full_like(frame, 128)
+ mask_3ch = mask[:, :, None].astype(bool)
+ frame = np.where(mask_3ch, frame, gray_bg)
+
+ cx, cy, s = bbx_xys[i][0].item(), bbx_xys[i][1].item(), bbx_xys[i][2].item()
+ half_s = s / 2
+ bbox = np.array([[cx - half_s, cy - half_s, cx + half_s, cy + half_s]], dtype=np.float32)
+
+ results = inference_topdown(self.pose, frame, bboxes=bbox)
+
+ if len(results) > 0:
+ kpts = results[0].pred_instances.keypoints[0] # (133, 2)
+ scores = results[0].pred_instances.keypoint_scores[0][:, None] # (133, 1)
+ kpts_full = np.concatenate([kpts, scores], axis=-1) # (133, 3)
+
+ # Extract hand keypoints
+ left_kpts = kpts_full[self.LEFT_HAND_INDICES] # (21, 3)
+ right_kpts = kpts_full[self.RIGHT_HAND_INDICES] # (21, 3)
+
+ # Get bboxes from keypoints
+ left_bbox = self._keypoints_to_bbox(left_kpts)
+ right_bbox = self._keypoints_to_bbox(right_kpts)
+ else:
+ left_kpts = np.zeros((21, 3))
+ right_kpts = np.zeros((21, 3))
+ left_bbox = None
+ right_bbox = None
+
+ left_hand_bboxes.append(left_bbox)
+ right_hand_bboxes.append(right_bbox)
+ left_hand_kpts.append(torch.from_numpy(left_kpts).float())
+ right_hand_kpts.append(torch.from_numpy(right_kpts).float())
+
+ left_hand_kpts = torch.stack(left_hand_kpts, dim=0) # (L, 21, 3)
+ right_hand_kpts = torch.stack(right_hand_kpts, dim=0) # (L, 21, 3)
+
+ return left_hand_bboxes, right_hand_bboxes, left_hand_kpts, right_hand_kpts
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/relpose/README.md b/third_party/GVHMR/hmr4d/utils/preproc/relpose/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..df692edf5d84a68d741c4768b54d2972cea1c442
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/relpose/README.md
@@ -0,0 +1,11 @@
+Follow https://github.com/zehongs/RelativePose for updates.
+
+
+requirements:
+```
+pip install pycolmap
+```
+
+
+
+
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/relpose/matcher_wrapper.py b/third_party/GVHMR/hmr4d/utils/preproc/relpose/matcher_wrapper.py
new file mode 100644
index 0000000000000000000000000000000000000000..2883f544fe0cd51c8e949d9dbab77f033f767047
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/relpose/matcher_wrapper.py
@@ -0,0 +1,24 @@
+from .model.base_matcher import BaseMatcher
+from .model.cv2_matcher import CV2SIFTMather, CV2ORBMather
+
+
+matcher_map = {
+ "sift": CV2SIFTMather,
+ "orb": CV2ORBMather,
+}
+
+
+class Matcher:
+ def __init__(self, matcher="sift", args=None):
+ self.matcher: BaseMatcher = matcher_map[matcher](args)
+
+ def match_np(self, img0, img1):
+ """
+ Args:
+ img0: np.ndarray, shape (H, W, 3), dtype=np.uint8
+ img1: np.ndarray, shape (H, W, 3), dtype=np.uint8
+ Returns:
+ pts0: np.ndarray, shape (N, 2), dtype=np.float32
+ pts1: np.ndarray, shape (N, 2), dtype=np.float32
+ """
+ return self.matcher.match_np(img0, img1)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/relpose/model/base_matcher.py b/third_party/GVHMR/hmr4d/utils/preproc/relpose/model/base_matcher.py
new file mode 100644
index 0000000000000000000000000000000000000000..3491c4189cc553e7637d086f1b1cf370f94c9ccd
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/relpose/model/base_matcher.py
@@ -0,0 +1,9 @@
+import cv2
+
+
+class BaseMatcher:
+ def __init__(self, args=None):
+ super().__init__()
+
+ def match_np(self, img0, img1):
+ pass
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/relpose/model/cv2_matcher.py b/third_party/GVHMR/hmr4d/utils/preproc/relpose/model/cv2_matcher.py
new file mode 100644
index 0000000000000000000000000000000000000000..325c8d12829e35747b26a02bb38e5231988cdab4
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/relpose/model/cv2_matcher.py
@@ -0,0 +1,78 @@
+import cv2
+import numpy as np
+from .base_matcher import BaseMatcher
+
+
+class CV2SIFTMather(BaseMatcher):
+ def __init__(self, args=None):
+ super().__init__()
+ self.sift = cv2.SIFT_create()
+
+ def match_np(self, img0, img1):
+ # Convert images to grayscale
+ gray0 = cv2.cvtColor(img0, cv2.COLOR_BGR2GRAY)
+ gray1 = cv2.cvtColor(img1, cv2.COLOR_BGR2GRAY)
+
+ # Find keypoints and descriptors
+ kp0, des0 = self.sift.detectAndCompute(gray0, None)
+ kp1, des1 = self.sift.detectAndCompute(gray1, None)
+
+ # Match descriptors using FLANN matcher (better for SIFT)
+ FLANN_INDEX_KDTREE = 1
+ index_params = dict(algorithm=FLANN_INDEX_KDTREE, trees=5)
+ search_params = dict(checks=50)
+ flann = cv2.FlannBasedMatcher(index_params, search_params)
+
+ matches = flann.knnMatch(des0, des1, k=2)
+
+ # Store all good matches as per Lowe's ratio test
+ good_matches = []
+ for m, n in matches:
+ if m.distance < 0.7 * n.distance:
+ good_matches.append(m)
+
+ if len(good_matches) < 8:
+ print(
+ f"Warning: Only {len(good_matches)} matches found, which might not be enough for reliable pose estimation"
+ )
+
+ # Extract matched point coordinates
+ pts0 = np.float32([kp0[m.queryIdx].pt for m in good_matches]).reshape(-1, 2)
+ pts1 = np.float32([kp1[m.trainIdx].pt for m in good_matches]).reshape(-1, 2)
+
+ return pts0, pts1
+
+
+class CV2ORBMather(BaseMatcher):
+ def __init__(self, args=None):
+ super().__init__()
+ self.orb = cv2.ORB_create()
+ self.num_matches = 1024
+
+ def match_np(self, img0, img1):
+ # Convert images to grayscale
+ gray0 = cv2.cvtColor(img0, cv2.COLOR_BGR2GRAY)
+ gray1 = cv2.cvtColor(img1, cv2.COLOR_BGR2GRAY)
+
+ # Original ORB method
+ kp0, des0 = self.orb.detectAndCompute(gray0, None)
+ kp1, des1 = self.orb.detectAndCompute(gray1, None)
+ bf = cv2.BFMatcher(cv2.NORM_HAMMING, crossCheck=True)
+ matches = bf.match(des0, des1)
+ matches = sorted(matches, key=lambda x: x.distance)
+ good_matches = matches[: self.num_matches]
+
+ # Ensure we have enough matches
+ if len(good_matches) < 8:
+ print(
+ f"Warning: Only {len(good_matches)} matches found, which might not be enough for reliable pose estimation"
+ )
+ # Pad with more matches if available
+ if len(matches) > len(good_matches):
+ good_matches = matches[: min(100, len(matches))]
+
+ # Extract matched point coordinates
+ pts0 = np.float32([kp0[m.queryIdx].pt for m in good_matches]).reshape(-1, 2)
+ pts1 = np.float32([kp1[m.trainIdx].pt for m in good_matches]).reshape(-1, 2)
+
+ return pts0, pts1
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/relpose/simple_vo.py b/third_party/GVHMR/hmr4d/utils/preproc/relpose/simple_vo.py
new file mode 100644
index 0000000000000000000000000000000000000000..1d54592021570e81d2032184bcbd0472eb936027
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/relpose/simple_vo.py
@@ -0,0 +1,59 @@
+import numpy as np
+from .utils import focal_length_from_mm
+from .matcher_wrapper import Matcher
+from .solver_two_view import TwoPairSolver, CameraParams, interpolate_missing_frames
+from tqdm import tqdm
+
+from hmr4d.utils.video_io_utils import get_video_lwh, read_video_np
+
+
+class SimpleVO:
+ def __init__(self, video_path, scale=0.5, step=8, method="sift", f_mm=None):
+ self.video_path = video_path
+ self.scale = scale
+ self.step = step
+ self.method = method
+ self.f_mm = 24 if f_mm is None else f_mm # fullframe camera focal length in mm
+
+ def compute(self):
+ # Read video
+ frames = read_video_np(self.video_path, scale=self.scale)
+
+ # Downsample frames, and interpolate missing frames
+ F_all = frames.shape[0]
+ sample_idxs = np.arange(0, F_all, self.step)
+ if sample_idxs[-1] != F_all - 1:
+ sample_idxs = np.concatenate([sample_idxs, [F_all - 1]])
+ frames = frames[sample_idxs]
+ F, H, W, C = frames.shape
+ print(f"[SimpleVO] Choosen frames shape: {frames.shape}")
+
+ matcher: Matcher = Matcher(self.method)
+ camera_params = CameraParams(W, H, focal_length=focal_length_from_mm(W, H, self.f_mm))
+ solver: TwoPairSolver = TwoPairSolver(camera_params, solver="pycolmap")
+
+ # TODO:We should use different pipelines for different methods
+ T_w2c_list = self.process_video_T_w2c_list_np(frames, matcher, solver)
+
+ # Interpolate missing frames
+ T_w2c_list = interpolate_missing_frames(T_w2c_list, sample_idxs)
+
+ return T_w2c_list
+
+ def process_video_T_w2c_list_np(self, frames, matcher: Matcher, solver: TwoPairSolver):
+ T_w2c_list = [np.eye(4)] # cam poses are defined as T_w2c @ p_w = p_c
+ prev_frame = frames[0]
+ for frame_idx in tqdm(range(1, len(frames))):
+ curr_frame = frames[frame_idx]
+
+ # Match frames
+ pts0, pts1 = matcher.match_np(prev_frame, curr_frame)
+ T_delta = solver.solve(pts0, pts1) # T_delta = T_curr @ T_last^-1
+
+ # Compute current frame's transformation matrix
+ T_w2c_list.append(T_delta @ T_w2c_list[-1])
+
+ # Current frame becomes previous frame for next iteration
+ prev_frame = curr_frame
+
+ return T_w2c_list
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/relpose/solver_two_view.py b/third_party/GVHMR/hmr4d/utils/preproc/relpose/solver_two_view.py
new file mode 100644
index 0000000000000000000000000000000000000000..929877e975176ec5d9421dee1686d00773ba48a1
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/relpose/solver_two_view.py
@@ -0,0 +1,194 @@
+import cv2
+import numpy as np
+from dataclasses import dataclass
+import pycolmap
+from .transformation_np import *
+
+
+@dataclass
+class CameraParams:
+ width: int
+ height: int
+ focal_length: float = None # Use sqrt(width^2 + height^2) if not provided FOV~=53°
+ cx: float = None # Use half of width if not provided
+ cy: float = None # Use half of height if not provided
+
+
+class Cv2RansacEssentialSolver:
+ def __init__(self, camera_params: CameraParams):
+ width = camera_params.width
+ height = camera_params.height
+ focal_length = camera_params.focal_length
+ if focal_length is None:
+ focal_length = (width**2 + height**2) ** 0.5
+ cx = camera_params.cx
+ cy = camera_params.cy
+ if cx is None:
+ cx = width / 2
+ if cy is None:
+ cy = height / 2
+
+ self.camera_matrix = np.array([[focal_length, 0, cx], [0, focal_length, cy], [0, 0, 1]])
+
+ def get_K(self):
+ """
+ Returns:
+ K: np.ndarray, shape (3, 3), dtype=np.float32
+ """
+ return self.camera_matrix
+
+ def solve(self, pts0, pts1):
+ # Find essential matrix with stricter RANSAC
+ E, mask = cv2.findEssentialMat(
+ pts0,
+ pts1,
+ self.camera_matrix,
+ method=cv2.RANSAC,
+ prob=0.999,
+ threshold=1.0,
+ )
+
+ # Recover pose
+ _, R, t, mask = cv2.recoverPose(E, pts0, pts1, self.camera_matrix, mask=mask)
+
+ return R, t
+
+
+class PycolmapRansacTwoViewGeometrySolver:
+ def __init__(self, camera_params: CameraParams):
+ width = camera_params.width
+ height = camera_params.height
+ focal_length = camera_params.focal_length
+ if focal_length is None:
+ focal_length = (width**2 + height**2) ** 0.5
+ cx = camera_params.cx
+ cy = camera_params.cy
+ if cx is None:
+ cx = width / 2
+ if cy is None:
+ cy = height / 2
+ self.camera_matrix = np.array([[focal_length, 0, cx], [0, focal_length, cy], [0, 0, 1]])
+
+ # Set up pycolmap
+ self.camera = pycolmap.Camera(
+ camera_id=0,
+ model="SIMPLE_PINHOLE",
+ width=width,
+ height=height,
+ params=[focal_length, cx, cy],
+ )
+
+ # Configure options for consecutive frames
+ self.options = pycolmap.TwoViewGeometryOptions(
+ min_num_inliers=10,
+ min_E_F_inlier_ratio=0.8,
+ max_H_inlier_ratio=0.9,
+ compute_relative_pose=True,
+ )
+ print(self.options.summary())
+
+ def get_K(self):
+ return self.camera_matrix
+
+ def solve(self, pts0, pts1):
+ matches = np.stack([np.arange(len(pts0)), np.arange(len(pts0))], axis=-1)
+ answer = pycolmap.estimate_calibrated_two_view_geometry(
+ self.camera,
+ pts0.astype(np.float64),
+ self.camera,
+ pts1.astype(np.float64),
+ matches=matches,
+ options=self.options,
+ )
+
+ # cam2_from_cam1 means T_0_to_1 in our language
+ Rt = answer.cam2_from_cam1.matrix().astype(np.float32) # shape (3, 4)
+ T = np.eye(4)
+ T[:3] = Rt
+ return T
+
+
+two_pair_solver_map = {
+ # "cv2": Cv2RansacEssentialSolver, # This is not stable
+ "pycolmap": PycolmapRansacTwoViewGeometrySolver, # Essential and Homography at the same time
+}
+
+
+class TwoPairSolver:
+ def __init__(self, params: CameraParams, solver: str = "pycolmap"):
+ self.solver = two_pair_solver_map[solver](params)
+
+ def get_K(self):
+ """
+ Returns:
+ K: np.ndarray, shape (3, 3), dtype=np.float32
+ """
+ return self.solver.get_K()
+
+ def solve(self, pts0, pts1):
+ """
+ Args:
+ pts0: np.ndarray, shape (N, 2), dtype=np.float32
+ pts1: np.ndarray, shape (N, 2), dtype=np.float32
+ Returns:
+ T: np.ndarray, shape (4, 4), dtype=np.float32
+ """
+ return self.solver.solve(pts0, pts1)
+
+
+########################################################
+# Interpolate missing frames
+########################################################
+
+
+def interpolate_missing_frames(T_w2c_list, sample_idxs):
+ """
+ 对给定的 T_w2c_list(已知帧的变换矩阵)进行平滑插值,生成所有帧的变换矩阵。
+ 其中:
+ - 平移部分采用线性插值;
+ - 旋转部分采用自实现的SLERP球面线性插值,保证旋转过渡平滑。
+
+ 参数:
+ T_w2c_list (numpy.ndarray): 形状为 (F, 4, 4) 的已知变换矩阵数组
+ sample_idxs (list 或 numpy.ndarray): 长度为 F 的已知帧在原始序列中的索引
+ (假设第一个索引为 0,最后一个为 F_all - 1)
+
+ 返回:
+ numpy.ndarray: 形状为 (F_all, 4, 4) 的所有帧的变换矩阵,缺失帧通过平滑插值填充。
+ """
+ sample_idxs = np.array(sample_idxs)
+ # 根据最后一个已知帧索引确定总帧数(假设索引从 0 开始)
+ F_all = sample_idxs[-1] + 1
+ new_T_list = []
+
+ # 分离出平移和旋转部分
+ translations = np.array([T[:3, 3] for T in T_w2c_list])
+ rotations = np.array([T[:3, :3] for T in T_w2c_list])
+ # 将旋转矩阵转换为四元数
+ quaternions = np.array([rotation_matrix_to_quaternion(R) for R in rotations])
+
+ for i in range(F_all):
+ # 如果该帧为已知帧,直接使用对应的变换矩阵
+ if i in sample_idxs:
+ known_index = np.where(sample_idxs == i)[0][0]
+ new_T_list.append(T_w2c_list[known_index])
+ else:
+ # 定位左右两侧已知帧
+ next_known = np.searchsorted(sample_idxs, i)
+ prev_known = next_known - 1
+ # 计算插值比例 t
+ t_interp = (i - sample_idxs[prev_known]) / (sample_idxs[next_known] - sample_idxs[prev_known])
+ # 平移部分:线性插值
+ trans_interp = (1 - t_interp) * translations[prev_known] + t_interp * translations[next_known]
+ # 旋转部分:自实现 SLERP 插值
+ q0 = quaternions[prev_known]
+ q1 = quaternions[next_known]
+ q_interp = slerp(q0, q1, t_interp)
+ rot_interp = quaternion_to_rotation_matrix(q_interp)
+ # 构造最终的 4x4 变换矩阵
+ T_interp = np.eye(4)
+ T_interp[:3, :3] = rot_interp
+ T_interp[:3, 3] = trans_interp
+ new_T_list.append(T_interp)
+
+ return np.array(new_T_list)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/relpose/transformation_np.py b/third_party/GVHMR/hmr4d/utils/preproc/relpose/transformation_np.py
new file mode 100644
index 0000000000000000000000000000000000000000..9a68b0aa06e571537a83bf90cbaad982936c8c8d
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/relpose/transformation_np.py
@@ -0,0 +1,140 @@
+import numpy as np
+
+
+def rotation_matrix_to_quaternion(R):
+ """
+ 将 3x3 旋转矩阵 R 转换为四元数 [w, x, y, z] 的形式。
+ """
+ m00, m01, m02 = R[0, 0], R[0, 1], R[0, 2]
+ m10, m11, m12 = R[1, 0], R[1, 1], R[1, 2]
+ m20, m21, m22 = R[2, 0], R[2, 1], R[2, 2]
+
+ tr = m00 + m11 + m22
+ if tr > 0:
+ S = np.sqrt(tr + 1.0) * 2 # S = 4 * qw
+ qw = 0.25 * S
+ qx = (m21 - m12) / S
+ qy = (m02 - m20) / S
+ qz = (m10 - m01) / S
+ elif (m00 > m11) and (m00 > m22):
+ S = np.sqrt(1.0 + m00 - m11 - m22) * 2 # S = 4 * qx
+ qw = (m21 - m12) / S
+ qx = 0.25 * S
+ qy = (m01 + m10) / S
+ qz = (m02 + m20) / S
+ elif m11 > m22:
+ S = np.sqrt(1.0 + m11 - m00 - m22) * 2 # S = 4 * qy
+ qw = (m02 - m20) / S
+ qx = (m01 + m10) / S
+ qy = 0.25 * S
+ qz = (m12 + m21) / S
+ else:
+ S = np.sqrt(1.0 + m22 - m00 - m11) * 2 # S = 4 * qz
+ qw = (m10 - m01) / S
+ qx = (m02 + m20) / S
+ qy = (m12 + m21) / S
+ qz = 0.25 * S
+ return np.array([qw, qx, qy, qz])
+
+
+def quaternion_to_rotation_matrix(q):
+ """
+ 将四元数 [w, x, y, z] 转换为 3x3 旋转矩阵。
+ """
+ qw, qx, qy, qz = q
+ R = np.array(
+ [
+ [1 - 2 * qy**2 - 2 * qz**2, 2 * qx * qy - 2 * qz * qw, 2 * qx * qz + 2 * qy * qw],
+ [2 * qx * qy + 2 * qz * qw, 1 - 2 * qx**2 - 2 * qz**2, 2 * qy * qz - 2 * qx * qw],
+ [2 * qx * qz - 2 * qy * qw, 2 * qy * qz + 2 * qx * qw, 1 - 2 * qx**2 - 2 * qy**2],
+ ]
+ )
+ return R
+
+
+def slerp(q0, q1, t):
+ """
+ 对两个四元数 q0 和 q1 进行球面线性插值(SLERP)。
+
+ 参数:
+ q0, q1: numpy 数组,形状为 (4,),表示四元数 [w, x, y, z]
+ t: 插值系数,0 <= t <= 1
+
+ 返回:
+ 插值后的四元数,形状为 (4,)
+ """
+ dot = np.dot(q0, q1)
+ # 如果点积为负,取相反数以保证取短路径
+ if dot < 0.0:
+ q1 = -q1
+ dot = -dot
+
+ DOT_THRESHOLD = 0.9995
+ if dot > DOT_THRESHOLD:
+ # 当两个四元数非常接近时,直接使用线性插值再归一化
+ result = q0 + t * (q1 - q0)
+ result = result / np.linalg.norm(result)
+ return result
+
+ theta_0 = np.arccos(dot) # 两个四元数之间的角度
+ theta = theta_0 * t # 插值后的角度
+ sin_theta = np.sin(theta)
+ sin_theta_0 = np.sin(theta_0)
+
+ s0 = np.cos(theta) - dot * sin_theta / sin_theta_0
+ s1 = sin_theta / sin_theta_0
+
+ return (s0 * q0) + (s1 * q1)
+
+
+def lerp_missing_frames(T_w2c_list, sample_idxs):
+ """
+ 对给定的 T_w2c_list(已知帧的变换矩阵)进行平滑插值,生成所有帧的变换矩阵。
+ 其中:
+ - 平移部分采用线性插值;
+ - 旋转部分采用自实现的SLERP球面线性插值,保证旋转过渡平滑。
+
+ 参数:
+ T_w2c_list (numpy.ndarray): 形状为 (F, 4, 4) 的已知变换矩阵数组
+ sample_idxs (list 或 numpy.ndarray): 长度为 F 的已知帧在原始序列中的索引
+ (假设第一个索引为 0,最后一个为 F_all - 1)
+
+ 返回:
+ numpy.ndarray: 形状为 (F_all, 4, 4) 的所有帧的变换矩阵,缺失帧通过平滑插值填充。
+ """
+ sample_idxs = np.array(sample_idxs)
+ # 根据最后一个已知帧索引确定总帧数(假设索引从 0 开始)
+ F_all = sample_idxs[-1] + 1
+ new_T_list = []
+
+ # 分离出平移和旋转部分
+ translations = np.array([T[:3, 3] for T in T_w2c_list])
+ rotations = np.array([T[:3, :3] for T in T_w2c_list])
+ # 将旋转矩阵转换为四元数
+ quaternions = np.array([rotation_matrix_to_quaternion(R) for R in rotations])
+
+ for i in range(F_all):
+ # 如果该帧为已知帧,直接使用对应的变换矩阵
+ if i in sample_idxs:
+ known_index = np.where(sample_idxs == i)[0][0]
+ new_T_list.append(T_w2c_list[known_index])
+ else:
+ # 定位左右两侧已知帧
+ next_known = np.searchsorted(sample_idxs, i)
+ prev_known = next_known - 1
+ # 计算插值比例 t
+ t_interp = (i - sample_idxs[prev_known]) / (sample_idxs[next_known] - sample_idxs[prev_known])
+ # 平移部分:线性插值
+ trans_interp = (1 - t_interp) * translations[prev_known] + t_interp * translations[next_known]
+ # 旋转部分:自实现 SLERP 插值
+ q0 = quaternions[prev_known]
+ q1 = quaternions[next_known]
+ q_interp = slerp(q0, q1, t_interp)
+ rot_interp = quaternion_to_rotation_matrix(q_interp)
+ # 构造最终的 4x4 变换矩阵
+ T_interp = np.eye(4)
+ T_interp[:3, :3] = rot_interp
+ T_interp[:3, 3] = trans_interp
+ new_T_list.append(T_interp)
+
+ return np.array(new_T_list)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/relpose/utils.py b/third_party/GVHMR/hmr4d/utils/preproc/relpose/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..a06bbcd027903f0d55bb0d8774d90dafb5fe00f9
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/relpose/utils.py
@@ -0,0 +1,183 @@
+import numpy as np
+import matplotlib.pyplot as plt
+import cv2
+from pathlib import Path
+from .viz2d import plot_matches, plot_images, save_plot
+
+# ================================
+# Visualization
+# ================================
+
+
+def visualize_matches(img0, img1, kp0, kp1, output_dir):
+ """Visualize the matched features between two images."""
+ plot_images([img0, img1], ["Image 0", "Image 1"])
+ plot_matches(kp0, kp1)
+ save_plot(Path(output_dir) / "matches.png")
+
+
+def visualize_T_w2c_rotations(T_w2c_list, output_dir):
+ """可视化相机旋转轨迹,并考虑 OpenCV 坐标系转换,
+ 使得相机的 x 轴保持右向,z 轴(光轴)变为水平前向,
+ 而相机的 y 轴(朝下)对应于 plt 的 -z 轴(即向下)。"""
+
+ # 定义转换矩阵:将 OpenCV 坐标 (x:right, y:down, z:forward)
+ # 转换为 world 坐标 (x:right, y:forward, z:up)
+ R_align = np.array([[0, 0, 1], [-1, 0, 0], [0, -1, 0]])
+
+ # 原始光轴:在 OpenCV 中通常为 [0, 0, 1]
+ normal_vector = np.array([0, 0, 1])
+ aligned_rotated_normals = []
+
+ # 对每一帧的旋转矩阵,先计算光轴旋转后的方向,再进行坐标转换
+ for T in T_w2c_list:
+ R = T[:3, :3]
+ rotated_normal = R.T @ normal_vector
+ # 应用对齐变换
+ aligned_normal = R_align @ rotated_normal
+ aligned_rotated_normals.append(aligned_normal)
+ aligned_rotated_normals = np.array(aligned_rotated_normals)
+
+ # ------------------ 3D 可视化 ------------------
+ fig = plt.figure(figsize=(12, 10))
+ ax = fig.add_subplot(111, projection="3d")
+
+ # 绘制调整后的旋转法向量轨迹
+ ax.plot(aligned_rotated_normals[:, 0], aligned_rotated_normals[:, 1], aligned_rotated_normals[:, 2], "b-")
+ ax.scatter(aligned_rotated_normals[:, 0], aligned_rotated_normals[:, 1], aligned_rotated_normals[:, 2], c="r", s=10)
+
+ # 标记起点与终点
+ ax.scatter(
+ aligned_rotated_normals[0, 0],
+ aligned_rotated_normals[0, 1],
+ aligned_rotated_normals[0, 2],
+ c="g",
+ s=100,
+ marker="o",
+ label="Start",
+ )
+ ax.scatter(
+ aligned_rotated_normals[-1, 0],
+ aligned_rotated_normals[-1, 1],
+ aligned_rotated_normals[-1, 2],
+ c="m",
+ s=100,
+ marker="o",
+ label="End",
+ )
+
+ # 绘制单位球以便参考:对球面上每个点也应用相同的转换
+ u, v = np.mgrid[0 : 2 * np.pi : 20j, 0 : np.pi : 10j]
+ x = np.cos(u) * np.sin(v)
+ y = np.sin(u) * np.sin(v)
+ z = np.cos(v)
+ sphere_points = np.stack([x, y, z], axis=-1)
+ sphere_points_aligned = sphere_points @ R_align.T
+ X_aligned = sphere_points_aligned[:, :, 0]
+ Y_aligned = sphere_points_aligned[:, :, 1]
+ Z_aligned = sphere_points_aligned[:, :, 2]
+ ax.plot_wireframe(X_aligned, Y_aligned, Z_aligned, color="gray", alpha=0.2)
+
+ # 设置坐标轴比例和范围
+ ax.set_box_aspect([1, 1, 1])
+ ax.set_xlim([-1.1, 1.1])
+ ax.set_ylim([-1.1, 1.1])
+ ax.set_zlim([-1.1, 1.1])
+
+ ax.set_xlabel("X")
+ ax.set_ylabel("Y (Forward)")
+ ax.set_zlabel("Z (Up)")
+ ax.set_title("Camera Rotation T_w2c_list (3D) - Aligned to Camera Conventions")
+ ax.legend()
+
+ # 添加帧数标记
+ frame_count = len(T_w2c_list)
+ interval = max(1, frame_count // 10)
+ for i in range(0, frame_count, interval):
+ ax.text(
+ aligned_rotated_normals[i, 0],
+ aligned_rotated_normals[i, 1],
+ aligned_rotated_normals[i, 2],
+ f"{i}",
+ fontsize=8,
+ )
+
+ plt.savefig(Path(output_dir) / "rotation_trajectory_aligned.png")
+ plt.show()
+
+
+def visualize_rotation_angles(T_w2c_list, output_dir):
+ """Visualize rotation as Euler angles over time."""
+ # Extract rotation matrices
+ rotations = [T[:3, :3] for T in T_w2c_list]
+
+ # Convert to Euler angles (in degrees)
+ euler_angles = []
+ for R in rotations:
+ # Convert rotation matrix to Euler angles
+ # Using the 'xyz' convention - roll, pitch, yaw
+ sy = np.sqrt(R[0, 0] * R[0, 0] + R[1, 0] * R[1, 0])
+ singular = sy < 1e-6
+
+ if not singular:
+ roll = np.arctan2(R[2, 1], R[2, 2])
+ pitch = np.arctan2(-R[2, 0], sy)
+ yaw = np.arctan2(R[1, 0], R[0, 0])
+ else:
+ roll = np.arctan2(-R[1, 2], R[1, 1])
+ pitch = np.arctan2(-R[2, 0], sy)
+ yaw = 0
+
+ # Convert to degrees
+ euler_angles.append([np.degrees(roll), np.degrees(pitch), np.degrees(yaw)])
+
+ euler_angles = np.array(euler_angles)
+
+ # Create plot
+ fig, ax = plt.subplots(figsize=(12, 8))
+ frame_indices = np.arange(len(T_w2c_list))
+
+ ax.plot(frame_indices, euler_angles[:, 0], "r-", label="Roll")
+ ax.plot(frame_indices, euler_angles[:, 1], "g-", label="Pitch")
+ ax.plot(frame_indices, euler_angles[:, 2], "b-", label="Yaw")
+
+ ax.set_xlabel("Frame Number")
+ ax.set_ylabel("Angle (degrees)")
+ ax.set_title("Camera Rotation: Euler Angles Over Time")
+ ax.legend()
+ ax.grid(True)
+
+ plt.savefig(Path(output_dir) / "rotation_angles.png")
+ plt.show()
+
+
+# ================================
+# Video Reader
+# ================================
+
+
+def read_video_frame_np(video_path, frame_index):
+ # Use opencv to read frame at frame_index
+ cap = cv2.VideoCapture(video_path)
+ cap.set(cv2.CAP_PROP_POS_FRAMES, frame_index)
+ ret, frame = cap.read()
+ cap.release()
+
+ # Convert to RGB
+ frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
+ return frame
+
+
+# ================================
+# Camera
+# ================================
+
+
+def focal_length_from_mm(width, height, mm=24):
+ """
+ Convert full-frame focal length to image sensor focal length.
+ """
+ diag_fullframe = (24**2 + 36**2) ** 0.5
+ diag_img = (width**2 + height**2) ** 0.5
+ focal_length = diag_img / diag_fullframe * mm
+ return focal_length
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/relpose/viz2d.py b/third_party/GVHMR/hmr4d/utils/preproc/relpose/viz2d.py
new file mode 100644
index 0000000000000000000000000000000000000000..317cad7ed71441c8465ecc013cec7713add0fe2c
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/relpose/viz2d.py
@@ -0,0 +1,178 @@
+"""
+2D visualization primitives based on Matplotlib.
+1) Plot images with `plot_images`.
+2) Call `plot_keypoints` or `plot_matches` any number of times.
+3) Optionally: save a .png or .pdf plot (nice in papers!) with `save_plot`.
+"""
+
+import matplotlib
+import matplotlib.patheffects as path_effects
+import matplotlib.pyplot as plt
+import numpy as np
+import torch
+
+
+def cm_RdGn(x):
+ """Custom colormap: red (0) -> yellow (0.5) -> green (1)."""
+ x = np.clip(x, 0, 1)[..., None] * 2
+ c = x * np.array([[0, 1.0, 0]]) + (2 - x) * np.array([[1.0, 0, 0]])
+ return np.clip(c, 0, 1)
+
+
+def cm_BlRdGn(x_):
+ """Custom colormap: blue (-1) -> red (0.0) -> green (1)."""
+ x = np.clip(x_, 0, 1)[..., None] * 2
+ c = x * np.array([[0, 1.0, 0, 1.0]]) + (2 - x) * np.array([[1.0, 0, 0, 1.0]])
+
+ xn = -np.clip(x_, -1, 0)[..., None] * 2
+ cn = xn * np.array([[0, 0.1, 1, 1.0]]) + (2 - xn) * np.array([[1.0, 0, 0, 1.0]])
+ out = np.clip(np.where(x_[..., None] < 0, cn, c), 0, 1)
+ return out
+
+
+def cm_prune(x_):
+ """Custom colormap to visualize pruning"""
+ if isinstance(x_, torch.Tensor):
+ x_ = x_.cpu().numpy()
+ max_i = max(x_)
+ norm_x = np.where(x_ == max_i, -1, (x_ - 1) / 9)
+ return cm_BlRdGn(norm_x)
+
+
+def plot_images(imgs, titles=None, cmaps="gray", dpi=100, pad=0.5, adaptive=True):
+ """Plot a set of images horizontally.
+ Args:
+ imgs: list of NumPy RGB (H, W, 3) or PyTorch RGB (3, H, W) or mono (H, W).
+ titles: a list of strings, as titles for each image.
+ cmaps: colormaps for monochrome images.
+ adaptive: whether the figure size should fit the image aspect ratios.
+ """
+ # conversion to (H, W, 3) for torch.Tensor
+ imgs = [
+ img.permute(1, 2, 0).cpu().numpy() if (isinstance(img, torch.Tensor) and img.dim() == 3) else img
+ for img in imgs
+ ]
+
+ n = len(imgs)
+ if not isinstance(cmaps, (list, tuple)):
+ cmaps = [cmaps] * n
+
+ if adaptive:
+ ratios = [i.shape[1] / i.shape[0] for i in imgs] # W / H
+ else:
+ ratios = [4 / 3] * n
+ figsize = [sum(ratios) * 4.5, 4.5]
+ fig, ax = plt.subplots(1, n, figsize=figsize, dpi=dpi, gridspec_kw={"width_ratios": ratios})
+ if n == 1:
+ ax = [ax]
+ for i in range(n):
+ ax[i].imshow(imgs[i], cmap=plt.get_cmap(cmaps[i]))
+ ax[i].get_yaxis().set_ticks([])
+ ax[i].get_xaxis().set_ticks([])
+ ax[i].set_axis_off()
+ for spine in ax[i].spines.values(): # remove frame
+ spine.set_visible(False)
+ if titles:
+ ax[i].set_title(titles[i])
+ fig.tight_layout(pad=pad)
+
+
+def plot_keypoints(kpts, colors="lime", ps=4, axes=None, a=1.0):
+ """Plot keypoints for existing images.
+ Args:
+ kpts: list of ndarrays of size (N, 2).
+ colors: string, or list of list of tuples (one for each keypoints).
+ ps: size of the keypoints as float.
+ """
+ if not isinstance(colors, list):
+ colors = [colors] * len(kpts)
+ if not isinstance(a, list):
+ a = [a] * len(kpts)
+ if axes is None:
+ axes = plt.gcf().axes
+ for ax, k, c, alpha in zip(axes, kpts, colors, a):
+ if isinstance(k, torch.Tensor):
+ k = k.cpu().numpy()
+ ax.scatter(k[:, 0], k[:, 1], c=c, s=ps, linewidths=0, alpha=alpha)
+
+
+def plot_matches(kpts0, kpts1, color=None, lw=1.5, ps=4, a=1.0, labels=None, axes=None):
+ """Plot matches for a pair of existing images.
+ Args:
+ kpts0, kpts1: corresponding keypoints of size (N, 2).
+ color: color of each match, string or RGB tuple. Random if not given.
+ lw: width of the lines.
+ ps: size of the end points (no endpoint if ps=0)
+ indices: indices of the images to draw the matches on.
+ a: alpha opacity of the match lines.
+ """
+ fig = plt.gcf()
+ if axes is None:
+ ax = fig.axes
+ ax0, ax1 = ax[0], ax[1]
+ else:
+ ax0, ax1 = axes
+ if isinstance(kpts0, torch.Tensor):
+ kpts0 = kpts0.cpu().numpy()
+ if isinstance(kpts1, torch.Tensor):
+ kpts1 = kpts1.cpu().numpy()
+ assert len(kpts0) == len(kpts1)
+ if color is None:
+ color = matplotlib.cm.hsv(np.random.rand(len(kpts0))).tolist()
+ elif len(color) > 0 and not isinstance(color[0], (tuple, list)):
+ color = [color] * len(kpts0)
+
+ if lw > 0:
+ for i in range(len(kpts0)):
+ line = matplotlib.patches.ConnectionPatch(
+ xyA=(kpts0[i, 0], kpts0[i, 1]),
+ xyB=(kpts1[i, 0], kpts1[i, 1]),
+ coordsA=ax0.transData,
+ coordsB=ax1.transData,
+ axesA=ax0,
+ axesB=ax1,
+ zorder=1,
+ color=color[i],
+ linewidth=lw,
+ clip_on=True,
+ alpha=a,
+ label=None if labels is None else labels[i],
+ picker=5.0,
+ )
+ line.set_annotation_clip(True)
+ fig.add_artist(line)
+
+ # freeze the axes to prevent the transform to change
+ ax0.autoscale(enable=False)
+ ax1.autoscale(enable=False)
+
+ if ps > 0:
+ ax0.scatter(kpts0[:, 0], kpts0[:, 1], c=color, s=ps)
+ ax1.scatter(kpts1[:, 0], kpts1[:, 1], c=color, s=ps)
+
+
+def add_text(
+ idx,
+ text,
+ pos=(0.01, 0.99),
+ fs=15,
+ color="w",
+ lcolor="k",
+ lwidth=2,
+ ha="left",
+ va="top",
+):
+ ax = plt.gcf().axes[idx]
+ t = ax.text(*pos, text, fontsize=fs, ha=ha, va=va, color=color, transform=ax.transAxes)
+ if lcolor is not None:
+ t.set_path_effects(
+ [
+ path_effects.Stroke(linewidth=lwidth, foreground=lcolor),
+ path_effects.Normal(),
+ ]
+ )
+
+
+def save_plot(path, **kw):
+ """Save the current figure without any white margin."""
+ plt.savefig(path, bbox_inches="tight", pad_inches=0, **kw)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/slam.py b/third_party/GVHMR/hmr4d/utils/preproc/slam.py
new file mode 100644
index 0000000000000000000000000000000000000000..ed4dbc94174b551b0899b8ca2b60907a41947a88
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/slam.py
@@ -0,0 +1,175 @@
+
+import cv2
+import time
+import torch
+import numpy as np
+from multiprocessing import Process, Queue
+
+try:
+ from dpvo.utils import Timer
+ from dpvo.dpvo import DPVO
+ from dpvo.config import cfg
+except:
+ pass
+
+
+from hmr4d import PROJ_ROOT
+from hmr4d.utils.geo.hmr_cam import estimate_focal_length
+
+
+class SLAMModel(object):
+ def __init__(
+ self,
+ video_path,
+ width,
+ height,
+ intrinsics=None,
+ stride=1,
+ skip=0,
+ buffer=2048,
+ resize=0.5,
+ masks=None,
+ dpvo_ckpt: str | None = None,
+ dpvo_cfg: str | None = None,
+ max_frames: int | None = None,
+ ):
+ """
+ Args:
+ intrinsics: [fx, fy, cx, cy]
+ masks: Optional list of (H, W) binary masks. 1=person (exclude), 0=background (track)
+ """
+ if intrinsics is None:
+ print("Estimating focal length")
+ focal_length = estimate_focal_length(width, height)
+ intrinsics = torch.tensor([focal_length, focal_length, width / 2.0, height / 2.0])
+ else:
+ intrinsics = intrinsics.clone()
+
+ # Resolve default DPVO paths relative to the GVHMR project root to avoid CWD-dependent failures.
+ self.dpvo_cfg = str((PROJ_ROOT / "third-party/DPVO/config/default.yaml")) if dpvo_cfg is None else str(dpvo_cfg)
+ self.dpvo_ckpt = str((PROJ_ROOT / "inputs/checkpoints/dpvo/dpvo.pth")) if dpvo_ckpt is None else str(dpvo_ckpt)
+
+ self.buffer = buffer
+ self.times = []
+ self.slam = None
+ self.queue = Queue(maxsize=8)
+
+ # Convert masks to numpy for multiprocessing
+ masks_np = None
+ if masks is not None:
+ masks_np = []
+ for m in masks:
+ if m is None:
+ masks_np.append(None)
+ elif isinstance(m, torch.Tensor):
+ masks_np.append(m.numpy())
+ else:
+ masks_np.append(m)
+
+ self.reader = Process(
+ target=video_stream,
+ args=(self.queue, video_path, intrinsics, stride, skip, resize, masks_np, max_frames)
+ )
+ self.reader.start()
+
+ def track(self):
+ (t, image, intrinsics) = self.queue.get()
+
+ if t < 0:
+ return False
+
+ image = torch.from_numpy(image).permute(2, 0, 1).cuda()
+ intrinsics = intrinsics.cuda() # [fx, fy, cx, cy]
+
+ if self.slam is None:
+ cfg.merge_from_file(self.dpvo_cfg)
+ cfg.BUFFER_SIZE = self.buffer
+ self.slam = DPVO(cfg, self.dpvo_ckpt, ht=image.shape[1], wd=image.shape[2], viz=False)
+
+ with Timer("SLAM", enabled=False):
+ t = time.time()
+ self.slam(t, image, intrinsics)
+ self.times.append(time.time() - t)
+
+ return True
+
+ def process(self):
+ for _ in range(12):
+ self.slam.update()
+
+ self.reader.join()
+ return self.slam.terminate()[0]
+
+
+def video_stream(queue, imagedir, intrinsics, stride, skip=0, resize=0.5, masks=None, max_frames: int | None = None):
+ """
+ Video generator with optional masking.
+
+ Args:
+ masks: Optional list of (H, W) binary masks. 1=person (exclude from tracking)
+ """
+ assert len(intrinsics) == 4, "intrinsics should be [fx, fy, cx, cy]"
+
+ cap = cv2.VideoCapture(imagedir)
+ t = 0
+ frame_idx = 0
+
+ for _ in range(skip):
+ ret, image = cap.read()
+ frame_idx += 1
+
+ while True:
+ # Capture frame-by-frame
+ for _ in range(stride):
+ ret, image = cap.read()
+ frame_idx += 1
+ # if frame is read correctly ret is True
+ if not ret:
+ break
+
+ if not ret:
+ break
+
+ # Store original frame index before stride adjustment
+ original_frame_idx = frame_idx - 1
+
+ image = cv2.resize(image, None, fx=resize, fy=resize, interpolation=cv2.INTER_AREA)
+ h, w, _ = image.shape
+ # Crop to be divisible by 16 (DPVO requirement)
+ h_cropped = h - h % 16
+ w_cropped = w - w % 16
+ image = image[:h_cropped, :w_cropped]
+
+ # Apply mask to exclude people from SLAM tracking (like PromptHMR)
+ # Set person pixels to BLACK (0) so DPVO doesn't track features on moving people
+ if masks is not None:
+ mask_idx = original_frame_idx // stride if stride > 1 else original_frame_idx
+ if mask_idx < len(masks) and masks[mask_idx] is not None:
+ mask = masks[mask_idx]
+
+ # Resize mask to match RESIZED frame (before 16-crop)
+ h_resized = int(mask.shape[0] * resize)
+ w_resized = int(mask.shape[1] * resize)
+ mask = cv2.resize(mask.astype(np.uint8), (w_resized, h_resized), interpolation=cv2.INTER_NEAREST)
+
+ # Crop mask the same way as image (divisible by 16)
+ mask = mask[:h_cropped, :w_cropped]
+
+ # PromptHMR style: multiply image by inverted mask (person → 0, background → keep)
+ # mask=1 is person, so we want (1 - mask) to zero out person pixels
+ mask_3ch = (1 - mask[:, :, None]).astype(np.float32)
+ image = (image.astype(np.float32) * mask_3ch).astype(np.uint8)
+
+ intrinsics_ = intrinsics.clone() * resize
+ queue.put((t, image, intrinsics_))
+
+ t += 1
+ if max_frames is not None and t >= int(max_frames):
+ break
+
+ queue.put((-1, image, intrinsics)) # -1 will terminate the process
+ cap.release()
+
+ # wait for the queue to be empty, otherwise the process will end immediately
+ while not queue.empty():
+ time.sleep(1)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/tracker.py b/third_party/GVHMR/hmr4d/utils/preproc/tracker.py
new file mode 100644
index 0000000000000000000000000000000000000000..2eded6b9850934adc1d3ccffd54351bfc7617de9
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/tracker.py
@@ -0,0 +1,297 @@
+import os
+import cv2
+import torch
+import numpy as np
+import shutil
+import tempfile
+import gc
+import hydra
+from hydra.core.global_hydra import GlobalHydra
+from PIL import Image
+from tqdm import tqdm
+from collections import defaultdict
+
+# --- Grounding DINO Import (upgraded to Base) ---
+from transformers import AutoProcessor, AutoModelForZeroShotObjectDetection
+
+# --- SAM2 Import ---
+from sam2.build_sam import build_sam2_video_predictor
+
+# --- HMR4D Imports ---
+from hmr4d.utils.seq_utils import (
+ get_frame_id_list_from_mask,
+ linear_interpolate_frame_ids,
+ frame_id_to_mask,
+ rearrange_by_mask,
+)
+from hmr4d.utils.video_io_utils import get_video_lwh
+from hmr4d.utils.net_utils import moving_average_smooth
+
+
+class Tracker:
+ def __init__(self) -> None:
+ self.device = "cuda" if torch.cuda.is_available() else "cpu"
+
+ # Use bfloat16 on Ampere+ GPUs for efficiency
+ if torch.cuda.is_available() and torch.cuda.get_device_properties(0).major >= 8:
+ self.dtype = torch.bfloat16
+ else:
+ self.dtype = torch.float16
+
+ print(f"Using device: {self.device} with precision: {self.dtype}")
+
+ # --- Initialize Grounding DINO BASE (UPGRADE from Tiny) ---
+ # Base is 3x larger than Tiny with better zero-shot performance
+ self.gd_model_id = "IDEA-Research/grounding-dino-base"
+ hf_cache_dir = os.path.abspath("./third_party/GVHMR/.cache/huggingface")
+ os.makedirs(hf_cache_dir, exist_ok=True)
+
+ try:
+ self.gd_processor = AutoProcessor.from_pretrained(
+ self.gd_model_id,
+ local_files_only=True,
+ cache_dir=hf_cache_dir,
+ )
+ self.gd_model = AutoModelForZeroShotObjectDetection.from_pretrained(
+ self.gd_model_id,
+ local_files_only=True,
+ cache_dir=hf_cache_dir,
+ ).to(self.device)
+ print("[Tracker] Loaded Grounding DINO Base from cache (offline mode)")
+ except Exception as e:
+ offline_env = os.environ.get("HF_HUB_OFFLINE") == "1" or os.environ.get("TRANSFORMERS_OFFLINE") == "1"
+ if offline_env:
+ raise RuntimeError(
+ "Grounding DINO Base not found in local cache and offline mode is enabled. "
+ f"Set internet on once to download into {hf_cache_dir}, or disable offline mode."
+ ) from e
+ print(f"[Tracker] Cache miss, downloading Grounding DINO Base (requires internet): {e}")
+ self.gd_processor = AutoProcessor.from_pretrained(
+ self.gd_model_id,
+ cache_dir=hf_cache_dir,
+ )
+ self.gd_model = AutoModelForZeroShotObjectDetection.from_pretrained(
+ self.gd_model_id,
+ cache_dir=hf_cache_dir,
+ ).to(self.device)
+ print("[Tracker] Grounding DINO Base downloaded and cached for future offline use")
+
+ # --- SAM 2 Video Predictor (lazy init to avoid GPU OOM with DINO) ---
+ self.sam2_checkpoint = "./third_party/GVHMR/Grounded-SAM-2/checkpoints/sam2.1_hiera_large.pt"
+ self.sam2_config_path = "./third_party/GVHMR/Grounded-SAM-2/sam2/configs/sam2.1/sam2.1_hiera_l.yaml"
+ self.video_predictor = None
+
+ # Detection config - expanded prompts for anime/character detection
+ self.text_prompt = "person. human. man. woman. character. anime character. boy. girl."
+ self.box_threshold = 0.20 # Slightly lower for anime
+ self.text_threshold = 0.25
+
+ def _ensure_sam2(self):
+ if self.video_predictor is not None:
+ return
+ config_abs_path = os.path.abspath(self.sam2_config_path)
+ config_dir = os.path.dirname(config_abs_path)
+ config_name = os.path.basename(config_abs_path)
+
+ if GlobalHydra.instance().is_initialized():
+ GlobalHydra.instance().clear()
+ hydra.initialize_config_dir(config_dir=config_dir, version_base="1.2")
+ self.video_predictor = build_sam2_video_predictor(config_name, self.sam2_checkpoint, device=self.device)
+
+ def _prepare_video_frames(self, video_path):
+ temp_dir = tempfile.mkdtemp()
+ cap = cv2.VideoCapture(str(video_path))
+ frame_idx = 0
+ max_frames = 10000
+ while True:
+ ret, frame = cap.read()
+ if not ret: break
+ cv2.imwrite(os.path.join(temp_dir, f"{frame_idx:05d}.jpg"), frame)
+ frame_idx += 1
+ if frame_idx >= max_frames:
+ print(f"WARNING: Extracted {max_frames} frames, stopping to prevent OOM")
+ break
+ cap.release()
+ print(f"[Tracker] Extracted {frame_idx} frames from {video_path}")
+ return temp_dir
+
+ def mask_to_xyxy(self, mask):
+ if not np.any(mask): return np.array([0, 0, 0, 0], dtype=np.float32)
+ y_indices, x_indices = np.where(mask)
+ x_min, x_max = x_indices.min(), x_indices.max()
+ y_min, y_max = y_indices.min(), y_indices.max()
+ return np.array([x_min, y_min, x_max, y_max], dtype=np.float32)
+
+ def track(self, video_path):
+ """Track with SAM2, return history with masks."""
+ video_dir = self._prepare_video_frames(video_path)
+ frame_names = sorted([p for p in os.listdir(video_dir) if p.endswith(".jpg")])
+ if not frame_names: return []
+
+ img_path = os.path.join(video_dir, frame_names[0])
+ image_pil = Image.open(img_path).convert("RGB")
+
+ # Get frame dimensions for fallback
+ frame_w, frame_h = image_pil.size
+
+ # --- STEP 1: Grounding DINO Base Detection ---
+ print("Running Grounding DINO Base...")
+ with torch.inference_mode():
+ inputs = self.gd_processor(images=image_pil, text=self.text_prompt, return_tensors="pt").to(self.device)
+ outputs = self.gd_model(**inputs)
+
+ results = self.gd_processor.post_process_grounded_object_detection(
+ outputs, inputs.input_ids,
+ threshold=self.box_threshold, text_threshold=self.text_threshold,
+ target_sizes=[image_pil.size[::-1]]
+ )
+
+ input_boxes = results[0]["boxes"].cpu().numpy()
+
+ if len(input_boxes) == 0:
+ print("[Tracker] No persons detected, falling back to full frame")
+ input_boxes = np.array([[0, 0, frame_w, frame_h]], dtype=np.float32)
+
+ # Keep only the largest detection
+ if len(input_boxes) > 1:
+ areas = (input_boxes[:, 2] - input_boxes[:, 0]) * (input_boxes[:, 3] - input_boxes[:, 1])
+ largest_idx = np.argmax(areas)
+ input_boxes = input_boxes[largest_idx:largest_idx+1]
+ print(f"[Tracker] Filtered to largest detection (area={areas[largest_idx]:.0f}px²)")
+
+ # Free DINO before SAM2 to reduce GPU pressure.
+ if self.gd_model is not None:
+ try:
+ self.gd_model.to("cpu")
+ except Exception:
+ pass
+ self.gd_model = None
+ self.gd_processor = None
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ # --- STEP 2: SAM 2 Tracking ---
+ print(f"Initializing SAM 2 with {len(input_boxes)} object(s)...")
+ self._ensure_sam2()
+
+ with torch.inference_mode(), torch.autocast(device_type="cuda", dtype=self.dtype):
+ inference_state = self.video_predictor.init_state(
+ video_path=video_dir,
+ offload_video_to_cpu=True,
+ offload_state_to_cpu=True,
+ )
+
+ for obj_id, box in enumerate(input_boxes, start=1):
+ _, out_obj_ids, out_mask_logits = self.video_predictor.add_new_points_or_box(
+ inference_state=inference_state, frame_idx=0, obj_id=obj_id, box=box
+ )
+
+ track_history = []
+ total_frames = len(frame_names)
+ temp_results = {}
+
+ print("Propagating tracking...")
+
+ for out_frame_idx, out_obj_ids, out_mask_logits in tqdm(
+ self.video_predictor.propagate_in_video(inference_state), total=total_frames
+ ):
+ frame_detections = []
+ for i, out_obj_id in enumerate(out_obj_ids):
+ mask = (out_mask_logits[i] > 0.0).cpu().numpy().squeeze()
+
+ if np.sum(mask) > 0:
+ bbox = self.mask_to_xyxy(mask)
+
+ # CRITICAL FIX: Clip mask to bbox to prevent bleeding onto other people
+ # This uses the SAM-derived bbox to constrain the mask
+ x1, y1, x2, y2 = [int(v) for v in bbox]
+ mask_h, mask_w = mask.shape
+ x1, y1 = max(0, x1), max(0, y1)
+ x2, y2 = min(mask_w, x2), min(mask_h, y2)
+
+ # Create clipped mask - only keep pixels within bbox
+ clipped_mask = np.zeros_like(mask)
+ clipped_mask[y1:y2, x1:x2] = mask[y1:y2, x1:x2]
+
+ frame_detections.append({
+ "id": int(out_obj_id),
+ "bbx_xyxy": bbox,
+ "mask": clipped_mask.astype(np.uint8) # Store clipped mask
+ })
+
+ temp_results[out_frame_idx] = frame_detections
+
+ # Re-assemble
+ for i in range(total_frames):
+ track_history.append(temp_results.get(i, []))
+
+ self.video_predictor.reset_state(inference_state)
+ shutil.rmtree(video_dir)
+ return track_history
+
+ @staticmethod
+ def sort_track_length(track_history, video_path):
+ id_to_frame_ids = defaultdict(list)
+ id_to_bbx_xyxys = defaultdict(list)
+ id_to_masks = defaultdict(list) # Store masks
+
+ for frame_id, frame in enumerate(track_history):
+ for det in frame:
+ id_to_frame_ids[det["id"]].append(frame_id)
+ id_to_bbx_xyxys[det["id"]].append(det["bbx_xyxy"])
+ if "mask" in det:
+ id_to_masks[det["id"]].append(det["mask"])
+
+ for k, v in id_to_bbx_xyxys.items():
+ id_to_bbx_xyxys[k] = np.array(v)
+
+ id_area_sum = {}
+ l, w, h = get_video_lwh(video_path)
+ for k, v in id_to_bbx_xyxys.items():
+ bbx_wh = v[:, 2:] - v[:, :2]
+ id_area_sum[k] = (bbx_wh[:, 0] * bbx_wh[:, 1] / w / h).sum()
+ id2area_sum = dict(sorted(id_area_sum.items(), key=lambda item: item[1], reverse=True))
+ id_sorted = list(id2area_sum.keys())
+
+ return id_to_frame_ids, id_to_bbx_xyxys, id_sorted, id_to_masks
+
+ def get_one_track(self, video_path):
+ """Original method - returns only bboxes for backward compatibility."""
+ bbx_xyxy, _ = self.get_one_track_with_masks(video_path)
+ return bbx_xyxy
+
+ def get_one_track_with_masks(self, video_path):
+ """Returns both bounding boxes AND per-frame masks."""
+ track_history = self.track(video_path)
+ if not track_history or all(len(x) == 0 for x in track_history):
+ raise ValueError("Grounding DINO + SAM 2 found no tracks.")
+
+ id_to_frame_ids, id_to_bbx_xyxys, id_sorted, id_to_masks = self.sort_track_length(track_history, video_path)
+ if not id_sorted:
+ raise ValueError("Grounding DINO + SAM 2 found no tracks.")
+
+ track_id = id_sorted[0]
+ frame_ids = torch.tensor(id_to_frame_ids[track_id])
+ bbx_xyxys = torch.tensor(id_to_bbx_xyxys[track_id])
+ masks_list = id_to_masks.get(track_id, [])
+
+ total_frames = get_video_lwh(video_path)[0]
+ mask = frame_id_to_mask(frame_ids, total_frames)
+
+ # Process bboxes
+ bbx_xyxy_one_track = rearrange_by_mask(bbx_xyxys, mask)
+ missing_frame_id_list = get_frame_id_list_from_mask(~mask)
+ bbx_xyxy_one_track = linear_interpolate_frame_ids(bbx_xyxy_one_track, missing_frame_id_list)
+ bbx_xyxy_one_track = moving_average_smooth(bbx_xyxy_one_track, window_size=5, dim=0)
+ bbx_xyxy_one_track = moving_average_smooth(bbx_xyxy_one_track, window_size=5, dim=0)
+
+ # Process masks - create full timeline with None for missing frames
+ if masks_list:
+ frame_id_list = id_to_frame_ids[track_id]
+ masks_full = [None] * total_frames
+ for fid, m in zip(frame_id_list, masks_list):
+ masks_full[fid] = m
+ else:
+ masks_full = [None] * total_frames
+
+ return bbx_xyxy_one_track, masks_full
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitfeat_extractor.py b/third_party/GVHMR/hmr4d/utils/preproc/vitfeat_extractor.py
new file mode 100644
index 0000000000000000000000000000000000000000..fe9c5ca43f37c95042559db2f36817b25a47c897
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitfeat_extractor.py
@@ -0,0 +1,123 @@
+import torch
+import cv2
+import numpy as np
+from tqdm import tqdm
+
+from hmr4d.network.hmr2 import load_hmr2, HMR2
+from hmr4d.utils.video_io_utils import read_video_np
+from hmr4d.network.hmr2.utils.preproc import crop_and_resize, IMAGE_MEAN, IMAGE_STD
+
+
+def get_batch(input_path, bbx_xys, masks=None, img_ds=0.5, img_dst_size=256, path_type="video"):
+ """
+ Get cropped image batch for feature extraction.
+
+ Args:
+ input_path: Video path or numpy array
+ bbx_xys: (F, 3) bounding boxes [cx, cy, size]
+ masks: Optional list of (H, W) binary masks per frame
+ img_ds: Downscale factor
+ img_dst_size: Output crop size
+ """
+ if path_type == "video":
+ imgs = read_video_np(input_path, scale=img_ds)
+ elif path_type == "image":
+ imgs = cv2.imread(str(input_path))[..., ::-1]
+ imgs = cv2.resize(imgs, (0, 0), fx=img_ds, fy=img_ds)
+ imgs = imgs[None]
+ elif path_type == "np":
+ assert isinstance(input_path, np.ndarray)
+ assert img_ds == 1.0
+ imgs = input_path
+
+ gt_center = bbx_xys[:, :2]
+ gt_bbx_size = bbx_xys[:, 2]
+
+ # Apply masks BEFORE cropping (but after downscaling)
+ if masks is not None:
+ for i in range(len(imgs)):
+ if i < len(masks) and masks[i] is not None:
+ mask = masks[i]
+ if isinstance(mask, torch.Tensor):
+ mask = mask.numpy()
+
+ # Resize mask to match downscaled image
+ img_h, img_w = imgs[i].shape[:2]
+ if mask.shape[0] != img_h or mask.shape[1] != img_w:
+ mask = cv2.resize(mask.astype(np.uint8), (img_w, img_h), interpolation=cv2.INTER_NEAREST)
+
+ # Gray background (mean color - neutral for neural networks)
+ gray_value = 128
+ gray_bg = np.full_like(imgs[i], gray_value)
+ mask_3ch = mask[:, :, None].astype(bool)
+ imgs[i] = np.where(mask_3ch, imgs[i], gray_bg)
+
+ # Blur image to avoid aliasing artifacts
+ if True:
+ gt_bbx_size_ds = gt_bbx_size * img_ds
+ ds_factors = ((gt_bbx_size_ds * 1.0) / img_dst_size / 2.0).numpy()
+ imgs = np.stack(
+ [
+ cv2.GaussianBlur(v, (5, 5), (d - 1) / 2) if d > 1.1 else v
+ for v, d in zip(imgs, ds_factors)
+ ]
+ )
+
+ # Output
+ imgs_list = []
+ bbx_xys_ds_list = []
+ for i in range(len(imgs)):
+ img, bbx_xys_ds = crop_and_resize(
+ imgs[i],
+ gt_center[i] * img_ds,
+ gt_bbx_size[i] * img_ds,
+ img_dst_size,
+ enlarge_ratio=1.0,
+ )
+ imgs_list.append(img)
+ bbx_xys_ds_list.append(bbx_xys_ds)
+ imgs = torch.from_numpy(np.stack(imgs_list)) # (F, 256, 256, 3), RGB
+ bbx_xys = torch.from_numpy(np.stack(bbx_xys_ds_list)) / img_ds # (F, 3)
+
+ imgs = ((imgs / 255.0 - IMAGE_MEAN) / IMAGE_STD).permute(0, 3, 1, 2) # (F, 3, 256, 256)
+ return imgs, bbx_xys
+
+
+class Extractor:
+ def __init__(self, tqdm_leave=True):
+ self.extractor: HMR2 = load_hmr2().cuda().eval()
+ self.tqdm_leave = tqdm_leave
+
+ def extract_video_features(self, video_path, bbx_xys, masks=None, img_ds=0.5):
+ """
+ Extract HMR2 features from video.
+
+ Args:
+ video_path: Video path or tensor
+ bbx_xys: (F, 3) bounding boxes
+ masks: Optional list of (H, W) binary masks per frame
+ img_ds: Image downscale factor
+ """
+ # Get the batch (with optional masking)
+ if isinstance(video_path, str):
+ imgs, bbx_xys = get_batch(video_path, bbx_xys, masks=masks, img_ds=img_ds)
+ else:
+ assert isinstance(video_path, torch.Tensor)
+ imgs = video_path
+
+ # Inference
+ F, _, H, W = imgs.shape # (F, 3, H, W)
+ imgs = imgs.cuda()
+ batch_size = 16
+ features = []
+
+ desc = "HMR2 Feature (masked)" if masks is not None else "HMR2 Feature"
+ for j in tqdm(range(0, F, batch_size), desc=desc, leave=self.tqdm_leave):
+ imgs_batch = imgs[j : j + batch_size]
+
+ with torch.no_grad():
+ feature = self.extractor({"img": imgs_batch})
+ features.append(feature.detach().cpu())
+
+ features = torch.cat(features, dim=0).clone() # (F, 1024)
+ return features
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose.py
new file mode 100644
index 0000000000000000000000000000000000000000..963117c66d813c1c83cf153d2f62816e475e1e54
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose.py
@@ -0,0 +1,115 @@
+import torch
+import numpy as np
+import cv2
+import mmcv
+from tqdm import tqdm
+from mmpose.apis import init_model, inference_topdown
+from pathlib import Path
+
+
+class VitPoseExtractor:
+ def __init__(self, tqdm_leave=True):
+ # Finetuned VitPose config (17 COCO keypoints)
+ # Resolve paths relative to the GVHMR repo root to avoid CWD-dependent failures.
+ repo_root = Path(__file__).resolve().parents[3]
+ self.config_file = str(
+ repo_root
+ / "mmpose"
+ / "configs"
+ / "body_2d_keypoint"
+ / "topdown_heatmap"
+ / "coco"
+ / "vitpose_huge_finetune.py"
+ )
+ self.ckpt_path = str(repo_root / "work_dirs" / "best_coco_AP_epoch_1.pth")
+
+ self.pose = init_model(self.config_file, self.ckpt_path, device='cuda:0')
+ self.pose.eval()
+ self.tqdm_leave = tqdm_leave
+
+ def _apply_mask_to_frame(self, frame, mask, bbx_xys):
+ """
+ Apply SAM mask to frame: keep person pixels, set background to gray.
+ This prevents other people's body parts from bleeding into keypoint detection.
+
+ Args:
+ frame: (H, W, 3) BGR image
+ mask: (H, W) binary mask where 1=person, 0=background
+ bbx_xys: (3,) tensor [cx, cy, size] - bounding box center and size
+
+ Returns:
+ masked_frame: (H, W, 3) frame with non-person pixels set to gray
+ """
+ if mask is None:
+ return frame
+
+ # Ensure mask matches frame size
+ frame_h, frame_w = frame.shape[:2]
+ mask_h, mask_w = mask.shape[:2]
+
+ if mask_h != frame_h or mask_w != frame_w:
+ # Resize mask to match frame
+ mask = cv2.resize(mask.astype(np.uint8), (frame_w, frame_h), interpolation=cv2.INTER_NEAREST)
+
+ # Gray background (mean color - neutral for neural networks)
+ gray_value = 128
+ gray_bg = np.full_like(frame, gray_value)
+ mask_3ch = mask[:, :, None].astype(bool)
+ masked_frame = np.where(mask_3ch, frame, gray_bg)
+
+ return masked_frame
+
+ @torch.no_grad()
+ def extract(self, video_path, bbx_xys, masks=None, img_ds=0.5):
+ """
+ Extract 2D keypoints from video.
+
+ Args:
+ video_path: Path to video file or tensor
+ bbx_xys: (L, 3) bounding boxes [cx, cy, size]
+ masks: Optional list of (H, W) binary masks, one per frame
+ img_ds: Image downscale factor (not used in current implementation)
+
+ Returns:
+ vitpose: (L, 17, 3) keypoints with confidence
+ """
+ if isinstance(video_path, str):
+ video = mmcv.VideoReader(video_path)
+ elif isinstance(video_path, torch.Tensor):
+ video = video_path.permute(0, 2, 3, 1).cpu().numpy()
+ if video.max() <= 1.0:
+ video = (video * 255).astype(np.uint8)
+ else:
+ video = video_path
+
+ L = len(bbx_xys)
+ vitpose_results = []
+
+ for i in tqdm(range(L), desc="ViTPose (masked)" if masks else "ViTPose", leave=self.tqdm_leave):
+ frame = video[i]
+
+ # NEW: Apply SAM mask to isolate target person
+ if masks is not None and i < len(masks) and masks[i] is not None:
+ frame = self._apply_mask_to_frame(frame, masks[i], bbx_xys[i])
+
+ cx, cy, s = bbx_xys[i][0].item(), bbx_xys[i][1].item(), bbx_xys[i][2].item()
+
+ # Keep the s / 1.25 logic as it matches the val_pipeline in finetune config
+ s_adjusted = s / 1.25
+ w = s_adjusted * 0.75
+ h = s_adjusted
+
+ bbox = np.array([[cx - w/2, cy - h/2, cx + w/2, cy + h/2]], dtype=np.float32)
+
+ results = inference_topdown(self.pose, frame, bboxes=bbox)
+
+ if len(results) > 0:
+ kpts = results[0].pred_instances.keypoints[0]
+ scores = results[0].pred_instances.keypoint_scores[0][:, None]
+ kp2d = np.concatenate([kpts, scores], axis=-1)
+ else:
+ kp2d = np.zeros((17, 3))
+
+ vitpose_results.append(torch.from_numpy(kp2d).float())
+
+ return torch.stack(vitpose_results, dim=0)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/__init__.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..df55ce4ddebd2a3784f75ee9975a941894ae9a59
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/__init__.py
@@ -0,0 +1 @@
+from .src.vitpose_infer.model_builder import build_model
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/__init__.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/__init__.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/__init__.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..44586e800a6b7e4b880cfe1bc35df3e9d99510bd
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/__init__.py
@@ -0,0 +1,35 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+# from .alexnet import AlexNet
+# from .cpm import CPM
+# from .hourglass import HourglassNet
+# from .hourglass_ae import HourglassAENet
+# from .hrformer import HRFormer
+# from .hrnet import HRNet
+# from .litehrnet import LiteHRNet
+# from .mobilenet_v2 import MobileNetV2
+# from .mobilenet_v3 import MobileNetV3
+# from .mspn import MSPN
+# from .regnet import RegNet
+# from .resnest import ResNeSt
+# from .resnet import ResNet, ResNetV1d
+# from .resnext import ResNeXt
+# from .rsn import RSN
+# from .scnet import SCNet
+# from .seresnet import SEResNet
+# from .seresnext import SEResNeXt
+# from .shufflenet_v1 import ShuffleNetV1
+# from .shufflenet_v2 import ShuffleNetV2
+# from .tcn import TCN
+# from .v2v_net import V2VNet
+# from .vgg import VGG
+# from .vipnas_mbv3 import ViPNAS_MobileNetV3
+# from .vipnas_resnet import ViPNAS_ResNet
+from .vit import ViT
+
+# __all__ = [
+# 'AlexNet', 'HourglassNet', 'HourglassAENet', 'HRNet', 'MobileNetV2',
+# 'MobileNetV3', 'RegNet', 'ResNet', 'ResNetV1d', 'ResNeXt', 'SCNet',
+# 'SEResNet', 'SEResNeXt', 'ShuffleNetV1', 'ShuffleNetV2', 'CPM', 'RSN',
+# 'MSPN', 'ResNeSt', 'VGG', 'TCN', 'ViPNAS_ResNet', 'ViPNAS_MobileNetV3',
+# 'LiteHRNet', 'V2VNet', 'HRFormer', 'ViT'
+# ]
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/alexnet.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/alexnet.py
new file mode 100644
index 0000000000000000000000000000000000000000..a8efd74d118f5abe4d9c880ebe80ce7cbd58c6b2
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/alexnet.py
@@ -0,0 +1,56 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import torch.nn as nn
+
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+
+
+@BACKBONES.register_module()
+class AlexNet(BaseBackbone):
+ """`AlexNet `__ backbone.
+
+ The input for AlexNet is a 224x224 RGB image.
+
+ Args:
+ num_classes (int): number of classes for classification.
+ The default value is -1, which uses the backbone as
+ a feature extractor without the top classifier.
+ """
+
+ def __init__(self, num_classes=-1):
+ super().__init__()
+ self.num_classes = num_classes
+ self.features = nn.Sequential(
+ nn.Conv2d(3, 64, kernel_size=11, stride=4, padding=2),
+ nn.ReLU(inplace=True),
+ nn.MaxPool2d(kernel_size=3, stride=2),
+ nn.Conv2d(64, 192, kernel_size=5, padding=2),
+ nn.ReLU(inplace=True),
+ nn.MaxPool2d(kernel_size=3, stride=2),
+ nn.Conv2d(192, 384, kernel_size=3, padding=1),
+ nn.ReLU(inplace=True),
+ nn.Conv2d(384, 256, kernel_size=3, padding=1),
+ nn.ReLU(inplace=True),
+ nn.Conv2d(256, 256, kernel_size=3, padding=1),
+ nn.ReLU(inplace=True),
+ nn.MaxPool2d(kernel_size=3, stride=2),
+ )
+ if self.num_classes > 0:
+ self.classifier = nn.Sequential(
+ nn.Dropout(),
+ nn.Linear(256 * 6 * 6, 4096),
+ nn.ReLU(inplace=True),
+ nn.Dropout(),
+ nn.Linear(4096, 4096),
+ nn.ReLU(inplace=True),
+ nn.Linear(4096, num_classes),
+ )
+
+ def forward(self, x):
+
+ x = self.features(x)
+ if self.num_classes > 0:
+ x = x.view(x.size(0), 256 * 6 * 6)
+ x = self.classifier(x)
+
+ return x
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/cpm.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/cpm.py
new file mode 100644
index 0000000000000000000000000000000000000000..458245d755f930f4ff625a754aadbab5c13494a6
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/cpm.py
@@ -0,0 +1,186 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+
+import torch
+import torch.nn as nn
+from mmcv.cnn import ConvModule, constant_init, normal_init
+from torch.nn.modules.batchnorm import _BatchNorm
+
+from mmpose.utils import get_root_logger
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+from .utils import load_checkpoint
+
+
+class CpmBlock(nn.Module):
+ """CpmBlock for Convolutional Pose Machine.
+
+ Args:
+ in_channels (int): Input channels of this block.
+ channels (list): Output channels of each conv module.
+ kernels (list): Kernel sizes of each conv module.
+ """
+
+ def __init__(self,
+ in_channels,
+ channels=(128, 128, 128),
+ kernels=(11, 11, 11),
+ norm_cfg=None):
+ super().__init__()
+
+ assert len(channels) == len(kernels)
+ layers = []
+ for i in range(len(channels)):
+ if i == 0:
+ input_channels = in_channels
+ else:
+ input_channels = channels[i - 1]
+ layers.append(
+ ConvModule(
+ input_channels,
+ channels[i],
+ kernels[i],
+ padding=(kernels[i] - 1) // 2,
+ norm_cfg=norm_cfg))
+ self.model = nn.Sequential(*layers)
+
+ def forward(self, x):
+ """Model forward function."""
+ out = self.model(x)
+ return out
+
+
+@BACKBONES.register_module()
+class CPM(BaseBackbone):
+ """CPM backbone.
+
+ Convolutional Pose Machines.
+ More details can be found in the `paper
+ `__ .
+
+ Args:
+ in_channels (int): The input channels of the CPM.
+ out_channels (int): The output channels of the CPM.
+ feat_channels (int): Feature channel of each CPM stage.
+ middle_channels (int): Feature channel of conv after the middle stage.
+ num_stages (int): Number of stages.
+ norm_cfg (dict): Dictionary to construct and config norm layer.
+
+ Example:
+ >>> from mmpose.models import CPM
+ >>> import torch
+ >>> self = CPM(3, 17)
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 3, 368, 368)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_output in level_outputs:
+ ... print(tuple(level_output.shape))
+ (1, 17, 46, 46)
+ (1, 17, 46, 46)
+ (1, 17, 46, 46)
+ (1, 17, 46, 46)
+ (1, 17, 46, 46)
+ (1, 17, 46, 46)
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ feat_channels=128,
+ middle_channels=32,
+ num_stages=6,
+ norm_cfg=dict(type='BN', requires_grad=True)):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+
+ assert in_channels == 3
+
+ self.num_stages = num_stages
+ assert self.num_stages >= 1
+
+ self.stem = nn.Sequential(
+ ConvModule(in_channels, 128, 9, padding=4, norm_cfg=norm_cfg),
+ nn.MaxPool2d(kernel_size=3, stride=2, padding=1),
+ ConvModule(128, 128, 9, padding=4, norm_cfg=norm_cfg),
+ nn.MaxPool2d(kernel_size=3, stride=2, padding=1),
+ ConvModule(128, 128, 9, padding=4, norm_cfg=norm_cfg),
+ nn.MaxPool2d(kernel_size=3, stride=2, padding=1),
+ ConvModule(128, 32, 5, padding=2, norm_cfg=norm_cfg),
+ ConvModule(32, 512, 9, padding=4, norm_cfg=norm_cfg),
+ ConvModule(512, 512, 1, padding=0, norm_cfg=norm_cfg),
+ ConvModule(512, out_channels, 1, padding=0, act_cfg=None))
+
+ self.middle = nn.Sequential(
+ ConvModule(in_channels, 128, 9, padding=4, norm_cfg=norm_cfg),
+ nn.MaxPool2d(kernel_size=3, stride=2, padding=1),
+ ConvModule(128, 128, 9, padding=4, norm_cfg=norm_cfg),
+ nn.MaxPool2d(kernel_size=3, stride=2, padding=1),
+ ConvModule(128, 128, 9, padding=4, norm_cfg=norm_cfg),
+ nn.MaxPool2d(kernel_size=3, stride=2, padding=1))
+
+ self.cpm_stages = nn.ModuleList([
+ CpmBlock(
+ middle_channels + out_channels,
+ channels=[feat_channels, feat_channels, feat_channels],
+ kernels=[11, 11, 11],
+ norm_cfg=norm_cfg) for _ in range(num_stages - 1)
+ ])
+
+ self.middle_conv = nn.ModuleList([
+ nn.Sequential(
+ ConvModule(
+ 128, middle_channels, 5, padding=2, norm_cfg=norm_cfg))
+ for _ in range(num_stages - 1)
+ ])
+
+ self.out_convs = nn.ModuleList([
+ nn.Sequential(
+ ConvModule(
+ feat_channels,
+ feat_channels,
+ 1,
+ padding=0,
+ norm_cfg=norm_cfg),
+ ConvModule(feat_channels, out_channels, 1, act_cfg=None))
+ for _ in range(num_stages - 1)
+ ])
+
+ def init_weights(self, pretrained=None):
+ """Initialize the weights in backbone.
+
+ Args:
+ pretrained (str, optional): Path to pre-trained weights.
+ Defaults to None.
+ """
+ if isinstance(pretrained, str):
+ logger = get_root_logger()
+ load_checkpoint(self, pretrained, strict=False, logger=logger)
+ elif pretrained is None:
+ for m in self.modules():
+ if isinstance(m, nn.Conv2d):
+ normal_init(m, std=0.001)
+ elif isinstance(m, (_BatchNorm, nn.GroupNorm)):
+ constant_init(m, 1)
+ else:
+ raise TypeError('pretrained must be a str or None')
+
+ def forward(self, x):
+ """Model forward function."""
+ stage1_out = self.stem(x)
+ middle_out = self.middle(x)
+ out_feats = []
+
+ out_feats.append(stage1_out)
+
+ for ind in range(self.num_stages - 1):
+ single_stage = self.cpm_stages[ind]
+ out_conv = self.out_convs[ind]
+
+ inp_feat = torch.cat(
+ [out_feats[-1], self.middle_conv[ind](middle_out)], 1)
+ cpm_feat = single_stage(inp_feat)
+ out_feat = out_conv(cpm_feat)
+ out_feats.append(out_feat)
+
+ return out_feats
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/hourglass.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/hourglass.py
new file mode 100644
index 0000000000000000000000000000000000000000..bf75fad9895ebfd3f3c2a6bffedb3d7e4cc77cba
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/hourglass.py
@@ -0,0 +1,212 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+
+import torch.nn as nn
+from mmcv.cnn import ConvModule, constant_init, normal_init
+from torch.nn.modules.batchnorm import _BatchNorm
+
+from mmpose.utils import get_root_logger
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+from .resnet import BasicBlock, ResLayer
+from .utils import load_checkpoint
+
+
+class HourglassModule(nn.Module):
+ """Hourglass Module for HourglassNet backbone.
+
+ Generate module recursively and use BasicBlock as the base unit.
+
+ Args:
+ depth (int): Depth of current HourglassModule.
+ stage_channels (list[int]): Feature channels of sub-modules in current
+ and follow-up HourglassModule.
+ stage_blocks (list[int]): Number of sub-modules stacked in current and
+ follow-up HourglassModule.
+ norm_cfg (dict): Dictionary to construct and config norm layer.
+ """
+
+ def __init__(self,
+ depth,
+ stage_channels,
+ stage_blocks,
+ norm_cfg=dict(type='BN', requires_grad=True)):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+
+ self.depth = depth
+
+ cur_block = stage_blocks[0]
+ next_block = stage_blocks[1]
+
+ cur_channel = stage_channels[0]
+ next_channel = stage_channels[1]
+
+ self.up1 = ResLayer(
+ BasicBlock, cur_block, cur_channel, cur_channel, norm_cfg=norm_cfg)
+
+ self.low1 = ResLayer(
+ BasicBlock,
+ cur_block,
+ cur_channel,
+ next_channel,
+ stride=2,
+ norm_cfg=norm_cfg)
+
+ if self.depth > 1:
+ self.low2 = HourglassModule(depth - 1, stage_channels[1:],
+ stage_blocks[1:])
+ else:
+ self.low2 = ResLayer(
+ BasicBlock,
+ next_block,
+ next_channel,
+ next_channel,
+ norm_cfg=norm_cfg)
+
+ self.low3 = ResLayer(
+ BasicBlock,
+ cur_block,
+ next_channel,
+ cur_channel,
+ norm_cfg=norm_cfg,
+ downsample_first=False)
+
+ self.up2 = nn.Upsample(scale_factor=2)
+
+ def forward(self, x):
+ """Model forward function."""
+ up1 = self.up1(x)
+ low1 = self.low1(x)
+ low2 = self.low2(low1)
+ low3 = self.low3(low2)
+ up2 = self.up2(low3)
+ return up1 + up2
+
+
+@BACKBONES.register_module()
+class HourglassNet(BaseBackbone):
+ """HourglassNet backbone.
+
+ Stacked Hourglass Networks for Human Pose Estimation.
+ More details can be found in the `paper
+ `__ .
+
+ Args:
+ downsample_times (int): Downsample times in a HourglassModule.
+ num_stacks (int): Number of HourglassModule modules stacked,
+ 1 for Hourglass-52, 2 for Hourglass-104.
+ stage_channels (list[int]): Feature channel of each sub-module in a
+ HourglassModule.
+ stage_blocks (list[int]): Number of sub-modules stacked in a
+ HourglassModule.
+ feat_channel (int): Feature channel of conv after a HourglassModule.
+ norm_cfg (dict): Dictionary to construct and config norm layer.
+
+ Example:
+ >>> from mmpose.models import HourglassNet
+ >>> import torch
+ >>> self = HourglassNet()
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 3, 511, 511)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_output in level_outputs:
+ ... print(tuple(level_output.shape))
+ (1, 256, 128, 128)
+ (1, 256, 128, 128)
+ """
+
+ def __init__(self,
+ downsample_times=5,
+ num_stacks=2,
+ stage_channels=(256, 256, 384, 384, 384, 512),
+ stage_blocks=(2, 2, 2, 2, 2, 4),
+ feat_channel=256,
+ norm_cfg=dict(type='BN', requires_grad=True)):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+
+ self.num_stacks = num_stacks
+ assert self.num_stacks >= 1
+ assert len(stage_channels) == len(stage_blocks)
+ assert len(stage_channels) > downsample_times
+
+ cur_channel = stage_channels[0]
+
+ self.stem = nn.Sequential(
+ ConvModule(3, 128, 7, padding=3, stride=2, norm_cfg=norm_cfg),
+ ResLayer(BasicBlock, 1, 128, 256, stride=2, norm_cfg=norm_cfg))
+
+ self.hourglass_modules = nn.ModuleList([
+ HourglassModule(downsample_times, stage_channels, stage_blocks)
+ for _ in range(num_stacks)
+ ])
+
+ self.inters = ResLayer(
+ BasicBlock,
+ num_stacks - 1,
+ cur_channel,
+ cur_channel,
+ norm_cfg=norm_cfg)
+
+ self.conv1x1s = nn.ModuleList([
+ ConvModule(
+ cur_channel, cur_channel, 1, norm_cfg=norm_cfg, act_cfg=None)
+ for _ in range(num_stacks - 1)
+ ])
+
+ self.out_convs = nn.ModuleList([
+ ConvModule(
+ cur_channel, feat_channel, 3, padding=1, norm_cfg=norm_cfg)
+ for _ in range(num_stacks)
+ ])
+
+ self.remap_convs = nn.ModuleList([
+ ConvModule(
+ feat_channel, cur_channel, 1, norm_cfg=norm_cfg, act_cfg=None)
+ for _ in range(num_stacks - 1)
+ ])
+
+ self.relu = nn.ReLU(inplace=True)
+
+ def init_weights(self, pretrained=None):
+ """Initialize the weights in backbone.
+
+ Args:
+ pretrained (str, optional): Path to pre-trained weights.
+ Defaults to None.
+ """
+ if isinstance(pretrained, str):
+ logger = get_root_logger()
+ load_checkpoint(self, pretrained, strict=False, logger=logger)
+ elif pretrained is None:
+ for m in self.modules():
+ if isinstance(m, nn.Conv2d):
+ normal_init(m, std=0.001)
+ elif isinstance(m, (_BatchNorm, nn.GroupNorm)):
+ constant_init(m, 1)
+ else:
+ raise TypeError('pretrained must be a str or None')
+
+ def forward(self, x):
+ """Model forward function."""
+ inter_feat = self.stem(x)
+ out_feats = []
+
+ for ind in range(self.num_stacks):
+ single_hourglass = self.hourglass_modules[ind]
+ out_conv = self.out_convs[ind]
+
+ hourglass_feat = single_hourglass(inter_feat)
+ out_feat = out_conv(hourglass_feat)
+ out_feats.append(out_feat)
+
+ if ind < self.num_stacks - 1:
+ inter_feat = self.conv1x1s[ind](
+ inter_feat) + self.remap_convs[ind](
+ out_feat)
+ inter_feat = self.inters[ind](self.relu(inter_feat))
+
+ return out_feats
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/hourglass_ae.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/hourglass_ae.py
new file mode 100644
index 0000000000000000000000000000000000000000..5a700e5cb2157fd1dc16771145f065e991b270ea
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/hourglass_ae.py
@@ -0,0 +1,212 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+
+import torch.nn as nn
+from mmcv.cnn import ConvModule, MaxPool2d, constant_init, normal_init
+from torch.nn.modules.batchnorm import _BatchNorm
+
+from mmpose.utils import get_root_logger
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+from .utils import load_checkpoint
+
+
+class HourglassAEModule(nn.Module):
+ """Modified Hourglass Module for HourglassNet_AE backbone.
+
+ Generate module recursively and use BasicBlock as the base unit.
+
+ Args:
+ depth (int): Depth of current HourglassModule.
+ stage_channels (list[int]): Feature channels of sub-modules in current
+ and follow-up HourglassModule.
+ norm_cfg (dict): Dictionary to construct and config norm layer.
+ """
+
+ def __init__(self,
+ depth,
+ stage_channels,
+ norm_cfg=dict(type='BN', requires_grad=True)):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+
+ self.depth = depth
+
+ cur_channel = stage_channels[0]
+ next_channel = stage_channels[1]
+
+ self.up1 = ConvModule(
+ cur_channel, cur_channel, 3, padding=1, norm_cfg=norm_cfg)
+
+ self.pool1 = MaxPool2d(2, 2)
+
+ self.low1 = ConvModule(
+ cur_channel, next_channel, 3, padding=1, norm_cfg=norm_cfg)
+
+ if self.depth > 1:
+ self.low2 = HourglassAEModule(depth - 1, stage_channels[1:])
+ else:
+ self.low2 = ConvModule(
+ next_channel, next_channel, 3, padding=1, norm_cfg=norm_cfg)
+
+ self.low3 = ConvModule(
+ next_channel, cur_channel, 3, padding=1, norm_cfg=norm_cfg)
+
+ self.up2 = nn.UpsamplingNearest2d(scale_factor=2)
+
+ def forward(self, x):
+ """Model forward function."""
+ up1 = self.up1(x)
+ pool1 = self.pool1(x)
+ low1 = self.low1(pool1)
+ low2 = self.low2(low1)
+ low3 = self.low3(low2)
+ up2 = self.up2(low3)
+ return up1 + up2
+
+
+@BACKBONES.register_module()
+class HourglassAENet(BaseBackbone):
+ """Hourglass-AE Network proposed by Newell et al.
+
+ Associative Embedding: End-to-End Learning for Joint
+ Detection and Grouping.
+
+ More details can be found in the `paper
+ `__ .
+
+ Args:
+ downsample_times (int): Downsample times in a HourglassModule.
+ num_stacks (int): Number of HourglassModule modules stacked,
+ 1 for Hourglass-52, 2 for Hourglass-104.
+ stage_channels (list[int]): Feature channel of each sub-module in a
+ HourglassModule.
+ stage_blocks (list[int]): Number of sub-modules stacked in a
+ HourglassModule.
+ feat_channels (int): Feature channel of conv after a HourglassModule.
+ norm_cfg (dict): Dictionary to construct and config norm layer.
+
+ Example:
+ >>> from mmpose.models import HourglassAENet
+ >>> import torch
+ >>> self = HourglassAENet()
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 3, 512, 512)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_output in level_outputs:
+ ... print(tuple(level_output.shape))
+ (1, 34, 128, 128)
+ """
+
+ def __init__(self,
+ downsample_times=4,
+ num_stacks=1,
+ out_channels=34,
+ stage_channels=(256, 384, 512, 640, 768),
+ feat_channels=256,
+ norm_cfg=dict(type='BN', requires_grad=True)):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+
+ self.num_stacks = num_stacks
+ assert self.num_stacks >= 1
+ assert len(stage_channels) > downsample_times
+
+ cur_channels = stage_channels[0]
+
+ self.stem = nn.Sequential(
+ ConvModule(3, 64, 7, padding=3, stride=2, norm_cfg=norm_cfg),
+ ConvModule(64, 128, 3, padding=1, norm_cfg=norm_cfg),
+ MaxPool2d(2, 2),
+ ConvModule(128, 128, 3, padding=1, norm_cfg=norm_cfg),
+ ConvModule(128, feat_channels, 3, padding=1, norm_cfg=norm_cfg),
+ )
+
+ self.hourglass_modules = nn.ModuleList([
+ nn.Sequential(
+ HourglassAEModule(
+ downsample_times, stage_channels, norm_cfg=norm_cfg),
+ ConvModule(
+ feat_channels,
+ feat_channels,
+ 3,
+ padding=1,
+ norm_cfg=norm_cfg),
+ ConvModule(
+ feat_channels,
+ feat_channels,
+ 3,
+ padding=1,
+ norm_cfg=norm_cfg)) for _ in range(num_stacks)
+ ])
+
+ self.out_convs = nn.ModuleList([
+ ConvModule(
+ cur_channels,
+ out_channels,
+ 1,
+ padding=0,
+ norm_cfg=None,
+ act_cfg=None) for _ in range(num_stacks)
+ ])
+
+ self.remap_out_convs = nn.ModuleList([
+ ConvModule(
+ out_channels,
+ feat_channels,
+ 1,
+ norm_cfg=norm_cfg,
+ act_cfg=None) for _ in range(num_stacks - 1)
+ ])
+
+ self.remap_feature_convs = nn.ModuleList([
+ ConvModule(
+ feat_channels,
+ feat_channels,
+ 1,
+ norm_cfg=norm_cfg,
+ act_cfg=None) for _ in range(num_stacks - 1)
+ ])
+
+ self.relu = nn.ReLU(inplace=True)
+
+ def init_weights(self, pretrained=None):
+ """Initialize the weights in backbone.
+
+ Args:
+ pretrained (str, optional): Path to pre-trained weights.
+ Defaults to None.
+ """
+ if isinstance(pretrained, str):
+ logger = get_root_logger()
+ load_checkpoint(self, pretrained, strict=False, logger=logger)
+ elif pretrained is None:
+ for m in self.modules():
+ if isinstance(m, nn.Conv2d):
+ normal_init(m, std=0.001)
+ elif isinstance(m, (_BatchNorm, nn.GroupNorm)):
+ constant_init(m, 1)
+ else:
+ raise TypeError('pretrained must be a str or None')
+
+ def forward(self, x):
+ """Model forward function."""
+ inter_feat = self.stem(x)
+ out_feats = []
+
+ for ind in range(self.num_stacks):
+ single_hourglass = self.hourglass_modules[ind]
+ out_conv = self.out_convs[ind]
+
+ hourglass_feat = single_hourglass(inter_feat)
+ out_feat = out_conv(hourglass_feat)
+ out_feats.append(out_feat)
+
+ if ind < self.num_stacks - 1:
+ inter_feat = inter_feat + self.remap_out_convs[ind](
+ out_feat) + self.remap_feature_convs[ind](
+ hourglass_feat)
+
+ return out_feats
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/hrformer.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/hrformer.py
new file mode 100644
index 0000000000000000000000000000000000000000..b843300a9fdb85908678c5a3fd45ce19e97ce2fe
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/hrformer.py
@@ -0,0 +1,746 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+
+import math
+
+import torch
+import torch.nn as nn
+# from timm.models.layers import to_2tuple, trunc_normal_
+from mmcv.cnn import (build_activation_layer, build_conv_layer,
+ build_norm_layer, trunc_normal_init)
+from mmcv.cnn.bricks.transformer import build_dropout
+from mmcv.runner import BaseModule
+from torch.nn.functional import pad
+
+from ..builder import BACKBONES
+from .hrnet import Bottleneck, HRModule, HRNet
+
+
+def nlc_to_nchw(x, hw_shape):
+ """Convert [N, L, C] shape tensor to [N, C, H, W] shape tensor.
+
+ Args:
+ x (Tensor): The input tensor of shape [N, L, C] before conversion.
+ hw_shape (Sequence[int]): The height and width of output feature map.
+
+ Returns:
+ Tensor: The output tensor of shape [N, C, H, W] after conversion.
+ """
+ H, W = hw_shape
+ assert len(x.shape) == 3
+ B, L, C = x.shape
+ assert L == H * W, 'The seq_len doesn\'t match H, W'
+ return x.transpose(1, 2).reshape(B, C, H, W)
+
+
+def nchw_to_nlc(x):
+ """Flatten [N, C, H, W] shape tensor to [N, L, C] shape tensor.
+
+ Args:
+ x (Tensor): The input tensor of shape [N, C, H, W] before conversion.
+
+ Returns:
+ Tensor: The output tensor of shape [N, L, C] after conversion.
+ """
+ assert len(x.shape) == 4
+ return x.flatten(2).transpose(1, 2).contiguous()
+
+
+def build_drop_path(drop_path_rate):
+ """Build drop path layer."""
+ return build_dropout(dict(type='DropPath', drop_prob=drop_path_rate))
+
+
+class WindowMSA(BaseModule):
+ """Window based multi-head self-attention (W-MSA) module with relative
+ position bias.
+
+ Args:
+ embed_dims (int): Number of input channels.
+ num_heads (int): Number of attention heads.
+ window_size (tuple[int]): The height and width of the window.
+ qkv_bias (bool, optional): If True, add a learnable bias to q, k, v.
+ Default: True.
+ qk_scale (float | None, optional): Override default qk scale of
+ head_dim ** -0.5 if set. Default: None.
+ attn_drop_rate (float, optional): Dropout ratio of attention weight.
+ Default: 0.0
+ proj_drop_rate (float, optional): Dropout ratio of output. Default: 0.
+ with_rpe (bool, optional): If True, use relative position bias.
+ Default: True.
+ init_cfg (dict | None, optional): The Config for initialization.
+ Default: None.
+ """
+
+ def __init__(self,
+ embed_dims,
+ num_heads,
+ window_size,
+ qkv_bias=True,
+ qk_scale=None,
+ attn_drop_rate=0.,
+ proj_drop_rate=0.,
+ with_rpe=True,
+ init_cfg=None):
+
+ super().__init__(init_cfg=init_cfg)
+ self.embed_dims = embed_dims
+ self.window_size = window_size # Wh, Ww
+ self.num_heads = num_heads
+ head_embed_dims = embed_dims // num_heads
+ self.scale = qk_scale or head_embed_dims**-0.5
+
+ self.with_rpe = with_rpe
+ if self.with_rpe:
+ # define a parameter table of relative position bias
+ self.relative_position_bias_table = nn.Parameter(
+ torch.zeros(
+ (2 * window_size[0] - 1) * (2 * window_size[1] - 1),
+ num_heads)) # 2*Wh-1 * 2*Ww-1, nH
+
+ Wh, Ww = self.window_size
+ rel_index_coords = self.double_step_seq(2 * Ww - 1, Wh, 1, Ww)
+ rel_position_index = rel_index_coords + rel_index_coords.T
+ rel_position_index = rel_position_index.flip(1).contiguous()
+ self.register_buffer('relative_position_index', rel_position_index)
+
+ self.qkv = nn.Linear(embed_dims, embed_dims * 3, bias=qkv_bias)
+ self.attn_drop = nn.Dropout(attn_drop_rate)
+ self.proj = nn.Linear(embed_dims, embed_dims)
+ self.proj_drop = nn.Dropout(proj_drop_rate)
+
+ self.softmax = nn.Softmax(dim=-1)
+
+ def init_weights(self):
+ trunc_normal_init(self.relative_position_bias_table, std=0.02)
+
+ def forward(self, x, mask=None):
+ """
+ Args:
+
+ x (tensor): input features with shape of (B*num_windows, N, C)
+ mask (tensor | None, Optional): mask with shape of (num_windows,
+ Wh*Ww, Wh*Ww), value should be between (-inf, 0].
+ """
+ B, N, C = x.shape
+ qkv = self.qkv(x).reshape(B, N, 3, self.num_heads,
+ C // self.num_heads).permute(2, 0, 3, 1, 4)
+ q, k, v = qkv[0], qkv[1], qkv[2]
+
+ q = q * self.scale
+ attn = (q @ k.transpose(-2, -1))
+
+ if self.with_rpe:
+ relative_position_bias = self.relative_position_bias_table[
+ self.relative_position_index.view(-1)].view(
+ self.window_size[0] * self.window_size[1],
+ self.window_size[0] * self.window_size[1],
+ -1) # Wh*Ww,Wh*Ww,nH
+ relative_position_bias = relative_position_bias.permute(
+ 2, 0, 1).contiguous() # nH, Wh*Ww, Wh*Ww
+ attn = attn + relative_position_bias.unsqueeze(0)
+
+ if mask is not None:
+ nW = mask.shape[0]
+ attn = attn.view(B // nW, nW, self.num_heads, N,
+ N) + mask.unsqueeze(1).unsqueeze(0)
+ attn = attn.view(-1, self.num_heads, N, N)
+ attn = self.softmax(attn)
+
+ attn = self.attn_drop(attn)
+
+ x = (attn @ v).transpose(1, 2).reshape(B, N, C)
+ x = self.proj(x)
+ x = self.proj_drop(x)
+ return x
+
+ @staticmethod
+ def double_step_seq(step1, len1, step2, len2):
+ seq1 = torch.arange(0, step1 * len1, step1)
+ seq2 = torch.arange(0, step2 * len2, step2)
+ return (seq1[:, None] + seq2[None, :]).reshape(1, -1)
+
+
+class LocalWindowSelfAttention(BaseModule):
+ r""" Local-window Self Attention (LSA) module with relative position bias.
+
+ This module is the short-range self-attention module in the
+ Interlaced Sparse Self-Attention `_.
+
+ Args:
+ embed_dims (int): Number of input channels.
+ num_heads (int): Number of attention heads.
+ window_size (tuple[int] | int): The height and width of the window.
+ qkv_bias (bool, optional): If True, add a learnable bias to q, k, v.
+ Default: True.
+ qk_scale (float | None, optional): Override default qk scale of
+ head_dim ** -0.5 if set. Default: None.
+ attn_drop_rate (float, optional): Dropout ratio of attention weight.
+ Default: 0.0
+ proj_drop_rate (float, optional): Dropout ratio of output. Default: 0.
+ with_rpe (bool, optional): If True, use relative position bias.
+ Default: True.
+ with_pad_mask (bool, optional): If True, mask out the padded tokens in
+ the attention process. Default: False.
+ init_cfg (dict | None, optional): The Config for initialization.
+ Default: None.
+ """
+
+ def __init__(self,
+ embed_dims,
+ num_heads,
+ window_size,
+ qkv_bias=True,
+ qk_scale=None,
+ attn_drop_rate=0.,
+ proj_drop_rate=0.,
+ with_rpe=True,
+ with_pad_mask=False,
+ init_cfg=None):
+ super().__init__(init_cfg=init_cfg)
+ if isinstance(window_size, int):
+ window_size = (window_size, window_size)
+ self.window_size = window_size
+ self.with_pad_mask = with_pad_mask
+ self.attn = WindowMSA(
+ embed_dims=embed_dims,
+ num_heads=num_heads,
+ window_size=window_size,
+ qkv_bias=qkv_bias,
+ qk_scale=qk_scale,
+ attn_drop_rate=attn_drop_rate,
+ proj_drop_rate=proj_drop_rate,
+ with_rpe=with_rpe,
+ init_cfg=init_cfg)
+
+ def forward(self, x, H, W, **kwargs):
+ """Forward function."""
+ B, N, C = x.shape
+ x = x.view(B, H, W, C)
+ Wh, Ww = self.window_size
+
+ # center-pad the feature on H and W axes
+ pad_h = math.ceil(H / Wh) * Wh - H
+ pad_w = math.ceil(W / Ww) * Ww - W
+ x = pad(x, (0, 0, pad_w // 2, pad_w - pad_w // 2, pad_h // 2,
+ pad_h - pad_h // 2))
+
+ # permute
+ x = x.view(B, math.ceil(H / Wh), Wh, math.ceil(W / Ww), Ww, C)
+ x = x.permute(0, 1, 3, 2, 4, 5)
+ x = x.reshape(-1, Wh * Ww, C) # (B*num_window, Wh*Ww, C)
+
+ # attention
+ if self.with_pad_mask and pad_h > 0 and pad_w > 0:
+ pad_mask = x.new_zeros(1, H, W, 1)
+ pad_mask = pad(
+ pad_mask, [
+ 0, 0, pad_w // 2, pad_w - pad_w // 2, pad_h // 2,
+ pad_h - pad_h // 2
+ ],
+ value=-float('inf'))
+ pad_mask = pad_mask.view(1, math.ceil(H / Wh), Wh,
+ math.ceil(W / Ww), Ww, 1)
+ pad_mask = pad_mask.permute(1, 3, 0, 2, 4, 5)
+ pad_mask = pad_mask.reshape(-1, Wh * Ww)
+ pad_mask = pad_mask[:, None, :].expand([-1, Wh * Ww, -1])
+ out = self.attn(x, pad_mask, **kwargs)
+ else:
+ out = self.attn(x, **kwargs)
+
+ # reverse permutation
+ out = out.reshape(B, math.ceil(H / Wh), math.ceil(W / Ww), Wh, Ww, C)
+ out = out.permute(0, 1, 3, 2, 4, 5)
+ out = out.reshape(B, H + pad_h, W + pad_w, C)
+
+ # de-pad
+ out = out[:, pad_h // 2:H + pad_h // 2, pad_w // 2:W + pad_w // 2]
+ return out.reshape(B, N, C)
+
+
+class CrossFFN(BaseModule):
+ r"""FFN with Depthwise Conv of HRFormer.
+
+ Args:
+ in_features (int): The feature dimension.
+ hidden_features (int, optional): The hidden dimension of FFNs.
+ Defaults: The same as in_features.
+ act_cfg (dict, optional): Config of activation layer.
+ Default: dict(type='GELU').
+ dw_act_cfg (dict, optional): Config of activation layer appended
+ right after DW Conv. Default: dict(type='GELU').
+ norm_cfg (dict, optional): Config of norm layer.
+ Default: dict(type='SyncBN').
+ init_cfg (dict | list | None, optional): The init config.
+ Default: None.
+ """
+
+ def __init__(self,
+ in_features,
+ hidden_features=None,
+ out_features=None,
+ act_cfg=dict(type='GELU'),
+ dw_act_cfg=dict(type='GELU'),
+ norm_cfg=dict(type='SyncBN'),
+ init_cfg=None):
+ super().__init__(init_cfg=init_cfg)
+ out_features = out_features or in_features
+ hidden_features = hidden_features or in_features
+ self.fc1 = nn.Conv2d(in_features, hidden_features, kernel_size=1)
+ self.act1 = build_activation_layer(act_cfg)
+ self.norm1 = build_norm_layer(norm_cfg, hidden_features)[1]
+ self.dw3x3 = nn.Conv2d(
+ hidden_features,
+ hidden_features,
+ kernel_size=3,
+ stride=1,
+ groups=hidden_features,
+ padding=1)
+ self.act2 = build_activation_layer(dw_act_cfg)
+ self.norm2 = build_norm_layer(norm_cfg, hidden_features)[1]
+ self.fc2 = nn.Conv2d(hidden_features, out_features, kernel_size=1)
+ self.act3 = build_activation_layer(act_cfg)
+ self.norm3 = build_norm_layer(norm_cfg, out_features)[1]
+
+ # put the modules togather
+ self.layers = [
+ self.fc1, self.norm1, self.act1, self.dw3x3, self.norm2, self.act2,
+ self.fc2, self.norm3, self.act3
+ ]
+
+ def forward(self, x, H, W):
+ """Forward function."""
+ x = nlc_to_nchw(x, (H, W))
+ for layer in self.layers:
+ x = layer(x)
+ x = nchw_to_nlc(x)
+ return x
+
+
+class HRFormerBlock(BaseModule):
+ """High-Resolution Block for HRFormer.
+
+ Args:
+ in_features (int): The input dimension.
+ out_features (int): The output dimension.
+ num_heads (int): The number of head within each LSA.
+ window_size (int, optional): The window size for the LSA.
+ Default: 7
+ mlp_ratio (int, optional): The expansion ration of FFN.
+ Default: 4
+ act_cfg (dict, optional): Config of activation layer.
+ Default: dict(type='GELU').
+ norm_cfg (dict, optional): Config of norm layer.
+ Default: dict(type='SyncBN').
+ transformer_norm_cfg (dict, optional): Config of transformer norm
+ layer. Default: dict(type='LN', eps=1e-6).
+ init_cfg (dict | list | None, optional): The init config.
+ Default: None.
+ """
+
+ expansion = 1
+
+ def __init__(self,
+ in_features,
+ out_features,
+ num_heads,
+ window_size=7,
+ mlp_ratio=4.0,
+ drop_path=0.0,
+ act_cfg=dict(type='GELU'),
+ norm_cfg=dict(type='SyncBN'),
+ transformer_norm_cfg=dict(type='LN', eps=1e-6),
+ init_cfg=None,
+ **kwargs):
+ super(HRFormerBlock, self).__init__(init_cfg=init_cfg)
+ self.num_heads = num_heads
+ self.window_size = window_size
+ self.mlp_ratio = mlp_ratio
+
+ self.norm1 = build_norm_layer(transformer_norm_cfg, in_features)[1]
+ self.attn = LocalWindowSelfAttention(
+ in_features,
+ num_heads=num_heads,
+ window_size=window_size,
+ init_cfg=None,
+ **kwargs)
+
+ self.norm2 = build_norm_layer(transformer_norm_cfg, out_features)[1]
+ self.ffn = CrossFFN(
+ in_features=in_features,
+ hidden_features=int(in_features * mlp_ratio),
+ out_features=out_features,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg,
+ dw_act_cfg=act_cfg,
+ init_cfg=None)
+
+ self.drop_path = build_drop_path(
+ drop_path) if drop_path > 0.0 else nn.Identity()
+
+ def forward(self, x):
+ """Forward function."""
+ B, C, H, W = x.size()
+ # Attention
+ x = x.view(B, C, -1).permute(0, 2, 1)
+ x = x + self.drop_path(self.attn(self.norm1(x), H, W))
+ # FFN
+ x = x + self.drop_path(self.ffn(self.norm2(x), H, W))
+ x = x.permute(0, 2, 1).view(B, C, H, W)
+ return x
+
+ def extra_repr(self):
+ """(Optional) Set the extra information about this module."""
+ return 'num_heads={}, window_size={}, mlp_ratio={}'.format(
+ self.num_heads, self.window_size, self.mlp_ratio)
+
+
+class HRFomerModule(HRModule):
+ """High-Resolution Module for HRFormer.
+
+ Args:
+ num_branches (int): The number of branches in the HRFormerModule.
+ block (nn.Module): The building block of HRFormer.
+ The block should be the HRFormerBlock.
+ num_blocks (tuple): The number of blocks in each branch.
+ The length must be equal to num_branches.
+ num_inchannels (tuple): The number of input channels in each branch.
+ The length must be equal to num_branches.
+ num_channels (tuple): The number of channels in each branch.
+ The length must be equal to num_branches.
+ num_heads (tuple): The number of heads within the LSAs.
+ num_window_sizes (tuple): The window size for the LSAs.
+ num_mlp_ratios (tuple): The expansion ratio for the FFNs.
+ drop_path (int, optional): The drop path rate of HRFomer.
+ Default: 0.0
+ multiscale_output (bool, optional): Whether to output multi-level
+ features produced by multiple branches. If False, only the first
+ level feature will be output. Default: True.
+ conv_cfg (dict, optional): Config of the conv layers.
+ Default: None.
+ norm_cfg (dict, optional): Config of the norm layers appended
+ right after conv. Default: dict(type='SyncBN', requires_grad=True)
+ transformer_norm_cfg (dict, optional): Config of the norm layers.
+ Default: dict(type='LN', eps=1e-6)
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False
+ upsample_cfg(dict, optional): The config of upsample layers in fuse
+ layers. Default: dict(mode='bilinear', align_corners=False)
+ """
+
+ def __init__(self,
+ num_branches,
+ block,
+ num_blocks,
+ num_inchannels,
+ num_channels,
+ num_heads,
+ num_window_sizes,
+ num_mlp_ratios,
+ multiscale_output=True,
+ drop_paths=0.0,
+ with_rpe=True,
+ with_pad_mask=False,
+ conv_cfg=None,
+ norm_cfg=dict(type='SyncBN', requires_grad=True),
+ transformer_norm_cfg=dict(type='LN', eps=1e-6),
+ with_cp=False,
+ upsample_cfg=dict(mode='bilinear', align_corners=False)):
+
+ self.transformer_norm_cfg = transformer_norm_cfg
+ self.drop_paths = drop_paths
+ self.num_heads = num_heads
+ self.num_window_sizes = num_window_sizes
+ self.num_mlp_ratios = num_mlp_ratios
+ self.with_rpe = with_rpe
+ self.with_pad_mask = with_pad_mask
+
+ super().__init__(num_branches, block, num_blocks, num_inchannels,
+ num_channels, multiscale_output, with_cp, conv_cfg,
+ norm_cfg, upsample_cfg)
+
+ def _make_one_branch(self,
+ branch_index,
+ block,
+ num_blocks,
+ num_channels,
+ stride=1):
+ """Build one branch."""
+ # HRFormerBlock does not support down sample layer yet.
+ assert stride == 1 and self.in_channels[branch_index] == num_channels[
+ branch_index]
+ layers = []
+ layers.append(
+ block(
+ self.in_channels[branch_index],
+ num_channels[branch_index],
+ num_heads=self.num_heads[branch_index],
+ window_size=self.num_window_sizes[branch_index],
+ mlp_ratio=self.num_mlp_ratios[branch_index],
+ drop_path=self.drop_paths[0],
+ norm_cfg=self.norm_cfg,
+ transformer_norm_cfg=self.transformer_norm_cfg,
+ init_cfg=None,
+ with_rpe=self.with_rpe,
+ with_pad_mask=self.with_pad_mask))
+
+ self.in_channels[
+ branch_index] = self.in_channels[branch_index] * block.expansion
+ for i in range(1, num_blocks[branch_index]):
+ layers.append(
+ block(
+ self.in_channels[branch_index],
+ num_channels[branch_index],
+ num_heads=self.num_heads[branch_index],
+ window_size=self.num_window_sizes[branch_index],
+ mlp_ratio=self.num_mlp_ratios[branch_index],
+ drop_path=self.drop_paths[i],
+ norm_cfg=self.norm_cfg,
+ transformer_norm_cfg=self.transformer_norm_cfg,
+ init_cfg=None,
+ with_rpe=self.with_rpe,
+ with_pad_mask=self.with_pad_mask))
+ return nn.Sequential(*layers)
+
+ def _make_fuse_layers(self):
+ """Build fuse layers."""
+ if self.num_branches == 1:
+ return None
+ num_branches = self.num_branches
+ num_inchannels = self.in_channels
+ fuse_layers = []
+ for i in range(num_branches if self.multiscale_output else 1):
+ fuse_layer = []
+ for j in range(num_branches):
+ if j > i:
+ fuse_layer.append(
+ nn.Sequential(
+ build_conv_layer(
+ self.conv_cfg,
+ num_inchannels[j],
+ num_inchannels[i],
+ kernel_size=1,
+ stride=1,
+ bias=False),
+ build_norm_layer(self.norm_cfg,
+ num_inchannels[i])[1],
+ nn.Upsample(
+ scale_factor=2**(j - i),
+ mode=self.upsample_cfg['mode'],
+ align_corners=self.
+ upsample_cfg['align_corners'])))
+ elif j == i:
+ fuse_layer.append(None)
+ else:
+ conv3x3s = []
+ for k in range(i - j):
+ if k == i - j - 1:
+ num_outchannels_conv3x3 = num_inchannels[i]
+ with_out_act = False
+ else:
+ num_outchannels_conv3x3 = num_inchannels[j]
+ with_out_act = True
+ sub_modules = [
+ build_conv_layer(
+ self.conv_cfg,
+ num_inchannels[j],
+ num_inchannels[j],
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ groups=num_inchannels[j],
+ bias=False,
+ ),
+ build_norm_layer(self.norm_cfg,
+ num_inchannels[j])[1],
+ build_conv_layer(
+ self.conv_cfg,
+ num_inchannels[j],
+ num_outchannels_conv3x3,
+ kernel_size=1,
+ stride=1,
+ bias=False,
+ ),
+ build_norm_layer(self.norm_cfg,
+ num_outchannels_conv3x3)[1]
+ ]
+ if with_out_act:
+ sub_modules.append(nn.ReLU(False))
+ conv3x3s.append(nn.Sequential(*sub_modules))
+ fuse_layer.append(nn.Sequential(*conv3x3s))
+ fuse_layers.append(nn.ModuleList(fuse_layer))
+
+ return nn.ModuleList(fuse_layers)
+
+ def get_num_inchannels(self):
+ """Return the number of input channels."""
+ return self.in_channels
+
+
+@BACKBONES.register_module()
+class HRFormer(HRNet):
+ """HRFormer backbone.
+
+ This backbone is the implementation of `HRFormer: High-Resolution
+ Transformer for Dense Prediction `_.
+
+ Args:
+ extra (dict): Detailed configuration for each stage of HRNet.
+ There must be 4 stages, the configuration for each stage must have
+ 5 keys:
+
+ - num_modules (int): The number of HRModule in this stage.
+ - num_branches (int): The number of branches in the HRModule.
+ - block (str): The type of block.
+ - num_blocks (tuple): The number of blocks in each branch.
+ The length must be equal to num_branches.
+ - num_channels (tuple): The number of channels in each branch.
+ The length must be equal to num_branches.
+ in_channels (int): Number of input image channels. Normally 3.
+ conv_cfg (dict): Dictionary to construct and config conv layer.
+ Default: None.
+ norm_cfg (dict): Config of norm layer.
+ Use `SyncBN` by default.
+ transformer_norm_cfg (dict): Config of transformer norm layer.
+ Use `LN` by default.
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ zero_init_residual (bool): Whether to use zero init for last norm layer
+ in resblocks to let them behave as identity. Default: False.
+ frozen_stages (int): Stages to be frozen (stop grad and set eval mode).
+ -1 means not freezing any parameters. Default: -1.
+ Example:
+ >>> from mmpose.models import HRFormer
+ >>> import torch
+ >>> extra = dict(
+ >>> stage1=dict(
+ >>> num_modules=1,
+ >>> num_branches=1,
+ >>> block='BOTTLENECK',
+ >>> num_blocks=(2, ),
+ >>> num_channels=(64, )),
+ >>> stage2=dict(
+ >>> num_modules=1,
+ >>> num_branches=2,
+ >>> block='HRFORMER',
+ >>> window_sizes=(7, 7),
+ >>> num_heads=(1, 2),
+ >>> mlp_ratios=(4, 4),
+ >>> num_blocks=(2, 2),
+ >>> num_channels=(32, 64)),
+ >>> stage3=dict(
+ >>> num_modules=4,
+ >>> num_branches=3,
+ >>> block='HRFORMER',
+ >>> window_sizes=(7, 7, 7),
+ >>> num_heads=(1, 2, 4),
+ >>> mlp_ratios=(4, 4, 4),
+ >>> num_blocks=(2, 2, 2),
+ >>> num_channels=(32, 64, 128)),
+ >>> stage4=dict(
+ >>> num_modules=2,
+ >>> num_branches=4,
+ >>> block='HRFORMER',
+ >>> window_sizes=(7, 7, 7, 7),
+ >>> num_heads=(1, 2, 4, 8),
+ >>> mlp_ratios=(4, 4, 4, 4),
+ >>> num_blocks=(2, 2, 2, 2),
+ >>> num_channels=(32, 64, 128, 256)))
+ >>> self = HRFormer(extra, in_channels=1)
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 1, 32, 32)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_out in level_outputs:
+ ... print(tuple(level_out.shape))
+ (1, 32, 8, 8)
+ (1, 64, 4, 4)
+ (1, 128, 2, 2)
+ (1, 256, 1, 1)
+ """
+
+ blocks_dict = {'BOTTLENECK': Bottleneck, 'HRFORMERBLOCK': HRFormerBlock}
+
+ def __init__(self,
+ extra,
+ in_channels=3,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN', requires_grad=True),
+ transformer_norm_cfg=dict(type='LN', eps=1e-6),
+ norm_eval=False,
+ with_cp=False,
+ zero_init_residual=False,
+ frozen_stages=-1):
+
+ # stochastic depth
+ depths = [
+ extra[stage]['num_blocks'][0] * extra[stage]['num_modules']
+ for stage in ['stage2', 'stage3', 'stage4']
+ ]
+ depth_s2, depth_s3, _ = depths
+ drop_path_rate = extra['drop_path_rate']
+ dpr = [
+ x.item() for x in torch.linspace(0, drop_path_rate, sum(depths))
+ ]
+ extra['stage2']['drop_path_rates'] = dpr[0:depth_s2]
+ extra['stage3']['drop_path_rates'] = dpr[depth_s2:depth_s2 + depth_s3]
+ extra['stage4']['drop_path_rates'] = dpr[depth_s2 + depth_s3:]
+
+ # HRFormer use bilinear upsample as default
+ upsample_cfg = extra.get('upsample', {
+ 'mode': 'bilinear',
+ 'align_corners': False
+ })
+ extra['upsample'] = upsample_cfg
+ self.transformer_norm_cfg = transformer_norm_cfg
+ self.with_rpe = extra.get('with_rpe', True)
+ self.with_pad_mask = extra.get('with_pad_mask', False)
+
+ super().__init__(extra, in_channels, conv_cfg, norm_cfg, norm_eval,
+ with_cp, zero_init_residual, frozen_stages)
+
+ def _make_stage(self,
+ layer_config,
+ num_inchannels,
+ multiscale_output=True):
+ """Make each stage."""
+ num_modules = layer_config['num_modules']
+ num_branches = layer_config['num_branches']
+ num_blocks = layer_config['num_blocks']
+ num_channels = layer_config['num_channels']
+ block = self.blocks_dict[layer_config['block']]
+ num_heads = layer_config['num_heads']
+ num_window_sizes = layer_config['window_sizes']
+ num_mlp_ratios = layer_config['mlp_ratios']
+ drop_path_rates = layer_config['drop_path_rates']
+
+ modules = []
+ for i in range(num_modules):
+ # multiscale_output is only used at the last module
+ if not multiscale_output and i == num_modules - 1:
+ reset_multiscale_output = False
+ else:
+ reset_multiscale_output = True
+
+ modules.append(
+ HRFomerModule(
+ num_branches,
+ block,
+ num_blocks,
+ num_inchannels,
+ num_channels,
+ num_heads,
+ num_window_sizes,
+ num_mlp_ratios,
+ reset_multiscale_output,
+ drop_paths=drop_path_rates[num_blocks[0] *
+ i:num_blocks[0] * (i + 1)],
+ with_rpe=self.with_rpe,
+ with_pad_mask=self.with_pad_mask,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ transformer_norm_cfg=self.transformer_norm_cfg,
+ with_cp=self.with_cp,
+ upsample_cfg=self.upsample_cfg))
+ num_inchannels = modules[-1].get_num_inchannels()
+
+ return nn.Sequential(*modules), num_inchannels
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/litehrnet.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/litehrnet.py
new file mode 100644
index 0000000000000000000000000000000000000000..954368841eb631e3dc6c77e9810f6980f3739bf3
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/litehrnet.py
@@ -0,0 +1,984 @@
+# ------------------------------------------------------------------------------
+# Adapted from https://github.com/HRNet/Lite-HRNet
+# Original licence: Apache License 2.0.
+# ------------------------------------------------------------------------------
+
+import mmcv
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import torch.utils.checkpoint as cp
+from mmcv.cnn import (ConvModule, DepthwiseSeparableConvModule,
+ build_conv_layer, build_norm_layer, constant_init,
+ normal_init)
+from torch.nn.modules.batchnorm import _BatchNorm
+
+from mmpose.utils import get_root_logger
+from ..builder import BACKBONES
+from .utils import channel_shuffle, load_checkpoint
+
+
+class SpatialWeighting(nn.Module):
+ """Spatial weighting module.
+
+ Args:
+ channels (int): The channels of the module.
+ ratio (int): channel reduction ratio.
+ conv_cfg (dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: None.
+ act_cfg (dict): Config dict for activation layer.
+ Default: (dict(type='ReLU'), dict(type='Sigmoid')).
+ The last ConvModule uses Sigmoid by default.
+ """
+
+ def __init__(self,
+ channels,
+ ratio=16,
+ conv_cfg=None,
+ norm_cfg=None,
+ act_cfg=(dict(type='ReLU'), dict(type='Sigmoid'))):
+ super().__init__()
+ if isinstance(act_cfg, dict):
+ act_cfg = (act_cfg, act_cfg)
+ assert len(act_cfg) == 2
+ assert mmcv.is_tuple_of(act_cfg, dict)
+ self.global_avgpool = nn.AdaptiveAvgPool2d(1)
+ self.conv1 = ConvModule(
+ in_channels=channels,
+ out_channels=int(channels / ratio),
+ kernel_size=1,
+ stride=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg[0])
+ self.conv2 = ConvModule(
+ in_channels=int(channels / ratio),
+ out_channels=channels,
+ kernel_size=1,
+ stride=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg[1])
+
+ def forward(self, x):
+ out = self.global_avgpool(x)
+ out = self.conv1(out)
+ out = self.conv2(out)
+ return x * out
+
+
+class CrossResolutionWeighting(nn.Module):
+ """Cross-resolution channel weighting module.
+
+ Args:
+ channels (int): The channels of the module.
+ ratio (int): channel reduction ratio.
+ conv_cfg (dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: None.
+ act_cfg (dict): Config dict for activation layer.
+ Default: (dict(type='ReLU'), dict(type='Sigmoid')).
+ The last ConvModule uses Sigmoid by default.
+ """
+
+ def __init__(self,
+ channels,
+ ratio=16,
+ conv_cfg=None,
+ norm_cfg=None,
+ act_cfg=(dict(type='ReLU'), dict(type='Sigmoid'))):
+ super().__init__()
+ if isinstance(act_cfg, dict):
+ act_cfg = (act_cfg, act_cfg)
+ assert len(act_cfg) == 2
+ assert mmcv.is_tuple_of(act_cfg, dict)
+ self.channels = channels
+ total_channel = sum(channels)
+ self.conv1 = ConvModule(
+ in_channels=total_channel,
+ out_channels=int(total_channel / ratio),
+ kernel_size=1,
+ stride=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg[0])
+ self.conv2 = ConvModule(
+ in_channels=int(total_channel / ratio),
+ out_channels=total_channel,
+ kernel_size=1,
+ stride=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg[1])
+
+ def forward(self, x):
+ mini_size = x[-1].size()[-2:]
+ out = [F.adaptive_avg_pool2d(s, mini_size) for s in x[:-1]] + [x[-1]]
+ out = torch.cat(out, dim=1)
+ out = self.conv1(out)
+ out = self.conv2(out)
+ out = torch.split(out, self.channels, dim=1)
+ out = [
+ s * F.interpolate(a, size=s.size()[-2:], mode='nearest')
+ for s, a in zip(x, out)
+ ]
+ return out
+
+
+class ConditionalChannelWeighting(nn.Module):
+ """Conditional channel weighting block.
+
+ Args:
+ in_channels (int): The input channels of the block.
+ stride (int): Stride of the 3x3 convolution layer.
+ reduce_ratio (int): channel reduction ratio.
+ conv_cfg (dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ """
+
+ def __init__(self,
+ in_channels,
+ stride,
+ reduce_ratio,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ with_cp=False):
+ super().__init__()
+ self.with_cp = with_cp
+ self.stride = stride
+ assert stride in [1, 2]
+
+ branch_channels = [channel // 2 for channel in in_channels]
+
+ self.cross_resolution_weighting = CrossResolutionWeighting(
+ branch_channels,
+ ratio=reduce_ratio,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg)
+
+ self.depthwise_convs = nn.ModuleList([
+ ConvModule(
+ channel,
+ channel,
+ kernel_size=3,
+ stride=self.stride,
+ padding=1,
+ groups=channel,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=None) for channel in branch_channels
+ ])
+
+ self.spatial_weighting = nn.ModuleList([
+ SpatialWeighting(channels=channel, ratio=4)
+ for channel in branch_channels
+ ])
+
+ def forward(self, x):
+
+ def _inner_forward(x):
+ x = [s.chunk(2, dim=1) for s in x]
+ x1 = [s[0] for s in x]
+ x2 = [s[1] for s in x]
+
+ x2 = self.cross_resolution_weighting(x2)
+ x2 = [dw(s) for s, dw in zip(x2, self.depthwise_convs)]
+ x2 = [sw(s) for s, sw in zip(x2, self.spatial_weighting)]
+
+ out = [torch.cat([s1, s2], dim=1) for s1, s2 in zip(x1, x2)]
+ out = [channel_shuffle(s, 2) for s in out]
+
+ return out
+
+ if self.with_cp and x.requires_grad:
+ out = cp.checkpoint(_inner_forward, x)
+ else:
+ out = _inner_forward(x)
+
+ return out
+
+
+class Stem(nn.Module):
+ """Stem network block.
+
+ Args:
+ in_channels (int): The input channels of the block.
+ stem_channels (int): Output channels of the stem layer.
+ out_channels (int): The output channels of the block.
+ expand_ratio (int): adjusts number of channels of the hidden layer
+ in InvertedResidual by this amount.
+ conv_cfg (dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ """
+
+ def __init__(self,
+ in_channels,
+ stem_channels,
+ out_channels,
+ expand_ratio,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ with_cp=False):
+ super().__init__()
+ self.in_channels = in_channels
+ self.out_channels = out_channels
+ self.conv_cfg = conv_cfg
+ self.norm_cfg = norm_cfg
+ self.with_cp = with_cp
+
+ self.conv1 = ConvModule(
+ in_channels=in_channels,
+ out_channels=stem_channels,
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ act_cfg=dict(type='ReLU'))
+
+ mid_channels = int(round(stem_channels * expand_ratio))
+ branch_channels = stem_channels // 2
+ if stem_channels == self.out_channels:
+ inc_channels = self.out_channels - branch_channels
+ else:
+ inc_channels = self.out_channels - stem_channels
+
+ self.branch1 = nn.Sequential(
+ ConvModule(
+ branch_channels,
+ branch_channels,
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ groups=branch_channels,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=None),
+ ConvModule(
+ branch_channels,
+ inc_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=dict(type='ReLU')),
+ )
+
+ self.expand_conv = ConvModule(
+ branch_channels,
+ mid_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=dict(type='ReLU'))
+ self.depthwise_conv = ConvModule(
+ mid_channels,
+ mid_channels,
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ groups=mid_channels,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=None)
+ self.linear_conv = ConvModule(
+ mid_channels,
+ branch_channels
+ if stem_channels == self.out_channels else stem_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=dict(type='ReLU'))
+
+ def forward(self, x):
+
+ def _inner_forward(x):
+ x = self.conv1(x)
+ x1, x2 = x.chunk(2, dim=1)
+
+ x2 = self.expand_conv(x2)
+ x2 = self.depthwise_conv(x2)
+ x2 = self.linear_conv(x2)
+
+ out = torch.cat((self.branch1(x1), x2), dim=1)
+
+ out = channel_shuffle(out, 2)
+
+ return out
+
+ if self.with_cp and x.requires_grad:
+ out = cp.checkpoint(_inner_forward, x)
+ else:
+ out = _inner_forward(x)
+
+ return out
+
+
+class IterativeHead(nn.Module):
+ """Extra iterative head for feature learning.
+
+ Args:
+ in_channels (int): The input channels of the block.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ """
+
+ def __init__(self, in_channels, norm_cfg=dict(type='BN')):
+ super().__init__()
+ projects = []
+ num_branchs = len(in_channels)
+ self.in_channels = in_channels[::-1]
+
+ for i in range(num_branchs):
+ if i != num_branchs - 1:
+ projects.append(
+ DepthwiseSeparableConvModule(
+ in_channels=self.in_channels[i],
+ out_channels=self.in_channels[i + 1],
+ kernel_size=3,
+ stride=1,
+ padding=1,
+ norm_cfg=norm_cfg,
+ act_cfg=dict(type='ReLU'),
+ dw_act_cfg=None,
+ pw_act_cfg=dict(type='ReLU')))
+ else:
+ projects.append(
+ DepthwiseSeparableConvModule(
+ in_channels=self.in_channels[i],
+ out_channels=self.in_channels[i],
+ kernel_size=3,
+ stride=1,
+ padding=1,
+ norm_cfg=norm_cfg,
+ act_cfg=dict(type='ReLU'),
+ dw_act_cfg=None,
+ pw_act_cfg=dict(type='ReLU')))
+ self.projects = nn.ModuleList(projects)
+
+ def forward(self, x):
+ x = x[::-1]
+
+ y = []
+ last_x = None
+ for i, s in enumerate(x):
+ if last_x is not None:
+ last_x = F.interpolate(
+ last_x,
+ size=s.size()[-2:],
+ mode='bilinear',
+ align_corners=True)
+ s = s + last_x
+ s = self.projects[i](s)
+ y.append(s)
+ last_x = s
+
+ return y[::-1]
+
+
+class ShuffleUnit(nn.Module):
+ """InvertedResidual block for ShuffleNetV2 backbone.
+
+ Args:
+ in_channels (int): The input channels of the block.
+ out_channels (int): The output channels of the block.
+ stride (int): Stride of the 3x3 convolution layer. Default: 1
+ conv_cfg (dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ act_cfg (dict): Config dict for activation layer.
+ Default: dict(type='ReLU').
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ stride=1,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ act_cfg=dict(type='ReLU'),
+ with_cp=False):
+ super().__init__()
+ self.stride = stride
+ self.with_cp = with_cp
+
+ branch_features = out_channels // 2
+ if self.stride == 1:
+ assert in_channels == branch_features * 2, (
+ f'in_channels ({in_channels}) should equal to '
+ f'branch_features * 2 ({branch_features * 2}) '
+ 'when stride is 1')
+
+ if in_channels != branch_features * 2:
+ assert self.stride != 1, (
+ f'stride ({self.stride}) should not equal 1 when '
+ f'in_channels != branch_features * 2')
+
+ if self.stride > 1:
+ self.branch1 = nn.Sequential(
+ ConvModule(
+ in_channels,
+ in_channels,
+ kernel_size=3,
+ stride=self.stride,
+ padding=1,
+ groups=in_channels,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=None),
+ ConvModule(
+ in_channels,
+ branch_features,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg),
+ )
+
+ self.branch2 = nn.Sequential(
+ ConvModule(
+ in_channels if (self.stride > 1) else branch_features,
+ branch_features,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg),
+ ConvModule(
+ branch_features,
+ branch_features,
+ kernel_size=3,
+ stride=self.stride,
+ padding=1,
+ groups=branch_features,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=None),
+ ConvModule(
+ branch_features,
+ branch_features,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg))
+
+ def forward(self, x):
+
+ def _inner_forward(x):
+ if self.stride > 1:
+ out = torch.cat((self.branch1(x), self.branch2(x)), dim=1)
+ else:
+ x1, x2 = x.chunk(2, dim=1)
+ out = torch.cat((x1, self.branch2(x2)), dim=1)
+
+ out = channel_shuffle(out, 2)
+
+ return out
+
+ if self.with_cp and x.requires_grad:
+ out = cp.checkpoint(_inner_forward, x)
+ else:
+ out = _inner_forward(x)
+
+ return out
+
+
+class LiteHRModule(nn.Module):
+ """High-Resolution Module for LiteHRNet.
+
+ It contains conditional channel weighting blocks and
+ shuffle blocks.
+
+
+ Args:
+ num_branches (int): Number of branches in the module.
+ num_blocks (int): Number of blocks in the module.
+ in_channels (list(int)): Number of input image channels.
+ reduce_ratio (int): Channel reduction ratio.
+ module_type (str): 'LITE' or 'NAIVE'
+ multiscale_output (bool): Whether to output multi-scale features.
+ with_fuse (bool): Whether to use fuse layers.
+ conv_cfg (dict): dictionary to construct and config conv layer.
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed.
+ """
+
+ def __init__(
+ self,
+ num_branches,
+ num_blocks,
+ in_channels,
+ reduce_ratio,
+ module_type,
+ multiscale_output=False,
+ with_fuse=True,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ with_cp=False,
+ ):
+ super().__init__()
+ self._check_branches(num_branches, in_channels)
+
+ self.in_channels = in_channels
+ self.num_branches = num_branches
+
+ self.module_type = module_type
+ self.multiscale_output = multiscale_output
+ self.with_fuse = with_fuse
+ self.norm_cfg = norm_cfg
+ self.conv_cfg = conv_cfg
+ self.with_cp = with_cp
+
+ if self.module_type.upper() == 'LITE':
+ self.layers = self._make_weighting_blocks(num_blocks, reduce_ratio)
+ elif self.module_type.upper() == 'NAIVE':
+ self.layers = self._make_naive_branches(num_branches, num_blocks)
+ else:
+ raise ValueError("module_type should be either 'LITE' or 'NAIVE'.")
+ if self.with_fuse:
+ self.fuse_layers = self._make_fuse_layers()
+ self.relu = nn.ReLU()
+
+ def _check_branches(self, num_branches, in_channels):
+ """Check input to avoid ValueError."""
+ if num_branches != len(in_channels):
+ error_msg = f'NUM_BRANCHES({num_branches}) ' \
+ f'!= NUM_INCHANNELS({len(in_channels)})'
+ raise ValueError(error_msg)
+
+ def _make_weighting_blocks(self, num_blocks, reduce_ratio, stride=1):
+ """Make channel weighting blocks."""
+ layers = []
+ for i in range(num_blocks):
+ layers.append(
+ ConditionalChannelWeighting(
+ self.in_channels,
+ stride=stride,
+ reduce_ratio=reduce_ratio,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ with_cp=self.with_cp))
+
+ return nn.Sequential(*layers)
+
+ def _make_one_branch(self, branch_index, num_blocks, stride=1):
+ """Make one branch."""
+ layers = []
+ layers.append(
+ ShuffleUnit(
+ self.in_channels[branch_index],
+ self.in_channels[branch_index],
+ stride=stride,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ act_cfg=dict(type='ReLU'),
+ with_cp=self.with_cp))
+ for i in range(1, num_blocks):
+ layers.append(
+ ShuffleUnit(
+ self.in_channels[branch_index],
+ self.in_channels[branch_index],
+ stride=1,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ act_cfg=dict(type='ReLU'),
+ with_cp=self.with_cp))
+
+ return nn.Sequential(*layers)
+
+ def _make_naive_branches(self, num_branches, num_blocks):
+ """Make branches."""
+ branches = []
+
+ for i in range(num_branches):
+ branches.append(self._make_one_branch(i, num_blocks))
+
+ return nn.ModuleList(branches)
+
+ def _make_fuse_layers(self):
+ """Make fuse layer."""
+ if self.num_branches == 1:
+ return None
+
+ num_branches = self.num_branches
+ in_channels = self.in_channels
+ fuse_layers = []
+ num_out_branches = num_branches if self.multiscale_output else 1
+ for i in range(num_out_branches):
+ fuse_layer = []
+ for j in range(num_branches):
+ if j > i:
+ fuse_layer.append(
+ nn.Sequential(
+ build_conv_layer(
+ self.conv_cfg,
+ in_channels[j],
+ in_channels[i],
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ bias=False),
+ build_norm_layer(self.norm_cfg, in_channels[i])[1],
+ nn.Upsample(
+ scale_factor=2**(j - i), mode='nearest')))
+ elif j == i:
+ fuse_layer.append(None)
+ else:
+ conv_downsamples = []
+ for k in range(i - j):
+ if k == i - j - 1:
+ conv_downsamples.append(
+ nn.Sequential(
+ build_conv_layer(
+ self.conv_cfg,
+ in_channels[j],
+ in_channels[j],
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ groups=in_channels[j],
+ bias=False),
+ build_norm_layer(self.norm_cfg,
+ in_channels[j])[1],
+ build_conv_layer(
+ self.conv_cfg,
+ in_channels[j],
+ in_channels[i],
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ bias=False),
+ build_norm_layer(self.norm_cfg,
+ in_channels[i])[1]))
+ else:
+ conv_downsamples.append(
+ nn.Sequential(
+ build_conv_layer(
+ self.conv_cfg,
+ in_channels[j],
+ in_channels[j],
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ groups=in_channels[j],
+ bias=False),
+ build_norm_layer(self.norm_cfg,
+ in_channels[j])[1],
+ build_conv_layer(
+ self.conv_cfg,
+ in_channels[j],
+ in_channels[j],
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ bias=False),
+ build_norm_layer(self.norm_cfg,
+ in_channels[j])[1],
+ nn.ReLU(inplace=True)))
+ fuse_layer.append(nn.Sequential(*conv_downsamples))
+ fuse_layers.append(nn.ModuleList(fuse_layer))
+
+ return nn.ModuleList(fuse_layers)
+
+ def forward(self, x):
+ """Forward function."""
+ if self.num_branches == 1:
+ return [self.layers[0](x[0])]
+
+ if self.module_type.upper() == 'LITE':
+ out = self.layers(x)
+ elif self.module_type.upper() == 'NAIVE':
+ for i in range(self.num_branches):
+ x[i] = self.layers[i](x[i])
+ out = x
+
+ if self.with_fuse:
+ out_fuse = []
+ for i in range(len(self.fuse_layers)):
+ # `y = 0` will lead to decreased accuracy (0.5~1 mAP)
+ y = out[0] if i == 0 else self.fuse_layers[i][0](out[0])
+ for j in range(self.num_branches):
+ if i == j:
+ y += out[j]
+ else:
+ y += self.fuse_layers[i][j](out[j])
+ out_fuse.append(self.relu(y))
+ out = out_fuse
+ if not self.multiscale_output:
+ out = [out[0]]
+ return out
+
+
+@BACKBONES.register_module()
+class LiteHRNet(nn.Module):
+ """Lite-HRNet backbone.
+
+ `Lite-HRNet: A Lightweight High-Resolution Network
+ `_.
+
+ Code adapted from 'https://github.com/HRNet/Lite-HRNet'.
+
+ Args:
+ extra (dict): detailed configuration for each stage of HRNet.
+ in_channels (int): Number of input image channels. Default: 3.
+ conv_cfg (dict): dictionary to construct and config conv layer.
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed.
+
+ Example:
+ >>> from mmpose.models import LiteHRNet
+ >>> import torch
+ >>> extra=dict(
+ >>> stem=dict(stem_channels=32, out_channels=32, expand_ratio=1),
+ >>> num_stages=3,
+ >>> stages_spec=dict(
+ >>> num_modules=(2, 4, 2),
+ >>> num_branches=(2, 3, 4),
+ >>> num_blocks=(2, 2, 2),
+ >>> module_type=('LITE', 'LITE', 'LITE'),
+ >>> with_fuse=(True, True, True),
+ >>> reduce_ratios=(8, 8, 8),
+ >>> num_channels=(
+ >>> (40, 80),
+ >>> (40, 80, 160),
+ >>> (40, 80, 160, 320),
+ >>> )),
+ >>> with_head=False)
+ >>> self = LiteHRNet(extra, in_channels=1)
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 1, 32, 32)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_out in level_outputs:
+ ... print(tuple(level_out.shape))
+ (1, 40, 8, 8)
+ """
+
+ def __init__(self,
+ extra,
+ in_channels=3,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ norm_eval=False,
+ with_cp=False):
+ super().__init__()
+ self.extra = extra
+ self.conv_cfg = conv_cfg
+ self.norm_cfg = norm_cfg
+ self.norm_eval = norm_eval
+ self.with_cp = with_cp
+
+ self.stem = Stem(
+ in_channels,
+ stem_channels=self.extra['stem']['stem_channels'],
+ out_channels=self.extra['stem']['out_channels'],
+ expand_ratio=self.extra['stem']['expand_ratio'],
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg)
+
+ self.num_stages = self.extra['num_stages']
+ self.stages_spec = self.extra['stages_spec']
+
+ num_channels_last = [
+ self.stem.out_channels,
+ ]
+ for i in range(self.num_stages):
+ num_channels = self.stages_spec['num_channels'][i]
+ num_channels = [num_channels[i] for i in range(len(num_channels))]
+ setattr(
+ self, f'transition{i}',
+ self._make_transition_layer(num_channels_last, num_channels))
+
+ stage, num_channels_last = self._make_stage(
+ self.stages_spec, i, num_channels, multiscale_output=True)
+ setattr(self, f'stage{i}', stage)
+
+ self.with_head = self.extra['with_head']
+ if self.with_head:
+ self.head_layer = IterativeHead(
+ in_channels=num_channels_last,
+ norm_cfg=self.norm_cfg,
+ )
+
+ def _make_transition_layer(self, num_channels_pre_layer,
+ num_channels_cur_layer):
+ """Make transition layer."""
+ num_branches_cur = len(num_channels_cur_layer)
+ num_branches_pre = len(num_channels_pre_layer)
+
+ transition_layers = []
+ for i in range(num_branches_cur):
+ if i < num_branches_pre:
+ if num_channels_cur_layer[i] != num_channels_pre_layer[i]:
+ transition_layers.append(
+ nn.Sequential(
+ build_conv_layer(
+ self.conv_cfg,
+ num_channels_pre_layer[i],
+ num_channels_pre_layer[i],
+ kernel_size=3,
+ stride=1,
+ padding=1,
+ groups=num_channels_pre_layer[i],
+ bias=False),
+ build_norm_layer(self.norm_cfg,
+ num_channels_pre_layer[i])[1],
+ build_conv_layer(
+ self.conv_cfg,
+ num_channels_pre_layer[i],
+ num_channels_cur_layer[i],
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ bias=False),
+ build_norm_layer(self.norm_cfg,
+ num_channels_cur_layer[i])[1],
+ nn.ReLU()))
+ else:
+ transition_layers.append(None)
+ else:
+ conv_downsamples = []
+ for j in range(i + 1 - num_branches_pre):
+ in_channels = num_channels_pre_layer[-1]
+ out_channels = num_channels_cur_layer[i] \
+ if j == i - num_branches_pre else in_channels
+ conv_downsamples.append(
+ nn.Sequential(
+ build_conv_layer(
+ self.conv_cfg,
+ in_channels,
+ in_channels,
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ groups=in_channels,
+ bias=False),
+ build_norm_layer(self.norm_cfg, in_channels)[1],
+ build_conv_layer(
+ self.conv_cfg,
+ in_channels,
+ out_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ bias=False),
+ build_norm_layer(self.norm_cfg, out_channels)[1],
+ nn.ReLU()))
+ transition_layers.append(nn.Sequential(*conv_downsamples))
+
+ return nn.ModuleList(transition_layers)
+
+ def _make_stage(self,
+ stages_spec,
+ stage_index,
+ in_channels,
+ multiscale_output=True):
+ num_modules = stages_spec['num_modules'][stage_index]
+ num_branches = stages_spec['num_branches'][stage_index]
+ num_blocks = stages_spec['num_blocks'][stage_index]
+ reduce_ratio = stages_spec['reduce_ratios'][stage_index]
+ with_fuse = stages_spec['with_fuse'][stage_index]
+ module_type = stages_spec['module_type'][stage_index]
+
+ modules = []
+ for i in range(num_modules):
+ # multi_scale_output is only used last module
+ if not multiscale_output and i == num_modules - 1:
+ reset_multiscale_output = False
+ else:
+ reset_multiscale_output = True
+
+ modules.append(
+ LiteHRModule(
+ num_branches,
+ num_blocks,
+ in_channels,
+ reduce_ratio,
+ module_type,
+ multiscale_output=reset_multiscale_output,
+ with_fuse=with_fuse,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ with_cp=self.with_cp))
+ in_channels = modules[-1].in_channels
+
+ return nn.Sequential(*modules), in_channels
+
+ def init_weights(self, pretrained=None):
+ """Initialize the weights in backbone.
+
+ Args:
+ pretrained (str, optional): Path to pre-trained weights.
+ Defaults to None.
+ """
+ if isinstance(pretrained, str):
+ logger = get_root_logger()
+ load_checkpoint(self, pretrained, strict=False, logger=logger)
+ elif pretrained is None:
+ for m in self.modules():
+ if isinstance(m, nn.Conv2d):
+ normal_init(m, std=0.001)
+ elif isinstance(m, (_BatchNorm, nn.GroupNorm)):
+ constant_init(m, 1)
+ else:
+ raise TypeError('pretrained must be a str or None')
+
+ def forward(self, x):
+ """Forward function."""
+ x = self.stem(x)
+
+ y_list = [x]
+ for i in range(self.num_stages):
+ x_list = []
+ transition = getattr(self, f'transition{i}')
+ for j in range(self.stages_spec['num_branches'][i]):
+ if transition[j]:
+ if j >= len(y_list):
+ x_list.append(transition[j](y_list[-1]))
+ else:
+ x_list.append(transition[j](y_list[j]))
+ else:
+ x_list.append(y_list[j])
+ y_list = getattr(self, f'stage{i}')(x_list)
+
+ x = y_list
+ if self.with_head:
+ x = self.head_layer(x)
+
+ return [x[0]]
+
+ def train(self, mode=True):
+ """Convert the model into training mode."""
+ super().train(mode)
+ if mode and self.norm_eval:
+ for m in self.modules():
+ if isinstance(m, _BatchNorm):
+ m.eval()
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/mobilenet_v2.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/mobilenet_v2.py
new file mode 100644
index 0000000000000000000000000000000000000000..5dc0cd1b7dfdec2aa751861e39fc1c1a45ec488e
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/mobilenet_v2.py
@@ -0,0 +1,275 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+import logging
+
+import torch.nn as nn
+import torch.utils.checkpoint as cp
+from mmcv.cnn import ConvModule, constant_init, kaiming_init
+from torch.nn.modules.batchnorm import _BatchNorm
+
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+from .utils import load_checkpoint, make_divisible
+
+
+class InvertedResidual(nn.Module):
+ """InvertedResidual block for MobileNetV2.
+
+ Args:
+ in_channels (int): The input channels of the InvertedResidual block.
+ out_channels (int): The output channels of the InvertedResidual block.
+ stride (int): Stride of the middle (first) 3x3 convolution.
+ expand_ratio (int): adjusts number of channels of the hidden layer
+ in InvertedResidual by this amount.
+ conv_cfg (dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ act_cfg (dict): Config dict for activation layer.
+ Default: dict(type='ReLU6').
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ stride,
+ expand_ratio,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ act_cfg=dict(type='ReLU6'),
+ with_cp=False):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ act_cfg = copy.deepcopy(act_cfg)
+ super().__init__()
+ self.stride = stride
+ assert stride in [1, 2], f'stride must in [1, 2]. ' \
+ f'But received {stride}.'
+ self.with_cp = with_cp
+ self.use_res_connect = self.stride == 1 and in_channels == out_channels
+ hidden_dim = int(round(in_channels * expand_ratio))
+
+ layers = []
+ if expand_ratio != 1:
+ layers.append(
+ ConvModule(
+ in_channels=in_channels,
+ out_channels=hidden_dim,
+ kernel_size=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg))
+ layers.extend([
+ ConvModule(
+ in_channels=hidden_dim,
+ out_channels=hidden_dim,
+ kernel_size=3,
+ stride=stride,
+ padding=1,
+ groups=hidden_dim,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg),
+ ConvModule(
+ in_channels=hidden_dim,
+ out_channels=out_channels,
+ kernel_size=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=None)
+ ])
+ self.conv = nn.Sequential(*layers)
+
+ def forward(self, x):
+
+ def _inner_forward(x):
+ if self.use_res_connect:
+ return x + self.conv(x)
+ return self.conv(x)
+
+ if self.with_cp and x.requires_grad:
+ out = cp.checkpoint(_inner_forward, x)
+ else:
+ out = _inner_forward(x)
+
+ return out
+
+
+@BACKBONES.register_module()
+class MobileNetV2(BaseBackbone):
+ """MobileNetV2 backbone.
+
+ Args:
+ widen_factor (float): Width multiplier, multiply number of
+ channels in each layer by this amount. Default: 1.0.
+ out_indices (None or Sequence[int]): Output from which stages.
+ Default: (7, ).
+ frozen_stages (int): Stages to be frozen (all param fixed).
+ Default: -1, which means not freezing any parameters.
+ conv_cfg (dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ act_cfg (dict): Config dict for activation layer.
+ Default: dict(type='ReLU6').
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ """
+
+ # Parameters to build layers. 4 parameters are needed to construct a
+ # layer, from left to right: expand_ratio, channel, num_blocks, stride.
+ arch_settings = [[1, 16, 1, 1], [6, 24, 2, 2], [6, 32, 3, 2],
+ [6, 64, 4, 2], [6, 96, 3, 1], [6, 160, 3, 2],
+ [6, 320, 1, 1]]
+
+ def __init__(self,
+ widen_factor=1.,
+ out_indices=(7, ),
+ frozen_stages=-1,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ act_cfg=dict(type='ReLU6'),
+ norm_eval=False,
+ with_cp=False):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ act_cfg = copy.deepcopy(act_cfg)
+ super().__init__()
+ self.widen_factor = widen_factor
+ self.out_indices = out_indices
+ for index in out_indices:
+ if index not in range(0, 8):
+ raise ValueError('the item in out_indices must in '
+ f'range(0, 8). But received {index}')
+
+ if frozen_stages not in range(-1, 8):
+ raise ValueError('frozen_stages must be in range(-1, 8). '
+ f'But received {frozen_stages}')
+ self.out_indices = out_indices
+ self.frozen_stages = frozen_stages
+ self.conv_cfg = conv_cfg
+ self.norm_cfg = norm_cfg
+ self.act_cfg = act_cfg
+ self.norm_eval = norm_eval
+ self.with_cp = with_cp
+
+ self.in_channels = make_divisible(32 * widen_factor, 8)
+
+ self.conv1 = ConvModule(
+ in_channels=3,
+ out_channels=self.in_channels,
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ act_cfg=self.act_cfg)
+
+ self.layers = []
+
+ for i, layer_cfg in enumerate(self.arch_settings):
+ expand_ratio, channel, num_blocks, stride = layer_cfg
+ out_channels = make_divisible(channel * widen_factor, 8)
+ inverted_res_layer = self.make_layer(
+ out_channels=out_channels,
+ num_blocks=num_blocks,
+ stride=stride,
+ expand_ratio=expand_ratio)
+ layer_name = f'layer{i + 1}'
+ self.add_module(layer_name, inverted_res_layer)
+ self.layers.append(layer_name)
+
+ if widen_factor > 1.0:
+ self.out_channel = int(1280 * widen_factor)
+ else:
+ self.out_channel = 1280
+
+ layer = ConvModule(
+ in_channels=self.in_channels,
+ out_channels=self.out_channel,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ act_cfg=self.act_cfg)
+ self.add_module('conv2', layer)
+ self.layers.append('conv2')
+
+ def make_layer(self, out_channels, num_blocks, stride, expand_ratio):
+ """Stack InvertedResidual blocks to build a layer for MobileNetV2.
+
+ Args:
+ out_channels (int): out_channels of block.
+ num_blocks (int): number of blocks.
+ stride (int): stride of the first block. Default: 1
+ expand_ratio (int): Expand the number of channels of the
+ hidden layer in InvertedResidual by this ratio. Default: 6.
+ """
+ layers = []
+ for i in range(num_blocks):
+ if i >= 1:
+ stride = 1
+ layers.append(
+ InvertedResidual(
+ self.in_channels,
+ out_channels,
+ stride,
+ expand_ratio=expand_ratio,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ act_cfg=self.act_cfg,
+ with_cp=self.with_cp))
+ self.in_channels = out_channels
+
+ return nn.Sequential(*layers)
+
+ def init_weights(self, pretrained=None):
+ if isinstance(pretrained, str):
+ logger = logging.getLogger()
+ load_checkpoint(self, pretrained, strict=False, logger=logger)
+ elif pretrained is None:
+ for m in self.modules():
+ if isinstance(m, nn.Conv2d):
+ kaiming_init(m)
+ elif isinstance(m, (_BatchNorm, nn.GroupNorm)):
+ constant_init(m, 1)
+ else:
+ raise TypeError('pretrained must be a str or None')
+
+ def forward(self, x):
+ x = self.conv1(x)
+
+ outs = []
+ for i, layer_name in enumerate(self.layers):
+ layer = getattr(self, layer_name)
+ x = layer(x)
+ if i in self.out_indices:
+ outs.append(x)
+
+ if len(outs) == 1:
+ return outs[0]
+ return tuple(outs)
+
+ def _freeze_stages(self):
+ if self.frozen_stages >= 0:
+ for param in self.conv1.parameters():
+ param.requires_grad = False
+ for i in range(1, self.frozen_stages + 1):
+ layer = getattr(self, f'layer{i}')
+ layer.eval()
+ for param in layer.parameters():
+ param.requires_grad = False
+
+ def train(self, mode=True):
+ super().train(mode)
+ self._freeze_stages()
+ if mode and self.norm_eval:
+ for m in self.modules():
+ if isinstance(m, _BatchNorm):
+ m.eval()
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/mobilenet_v3.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/mobilenet_v3.py
new file mode 100644
index 0000000000000000000000000000000000000000..d640abec79f06d689f2d4bc1e92999946bc07261
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/mobilenet_v3.py
@@ -0,0 +1,188 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+import logging
+
+import torch.nn as nn
+from mmcv.cnn import ConvModule, constant_init, kaiming_init
+from torch.nn.modules.batchnorm import _BatchNorm
+
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+from .utils import InvertedResidual, load_checkpoint
+
+
+@BACKBONES.register_module()
+class MobileNetV3(BaseBackbone):
+ """MobileNetV3 backbone.
+
+ Args:
+ arch (str): Architecture of mobilnetv3, from {small, big}.
+ Default: small.
+ conv_cfg (dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ out_indices (None or Sequence[int]): Output from which stages.
+ Default: (-1, ), which means output tensors from final stage.
+ frozen_stages (int): Stages to be frozen (all param fixed).
+ Default: -1, which means not freezing any parameters.
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save
+ some memory while slowing down the training speed.
+ Default: False.
+ """
+ # Parameters to build each block:
+ # [kernel size, mid channels, out channels, with_se, act type, stride]
+ arch_settings = {
+ 'small': [[3, 16, 16, True, 'ReLU', 2],
+ [3, 72, 24, False, 'ReLU', 2],
+ [3, 88, 24, False, 'ReLU', 1],
+ [5, 96, 40, True, 'HSwish', 2],
+ [5, 240, 40, True, 'HSwish', 1],
+ [5, 240, 40, True, 'HSwish', 1],
+ [5, 120, 48, True, 'HSwish', 1],
+ [5, 144, 48, True, 'HSwish', 1],
+ [5, 288, 96, True, 'HSwish', 2],
+ [5, 576, 96, True, 'HSwish', 1],
+ [5, 576, 96, True, 'HSwish', 1]],
+ 'big': [[3, 16, 16, False, 'ReLU', 1],
+ [3, 64, 24, False, 'ReLU', 2],
+ [3, 72, 24, False, 'ReLU', 1],
+ [5, 72, 40, True, 'ReLU', 2],
+ [5, 120, 40, True, 'ReLU', 1],
+ [5, 120, 40, True, 'ReLU', 1],
+ [3, 240, 80, False, 'HSwish', 2],
+ [3, 200, 80, False, 'HSwish', 1],
+ [3, 184, 80, False, 'HSwish', 1],
+ [3, 184, 80, False, 'HSwish', 1],
+ [3, 480, 112, True, 'HSwish', 1],
+ [3, 672, 112, True, 'HSwish', 1],
+ [5, 672, 160, True, 'HSwish', 1],
+ [5, 672, 160, True, 'HSwish', 2],
+ [5, 960, 160, True, 'HSwish', 1]]
+ } # yapf: disable
+
+ def __init__(self,
+ arch='small',
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ out_indices=(-1, ),
+ frozen_stages=-1,
+ norm_eval=False,
+ with_cp=False):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+ assert arch in self.arch_settings
+ for index in out_indices:
+ if index not in range(-len(self.arch_settings[arch]),
+ len(self.arch_settings[arch])):
+ raise ValueError('the item in out_indices must in '
+ f'range(0, {len(self.arch_settings[arch])}). '
+ f'But received {index}')
+
+ if frozen_stages not in range(-1, len(self.arch_settings[arch])):
+ raise ValueError('frozen_stages must be in range(-1, '
+ f'{len(self.arch_settings[arch])}). '
+ f'But received {frozen_stages}')
+ self.arch = arch
+ self.conv_cfg = conv_cfg
+ self.norm_cfg = norm_cfg
+ self.out_indices = out_indices
+ self.frozen_stages = frozen_stages
+ self.norm_eval = norm_eval
+ self.with_cp = with_cp
+
+ self.in_channels = 16
+ self.conv1 = ConvModule(
+ in_channels=3,
+ out_channels=self.in_channels,
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=dict(type='HSwish'))
+
+ self.layers = self._make_layer()
+ self.feat_dim = self.arch_settings[arch][-1][2]
+
+ def _make_layer(self):
+ layers = []
+ layer_setting = self.arch_settings[self.arch]
+ for i, params in enumerate(layer_setting):
+ (kernel_size, mid_channels, out_channels, with_se, act,
+ stride) = params
+ if with_se:
+ se_cfg = dict(
+ channels=mid_channels,
+ ratio=4,
+ act_cfg=(dict(type='ReLU'), dict(type='HSigmoid')))
+ else:
+ se_cfg = None
+
+ layer = InvertedResidual(
+ in_channels=self.in_channels,
+ out_channels=out_channels,
+ mid_channels=mid_channels,
+ kernel_size=kernel_size,
+ stride=stride,
+ se_cfg=se_cfg,
+ with_expand_conv=True,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ act_cfg=dict(type=act),
+ with_cp=self.with_cp)
+ self.in_channels = out_channels
+ layer_name = f'layer{i + 1}'
+ self.add_module(layer_name, layer)
+ layers.append(layer_name)
+ return layers
+
+ def init_weights(self, pretrained=None):
+ if isinstance(pretrained, str):
+ logger = logging.getLogger()
+ load_checkpoint(self, pretrained, strict=False, logger=logger)
+ elif pretrained is None:
+ for m in self.modules():
+ if isinstance(m, nn.Conv2d):
+ kaiming_init(m)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
+ else:
+ raise TypeError('pretrained must be a str or None')
+
+ def forward(self, x):
+ x = self.conv1(x)
+
+ outs = []
+ for i, layer_name in enumerate(self.layers):
+ layer = getattr(self, layer_name)
+ x = layer(x)
+ if i in self.out_indices or \
+ i - len(self.layers) in self.out_indices:
+ outs.append(x)
+
+ if len(outs) == 1:
+ return outs[0]
+ return tuple(outs)
+
+ def _freeze_stages(self):
+ if self.frozen_stages >= 0:
+ for param in self.conv1.parameters():
+ param.requires_grad = False
+ for i in range(1, self.frozen_stages + 1):
+ layer = getattr(self, f'layer{i}')
+ layer.eval()
+ for param in layer.parameters():
+ param.requires_grad = False
+
+ def train(self, mode=True):
+ super().train(mode)
+ self._freeze_stages()
+ if mode and self.norm_eval:
+ for m in self.modules():
+ if isinstance(m, _BatchNorm):
+ m.eval()
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/mspn.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/mspn.py
new file mode 100644
index 0000000000000000000000000000000000000000..71cee34e399780e8b67eac43d862b65a3ce05412
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/mspn.py
@@ -0,0 +1,513 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy as cp
+from collections import OrderedDict
+
+import torch.nn as nn
+import torch.nn.functional as F
+from mmcv.cnn import (ConvModule, MaxPool2d, constant_init, kaiming_init,
+ normal_init)
+from mmcv.runner.checkpoint import load_state_dict
+
+from mmpose.utils import get_root_logger
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+from .resnet import Bottleneck as _Bottleneck
+from .utils.utils import get_state_dict
+
+
+class Bottleneck(_Bottleneck):
+ expansion = 4
+ """Bottleneck block for MSPN.
+
+ Args:
+ in_channels (int): Input channels of this block.
+ out_channels (int): Output channels of this block.
+ stride (int): stride of the block. Default: 1
+ downsample (nn.Module): downsample operation on identity branch.
+ Default: None
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ """
+
+ def __init__(self, in_channels, out_channels, **kwargs):
+ super().__init__(in_channels, out_channels * 4, **kwargs)
+
+
+class DownsampleModule(nn.Module):
+ """Downsample module for MSPN.
+
+ Args:
+ block (nn.Module): Downsample block.
+ num_blocks (list): Number of blocks in each downsample unit.
+ num_units (int): Numbers of downsample units. Default: 4
+ has_skip (bool): Have skip connections from prior upsample
+ module or not. Default:False
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ in_channels (int): Number of channels of the input feature to
+ downsample module. Default: 64
+ """
+
+ def __init__(self,
+ block,
+ num_blocks,
+ num_units=4,
+ has_skip=False,
+ norm_cfg=dict(type='BN'),
+ in_channels=64):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+ self.has_skip = has_skip
+ self.in_channels = in_channels
+ assert len(num_blocks) == num_units
+ self.num_blocks = num_blocks
+ self.num_units = num_units
+ self.norm_cfg = norm_cfg
+ self.layer1 = self._make_layer(block, in_channels, num_blocks[0])
+ for i in range(1, num_units):
+ module_name = f'layer{i + 1}'
+ self.add_module(
+ module_name,
+ self._make_layer(
+ block, in_channels * pow(2, i), num_blocks[i], stride=2))
+
+ def _make_layer(self, block, out_channels, blocks, stride=1):
+ downsample = None
+ if stride != 1 or self.in_channels != out_channels * block.expansion:
+ downsample = ConvModule(
+ self.in_channels,
+ out_channels * block.expansion,
+ kernel_size=1,
+ stride=stride,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ act_cfg=None,
+ inplace=True)
+
+ units = list()
+ units.append(
+ block(
+ self.in_channels,
+ out_channels,
+ stride=stride,
+ downsample=downsample,
+ norm_cfg=self.norm_cfg))
+ self.in_channels = out_channels * block.expansion
+ for _ in range(1, blocks):
+ units.append(block(self.in_channels, out_channels))
+
+ return nn.Sequential(*units)
+
+ def forward(self, x, skip1, skip2):
+ out = list()
+ for i in range(self.num_units):
+ module_name = f'layer{i + 1}'
+ module_i = getattr(self, module_name)
+ x = module_i(x)
+ if self.has_skip:
+ x = x + skip1[i] + skip2[i]
+ out.append(x)
+ out.reverse()
+
+ return tuple(out)
+
+
+class UpsampleUnit(nn.Module):
+ """Upsample unit for upsample module.
+
+ Args:
+ ind (int): Indicates whether to interpolate (>0) and whether to
+ generate feature map for the next hourglass-like module.
+ num_units (int): Number of units that form a upsample module. Along
+ with ind and gen_cross_conv, nm_units is used to decide whether
+ to generate feature map for the next hourglass-like module.
+ in_channels (int): Channel number of the skip-in feature maps from
+ the corresponding downsample unit.
+ unit_channels (int): Channel number in this unit. Default:256.
+ gen_skip: (bool): Whether or not to generate skips for the posterior
+ downsample module. Default:False
+ gen_cross_conv (bool): Whether to generate feature map for the next
+ hourglass-like module. Default:False
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ out_channels (int): Number of channels of feature output by upsample
+ module. Must equal to in_channels of downsample module. Default:64
+ """
+
+ def __init__(self,
+ ind,
+ num_units,
+ in_channels,
+ unit_channels=256,
+ gen_skip=False,
+ gen_cross_conv=False,
+ norm_cfg=dict(type='BN'),
+ out_channels=64):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+ self.num_units = num_units
+ self.norm_cfg = norm_cfg
+ self.in_skip = ConvModule(
+ in_channels,
+ unit_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ act_cfg=None,
+ inplace=True)
+ self.relu = nn.ReLU(inplace=True)
+
+ self.ind = ind
+ if self.ind > 0:
+ self.up_conv = ConvModule(
+ unit_channels,
+ unit_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ act_cfg=None,
+ inplace=True)
+
+ self.gen_skip = gen_skip
+ if self.gen_skip:
+ self.out_skip1 = ConvModule(
+ in_channels,
+ in_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ inplace=True)
+
+ self.out_skip2 = ConvModule(
+ unit_channels,
+ in_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ inplace=True)
+
+ self.gen_cross_conv = gen_cross_conv
+ if self.ind == num_units - 1 and self.gen_cross_conv:
+ self.cross_conv = ConvModule(
+ unit_channels,
+ out_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ inplace=True)
+
+ def forward(self, x, up_x):
+ out = self.in_skip(x)
+
+ if self.ind > 0:
+ up_x = F.interpolate(
+ up_x,
+ size=(x.size(2), x.size(3)),
+ mode='bilinear',
+ align_corners=True)
+ up_x = self.up_conv(up_x)
+ out = out + up_x
+ out = self.relu(out)
+
+ skip1 = None
+ skip2 = None
+ if self.gen_skip:
+ skip1 = self.out_skip1(x)
+ skip2 = self.out_skip2(out)
+
+ cross_conv = None
+ if self.ind == self.num_units - 1 and self.gen_cross_conv:
+ cross_conv = self.cross_conv(out)
+
+ return out, skip1, skip2, cross_conv
+
+
+class UpsampleModule(nn.Module):
+ """Upsample module for MSPN.
+
+ Args:
+ unit_channels (int): Channel number in the upsample units.
+ Default:256.
+ num_units (int): Numbers of upsample units. Default: 4
+ gen_skip (bool): Whether to generate skip for posterior downsample
+ module or not. Default:False
+ gen_cross_conv (bool): Whether to generate feature map for the next
+ hourglass-like module. Default:False
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ out_channels (int): Number of channels of feature output by upsample
+ module. Must equal to in_channels of downsample module. Default:64
+ """
+
+ def __init__(self,
+ unit_channels=256,
+ num_units=4,
+ gen_skip=False,
+ gen_cross_conv=False,
+ norm_cfg=dict(type='BN'),
+ out_channels=64):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+ self.in_channels = list()
+ for i in range(num_units):
+ self.in_channels.append(Bottleneck.expansion * out_channels *
+ pow(2, i))
+ self.in_channels.reverse()
+ self.num_units = num_units
+ self.gen_skip = gen_skip
+ self.gen_cross_conv = gen_cross_conv
+ self.norm_cfg = norm_cfg
+ for i in range(num_units):
+ module_name = f'up{i + 1}'
+ self.add_module(
+ module_name,
+ UpsampleUnit(
+ i,
+ self.num_units,
+ self.in_channels[i],
+ unit_channels,
+ self.gen_skip,
+ self.gen_cross_conv,
+ norm_cfg=self.norm_cfg,
+ out_channels=64))
+
+ def forward(self, x):
+ out = list()
+ skip1 = list()
+ skip2 = list()
+ cross_conv = None
+ for i in range(self.num_units):
+ module_i = getattr(self, f'up{i + 1}')
+ if i == 0:
+ outi, skip1_i, skip2_i, _ = module_i(x[i], None)
+ elif i == self.num_units - 1:
+ outi, skip1_i, skip2_i, cross_conv = module_i(x[i], out[i - 1])
+ else:
+ outi, skip1_i, skip2_i, _ = module_i(x[i], out[i - 1])
+ out.append(outi)
+ skip1.append(skip1_i)
+ skip2.append(skip2_i)
+ skip1.reverse()
+ skip2.reverse()
+
+ return out, skip1, skip2, cross_conv
+
+
+class SingleStageNetwork(nn.Module):
+ """Single_stage Network.
+
+ Args:
+ unit_channels (int): Channel number in the upsample units. Default:256.
+ num_units (int): Numbers of downsample/upsample units. Default: 4
+ gen_skip (bool): Whether to generate skip for posterior downsample
+ module or not. Default:False
+ gen_cross_conv (bool): Whether to generate feature map for the next
+ hourglass-like module. Default:False
+ has_skip (bool): Have skip connections from prior upsample
+ module or not. Default:False
+ num_blocks (list): Number of blocks in each downsample unit.
+ Default: [2, 2, 2, 2] Note: Make sure num_units==len(num_blocks)
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ in_channels (int): Number of channels of the feature from ResNetTop.
+ Default: 64.
+ """
+
+ def __init__(self,
+ has_skip=False,
+ gen_skip=False,
+ gen_cross_conv=False,
+ unit_channels=256,
+ num_units=4,
+ num_blocks=[2, 2, 2, 2],
+ norm_cfg=dict(type='BN'),
+ in_channels=64):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ num_blocks = cp.deepcopy(num_blocks)
+ super().__init__()
+ assert len(num_blocks) == num_units
+ self.has_skip = has_skip
+ self.gen_skip = gen_skip
+ self.gen_cross_conv = gen_cross_conv
+ self.num_units = num_units
+ self.unit_channels = unit_channels
+ self.num_blocks = num_blocks
+ self.norm_cfg = norm_cfg
+
+ self.downsample = DownsampleModule(Bottleneck, num_blocks, num_units,
+ has_skip, norm_cfg, in_channels)
+ self.upsample = UpsampleModule(unit_channels, num_units, gen_skip,
+ gen_cross_conv, norm_cfg, in_channels)
+
+ def forward(self, x, skip1, skip2):
+ mid = self.downsample(x, skip1, skip2)
+ out, skip1, skip2, cross_conv = self.upsample(mid)
+
+ return out, skip1, skip2, cross_conv
+
+
+class ResNetTop(nn.Module):
+ """ResNet top for MSPN.
+
+ Args:
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ channels (int): Number of channels of the feature output by ResNetTop.
+ """
+
+ def __init__(self, norm_cfg=dict(type='BN'), channels=64):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+ self.top = nn.Sequential(
+ ConvModule(
+ 3,
+ channels,
+ kernel_size=7,
+ stride=2,
+ padding=3,
+ norm_cfg=norm_cfg,
+ inplace=True), MaxPool2d(kernel_size=3, stride=2, padding=1))
+
+ def forward(self, img):
+ return self.top(img)
+
+
+@BACKBONES.register_module()
+class MSPN(BaseBackbone):
+ """MSPN backbone. Paper ref: Li et al. "Rethinking on Multi-Stage Networks
+ for Human Pose Estimation" (CVPR 2020).
+
+ Args:
+ unit_channels (int): Number of Channels in an upsample unit.
+ Default: 256
+ num_stages (int): Number of stages in a multi-stage MSPN. Default: 4
+ num_units (int): Number of downsample/upsample units in a single-stage
+ network. Default: 4
+ Note: Make sure num_units == len(self.num_blocks)
+ num_blocks (list): Number of bottlenecks in each
+ downsample unit. Default: [2, 2, 2, 2]
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ res_top_channels (int): Number of channels of feature from ResNetTop.
+ Default: 64.
+
+ Example:
+ >>> from mmpose.models import MSPN
+ >>> import torch
+ >>> self = MSPN(num_stages=2,num_units=2,num_blocks=[2,2])
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 3, 511, 511)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_output in level_outputs:
+ ... for feature in level_output:
+ ... print(tuple(feature.shape))
+ ...
+ (1, 256, 64, 64)
+ (1, 256, 128, 128)
+ (1, 256, 64, 64)
+ (1, 256, 128, 128)
+ """
+
+ def __init__(self,
+ unit_channels=256,
+ num_stages=4,
+ num_units=4,
+ num_blocks=[2, 2, 2, 2],
+ norm_cfg=dict(type='BN'),
+ res_top_channels=64):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ num_blocks = cp.deepcopy(num_blocks)
+ super().__init__()
+ self.unit_channels = unit_channels
+ self.num_stages = num_stages
+ self.num_units = num_units
+ self.num_blocks = num_blocks
+ self.norm_cfg = norm_cfg
+
+ assert self.num_stages > 0
+ assert self.num_units > 1
+ assert self.num_units == len(self.num_blocks)
+ self.top = ResNetTop(norm_cfg=norm_cfg)
+ self.multi_stage_mspn = nn.ModuleList([])
+ for i in range(self.num_stages):
+ if i == 0:
+ has_skip = False
+ else:
+ has_skip = True
+ if i != self.num_stages - 1:
+ gen_skip = True
+ gen_cross_conv = True
+ else:
+ gen_skip = False
+ gen_cross_conv = False
+ self.multi_stage_mspn.append(
+ SingleStageNetwork(has_skip, gen_skip, gen_cross_conv,
+ unit_channels, num_units, num_blocks,
+ norm_cfg, res_top_channels))
+
+ def forward(self, x):
+ """Model forward function."""
+ out_feats = []
+ skip1 = None
+ skip2 = None
+ x = self.top(x)
+ for i in range(self.num_stages):
+ out, skip1, skip2, x = self.multi_stage_mspn[i](x, skip1, skip2)
+ out_feats.append(out)
+
+ return out_feats
+
+ def init_weights(self, pretrained=None):
+ """Initialize model weights."""
+ if isinstance(pretrained, str):
+ logger = get_root_logger()
+ state_dict_tmp = get_state_dict(pretrained)
+ state_dict = OrderedDict()
+ state_dict['top'] = OrderedDict()
+ state_dict['bottlenecks'] = OrderedDict()
+ for k, v in state_dict_tmp.items():
+ if k.startswith('layer'):
+ if 'downsample.0' in k:
+ state_dict['bottlenecks'][k.replace(
+ 'downsample.0', 'downsample.conv')] = v
+ elif 'downsample.1' in k:
+ state_dict['bottlenecks'][k.replace(
+ 'downsample.1', 'downsample.bn')] = v
+ else:
+ state_dict['bottlenecks'][k] = v
+ elif k.startswith('conv1'):
+ state_dict['top'][k.replace('conv1', 'top.0.conv')] = v
+ elif k.startswith('bn1'):
+ state_dict['top'][k.replace('bn1', 'top.0.bn')] = v
+
+ load_state_dict(
+ self.top, state_dict['top'], strict=False, logger=logger)
+ for i in range(self.num_stages):
+ load_state_dict(
+ self.multi_stage_mspn[i].downsample,
+ state_dict['bottlenecks'],
+ strict=False,
+ logger=logger)
+ else:
+ for m in self.multi_stage_mspn.modules():
+ if isinstance(m, nn.Conv2d):
+ kaiming_init(m)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
+ elif isinstance(m, nn.Linear):
+ normal_init(m, std=0.01)
+
+ for m in self.top.modules():
+ if isinstance(m, nn.Conv2d):
+ kaiming_init(m)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/regnet.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/regnet.py
new file mode 100644
index 0000000000000000000000000000000000000000..693417c2d61066e4e9a90989ad61700448028e58
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/regnet.py
@@ -0,0 +1,317 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+
+import numpy as np
+import torch.nn as nn
+from mmcv.cnn import build_conv_layer, build_norm_layer
+
+from ..builder import BACKBONES
+from .resnet import ResNet
+from .resnext import Bottleneck
+
+
+@BACKBONES.register_module()
+class RegNet(ResNet):
+ """RegNet backbone.
+
+ More details can be found in `paper `__ .
+
+ Args:
+ arch (dict): The parameter of RegNets.
+ - w0 (int): initial width
+ - wa (float): slope of width
+ - wm (float): quantization parameter to quantize the width
+ - depth (int): depth of the backbone
+ - group_w (int): width of group
+ - bot_mul (float): bottleneck ratio, i.e. expansion of bottleneck.
+ strides (Sequence[int]): Strides of the first block of each stage.
+ base_channels (int): Base channels after stem layer.
+ in_channels (int): Number of input image channels. Default: 3.
+ dilations (Sequence[int]): Dilation of each stage.
+ out_indices (Sequence[int]): Output from which stages.
+ style (str): `pytorch` or `caffe`. If set to "pytorch", the stride-two
+ layer is the 3x3 conv layer, otherwise the stride-two layer is
+ the first 1x1 conv layer. Default: "pytorch".
+ frozen_stages (int): Stages to be frozen (all param fixed). -1 means
+ not freezing any parameters. Default: -1.
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN', requires_grad=True).
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ zero_init_residual (bool): whether to use zero init for last norm layer
+ in resblocks to let them behave as identity. Default: True.
+
+ Example:
+ >>> from mmpose.models import RegNet
+ >>> import torch
+ >>> self = RegNet(
+ arch=dict(
+ w0=88,
+ wa=26.31,
+ wm=2.25,
+ group_w=48,
+ depth=25,
+ bot_mul=1.0),
+ out_indices=(0, 1, 2, 3))
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 3, 32, 32)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_out in level_outputs:
+ ... print(tuple(level_out.shape))
+ (1, 96, 8, 8)
+ (1, 192, 4, 4)
+ (1, 432, 2, 2)
+ (1, 1008, 1, 1)
+ """
+ arch_settings = {
+ 'regnetx_400mf':
+ dict(w0=24, wa=24.48, wm=2.54, group_w=16, depth=22, bot_mul=1.0),
+ 'regnetx_800mf':
+ dict(w0=56, wa=35.73, wm=2.28, group_w=16, depth=16, bot_mul=1.0),
+ 'regnetx_1.6gf':
+ dict(w0=80, wa=34.01, wm=2.25, group_w=24, depth=18, bot_mul=1.0),
+ 'regnetx_3.2gf':
+ dict(w0=88, wa=26.31, wm=2.25, group_w=48, depth=25, bot_mul=1.0),
+ 'regnetx_4.0gf':
+ dict(w0=96, wa=38.65, wm=2.43, group_w=40, depth=23, bot_mul=1.0),
+ 'regnetx_6.4gf':
+ dict(w0=184, wa=60.83, wm=2.07, group_w=56, depth=17, bot_mul=1.0),
+ 'regnetx_8.0gf':
+ dict(w0=80, wa=49.56, wm=2.88, group_w=120, depth=23, bot_mul=1.0),
+ 'regnetx_12gf':
+ dict(w0=168, wa=73.36, wm=2.37, group_w=112, depth=19, bot_mul=1.0),
+ }
+
+ def __init__(self,
+ arch,
+ in_channels=3,
+ stem_channels=32,
+ base_channels=32,
+ strides=(2, 2, 2, 2),
+ dilations=(1, 1, 1, 1),
+ out_indices=(3, ),
+ style='pytorch',
+ deep_stem=False,
+ avg_down=False,
+ frozen_stages=-1,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN', requires_grad=True),
+ norm_eval=False,
+ with_cp=False,
+ zero_init_residual=True):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super(ResNet, self).__init__()
+
+ # Generate RegNet parameters first
+ if isinstance(arch, str):
+ assert arch in self.arch_settings, \
+ f'"arch": "{arch}" is not one of the' \
+ ' arch_settings'
+ arch = self.arch_settings[arch]
+ elif not isinstance(arch, dict):
+ raise TypeError('Expect "arch" to be either a string '
+ f'or a dict, got {type(arch)}')
+
+ widths, num_stages = self.generate_regnet(
+ arch['w0'],
+ arch['wa'],
+ arch['wm'],
+ arch['depth'],
+ )
+ # Convert to per stage format
+ stage_widths, stage_blocks = self.get_stages_from_blocks(widths)
+ # Generate group widths and bot muls
+ group_widths = [arch['group_w'] for _ in range(num_stages)]
+ self.bottleneck_ratio = [arch['bot_mul'] for _ in range(num_stages)]
+ # Adjust the compatibility of stage_widths and group_widths
+ stage_widths, group_widths = self.adjust_width_group(
+ stage_widths, self.bottleneck_ratio, group_widths)
+
+ # Group params by stage
+ self.stage_widths = stage_widths
+ self.group_widths = group_widths
+ self.depth = sum(stage_blocks)
+ self.stem_channels = stem_channels
+ self.base_channels = base_channels
+ self.num_stages = num_stages
+ assert 1 <= num_stages <= 4
+ self.strides = strides
+ self.dilations = dilations
+ assert len(strides) == len(dilations) == num_stages
+ self.out_indices = out_indices
+ assert max(out_indices) < num_stages
+ self.style = style
+ self.deep_stem = deep_stem
+ if self.deep_stem:
+ raise NotImplementedError(
+ 'deep_stem has not been implemented for RegNet')
+ self.avg_down = avg_down
+ self.frozen_stages = frozen_stages
+ self.conv_cfg = conv_cfg
+ self.norm_cfg = norm_cfg
+ self.with_cp = with_cp
+ self.norm_eval = norm_eval
+ self.zero_init_residual = zero_init_residual
+ self.stage_blocks = stage_blocks[:num_stages]
+
+ self._make_stem_layer(in_channels, stem_channels)
+
+ _in_channels = stem_channels
+ self.res_layers = []
+ for i, num_blocks in enumerate(self.stage_blocks):
+ stride = self.strides[i]
+ dilation = self.dilations[i]
+ group_width = self.group_widths[i]
+ width = int(round(self.stage_widths[i] * self.bottleneck_ratio[i]))
+ stage_groups = width // group_width
+
+ res_layer = self.make_res_layer(
+ block=Bottleneck,
+ num_blocks=num_blocks,
+ in_channels=_in_channels,
+ out_channels=self.stage_widths[i],
+ expansion=1,
+ stride=stride,
+ dilation=dilation,
+ style=self.style,
+ avg_down=self.avg_down,
+ with_cp=self.with_cp,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ base_channels=self.stage_widths[i],
+ groups=stage_groups,
+ width_per_group=group_width)
+ _in_channels = self.stage_widths[i]
+ layer_name = f'layer{i + 1}'
+ self.add_module(layer_name, res_layer)
+ self.res_layers.append(layer_name)
+
+ self._freeze_stages()
+
+ self.feat_dim = stage_widths[-1]
+
+ def _make_stem_layer(self, in_channels, base_channels):
+ self.conv1 = build_conv_layer(
+ self.conv_cfg,
+ in_channels,
+ base_channels,
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ bias=False)
+ self.norm1_name, norm1 = build_norm_layer(
+ self.norm_cfg, base_channels, postfix=1)
+ self.add_module(self.norm1_name, norm1)
+ self.relu = nn.ReLU(inplace=True)
+
+ @staticmethod
+ def generate_regnet(initial_width,
+ width_slope,
+ width_parameter,
+ depth,
+ divisor=8):
+ """Generates per block width from RegNet parameters.
+
+ Args:
+ initial_width ([int]): Initial width of the backbone
+ width_slope ([float]): Slope of the quantized linear function
+ width_parameter ([int]): Parameter used to quantize the width.
+ depth ([int]): Depth of the backbone.
+ divisor (int, optional): The divisor of channels. Defaults to 8.
+
+ Returns:
+ list, int: return a list of widths of each stage and the number of
+ stages
+ """
+ assert width_slope >= 0
+ assert initial_width > 0
+ assert width_parameter > 1
+ assert initial_width % divisor == 0
+ widths_cont = np.arange(depth) * width_slope + initial_width
+ ks = np.round(
+ np.log(widths_cont / initial_width) / np.log(width_parameter))
+ widths = initial_width * np.power(width_parameter, ks)
+ widths = np.round(np.divide(widths, divisor)) * divisor
+ num_stages = len(np.unique(widths))
+ widths, widths_cont = widths.astype(int).tolist(), widths_cont.tolist()
+ return widths, num_stages
+
+ @staticmethod
+ def quantize_float(number, divisor):
+ """Converts a float to closest non-zero int divisible by divior.
+
+ Args:
+ number (int): Original number to be quantized.
+ divisor (int): Divisor used to quantize the number.
+
+ Returns:
+ int: quantized number that is divisible by devisor.
+ """
+ return int(round(number / divisor) * divisor)
+
+ def adjust_width_group(self, widths, bottleneck_ratio, groups):
+ """Adjusts the compatibility of widths and groups.
+
+ Args:
+ widths (list[int]): Width of each stage.
+ bottleneck_ratio (float): Bottleneck ratio.
+ groups (int): number of groups in each stage
+
+ Returns:
+ tuple(list): The adjusted widths and groups of each stage.
+ """
+ bottleneck_width = [
+ int(w * b) for w, b in zip(widths, bottleneck_ratio)
+ ]
+ groups = [min(g, w_bot) for g, w_bot in zip(groups, bottleneck_width)]
+ bottleneck_width = [
+ self.quantize_float(w_bot, g)
+ for w_bot, g in zip(bottleneck_width, groups)
+ ]
+ widths = [
+ int(w_bot / b)
+ for w_bot, b in zip(bottleneck_width, bottleneck_ratio)
+ ]
+ return widths, groups
+
+ def get_stages_from_blocks(self, widths):
+ """Gets widths/stage_blocks of network at each stage.
+
+ Args:
+ widths (list[int]): Width in each stage.
+
+ Returns:
+ tuple(list): width and depth of each stage
+ """
+ width_diff = [
+ width != width_prev
+ for width, width_prev in zip(widths + [0], [0] + widths)
+ ]
+ stage_widths = [
+ width for width, diff in zip(widths, width_diff[:-1]) if diff
+ ]
+ stage_blocks = np.diff([
+ depth for depth, diff in zip(range(len(width_diff)), width_diff)
+ if diff
+ ]).tolist()
+ return stage_widths, stage_blocks
+
+ def forward(self, x):
+ x = self.conv1(x)
+ x = self.norm1(x)
+ x = self.relu(x)
+
+ outs = []
+ for i, layer_name in enumerate(self.res_layers):
+ res_layer = getattr(self, layer_name)
+ x = res_layer(x)
+ if i in self.out_indices:
+ outs.append(x)
+
+ if len(outs) == 1:
+ return outs[0]
+ return tuple(outs)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/resnest.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/resnest.py
new file mode 100644
index 0000000000000000000000000000000000000000..0a2d4081df1417155f0626646f5fe3d0dbfc2864
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/resnest.py
@@ -0,0 +1,338 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import torch.utils.checkpoint as cp
+from mmcv.cnn import build_conv_layer, build_norm_layer
+
+from ..builder import BACKBONES
+from .resnet import Bottleneck as _Bottleneck
+from .resnet import ResLayer, ResNetV1d
+
+
+class RSoftmax(nn.Module):
+ """Radix Softmax module in ``SplitAttentionConv2d``.
+
+ Args:
+ radix (int): Radix of input.
+ groups (int): Groups of input.
+ """
+
+ def __init__(self, radix, groups):
+ super().__init__()
+ self.radix = radix
+ self.groups = groups
+
+ def forward(self, x):
+ batch = x.size(0)
+ if self.radix > 1:
+ x = x.view(batch, self.groups, self.radix, -1).transpose(1, 2)
+ x = F.softmax(x, dim=1)
+ x = x.reshape(batch, -1)
+ else:
+ x = torch.sigmoid(x)
+ return x
+
+
+class SplitAttentionConv2d(nn.Module):
+ """Split-Attention Conv2d.
+
+ Args:
+ in_channels (int): Same as nn.Conv2d.
+ out_channels (int): Same as nn.Conv2d.
+ kernel_size (int | tuple[int]): Same as nn.Conv2d.
+ stride (int | tuple[int]): Same as nn.Conv2d.
+ padding (int | tuple[int]): Same as nn.Conv2d.
+ dilation (int | tuple[int]): Same as nn.Conv2d.
+ groups (int): Same as nn.Conv2d.
+ radix (int): Radix of SpltAtConv2d. Default: 2
+ reduction_factor (int): Reduction factor of SplitAttentionConv2d.
+ Default: 4.
+ conv_cfg (dict): Config dict for convolution layer. Default: None,
+ which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer. Default: None.
+ """
+
+ def __init__(self,
+ in_channels,
+ channels,
+ kernel_size,
+ stride=1,
+ padding=0,
+ dilation=1,
+ groups=1,
+ radix=2,
+ reduction_factor=4,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN')):
+ super().__init__()
+ inter_channels = max(in_channels * radix // reduction_factor, 32)
+ self.radix = radix
+ self.groups = groups
+ self.channels = channels
+ self.conv = build_conv_layer(
+ conv_cfg,
+ in_channels,
+ channels * radix,
+ kernel_size,
+ stride=stride,
+ padding=padding,
+ dilation=dilation,
+ groups=groups * radix,
+ bias=False)
+ self.norm0_name, norm0 = build_norm_layer(
+ norm_cfg, channels * radix, postfix=0)
+ self.add_module(self.norm0_name, norm0)
+ self.relu = nn.ReLU(inplace=True)
+ self.fc1 = build_conv_layer(
+ None, channels, inter_channels, 1, groups=self.groups)
+ self.norm1_name, norm1 = build_norm_layer(
+ norm_cfg, inter_channels, postfix=1)
+ self.add_module(self.norm1_name, norm1)
+ self.fc2 = build_conv_layer(
+ None, inter_channels, channels * radix, 1, groups=self.groups)
+ self.rsoftmax = RSoftmax(radix, groups)
+
+ @property
+ def norm0(self):
+ return getattr(self, self.norm0_name)
+
+ @property
+ def norm1(self):
+ return getattr(self, self.norm1_name)
+
+ def forward(self, x):
+ x = self.conv(x)
+ x = self.norm0(x)
+ x = self.relu(x)
+
+ batch, rchannel = x.shape[:2]
+ if self.radix > 1:
+ splits = x.view(batch, self.radix, -1, *x.shape[2:])
+ gap = splits.sum(dim=1)
+ else:
+ gap = x
+ gap = F.adaptive_avg_pool2d(gap, 1)
+ gap = self.fc1(gap)
+
+ gap = self.norm1(gap)
+ gap = self.relu(gap)
+
+ atten = self.fc2(gap)
+ atten = self.rsoftmax(atten).view(batch, -1, 1, 1)
+
+ if self.radix > 1:
+ attens = atten.view(batch, self.radix, -1, *atten.shape[2:])
+ out = torch.sum(attens * splits, dim=1)
+ else:
+ out = atten * x
+ return out.contiguous()
+
+
+class Bottleneck(_Bottleneck):
+ """Bottleneck block for ResNeSt.
+
+ Args:
+ in_channels (int): Input channels of this block.
+ out_channels (int): Output channels of this block.
+ groups (int): Groups of conv2.
+ width_per_group (int): Width per group of conv2. 64x4d indicates
+ ``groups=64, width_per_group=4`` and 32x8d indicates
+ ``groups=32, width_per_group=8``.
+ radix (int): Radix of SpltAtConv2d. Default: 2
+ reduction_factor (int): Reduction factor of SplitAttentionConv2d.
+ Default: 4.
+ avg_down_stride (bool): Whether to use average pool for stride in
+ Bottleneck. Default: True.
+ stride (int): stride of the block. Default: 1
+ dilation (int): dilation of convolution. Default: 1
+ downsample (nn.Module): downsample operation on identity branch.
+ Default: None
+ style (str): `pytorch` or `caffe`. If set to "pytorch", the stride-two
+ layer is the 3x3 conv layer, otherwise the stride-two layer is
+ the first 1x1 conv layer.
+ conv_cfg (dict): dictionary to construct and config conv layer.
+ Default: None
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed.
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ groups=1,
+ width_per_group=4,
+ base_channels=64,
+ radix=2,
+ reduction_factor=4,
+ avg_down_stride=True,
+ **kwargs):
+ super().__init__(in_channels, out_channels, **kwargs)
+
+ self.groups = groups
+ self.width_per_group = width_per_group
+
+ # For ResNet bottleneck, middle channels are determined by expansion
+ # and out_channels, but for ResNeXt bottleneck, it is determined by
+ # groups and width_per_group and the stage it is located in.
+ if groups != 1:
+ assert self.mid_channels % base_channels == 0
+ self.mid_channels = (
+ groups * width_per_group * self.mid_channels // base_channels)
+
+ self.avg_down_stride = avg_down_stride and self.conv2_stride > 1
+
+ self.norm1_name, norm1 = build_norm_layer(
+ self.norm_cfg, self.mid_channels, postfix=1)
+ self.norm3_name, norm3 = build_norm_layer(
+ self.norm_cfg, self.out_channels, postfix=3)
+
+ self.conv1 = build_conv_layer(
+ self.conv_cfg,
+ self.in_channels,
+ self.mid_channels,
+ kernel_size=1,
+ stride=self.conv1_stride,
+ bias=False)
+ self.add_module(self.norm1_name, norm1)
+ self.conv2 = SplitAttentionConv2d(
+ self.mid_channels,
+ self.mid_channels,
+ kernel_size=3,
+ stride=1 if self.avg_down_stride else self.conv2_stride,
+ padding=self.dilation,
+ dilation=self.dilation,
+ groups=groups,
+ radix=radix,
+ reduction_factor=reduction_factor,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg)
+ delattr(self, self.norm2_name)
+
+ if self.avg_down_stride:
+ self.avd_layer = nn.AvgPool2d(3, self.conv2_stride, padding=1)
+
+ self.conv3 = build_conv_layer(
+ self.conv_cfg,
+ self.mid_channels,
+ self.out_channels,
+ kernel_size=1,
+ bias=False)
+ self.add_module(self.norm3_name, norm3)
+
+ def forward(self, x):
+
+ def _inner_forward(x):
+ identity = x
+
+ out = self.conv1(x)
+ out = self.norm1(out)
+ out = self.relu(out)
+
+ out = self.conv2(out)
+
+ if self.avg_down_stride:
+ out = self.avd_layer(out)
+
+ out = self.conv3(out)
+ out = self.norm3(out)
+
+ if self.downsample is not None:
+ identity = self.downsample(x)
+
+ out += identity
+
+ return out
+
+ if self.with_cp and x.requires_grad:
+ out = cp.checkpoint(_inner_forward, x)
+ else:
+ out = _inner_forward(x)
+
+ out = self.relu(out)
+
+ return out
+
+
+@BACKBONES.register_module()
+class ResNeSt(ResNetV1d):
+ """ResNeSt backbone.
+
+ Please refer to the `paper `__
+ for details.
+
+ Args:
+ depth (int): Network depth, from {50, 101, 152, 200}.
+ groups (int): Groups of conv2 in Bottleneck. Default: 32.
+ width_per_group (int): Width per group of conv2 in Bottleneck.
+ Default: 4.
+ radix (int): Radix of SpltAtConv2d. Default: 2
+ reduction_factor (int): Reduction factor of SplitAttentionConv2d.
+ Default: 4.
+ avg_down_stride (bool): Whether to use average pool for stride in
+ Bottleneck. Default: True.
+ in_channels (int): Number of input image channels. Default: 3.
+ stem_channels (int): Output channels of the stem layer. Default: 64.
+ num_stages (int): Stages of the network. Default: 4.
+ strides (Sequence[int]): Strides of the first block of each stage.
+ Default: ``(1, 2, 2, 2)``.
+ dilations (Sequence[int]): Dilation of each stage.
+ Default: ``(1, 1, 1, 1)``.
+ out_indices (Sequence[int]): Output from which stages. If only one
+ stage is specified, a single tensor (feature map) is returned,
+ otherwise multiple stages are specified, a tuple of tensors will
+ be returned. Default: ``(3, )``.
+ style (str): `pytorch` or `caffe`. If set to "pytorch", the stride-two
+ layer is the 3x3 conv layer, otherwise the stride-two layer is
+ the first 1x1 conv layer.
+ deep_stem (bool): Replace 7x7 conv in input stem with 3 3x3 conv.
+ Default: False.
+ avg_down (bool): Use AvgPool instead of stride conv when
+ downsampling in the bottleneck. Default: False.
+ frozen_stages (int): Stages to be frozen (stop grad and set eval mode).
+ -1 means not freezing any parameters. Default: -1.
+ conv_cfg (dict | None): The config dict for conv layers. Default: None.
+ norm_cfg (dict): The config dict for norm layers.
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ zero_init_residual (bool): Whether to use zero init for last norm layer
+ in resblocks to let them behave as identity. Default: True.
+ """
+
+ arch_settings = {
+ 50: (Bottleneck, (3, 4, 6, 3)),
+ 101: (Bottleneck, (3, 4, 23, 3)),
+ 152: (Bottleneck, (3, 8, 36, 3)),
+ 200: (Bottleneck, (3, 24, 36, 3)),
+ 269: (Bottleneck, (3, 30, 48, 8))
+ }
+
+ def __init__(self,
+ depth,
+ groups=1,
+ width_per_group=4,
+ radix=2,
+ reduction_factor=4,
+ avg_down_stride=True,
+ **kwargs):
+ self.groups = groups
+ self.width_per_group = width_per_group
+ self.radix = radix
+ self.reduction_factor = reduction_factor
+ self.avg_down_stride = avg_down_stride
+ super().__init__(depth=depth, **kwargs)
+
+ def make_res_layer(self, **kwargs):
+ return ResLayer(
+ groups=self.groups,
+ width_per_group=self.width_per_group,
+ base_channels=self.base_channels,
+ radix=self.radix,
+ reduction_factor=self.reduction_factor,
+ avg_down_stride=self.avg_down_stride,
+ **kwargs)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/resnext.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/resnext.py
new file mode 100644
index 0000000000000000000000000000000000000000..c10dc33f98ac3229c77bf306acf19950c295f904
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/resnext.py
@@ -0,0 +1,162 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+from mmcv.cnn import build_conv_layer, build_norm_layer
+
+from ..builder import BACKBONES
+from .resnet import Bottleneck as _Bottleneck
+from .resnet import ResLayer, ResNet
+
+
+class Bottleneck(_Bottleneck):
+ """Bottleneck block for ResNeXt.
+
+ Args:
+ in_channels (int): Input channels of this block.
+ out_channels (int): Output channels of this block.
+ groups (int): Groups of conv2.
+ width_per_group (int): Width per group of conv2. 64x4d indicates
+ ``groups=64, width_per_group=4`` and 32x8d indicates
+ ``groups=32, width_per_group=8``.
+ stride (int): stride of the block. Default: 1
+ dilation (int): dilation of convolution. Default: 1
+ downsample (nn.Module): downsample operation on identity branch.
+ Default: None
+ style (str): `pytorch` or `caffe`. If set to "pytorch", the stride-two
+ layer is the 3x3 conv layer, otherwise the stride-two layer is
+ the first 1x1 conv layer.
+ conv_cfg (dict): dictionary to construct and config conv layer.
+ Default: None
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed.
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ base_channels=64,
+ groups=32,
+ width_per_group=4,
+ **kwargs):
+ super().__init__(in_channels, out_channels, **kwargs)
+ self.groups = groups
+ self.width_per_group = width_per_group
+
+ # For ResNet bottleneck, middle channels are determined by expansion
+ # and out_channels, but for ResNeXt bottleneck, it is determined by
+ # groups and width_per_group and the stage it is located in.
+ if groups != 1:
+ assert self.mid_channels % base_channels == 0
+ self.mid_channels = (
+ groups * width_per_group * self.mid_channels // base_channels)
+
+ self.norm1_name, norm1 = build_norm_layer(
+ self.norm_cfg, self.mid_channels, postfix=1)
+ self.norm2_name, norm2 = build_norm_layer(
+ self.norm_cfg, self.mid_channels, postfix=2)
+ self.norm3_name, norm3 = build_norm_layer(
+ self.norm_cfg, self.out_channels, postfix=3)
+
+ self.conv1 = build_conv_layer(
+ self.conv_cfg,
+ self.in_channels,
+ self.mid_channels,
+ kernel_size=1,
+ stride=self.conv1_stride,
+ bias=False)
+ self.add_module(self.norm1_name, norm1)
+ self.conv2 = build_conv_layer(
+ self.conv_cfg,
+ self.mid_channels,
+ self.mid_channels,
+ kernel_size=3,
+ stride=self.conv2_stride,
+ padding=self.dilation,
+ dilation=self.dilation,
+ groups=groups,
+ bias=False)
+
+ self.add_module(self.norm2_name, norm2)
+ self.conv3 = build_conv_layer(
+ self.conv_cfg,
+ self.mid_channels,
+ self.out_channels,
+ kernel_size=1,
+ bias=False)
+ self.add_module(self.norm3_name, norm3)
+
+
+@BACKBONES.register_module()
+class ResNeXt(ResNet):
+ """ResNeXt backbone.
+
+ Please refer to the `paper `__ for
+ details.
+
+ Args:
+ depth (int): Network depth, from {50, 101, 152}.
+ groups (int): Groups of conv2 in Bottleneck. Default: 32.
+ width_per_group (int): Width per group of conv2 in Bottleneck.
+ Default: 4.
+ in_channels (int): Number of input image channels. Default: 3.
+ stem_channels (int): Output channels of the stem layer. Default: 64.
+ num_stages (int): Stages of the network. Default: 4.
+ strides (Sequence[int]): Strides of the first block of each stage.
+ Default: ``(1, 2, 2, 2)``.
+ dilations (Sequence[int]): Dilation of each stage.
+ Default: ``(1, 1, 1, 1)``.
+ out_indices (Sequence[int]): Output from which stages. If only one
+ stage is specified, a single tensor (feature map) is returned,
+ otherwise multiple stages are specified, a tuple of tensors will
+ be returned. Default: ``(3, )``.
+ style (str): `pytorch` or `caffe`. If set to "pytorch", the stride-two
+ layer is the 3x3 conv layer, otherwise the stride-two layer is
+ the first 1x1 conv layer.
+ deep_stem (bool): Replace 7x7 conv in input stem with 3 3x3 conv.
+ Default: False.
+ avg_down (bool): Use AvgPool instead of stride conv when
+ downsampling in the bottleneck. Default: False.
+ frozen_stages (int): Stages to be frozen (stop grad and set eval mode).
+ -1 means not freezing any parameters. Default: -1.
+ conv_cfg (dict | None): The config dict for conv layers. Default: None.
+ norm_cfg (dict): The config dict for norm layers.
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ zero_init_residual (bool): Whether to use zero init for last norm layer
+ in resblocks to let them behave as identity. Default: True.
+
+ Example:
+ >>> from mmpose.models import ResNeXt
+ >>> import torch
+ >>> self = ResNeXt(depth=50, out_indices=(0, 1, 2, 3))
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 3, 32, 32)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_out in level_outputs:
+ ... print(tuple(level_out.shape))
+ (1, 256, 8, 8)
+ (1, 512, 4, 4)
+ (1, 1024, 2, 2)
+ (1, 2048, 1, 1)
+ """
+
+ arch_settings = {
+ 50: (Bottleneck, (3, 4, 6, 3)),
+ 101: (Bottleneck, (3, 4, 23, 3)),
+ 152: (Bottleneck, (3, 8, 36, 3))
+ }
+
+ def __init__(self, depth, groups=32, width_per_group=4, **kwargs):
+ self.groups = groups
+ self.width_per_group = width_per_group
+ super().__init__(depth, **kwargs)
+
+ def make_res_layer(self, **kwargs):
+ return ResLayer(
+ groups=self.groups,
+ width_per_group=self.width_per_group,
+ base_channels=self.base_channels,
+ **kwargs)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/rsn.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/rsn.py
new file mode 100644
index 0000000000000000000000000000000000000000..29038afe2a77dcb3d3b027b1549d478916a50727
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/rsn.py
@@ -0,0 +1,616 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy as cp
+
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from mmcv.cnn import (ConvModule, MaxPool2d, constant_init, kaiming_init,
+ normal_init)
+
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+
+
+class RSB(nn.Module):
+ """Residual Steps block for RSN. Paper ref: Cai et al. "Learning Delicate
+ Local Representations for Multi-Person Pose Estimation" (ECCV 2020).
+
+ Args:
+ in_channels (int): Input channels of this block.
+ out_channels (int): Output channels of this block.
+ num_steps (int): Numbers of steps in RSB
+ stride (int): stride of the block. Default: 1
+ downsample (nn.Module): downsample operation on identity branch.
+ Default: None.
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ expand_times (int): Times by which the in_channels are expanded.
+ Default:26.
+ res_top_channels (int): Number of channels of feature output by
+ ResNet_top. Default:64.
+ """
+
+ expansion = 1
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ num_steps=4,
+ stride=1,
+ downsample=None,
+ with_cp=False,
+ norm_cfg=dict(type='BN'),
+ expand_times=26,
+ res_top_channels=64):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+ assert num_steps > 1
+ self.in_channels = in_channels
+ self.branch_channels = self.in_channels * expand_times
+ self.branch_channels //= res_top_channels
+ self.out_channels = out_channels
+ self.stride = stride
+ self.downsample = downsample
+ self.with_cp = with_cp
+ self.norm_cfg = norm_cfg
+ self.num_steps = num_steps
+ self.conv_bn_relu1 = ConvModule(
+ self.in_channels,
+ self.num_steps * self.branch_channels,
+ kernel_size=1,
+ stride=self.stride,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ inplace=False)
+ for i in range(self.num_steps):
+ for j in range(i + 1):
+ module_name = f'conv_bn_relu2_{i + 1}_{j + 1}'
+ self.add_module(
+ module_name,
+ ConvModule(
+ self.branch_channels,
+ self.branch_channels,
+ kernel_size=3,
+ stride=1,
+ padding=1,
+ norm_cfg=self.norm_cfg,
+ inplace=False))
+ self.conv_bn3 = ConvModule(
+ self.num_steps * self.branch_channels,
+ self.out_channels * self.expansion,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ act_cfg=None,
+ norm_cfg=self.norm_cfg,
+ inplace=False)
+ self.relu = nn.ReLU(inplace=False)
+
+ def forward(self, x):
+ """Forward function."""
+
+ identity = x
+ x = self.conv_bn_relu1(x)
+ spx = torch.split(x, self.branch_channels, 1)
+ outputs = list()
+ outs = list()
+ for i in range(self.num_steps):
+ outputs_i = list()
+ outputs.append(outputs_i)
+ for j in range(i + 1):
+ if j == 0:
+ inputs = spx[i]
+ else:
+ inputs = outputs[i][j - 1]
+ if i > j:
+ inputs = inputs + outputs[i - 1][j]
+ module_name = f'conv_bn_relu2_{i + 1}_{j + 1}'
+ module_i_j = getattr(self, module_name)
+ outputs[i].append(module_i_j(inputs))
+
+ outs.append(outputs[i][i])
+ out = torch.cat(tuple(outs), 1)
+ out = self.conv_bn3(out)
+
+ if self.downsample is not None:
+ identity = self.downsample(identity)
+ out = out + identity
+
+ out = self.relu(out)
+
+ return out
+
+
+class Downsample_module(nn.Module):
+ """Downsample module for RSN.
+
+ Args:
+ block (nn.Module): Downsample block.
+ num_blocks (list): Number of blocks in each downsample unit.
+ num_units (int): Numbers of downsample units. Default: 4
+ has_skip (bool): Have skip connections from prior upsample
+ module or not. Default:False
+ num_steps (int): Number of steps in a block. Default:4
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ in_channels (int): Number of channels of the input feature to
+ downsample module. Default: 64
+ expand_times (int): Times by which the in_channels are expanded.
+ Default:26.
+ """
+
+ def __init__(self,
+ block,
+ num_blocks,
+ num_steps=4,
+ num_units=4,
+ has_skip=False,
+ norm_cfg=dict(type='BN'),
+ in_channels=64,
+ expand_times=26):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+ self.has_skip = has_skip
+ self.in_channels = in_channels
+ assert len(num_blocks) == num_units
+ self.num_blocks = num_blocks
+ self.num_units = num_units
+ self.num_steps = num_steps
+ self.norm_cfg = norm_cfg
+ self.layer1 = self._make_layer(
+ block,
+ in_channels,
+ num_blocks[0],
+ expand_times=expand_times,
+ res_top_channels=in_channels)
+ for i in range(1, num_units):
+ module_name = f'layer{i + 1}'
+ self.add_module(
+ module_name,
+ self._make_layer(
+ block,
+ in_channels * pow(2, i),
+ num_blocks[i],
+ stride=2,
+ expand_times=expand_times,
+ res_top_channels=in_channels))
+
+ def _make_layer(self,
+ block,
+ out_channels,
+ blocks,
+ stride=1,
+ expand_times=26,
+ res_top_channels=64):
+ downsample = None
+ if stride != 1 or self.in_channels != out_channels * block.expansion:
+ downsample = ConvModule(
+ self.in_channels,
+ out_channels * block.expansion,
+ kernel_size=1,
+ stride=stride,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ act_cfg=None,
+ inplace=True)
+
+ units = list()
+ units.append(
+ block(
+ self.in_channels,
+ out_channels,
+ num_steps=self.num_steps,
+ stride=stride,
+ downsample=downsample,
+ norm_cfg=self.norm_cfg,
+ expand_times=expand_times,
+ res_top_channels=res_top_channels))
+ self.in_channels = out_channels * block.expansion
+ for _ in range(1, blocks):
+ units.append(
+ block(
+ self.in_channels,
+ out_channels,
+ num_steps=self.num_steps,
+ expand_times=expand_times,
+ res_top_channels=res_top_channels))
+
+ return nn.Sequential(*units)
+
+ def forward(self, x, skip1, skip2):
+ out = list()
+ for i in range(self.num_units):
+ module_name = f'layer{i + 1}'
+ module_i = getattr(self, module_name)
+ x = module_i(x)
+ if self.has_skip:
+ x = x + skip1[i] + skip2[i]
+ out.append(x)
+ out.reverse()
+
+ return tuple(out)
+
+
+class Upsample_unit(nn.Module):
+ """Upsample unit for upsample module.
+
+ Args:
+ ind (int): Indicates whether to interpolate (>0) and whether to
+ generate feature map for the next hourglass-like module.
+ num_units (int): Number of units that form a upsample module. Along
+ with ind and gen_cross_conv, nm_units is used to decide whether
+ to generate feature map for the next hourglass-like module.
+ in_channels (int): Channel number of the skip-in feature maps from
+ the corresponding downsample unit.
+ unit_channels (int): Channel number in this unit. Default:256.
+ gen_skip: (bool): Whether or not to generate skips for the posterior
+ downsample module. Default:False
+ gen_cross_conv (bool): Whether to generate feature map for the next
+ hourglass-like module. Default:False
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ out_channels (in): Number of channels of feature output by upsample
+ module. Must equal to in_channels of downsample module. Default:64
+ """
+
+ def __init__(self,
+ ind,
+ num_units,
+ in_channels,
+ unit_channels=256,
+ gen_skip=False,
+ gen_cross_conv=False,
+ norm_cfg=dict(type='BN'),
+ out_channels=64):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+ self.num_units = num_units
+ self.norm_cfg = norm_cfg
+ self.in_skip = ConvModule(
+ in_channels,
+ unit_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ act_cfg=None,
+ inplace=True)
+ self.relu = nn.ReLU(inplace=True)
+
+ self.ind = ind
+ if self.ind > 0:
+ self.up_conv = ConvModule(
+ unit_channels,
+ unit_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ act_cfg=None,
+ inplace=True)
+
+ self.gen_skip = gen_skip
+ if self.gen_skip:
+ self.out_skip1 = ConvModule(
+ in_channels,
+ in_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ inplace=True)
+
+ self.out_skip2 = ConvModule(
+ unit_channels,
+ in_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ inplace=True)
+
+ self.gen_cross_conv = gen_cross_conv
+ if self.ind == num_units - 1 and self.gen_cross_conv:
+ self.cross_conv = ConvModule(
+ unit_channels,
+ out_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=self.norm_cfg,
+ inplace=True)
+
+ def forward(self, x, up_x):
+ out = self.in_skip(x)
+
+ if self.ind > 0:
+ up_x = F.interpolate(
+ up_x,
+ size=(x.size(2), x.size(3)),
+ mode='bilinear',
+ align_corners=True)
+ up_x = self.up_conv(up_x)
+ out = out + up_x
+ out = self.relu(out)
+
+ skip1 = None
+ skip2 = None
+ if self.gen_skip:
+ skip1 = self.out_skip1(x)
+ skip2 = self.out_skip2(out)
+
+ cross_conv = None
+ if self.ind == self.num_units - 1 and self.gen_cross_conv:
+ cross_conv = self.cross_conv(out)
+
+ return out, skip1, skip2, cross_conv
+
+
+class Upsample_module(nn.Module):
+ """Upsample module for RSN.
+
+ Args:
+ unit_channels (int): Channel number in the upsample units.
+ Default:256.
+ num_units (int): Numbers of upsample units. Default: 4
+ gen_skip (bool): Whether to generate skip for posterior downsample
+ module or not. Default:False
+ gen_cross_conv (bool): Whether to generate feature map for the next
+ hourglass-like module. Default:False
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ out_channels (int): Number of channels of feature output by upsample
+ module. Must equal to in_channels of downsample module. Default:64
+ """
+
+ def __init__(self,
+ unit_channels=256,
+ num_units=4,
+ gen_skip=False,
+ gen_cross_conv=False,
+ norm_cfg=dict(type='BN'),
+ out_channels=64):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+ self.in_channels = list()
+ for i in range(num_units):
+ self.in_channels.append(RSB.expansion * out_channels * pow(2, i))
+ self.in_channels.reverse()
+ self.num_units = num_units
+ self.gen_skip = gen_skip
+ self.gen_cross_conv = gen_cross_conv
+ self.norm_cfg = norm_cfg
+ for i in range(num_units):
+ module_name = f'up{i + 1}'
+ self.add_module(
+ module_name,
+ Upsample_unit(
+ i,
+ self.num_units,
+ self.in_channels[i],
+ unit_channels,
+ self.gen_skip,
+ self.gen_cross_conv,
+ norm_cfg=self.norm_cfg,
+ out_channels=64))
+
+ def forward(self, x):
+ out = list()
+ skip1 = list()
+ skip2 = list()
+ cross_conv = None
+ for i in range(self.num_units):
+ module_i = getattr(self, f'up{i + 1}')
+ if i == 0:
+ outi, skip1_i, skip2_i, _ = module_i(x[i], None)
+ elif i == self.num_units - 1:
+ outi, skip1_i, skip2_i, cross_conv = module_i(x[i], out[i - 1])
+ else:
+ outi, skip1_i, skip2_i, _ = module_i(x[i], out[i - 1])
+ out.append(outi)
+ skip1.append(skip1_i)
+ skip2.append(skip2_i)
+ skip1.reverse()
+ skip2.reverse()
+
+ return out, skip1, skip2, cross_conv
+
+
+class Single_stage_RSN(nn.Module):
+ """Single_stage Residual Steps Network.
+
+ Args:
+ unit_channels (int): Channel number in the upsample units. Default:256.
+ num_units (int): Numbers of downsample/upsample units. Default: 4
+ gen_skip (bool): Whether to generate skip for posterior downsample
+ module or not. Default:False
+ gen_cross_conv (bool): Whether to generate feature map for the next
+ hourglass-like module. Default:False
+ has_skip (bool): Have skip connections from prior upsample
+ module or not. Default:False
+ num_steps (int): Number of steps in RSB. Default: 4
+ num_blocks (list): Number of blocks in each downsample unit.
+ Default: [2, 2, 2, 2] Note: Make sure num_units==len(num_blocks)
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ in_channels (int): Number of channels of the feature from ResNet_Top.
+ Default: 64.
+ expand_times (int): Times by which the in_channels are expanded in RSB.
+ Default:26.
+ """
+
+ def __init__(self,
+ has_skip=False,
+ gen_skip=False,
+ gen_cross_conv=False,
+ unit_channels=256,
+ num_units=4,
+ num_steps=4,
+ num_blocks=[2, 2, 2, 2],
+ norm_cfg=dict(type='BN'),
+ in_channels=64,
+ expand_times=26):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ num_blocks = cp.deepcopy(num_blocks)
+ super().__init__()
+ assert len(num_blocks) == num_units
+ self.has_skip = has_skip
+ self.gen_skip = gen_skip
+ self.gen_cross_conv = gen_cross_conv
+ self.num_units = num_units
+ self.num_steps = num_steps
+ self.unit_channels = unit_channels
+ self.num_blocks = num_blocks
+ self.norm_cfg = norm_cfg
+
+ self.downsample = Downsample_module(RSB, num_blocks, num_steps,
+ num_units, has_skip, norm_cfg,
+ in_channels, expand_times)
+ self.upsample = Upsample_module(unit_channels, num_units, gen_skip,
+ gen_cross_conv, norm_cfg, in_channels)
+
+ def forward(self, x, skip1, skip2):
+ mid = self.downsample(x, skip1, skip2)
+ out, skip1, skip2, cross_conv = self.upsample(mid)
+
+ return out, skip1, skip2, cross_conv
+
+
+class ResNet_top(nn.Module):
+ """ResNet top for RSN.
+
+ Args:
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ channels (int): Number of channels of the feature output by ResNet_top.
+ """
+
+ def __init__(self, norm_cfg=dict(type='BN'), channels=64):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+ self.top = nn.Sequential(
+ ConvModule(
+ 3,
+ channels,
+ kernel_size=7,
+ stride=2,
+ padding=3,
+ norm_cfg=norm_cfg,
+ inplace=True), MaxPool2d(kernel_size=3, stride=2, padding=1))
+
+ def forward(self, img):
+ return self.top(img)
+
+
+@BACKBONES.register_module()
+class RSN(BaseBackbone):
+ """Residual Steps Network backbone. Paper ref: Cai et al. "Learning
+ Delicate Local Representations for Multi-Person Pose Estimation" (ECCV
+ 2020).
+
+ Args:
+ unit_channels (int): Number of Channels in an upsample unit.
+ Default: 256
+ num_stages (int): Number of stages in a multi-stage RSN. Default: 4
+ num_units (int): NUmber of downsample/upsample units in a single-stage
+ RSN. Default: 4 Note: Make sure num_units == len(self.num_blocks)
+ num_blocks (list): Number of RSBs (Residual Steps Block) in each
+ downsample unit. Default: [2, 2, 2, 2]
+ num_steps (int): Number of steps in a RSB. Default:4
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ res_top_channels (int): Number of channels of feature from ResNet_top.
+ Default: 64.
+ expand_times (int): Times by which the in_channels are expanded in RSB.
+ Default:26.
+ Example:
+ >>> from mmpose.models import RSN
+ >>> import torch
+ >>> self = RSN(num_stages=2,num_units=2,num_blocks=[2,2])
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 3, 511, 511)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_output in level_outputs:
+ ... for feature in level_output:
+ ... print(tuple(feature.shape))
+ ...
+ (1, 256, 64, 64)
+ (1, 256, 128, 128)
+ (1, 256, 64, 64)
+ (1, 256, 128, 128)
+ """
+
+ def __init__(self,
+ unit_channels=256,
+ num_stages=4,
+ num_units=4,
+ num_blocks=[2, 2, 2, 2],
+ num_steps=4,
+ norm_cfg=dict(type='BN'),
+ res_top_channels=64,
+ expand_times=26):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ num_blocks = cp.deepcopy(num_blocks)
+ super().__init__()
+ self.unit_channels = unit_channels
+ self.num_stages = num_stages
+ self.num_units = num_units
+ self.num_blocks = num_blocks
+ self.num_steps = num_steps
+ self.norm_cfg = norm_cfg
+
+ assert self.num_stages > 0
+ assert self.num_steps > 1
+ assert self.num_units > 1
+ assert self.num_units == len(self.num_blocks)
+ self.top = ResNet_top(norm_cfg=norm_cfg)
+ self.multi_stage_rsn = nn.ModuleList([])
+ for i in range(self.num_stages):
+ if i == 0:
+ has_skip = False
+ else:
+ has_skip = True
+ if i != self.num_stages - 1:
+ gen_skip = True
+ gen_cross_conv = True
+ else:
+ gen_skip = False
+ gen_cross_conv = False
+ self.multi_stage_rsn.append(
+ Single_stage_RSN(has_skip, gen_skip, gen_cross_conv,
+ unit_channels, num_units, num_steps,
+ num_blocks, norm_cfg, res_top_channels,
+ expand_times))
+
+ def forward(self, x):
+ """Model forward function."""
+ out_feats = []
+ skip1 = None
+ skip2 = None
+ x = self.top(x)
+ for i in range(self.num_stages):
+ out, skip1, skip2, x = self.multi_stage_rsn[i](x, skip1, skip2)
+ out_feats.append(out)
+
+ return out_feats
+
+ def init_weights(self, pretrained=None):
+ """Initialize model weights."""
+ for m in self.multi_stage_rsn.modules():
+ if isinstance(m, nn.Conv2d):
+ kaiming_init(m)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
+ elif isinstance(m, nn.Linear):
+ normal_init(m, std=0.01)
+
+ for m in self.top.modules():
+ if isinstance(m, nn.Conv2d):
+ kaiming_init(m)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/scnet.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/scnet.py
new file mode 100644
index 0000000000000000000000000000000000000000..3786c5731d685638cfa64a83e5d4a5e2eee545de
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/scnet.py
@@ -0,0 +1,248 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import torch.utils.checkpoint as cp
+from mmcv.cnn import build_conv_layer, build_norm_layer
+
+from ..builder import BACKBONES
+from .resnet import Bottleneck, ResNet
+
+
+class SCConv(nn.Module):
+ """SCConv (Self-calibrated Convolution)
+
+ Args:
+ in_channels (int): The input channels of the SCConv.
+ out_channels (int): The output channel of the SCConv.
+ stride (int): stride of SCConv.
+ pooling_r (int): size of pooling for scconv.
+ conv_cfg (dict): dictionary to construct and config conv layer.
+ Default: None
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ stride,
+ pooling_r,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN', momentum=0.1)):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+
+ assert in_channels == out_channels
+
+ self.k2 = nn.Sequential(
+ nn.AvgPool2d(kernel_size=pooling_r, stride=pooling_r),
+ build_conv_layer(
+ conv_cfg,
+ in_channels,
+ in_channels,
+ kernel_size=3,
+ stride=1,
+ padding=1,
+ bias=False),
+ build_norm_layer(norm_cfg, in_channels)[1],
+ )
+ self.k3 = nn.Sequential(
+ build_conv_layer(
+ conv_cfg,
+ in_channels,
+ in_channels,
+ kernel_size=3,
+ stride=1,
+ padding=1,
+ bias=False),
+ build_norm_layer(norm_cfg, in_channels)[1],
+ )
+ self.k4 = nn.Sequential(
+ build_conv_layer(
+ conv_cfg,
+ in_channels,
+ in_channels,
+ kernel_size=3,
+ stride=stride,
+ padding=1,
+ bias=False),
+ build_norm_layer(norm_cfg, out_channels)[1],
+ nn.ReLU(inplace=True),
+ )
+
+ def forward(self, x):
+ """Forward function."""
+ identity = x
+
+ out = torch.sigmoid(
+ torch.add(identity, F.interpolate(self.k2(x),
+ identity.size()[2:])))
+ out = torch.mul(self.k3(x), out)
+ out = self.k4(out)
+
+ return out
+
+
+class SCBottleneck(Bottleneck):
+ """SC(Self-calibrated) Bottleneck.
+
+ Args:
+ in_channels (int): The input channels of the SCBottleneck block.
+ out_channels (int): The output channel of the SCBottleneck block.
+ """
+
+ pooling_r = 4
+
+ def __init__(self, in_channels, out_channels, **kwargs):
+ super().__init__(in_channels, out_channels, **kwargs)
+ self.mid_channels = out_channels // self.expansion // 2
+
+ self.norm1_name, norm1 = build_norm_layer(
+ self.norm_cfg, self.mid_channels, postfix=1)
+ self.norm2_name, norm2 = build_norm_layer(
+ self.norm_cfg, self.mid_channels, postfix=2)
+ self.norm3_name, norm3 = build_norm_layer(
+ self.norm_cfg, out_channels, postfix=3)
+
+ self.conv1 = build_conv_layer(
+ self.conv_cfg,
+ in_channels,
+ self.mid_channels,
+ kernel_size=1,
+ stride=1,
+ bias=False)
+ self.add_module(self.norm1_name, norm1)
+
+ self.k1 = nn.Sequential(
+ build_conv_layer(
+ self.conv_cfg,
+ self.mid_channels,
+ self.mid_channels,
+ kernel_size=3,
+ stride=self.stride,
+ padding=1,
+ bias=False),
+ build_norm_layer(self.norm_cfg, self.mid_channels)[1],
+ nn.ReLU(inplace=True))
+
+ self.conv2 = build_conv_layer(
+ self.conv_cfg,
+ in_channels,
+ self.mid_channels,
+ kernel_size=1,
+ stride=1,
+ bias=False)
+ self.add_module(self.norm2_name, norm2)
+
+ self.scconv = SCConv(self.mid_channels, self.mid_channels, self.stride,
+ self.pooling_r, self.conv_cfg, self.norm_cfg)
+
+ self.conv3 = build_conv_layer(
+ self.conv_cfg,
+ self.mid_channels * 2,
+ out_channels,
+ kernel_size=1,
+ stride=1,
+ bias=False)
+ self.add_module(self.norm3_name, norm3)
+
+ def forward(self, x):
+ """Forward function."""
+
+ def _inner_forward(x):
+ identity = x
+
+ out_a = self.conv1(x)
+ out_a = self.norm1(out_a)
+ out_a = self.relu(out_a)
+
+ out_a = self.k1(out_a)
+
+ out_b = self.conv2(x)
+ out_b = self.norm2(out_b)
+ out_b = self.relu(out_b)
+
+ out_b = self.scconv(out_b)
+
+ out = self.conv3(torch.cat([out_a, out_b], dim=1))
+ out = self.norm3(out)
+
+ if self.downsample is not None:
+ identity = self.downsample(x)
+
+ out += identity
+
+ return out
+
+ if self.with_cp and x.requires_grad:
+ out = cp.checkpoint(_inner_forward, x)
+ else:
+ out = _inner_forward(x)
+
+ out = self.relu(out)
+
+ return out
+
+
+@BACKBONES.register_module()
+class SCNet(ResNet):
+ """SCNet backbone.
+
+ Improving Convolutional Networks with Self-Calibrated Convolutions,
+ Jiang-Jiang Liu, Qibin Hou, Ming-Ming Cheng, Changhu Wang, Jiashi Feng,
+ IEEE CVPR, 2020.
+ http://mftp.mmcheng.net/Papers/20cvprSCNet.pdf
+
+ Args:
+ depth (int): Depth of scnet, from {50, 101}.
+ in_channels (int): Number of input image channels. Normally 3.
+ base_channels (int): Number of base channels of hidden layer.
+ num_stages (int): SCNet stages, normally 4.
+ strides (Sequence[int]): Strides of the first block of each stage.
+ dilations (Sequence[int]): Dilation of each stage.
+ out_indices (Sequence[int]): Output from which stages.
+ style (str): `pytorch` or `caffe`. If set to "pytorch", the stride-two
+ layer is the 3x3 conv layer, otherwise the stride-two layer is
+ the first 1x1 conv layer.
+ deep_stem (bool): Replace 7x7 conv in input stem with 3 3x3 conv
+ avg_down (bool): Use AvgPool instead of stride conv when
+ downsampling in the bottleneck.
+ frozen_stages (int): Stages to be frozen (stop grad and set eval mode).
+ -1 means not freezing any parameters.
+ norm_cfg (dict): Dictionary to construct and config norm layer.
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed.
+ zero_init_residual (bool): Whether to use zero init for last norm layer
+ in resblocks to let them behave as identity.
+
+ Example:
+ >>> from mmpose.models import SCNet
+ >>> import torch
+ >>> self = SCNet(depth=50, out_indices=(0, 1, 2, 3))
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 3, 224, 224)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_out in level_outputs:
+ ... print(tuple(level_out.shape))
+ (1, 256, 56, 56)
+ (1, 512, 28, 28)
+ (1, 1024, 14, 14)
+ (1, 2048, 7, 7)
+ """
+
+ arch_settings = {
+ 50: (SCBottleneck, [3, 4, 6, 3]),
+ 101: (SCBottleneck, [3, 4, 23, 3])
+ }
+
+ def __init__(self, depth, **kwargs):
+ if depth not in self.arch_settings:
+ raise KeyError(f'invalid depth {depth} for SCNet')
+ super().__init__(depth, **kwargs)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/seresnet.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/seresnet.py
new file mode 100644
index 0000000000000000000000000000000000000000..ac2d53b40a4593bce96d5c7c3bb4e06d38353d0b
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/seresnet.py
@@ -0,0 +1,125 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import torch.utils.checkpoint as cp
+
+from ..builder import BACKBONES
+from .resnet import Bottleneck, ResLayer, ResNet
+from .utils.se_layer import SELayer
+
+
+class SEBottleneck(Bottleneck):
+ """SEBottleneck block for SEResNet.
+
+ Args:
+ in_channels (int): The input channels of the SEBottleneck block.
+ out_channels (int): The output channel of the SEBottleneck block.
+ se_ratio (int): Squeeze ratio in SELayer. Default: 16
+ """
+
+ def __init__(self, in_channels, out_channels, se_ratio=16, **kwargs):
+ super().__init__(in_channels, out_channels, **kwargs)
+ self.se_layer = SELayer(out_channels, ratio=se_ratio)
+
+ def forward(self, x):
+
+ def _inner_forward(x):
+ identity = x
+
+ out = self.conv1(x)
+ out = self.norm1(out)
+ out = self.relu(out)
+
+ out = self.conv2(out)
+ out = self.norm2(out)
+ out = self.relu(out)
+
+ out = self.conv3(out)
+ out = self.norm3(out)
+
+ out = self.se_layer(out)
+
+ if self.downsample is not None:
+ identity = self.downsample(x)
+
+ out += identity
+
+ return out
+
+ if self.with_cp and x.requires_grad:
+ out = cp.checkpoint(_inner_forward, x)
+ else:
+ out = _inner_forward(x)
+
+ out = self.relu(out)
+
+ return out
+
+
+@BACKBONES.register_module()
+class SEResNet(ResNet):
+ """SEResNet backbone.
+
+ Please refer to the `paper `__ for
+ details.
+
+ Args:
+ depth (int): Network depth, from {50, 101, 152}.
+ se_ratio (int): Squeeze ratio in SELayer. Default: 16.
+ in_channels (int): Number of input image channels. Default: 3.
+ stem_channels (int): Output channels of the stem layer. Default: 64.
+ num_stages (int): Stages of the network. Default: 4.
+ strides (Sequence[int]): Strides of the first block of each stage.
+ Default: ``(1, 2, 2, 2)``.
+ dilations (Sequence[int]): Dilation of each stage.
+ Default: ``(1, 1, 1, 1)``.
+ out_indices (Sequence[int]): Output from which stages. If only one
+ stage is specified, a single tensor (feature map) is returned,
+ otherwise multiple stages are specified, a tuple of tensors will
+ be returned. Default: ``(3, )``.
+ style (str): `pytorch` or `caffe`. If set to "pytorch", the stride-two
+ layer is the 3x3 conv layer, otherwise the stride-two layer is
+ the first 1x1 conv layer.
+ deep_stem (bool): Replace 7x7 conv in input stem with 3 3x3 conv.
+ Default: False.
+ avg_down (bool): Use AvgPool instead of stride conv when
+ downsampling in the bottleneck. Default: False.
+ frozen_stages (int): Stages to be frozen (stop grad and set eval mode).
+ -1 means not freezing any parameters. Default: -1.
+ conv_cfg (dict | None): The config dict for conv layers. Default: None.
+ norm_cfg (dict): The config dict for norm layers.
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ zero_init_residual (bool): Whether to use zero init for last norm layer
+ in resblocks to let them behave as identity. Default: True.
+
+ Example:
+ >>> from mmpose.models import SEResNet
+ >>> import torch
+ >>> self = SEResNet(depth=50, out_indices=(0, 1, 2, 3))
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 3, 224, 224)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_out in level_outputs:
+ ... print(tuple(level_out.shape))
+ (1, 256, 56, 56)
+ (1, 512, 28, 28)
+ (1, 1024, 14, 14)
+ (1, 2048, 7, 7)
+ """
+
+ arch_settings = {
+ 50: (SEBottleneck, (3, 4, 6, 3)),
+ 101: (SEBottleneck, (3, 4, 23, 3)),
+ 152: (SEBottleneck, (3, 8, 36, 3))
+ }
+
+ def __init__(self, depth, se_ratio=16, **kwargs):
+ if depth not in self.arch_settings:
+ raise KeyError(f'invalid depth {depth} for SEResNet')
+ self.se_ratio = se_ratio
+ super().__init__(depth, **kwargs)
+
+ def make_res_layer(self, **kwargs):
+ return ResLayer(se_ratio=self.se_ratio, **kwargs)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/seresnext.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/seresnext.py
new file mode 100644
index 0000000000000000000000000000000000000000..c5c4e4ce03684f8a9bd0c6166969c01bace54bd2
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/seresnext.py
@@ -0,0 +1,168 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+from mmcv.cnn import build_conv_layer, build_norm_layer
+
+from ..builder import BACKBONES
+from .resnet import ResLayer
+from .seresnet import SEBottleneck as _SEBottleneck
+from .seresnet import SEResNet
+
+
+class SEBottleneck(_SEBottleneck):
+ """SEBottleneck block for SEResNeXt.
+
+ Args:
+ in_channels (int): Input channels of this block.
+ out_channels (int): Output channels of this block.
+ base_channels (int): Middle channels of the first stage. Default: 64.
+ groups (int): Groups of conv2.
+ width_per_group (int): Width per group of conv2. 64x4d indicates
+ ``groups=64, width_per_group=4`` and 32x8d indicates
+ ``groups=32, width_per_group=8``.
+ stride (int): stride of the block. Default: 1
+ dilation (int): dilation of convolution. Default: 1
+ downsample (nn.Module): downsample operation on identity branch.
+ Default: None
+ se_ratio (int): Squeeze ratio in SELayer. Default: 16
+ style (str): `pytorch` or `caffe`. If set to "pytorch", the stride-two
+ layer is the 3x3 conv layer, otherwise the stride-two layer is
+ the first 1x1 conv layer.
+ conv_cfg (dict): dictionary to construct and config conv layer.
+ Default: None
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed.
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ base_channels=64,
+ groups=32,
+ width_per_group=4,
+ se_ratio=16,
+ **kwargs):
+ super().__init__(in_channels, out_channels, se_ratio, **kwargs)
+ self.groups = groups
+ self.width_per_group = width_per_group
+
+ # We follow the same rational of ResNext to compute mid_channels.
+ # For SEResNet bottleneck, middle channels are determined by expansion
+ # and out_channels, but for SEResNeXt bottleneck, it is determined by
+ # groups and width_per_group and the stage it is located in.
+ if groups != 1:
+ assert self.mid_channels % base_channels == 0
+ self.mid_channels = (
+ groups * width_per_group * self.mid_channels // base_channels)
+
+ self.norm1_name, norm1 = build_norm_layer(
+ self.norm_cfg, self.mid_channels, postfix=1)
+ self.norm2_name, norm2 = build_norm_layer(
+ self.norm_cfg, self.mid_channels, postfix=2)
+ self.norm3_name, norm3 = build_norm_layer(
+ self.norm_cfg, self.out_channels, postfix=3)
+
+ self.conv1 = build_conv_layer(
+ self.conv_cfg,
+ self.in_channels,
+ self.mid_channels,
+ kernel_size=1,
+ stride=self.conv1_stride,
+ bias=False)
+ self.add_module(self.norm1_name, norm1)
+ self.conv2 = build_conv_layer(
+ self.conv_cfg,
+ self.mid_channels,
+ self.mid_channels,
+ kernel_size=3,
+ stride=self.conv2_stride,
+ padding=self.dilation,
+ dilation=self.dilation,
+ groups=groups,
+ bias=False)
+
+ self.add_module(self.norm2_name, norm2)
+ self.conv3 = build_conv_layer(
+ self.conv_cfg,
+ self.mid_channels,
+ self.out_channels,
+ kernel_size=1,
+ bias=False)
+ self.add_module(self.norm3_name, norm3)
+
+
+@BACKBONES.register_module()
+class SEResNeXt(SEResNet):
+ """SEResNeXt backbone.
+
+ Please refer to the `paper `__ for
+ details.
+
+ Args:
+ depth (int): Network depth, from {50, 101, 152}.
+ groups (int): Groups of conv2 in Bottleneck. Default: 32.
+ width_per_group (int): Width per group of conv2 in Bottleneck.
+ Default: 4.
+ se_ratio (int): Squeeze ratio in SELayer. Default: 16.
+ in_channels (int): Number of input image channels. Default: 3.
+ stem_channels (int): Output channels of the stem layer. Default: 64.
+ num_stages (int): Stages of the network. Default: 4.
+ strides (Sequence[int]): Strides of the first block of each stage.
+ Default: ``(1, 2, 2, 2)``.
+ dilations (Sequence[int]): Dilation of each stage.
+ Default: ``(1, 1, 1, 1)``.
+ out_indices (Sequence[int]): Output from which stages. If only one
+ stage is specified, a single tensor (feature map) is returned,
+ otherwise multiple stages are specified, a tuple of tensors will
+ be returned. Default: ``(3, )``.
+ style (str): `pytorch` or `caffe`. If set to "pytorch", the stride-two
+ layer is the 3x3 conv layer, otherwise the stride-two layer is
+ the first 1x1 conv layer.
+ deep_stem (bool): Replace 7x7 conv in input stem with 3 3x3 conv.
+ Default: False.
+ avg_down (bool): Use AvgPool instead of stride conv when
+ downsampling in the bottleneck. Default: False.
+ frozen_stages (int): Stages to be frozen (stop grad and set eval mode).
+ -1 means not freezing any parameters. Default: -1.
+ conv_cfg (dict | None): The config dict for conv layers. Default: None.
+ norm_cfg (dict): The config dict for norm layers.
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ zero_init_residual (bool): Whether to use zero init for last norm layer
+ in resblocks to let them behave as identity. Default: True.
+
+ Example:
+ >>> from mmpose.models import SEResNeXt
+ >>> import torch
+ >>> self = SEResNet(depth=50, out_indices=(0, 1, 2, 3))
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 3, 224, 224)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_out in level_outputs:
+ ... print(tuple(level_out.shape))
+ (1, 256, 56, 56)
+ (1, 512, 28, 28)
+ (1, 1024, 14, 14)
+ (1, 2048, 7, 7)
+ """
+
+ arch_settings = {
+ 50: (SEBottleneck, (3, 4, 6, 3)),
+ 101: (SEBottleneck, (3, 4, 23, 3)),
+ 152: (SEBottleneck, (3, 8, 36, 3))
+ }
+
+ def __init__(self, depth, groups=32, width_per_group=4, **kwargs):
+ self.groups = groups
+ self.width_per_group = width_per_group
+ super().__init__(depth, **kwargs)
+
+ def make_res_layer(self, **kwargs):
+ return ResLayer(
+ groups=self.groups,
+ width_per_group=self.width_per_group,
+ base_channels=self.base_channels,
+ **kwargs)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/shufflenet_v1.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/shufflenet_v1.py
new file mode 100644
index 0000000000000000000000000000000000000000..9f98cbd2132250ec13adcce6e642c966b0dbd7cc
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/shufflenet_v1.py
@@ -0,0 +1,329 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+import logging
+
+import torch
+import torch.nn as nn
+import torch.utils.checkpoint as cp
+from mmcv.cnn import (ConvModule, build_activation_layer, constant_init,
+ normal_init)
+from torch.nn.modules.batchnorm import _BatchNorm
+
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+from .utils import channel_shuffle, load_checkpoint, make_divisible
+
+
+class ShuffleUnit(nn.Module):
+ """ShuffleUnit block.
+
+ ShuffleNet unit with pointwise group convolution (GConv) and channel
+ shuffle.
+
+ Args:
+ in_channels (int): The input channels of the ShuffleUnit.
+ out_channels (int): The output channels of the ShuffleUnit.
+ groups (int, optional): The number of groups to be used in grouped 1x1
+ convolutions in each ShuffleUnit. Default: 3
+ first_block (bool, optional): Whether it is the first ShuffleUnit of a
+ sequential ShuffleUnits. Default: True, which means not using the
+ grouped 1x1 convolution.
+ combine (str, optional): The ways to combine the input and output
+ branches. Default: 'add'.
+ conv_cfg (dict): Config dict for convolution layer. Default: None,
+ which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ act_cfg (dict): Config dict for activation layer.
+ Default: dict(type='ReLU').
+ with_cp (bool, optional): Use checkpoint or not. Using checkpoint
+ will save some memory while slowing down the training speed.
+ Default: False.
+
+ Returns:
+ Tensor: The output tensor.
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ groups=3,
+ first_block=True,
+ combine='add',
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ act_cfg=dict(type='ReLU'),
+ with_cp=False):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ act_cfg = copy.deepcopy(act_cfg)
+ super().__init__()
+ self.in_channels = in_channels
+ self.out_channels = out_channels
+ self.first_block = first_block
+ self.combine = combine
+ self.groups = groups
+ self.bottleneck_channels = self.out_channels // 4
+ self.with_cp = with_cp
+
+ if self.combine == 'add':
+ self.depthwise_stride = 1
+ self._combine_func = self._add
+ assert in_channels == out_channels, (
+ 'in_channels must be equal to out_channels when combine '
+ 'is add')
+ elif self.combine == 'concat':
+ self.depthwise_stride = 2
+ self._combine_func = self._concat
+ self.out_channels -= self.in_channels
+ self.avgpool = nn.AvgPool2d(kernel_size=3, stride=2, padding=1)
+ else:
+ raise ValueError(f'Cannot combine tensors with {self.combine}. '
+ 'Only "add" and "concat" are supported')
+
+ self.first_1x1_groups = 1 if first_block else self.groups
+ self.g_conv_1x1_compress = ConvModule(
+ in_channels=self.in_channels,
+ out_channels=self.bottleneck_channels,
+ kernel_size=1,
+ groups=self.first_1x1_groups,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg)
+
+ self.depthwise_conv3x3_bn = ConvModule(
+ in_channels=self.bottleneck_channels,
+ out_channels=self.bottleneck_channels,
+ kernel_size=3,
+ stride=self.depthwise_stride,
+ padding=1,
+ groups=self.bottleneck_channels,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=None)
+
+ self.g_conv_1x1_expand = ConvModule(
+ in_channels=self.bottleneck_channels,
+ out_channels=self.out_channels,
+ kernel_size=1,
+ groups=self.groups,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=None)
+
+ self.act = build_activation_layer(act_cfg)
+
+ @staticmethod
+ def _add(x, out):
+ # residual connection
+ return x + out
+
+ @staticmethod
+ def _concat(x, out):
+ # concatenate along channel axis
+ return torch.cat((x, out), 1)
+
+ def forward(self, x):
+
+ def _inner_forward(x):
+ residual = x
+
+ out = self.g_conv_1x1_compress(x)
+ out = self.depthwise_conv3x3_bn(out)
+
+ if self.groups > 1:
+ out = channel_shuffle(out, self.groups)
+
+ out = self.g_conv_1x1_expand(out)
+
+ if self.combine == 'concat':
+ residual = self.avgpool(residual)
+ out = self.act(out)
+ out = self._combine_func(residual, out)
+ else:
+ out = self._combine_func(residual, out)
+ out = self.act(out)
+ return out
+
+ if self.with_cp and x.requires_grad:
+ out = cp.checkpoint(_inner_forward, x)
+ else:
+ out = _inner_forward(x)
+
+ return out
+
+
+@BACKBONES.register_module()
+class ShuffleNetV1(BaseBackbone):
+ """ShuffleNetV1 backbone.
+
+ Args:
+ groups (int, optional): The number of groups to be used in grouped 1x1
+ convolutions in each ShuffleUnit. Default: 3.
+ widen_factor (float, optional): Width multiplier - adjusts the number
+ of channels in each layer by this amount. Default: 1.0.
+ out_indices (Sequence[int]): Output from which stages.
+ Default: (2, )
+ frozen_stages (int): Stages to be frozen (all param fixed).
+ Default: -1, which means not freezing any parameters.
+ conv_cfg (dict): Config dict for convolution layer. Default: None,
+ which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ act_cfg (dict): Config dict for activation layer.
+ Default: dict(type='ReLU').
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ """
+
+ def __init__(self,
+ groups=3,
+ widen_factor=1.0,
+ out_indices=(2, ),
+ frozen_stages=-1,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ act_cfg=dict(type='ReLU'),
+ norm_eval=False,
+ with_cp=False):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ act_cfg = copy.deepcopy(act_cfg)
+ super().__init__()
+ self.stage_blocks = [4, 8, 4]
+ self.groups = groups
+
+ for index in out_indices:
+ if index not in range(0, 3):
+ raise ValueError('the item in out_indices must in '
+ f'range(0, 3). But received {index}')
+
+ if frozen_stages not in range(-1, 3):
+ raise ValueError('frozen_stages must be in range(-1, 3). '
+ f'But received {frozen_stages}')
+ self.out_indices = out_indices
+ self.frozen_stages = frozen_stages
+ self.conv_cfg = conv_cfg
+ self.norm_cfg = norm_cfg
+ self.act_cfg = act_cfg
+ self.norm_eval = norm_eval
+ self.with_cp = with_cp
+
+ if groups == 1:
+ channels = (144, 288, 576)
+ elif groups == 2:
+ channels = (200, 400, 800)
+ elif groups == 3:
+ channels = (240, 480, 960)
+ elif groups == 4:
+ channels = (272, 544, 1088)
+ elif groups == 8:
+ channels = (384, 768, 1536)
+ else:
+ raise ValueError(f'{groups} groups is not supported for 1x1 '
+ 'Grouped Convolutions')
+
+ channels = [make_divisible(ch * widen_factor, 8) for ch in channels]
+
+ self.in_channels = int(24 * widen_factor)
+
+ self.conv1 = ConvModule(
+ in_channels=3,
+ out_channels=self.in_channels,
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg)
+ self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)
+
+ self.layers = nn.ModuleList()
+ for i, num_blocks in enumerate(self.stage_blocks):
+ first_block = (i == 0)
+ layer = self.make_layer(channels[i], num_blocks, first_block)
+ self.layers.append(layer)
+
+ def _freeze_stages(self):
+ if self.frozen_stages >= 0:
+ for param in self.conv1.parameters():
+ param.requires_grad = False
+ for i in range(self.frozen_stages):
+ layer = self.layers[i]
+ layer.eval()
+ for param in layer.parameters():
+ param.requires_grad = False
+
+ def init_weights(self, pretrained=None):
+ if isinstance(pretrained, str):
+ logger = logging.getLogger()
+ load_checkpoint(self, pretrained, strict=False, logger=logger)
+ elif pretrained is None:
+ for name, m in self.named_modules():
+ if isinstance(m, nn.Conv2d):
+ if 'conv1' in name:
+ normal_init(m, mean=0, std=0.01)
+ else:
+ normal_init(m, mean=0, std=1.0 / m.weight.shape[1])
+ elif isinstance(m, (_BatchNorm, nn.GroupNorm)):
+ constant_init(m, val=1, bias=0.0001)
+ if isinstance(m, _BatchNorm):
+ if m.running_mean is not None:
+ nn.init.constant_(m.running_mean, 0)
+ else:
+ raise TypeError('pretrained must be a str or None. But received '
+ f'{type(pretrained)}')
+
+ def make_layer(self, out_channels, num_blocks, first_block=False):
+ """Stack ShuffleUnit blocks to make a layer.
+
+ Args:
+ out_channels (int): out_channels of the block.
+ num_blocks (int): Number of blocks.
+ first_block (bool, optional): Whether is the first ShuffleUnit of a
+ sequential ShuffleUnits. Default: False, which means using
+ the grouped 1x1 convolution.
+ """
+ layers = []
+ for i in range(num_blocks):
+ first_block = first_block if i == 0 else False
+ combine_mode = 'concat' if i == 0 else 'add'
+ layers.append(
+ ShuffleUnit(
+ self.in_channels,
+ out_channels,
+ groups=self.groups,
+ first_block=first_block,
+ combine=combine_mode,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ act_cfg=self.act_cfg,
+ with_cp=self.with_cp))
+ self.in_channels = out_channels
+
+ return nn.Sequential(*layers)
+
+ def forward(self, x):
+ x = self.conv1(x)
+ x = self.maxpool(x)
+
+ outs = []
+ for i, layer in enumerate(self.layers):
+ x = layer(x)
+ if i in self.out_indices:
+ outs.append(x)
+
+ if len(outs) == 1:
+ return outs[0]
+ return tuple(outs)
+
+ def train(self, mode=True):
+ super().train(mode)
+ self._freeze_stages()
+ if mode and self.norm_eval:
+ for m in self.modules():
+ if isinstance(m, _BatchNorm):
+ m.eval()
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/shufflenet_v2.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/shufflenet_v2.py
new file mode 100644
index 0000000000000000000000000000000000000000..e93533367afe4efa01fa67d14cafcca006c990e8
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/shufflenet_v2.py
@@ -0,0 +1,302 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+import logging
+
+import torch
+import torch.nn as nn
+import torch.utils.checkpoint as cp
+from mmcv.cnn import ConvModule, constant_init, normal_init
+from torch.nn.modules.batchnorm import _BatchNorm
+
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+from .utils import channel_shuffle, load_checkpoint
+
+
+class InvertedResidual(nn.Module):
+ """InvertedResidual block for ShuffleNetV2 backbone.
+
+ Args:
+ in_channels (int): The input channels of the block.
+ out_channels (int): The output channels of the block.
+ stride (int): Stride of the 3x3 convolution layer. Default: 1
+ conv_cfg (dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ act_cfg (dict): Config dict for activation layer.
+ Default: dict(type='ReLU').
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ stride=1,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ act_cfg=dict(type='ReLU'),
+ with_cp=False):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ act_cfg = copy.deepcopy(act_cfg)
+ super().__init__()
+ self.stride = stride
+ self.with_cp = with_cp
+
+ branch_features = out_channels // 2
+ if self.stride == 1:
+ assert in_channels == branch_features * 2, (
+ f'in_channels ({in_channels}) should equal to '
+ f'branch_features * 2 ({branch_features * 2}) '
+ 'when stride is 1')
+
+ if in_channels != branch_features * 2:
+ assert self.stride != 1, (
+ f'stride ({self.stride}) should not equal 1 when '
+ f'in_channels != branch_features * 2')
+
+ if self.stride > 1:
+ self.branch1 = nn.Sequential(
+ ConvModule(
+ in_channels,
+ in_channels,
+ kernel_size=3,
+ stride=self.stride,
+ padding=1,
+ groups=in_channels,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=None),
+ ConvModule(
+ in_channels,
+ branch_features,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg),
+ )
+
+ self.branch2 = nn.Sequential(
+ ConvModule(
+ in_channels if (self.stride > 1) else branch_features,
+ branch_features,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg),
+ ConvModule(
+ branch_features,
+ branch_features,
+ kernel_size=3,
+ stride=self.stride,
+ padding=1,
+ groups=branch_features,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=None),
+ ConvModule(
+ branch_features,
+ branch_features,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg))
+
+ def forward(self, x):
+
+ def _inner_forward(x):
+ if self.stride > 1:
+ out = torch.cat((self.branch1(x), self.branch2(x)), dim=1)
+ else:
+ x1, x2 = x.chunk(2, dim=1)
+ out = torch.cat((x1, self.branch2(x2)), dim=1)
+
+ out = channel_shuffle(out, 2)
+
+ return out
+
+ if self.with_cp and x.requires_grad:
+ out = cp.checkpoint(_inner_forward, x)
+ else:
+ out = _inner_forward(x)
+
+ return out
+
+
+@BACKBONES.register_module()
+class ShuffleNetV2(BaseBackbone):
+ """ShuffleNetV2 backbone.
+
+ Args:
+ widen_factor (float): Width multiplier - adjusts the number of
+ channels in each layer by this amount. Default: 1.0.
+ out_indices (Sequence[int]): Output from which stages.
+ Default: (0, 1, 2, 3).
+ frozen_stages (int): Stages to be frozen (all param fixed).
+ Default: -1, which means not freezing any parameters.
+ conv_cfg (dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ act_cfg (dict): Config dict for activation layer.
+ Default: dict(type='ReLU').
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ """
+
+ def __init__(self,
+ widen_factor=1.0,
+ out_indices=(3, ),
+ frozen_stages=-1,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ act_cfg=dict(type='ReLU'),
+ norm_eval=False,
+ with_cp=False):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ act_cfg = copy.deepcopy(act_cfg)
+ super().__init__()
+ self.stage_blocks = [4, 8, 4]
+ for index in out_indices:
+ if index not in range(0, 4):
+ raise ValueError('the item in out_indices must in '
+ f'range(0, 4). But received {index}')
+
+ if frozen_stages not in range(-1, 4):
+ raise ValueError('frozen_stages must be in range(-1, 4). '
+ f'But received {frozen_stages}')
+ self.out_indices = out_indices
+ self.frozen_stages = frozen_stages
+ self.conv_cfg = conv_cfg
+ self.norm_cfg = norm_cfg
+ self.act_cfg = act_cfg
+ self.norm_eval = norm_eval
+ self.with_cp = with_cp
+
+ if widen_factor == 0.5:
+ channels = [48, 96, 192, 1024]
+ elif widen_factor == 1.0:
+ channels = [116, 232, 464, 1024]
+ elif widen_factor == 1.5:
+ channels = [176, 352, 704, 1024]
+ elif widen_factor == 2.0:
+ channels = [244, 488, 976, 2048]
+ else:
+ raise ValueError('widen_factor must be in [0.5, 1.0, 1.5, 2.0]. '
+ f'But received {widen_factor}')
+
+ self.in_channels = 24
+ self.conv1 = ConvModule(
+ in_channels=3,
+ out_channels=self.in_channels,
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg)
+
+ self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)
+
+ self.layers = nn.ModuleList()
+ for i, num_blocks in enumerate(self.stage_blocks):
+ layer = self._make_layer(channels[i], num_blocks)
+ self.layers.append(layer)
+
+ output_channels = channels[-1]
+ self.layers.append(
+ ConvModule(
+ in_channels=self.in_channels,
+ out_channels=output_channels,
+ kernel_size=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg))
+
+ def _make_layer(self, out_channels, num_blocks):
+ """Stack blocks to make a layer.
+
+ Args:
+ out_channels (int): out_channels of the block.
+ num_blocks (int): number of blocks.
+ """
+ layers = []
+ for i in range(num_blocks):
+ stride = 2 if i == 0 else 1
+ layers.append(
+ InvertedResidual(
+ in_channels=self.in_channels,
+ out_channels=out_channels,
+ stride=stride,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ act_cfg=self.act_cfg,
+ with_cp=self.with_cp))
+ self.in_channels = out_channels
+
+ return nn.Sequential(*layers)
+
+ def _freeze_stages(self):
+ if self.frozen_stages >= 0:
+ for param in self.conv1.parameters():
+ param.requires_grad = False
+
+ for i in range(self.frozen_stages):
+ m = self.layers[i]
+ m.eval()
+ for param in m.parameters():
+ param.requires_grad = False
+
+ def init_weights(self, pretrained=None):
+ if isinstance(pretrained, str):
+ logger = logging.getLogger()
+ load_checkpoint(self, pretrained, strict=False, logger=logger)
+ elif pretrained is None:
+ for name, m in self.named_modules():
+ if isinstance(m, nn.Conv2d):
+ if 'conv1' in name:
+ normal_init(m, mean=0, std=0.01)
+ else:
+ normal_init(m, mean=0, std=1.0 / m.weight.shape[1])
+ elif isinstance(m, (_BatchNorm, nn.GroupNorm)):
+ constant_init(m.weight, val=1, bias=0.0001)
+ if isinstance(m, _BatchNorm):
+ if m.running_mean is not None:
+ nn.init.constant_(m.running_mean, 0)
+ else:
+ raise TypeError('pretrained must be a str or None. But received '
+ f'{type(pretrained)}')
+
+ def forward(self, x):
+ x = self.conv1(x)
+ x = self.maxpool(x)
+
+ outs = []
+ for i, layer in enumerate(self.layers):
+ x = layer(x)
+ if i in self.out_indices:
+ outs.append(x)
+
+ if len(outs) == 1:
+ return outs[0]
+ return tuple(outs)
+
+ def train(self, mode=True):
+ super().train(mode)
+ self._freeze_stages()
+ if mode and self.norm_eval:
+ for m in self.modules():
+ if isinstance(m, nn.BatchNorm2d):
+ m.eval()
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/tcn.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/tcn.py
new file mode 100644
index 0000000000000000000000000000000000000000..deca2290aeb1830bc3e241b819157369371aaf27
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/tcn.py
@@ -0,0 +1,267 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+
+import torch.nn as nn
+from mmcv.cnn import ConvModule, build_conv_layer, constant_init, kaiming_init
+from mmcv.utils.parrots_wrapper import _BatchNorm
+
+from mmpose.core import WeightNormClipHook
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+
+
+class BasicTemporalBlock(nn.Module):
+ """Basic block for VideoPose3D.
+
+ Args:
+ in_channels (int): Input channels of this block.
+ out_channels (int): Output channels of this block.
+ mid_channels (int): The output channels of conv1. Default: 1024.
+ kernel_size (int): Size of the convolving kernel. Default: 3.
+ dilation (int): Spacing between kernel elements. Default: 3.
+ dropout (float): Dropout rate. Default: 0.25.
+ causal (bool): Use causal convolutions instead of symmetric
+ convolutions (for real-time applications). Default: False.
+ residual (bool): Use residual connection. Default: True.
+ use_stride_conv (bool): Use optimized TCN that designed
+ specifically for single-frame batching, i.e. where batches have
+ input length = receptive field, and output length = 1. This
+ implementation replaces dilated convolutions with strided
+ convolutions to avoid generating unused intermediate results.
+ Default: False.
+ conv_cfg (dict): dictionary to construct and config conv layer.
+ Default: dict(type='Conv1d').
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN1d').
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ mid_channels=1024,
+ kernel_size=3,
+ dilation=3,
+ dropout=0.25,
+ causal=False,
+ residual=True,
+ use_stride_conv=False,
+ conv_cfg=dict(type='Conv1d'),
+ norm_cfg=dict(type='BN1d')):
+ # Protect mutable default arguments
+ conv_cfg = copy.deepcopy(conv_cfg)
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+ self.in_channels = in_channels
+ self.out_channels = out_channels
+ self.mid_channels = mid_channels
+ self.kernel_size = kernel_size
+ self.dilation = dilation
+ self.dropout = dropout
+ self.causal = causal
+ self.residual = residual
+ self.use_stride_conv = use_stride_conv
+
+ self.pad = (kernel_size - 1) * dilation // 2
+ if use_stride_conv:
+ self.stride = kernel_size
+ self.causal_shift = kernel_size // 2 if causal else 0
+ self.dilation = 1
+ else:
+ self.stride = 1
+ self.causal_shift = kernel_size // 2 * dilation if causal else 0
+
+ self.conv1 = nn.Sequential(
+ ConvModule(
+ in_channels,
+ mid_channels,
+ kernel_size=kernel_size,
+ stride=self.stride,
+ dilation=self.dilation,
+ bias='auto',
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg))
+ self.conv2 = nn.Sequential(
+ ConvModule(
+ mid_channels,
+ out_channels,
+ kernel_size=1,
+ bias='auto',
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg))
+
+ if residual and in_channels != out_channels:
+ self.short_cut = build_conv_layer(conv_cfg, in_channels,
+ out_channels, 1)
+ else:
+ self.short_cut = None
+
+ self.dropout = nn.Dropout(dropout) if dropout > 0 else None
+
+ def forward(self, x):
+ """Forward function."""
+ if self.use_stride_conv:
+ assert self.causal_shift + self.kernel_size // 2 < x.shape[2]
+ else:
+ assert 0 <= self.pad + self.causal_shift < x.shape[2] - \
+ self.pad + self.causal_shift <= x.shape[2]
+
+ out = self.conv1(x)
+ if self.dropout is not None:
+ out = self.dropout(out)
+
+ out = self.conv2(out)
+ if self.dropout is not None:
+ out = self.dropout(out)
+
+ if self.residual:
+ if self.use_stride_conv:
+ res = x[:, :, self.causal_shift +
+ self.kernel_size // 2::self.kernel_size]
+ else:
+ res = x[:, :,
+ (self.pad + self.causal_shift):(x.shape[2] - self.pad +
+ self.causal_shift)]
+
+ if self.short_cut is not None:
+ res = self.short_cut(res)
+ out = out + res
+
+ return out
+
+
+@BACKBONES.register_module()
+class TCN(BaseBackbone):
+ """TCN backbone.
+
+ Temporal Convolutional Networks.
+ More details can be found in the
+ `paper `__ .
+
+ Args:
+ in_channels (int): Number of input channels, which equals to
+ num_keypoints * num_features.
+ stem_channels (int): Number of feature channels. Default: 1024.
+ num_blocks (int): NUmber of basic temporal convolutional blocks.
+ Default: 2.
+ kernel_sizes (Sequence[int]): Sizes of the convolving kernel of
+ each basic block. Default: ``(3, 3, 3)``.
+ dropout (float): Dropout rate. Default: 0.25.
+ causal (bool): Use causal convolutions instead of symmetric
+ convolutions (for real-time applications).
+ Default: False.
+ residual (bool): Use residual connection. Default: True.
+ use_stride_conv (bool): Use TCN backbone optimized for
+ single-frame batching, i.e. where batches have input length =
+ receptive field, and output length = 1. This implementation
+ replaces dilated convolutions with strided convolutions to avoid
+ generating unused intermediate results. The weights are
+ interchangeable with the reference implementation. Default: False
+ conv_cfg (dict): dictionary to construct and config conv layer.
+ Default: dict(type='Conv1d').
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN1d').
+ max_norm (float|None): if not None, the weight of convolution layers
+ will be clipped to have a maximum norm of max_norm.
+
+ Example:
+ >>> from mmpose.models import TCN
+ >>> import torch
+ >>> self = TCN(in_channels=34)
+ >>> self.eval()
+ >>> inputs = torch.rand(1, 34, 243)
+ >>> level_outputs = self.forward(inputs)
+ >>> for level_out in level_outputs:
+ ... print(tuple(level_out.shape))
+ (1, 1024, 235)
+ (1, 1024, 217)
+ """
+
+ def __init__(self,
+ in_channels,
+ stem_channels=1024,
+ num_blocks=2,
+ kernel_sizes=(3, 3, 3),
+ dropout=0.25,
+ causal=False,
+ residual=True,
+ use_stride_conv=False,
+ conv_cfg=dict(type='Conv1d'),
+ norm_cfg=dict(type='BN1d'),
+ max_norm=None):
+ # Protect mutable default arguments
+ conv_cfg = copy.deepcopy(conv_cfg)
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+ self.in_channels = in_channels
+ self.stem_channels = stem_channels
+ self.num_blocks = num_blocks
+ self.kernel_sizes = kernel_sizes
+ self.dropout = dropout
+ self.causal = causal
+ self.residual = residual
+ self.use_stride_conv = use_stride_conv
+ self.max_norm = max_norm
+
+ assert num_blocks == len(kernel_sizes) - 1
+ for ks in kernel_sizes:
+ assert ks % 2 == 1, 'Only odd filter widths are supported.'
+
+ self.expand_conv = ConvModule(
+ in_channels,
+ stem_channels,
+ kernel_size=kernel_sizes[0],
+ stride=kernel_sizes[0] if use_stride_conv else 1,
+ bias='auto',
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg)
+
+ dilation = kernel_sizes[0]
+ self.tcn_blocks = nn.ModuleList()
+ for i in range(1, num_blocks + 1):
+ self.tcn_blocks.append(
+ BasicTemporalBlock(
+ in_channels=stem_channels,
+ out_channels=stem_channels,
+ mid_channels=stem_channels,
+ kernel_size=kernel_sizes[i],
+ dilation=dilation,
+ dropout=dropout,
+ causal=causal,
+ residual=residual,
+ use_stride_conv=use_stride_conv,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg))
+ dilation *= kernel_sizes[i]
+
+ if self.max_norm is not None:
+ # Apply weight norm clip to conv layers
+ weight_clip = WeightNormClipHook(self.max_norm)
+ for module in self.modules():
+ if isinstance(module, nn.modules.conv._ConvNd):
+ weight_clip.register(module)
+
+ self.dropout = nn.Dropout(dropout) if dropout > 0 else None
+
+ def forward(self, x):
+ """Forward function."""
+ x = self.expand_conv(x)
+
+ if self.dropout is not None:
+ x = self.dropout(x)
+
+ outs = []
+ for i in range(self.num_blocks):
+ x = self.tcn_blocks[i](x)
+ outs.append(x)
+
+ return tuple(outs)
+
+ def init_weights(self, pretrained=None):
+ """Initialize the weights."""
+ super().init_weights(pretrained)
+ if pretrained is None:
+ for m in self.modules():
+ if isinstance(m, nn.modules.conv._ConvNd):
+ kaiming_init(m, mode='fan_in', nonlinearity='relu')
+ elif isinstance(m, _BatchNorm):
+ constant_init(m, 1)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/test_torch.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/test_torch.py
new file mode 100644
index 0000000000000000000000000000000000000000..c6833af88fa89a417aa85aef28eb4d81ae7b762b
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/test_torch.py
@@ -0,0 +1,60 @@
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+
+
+class Net(nn.Module):
+
+ def __init__(self):
+ super(Net, self).__init__()
+ # 1 input image channel, 6 output channels, 5x5 square convolution
+ # kernel
+ self.conv1 = nn.Conv2d(1, 6, 5)
+ self.conv2 = nn.Conv2d(6, 16, 5)
+ # an affine operation: y = Wx + b
+ self.fc1 = nn.Linear(16 * 5 * 5, 120) # 5*5 from image dimension
+ self.fc2 = nn.Linear(120, 84)
+ self.fc3 = nn.Linear(84, 10)
+
+ def forward(self, x):
+ # Max pooling over a (2, 2) window
+ x = F.max_pool2d(F.relu(self.conv1(x)), (2, 2))
+ # If the size is a square, you can specify with a single number
+ x = F.max_pool2d(F.relu(self.conv2(x)), 2)
+ x = torch.flatten(x, 1) # flatten all dimensions except the batch dimension
+ x = F.relu(self.fc1(x))
+ x = F.relu(self.fc2(x))
+ x = self.fc3(x)
+ return x
+
+
+net = Net()
+# print(net)
+
+net.train()
+
+input = torch.randn(1, 1, 32, 32)
+# out = net(input)
+# print(out)
+output = net(input)
+target = torch.randn(10) # a dummy target, for example
+target = target.view(1, -1) # make it the same shape as output
+criterion = nn.MSELoss()
+
+# loss = criterion(output.cuda(), target.cuda())
+
+import torch.optim as optim
+
+# create your optimizer
+optimizer = optim.SGD(net.parameters(), lr=0.01)
+
+# in your training loop:
+optimizer.zero_grad() # zero the gradient buffers
+output = net(input)
+loss = criterion(output, target)
+
+loss.backward()
+
+optimizer.step()
+
+# print(loss)
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/__init__.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..52a30ca9f7c8e90b6c6fa2fd8a9705ca0403b259
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/__init__.py
@@ -0,0 +1,11 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+from .channel_shuffle import channel_shuffle
+from .inverted_residual import InvertedResidual
+from .make_divisible import make_divisible
+from .se_layer import SELayer
+from .utils import load_checkpoint
+
+__all__ = [
+ 'channel_shuffle', 'make_divisible', 'InvertedResidual', 'SELayer',
+ 'load_checkpoint'
+]
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/channel_shuffle.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/channel_shuffle.py
new file mode 100644
index 0000000000000000000000000000000000000000..27006a8065db35a14c4207ce6613104374b064ad
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/channel_shuffle.py
@@ -0,0 +1,29 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import torch
+
+
+def channel_shuffle(x, groups):
+ """Channel Shuffle operation.
+
+ This function enables cross-group information flow for multiple groups
+ convolution layers.
+
+ Args:
+ x (Tensor): The input tensor.
+ groups (int): The number of groups to divide the input tensor
+ in the channel dimension.
+
+ Returns:
+ Tensor: The output tensor after channel shuffle operation.
+ """
+
+ batch_size, num_channels, height, width = x.size()
+ assert (num_channels % groups == 0), ('num_channels should be '
+ 'divisible by groups')
+ channels_per_group = num_channels // groups
+
+ x = x.view(batch_size, groups, channels_per_group, height, width)
+ x = torch.transpose(x, 1, 2).contiguous()
+ x = x.view(batch_size, -1, height, width)
+
+ return x
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/inverted_residual.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/inverted_residual.py
new file mode 100644
index 0000000000000000000000000000000000000000..dff762c570550e4a738ae1833a4c82c18777115d
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/inverted_residual.py
@@ -0,0 +1,128 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+
+import torch.nn as nn
+import torch.utils.checkpoint as cp
+from mmcv.cnn import ConvModule
+
+from .se_layer import SELayer
+
+
+class InvertedResidual(nn.Module):
+ """Inverted Residual Block.
+
+ Args:
+ in_channels (int): The input channels of this Module.
+ out_channels (int): The output channels of this Module.
+ mid_channels (int): The input channels of the depthwise convolution.
+ kernel_size (int): The kernel size of the depthwise convolution.
+ Default: 3.
+ groups (None or int): The group number of the depthwise convolution.
+ Default: None, which means group number = mid_channels.
+ stride (int): The stride of the depthwise convolution. Default: 1.
+ se_cfg (dict): Config dict for se layer. Default: None, which means no
+ se layer.
+ with_expand_conv (bool): Use expand conv or not. If set False,
+ mid_channels must be the same with in_channels.
+ Default: True.
+ conv_cfg (dict): Config dict for convolution layer. Default: None,
+ which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ act_cfg (dict): Config dict for activation layer.
+ Default: dict(type='ReLU').
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+
+ Returns:
+ Tensor: The output tensor.
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ mid_channels,
+ kernel_size=3,
+ groups=None,
+ stride=1,
+ se_cfg=None,
+ with_expand_conv=True,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ act_cfg=dict(type='ReLU'),
+ with_cp=False):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ act_cfg = copy.deepcopy(act_cfg)
+ super().__init__()
+ self.with_res_shortcut = (stride == 1 and in_channels == out_channels)
+ assert stride in [1, 2]
+ self.with_cp = with_cp
+ self.with_se = se_cfg is not None
+ self.with_expand_conv = with_expand_conv
+
+ if groups is None:
+ groups = mid_channels
+
+ if self.with_se:
+ assert isinstance(se_cfg, dict)
+ if not self.with_expand_conv:
+ assert mid_channels == in_channels
+
+ if self.with_expand_conv:
+ self.expand_conv = ConvModule(
+ in_channels=in_channels,
+ out_channels=mid_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg)
+ self.depthwise_conv = ConvModule(
+ in_channels=mid_channels,
+ out_channels=mid_channels,
+ kernel_size=kernel_size,
+ stride=stride,
+ padding=kernel_size // 2,
+ groups=groups,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg)
+ if self.with_se:
+ self.se = SELayer(**se_cfg)
+ self.linear_conv = ConvModule(
+ in_channels=mid_channels,
+ out_channels=out_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=None)
+
+ def forward(self, x):
+
+ def _inner_forward(x):
+ out = x
+
+ if self.with_expand_conv:
+ out = self.expand_conv(out)
+
+ out = self.depthwise_conv(out)
+
+ if self.with_se:
+ out = self.se(out)
+
+ out = self.linear_conv(out)
+
+ if self.with_res_shortcut:
+ return x + out
+ return out
+
+ if self.with_cp and x.requires_grad:
+ out = cp.checkpoint(_inner_forward, x)
+ else:
+ out = _inner_forward(x)
+
+ return out
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/make_divisible.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/make_divisible.py
new file mode 100644
index 0000000000000000000000000000000000000000..b7666be65939d5c76057e73927c230029cb1871d
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/make_divisible.py
@@ -0,0 +1,25 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+def make_divisible(value, divisor, min_value=None, min_ratio=0.9):
+ """Make divisible function.
+
+ This function rounds the channel number down to the nearest value that can
+ be divisible by the divisor.
+
+ Args:
+ value (int): The original channel number.
+ divisor (int): The divisor to fully divide the channel number.
+ min_value (int, optional): The minimum value of the output channel.
+ Default: None, means that the minimum value equal to the divisor.
+ min_ratio (float, optional): The minimum ratio of the rounded channel
+ number to the original channel number. Default: 0.9.
+ Returns:
+ int: The modified output channel number
+ """
+
+ if min_value is None:
+ min_value = divisor
+ new_value = max(min_value, int(value + divisor / 2) // divisor * divisor)
+ # Make sure that round down does not go down by more than (1-min_ratio).
+ if new_value < min_ratio * value:
+ new_value += divisor
+ return new_value
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/se_layer.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/se_layer.py
new file mode 100644
index 0000000000000000000000000000000000000000..07f70802eb1b98b1f22516ba62b1533557f428ed
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/se_layer.py
@@ -0,0 +1,54 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import mmcv
+import torch.nn as nn
+from mmcv.cnn import ConvModule
+
+
+class SELayer(nn.Module):
+ """Squeeze-and-Excitation Module.
+
+ Args:
+ channels (int): The input (and output) channels of the SE layer.
+ ratio (int): Squeeze ratio in SELayer, the intermediate channel will be
+ ``int(channels/ratio)``. Default: 16.
+ conv_cfg (None or dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ act_cfg (dict or Sequence[dict]): Config dict for activation layer.
+ If act_cfg is a dict, two activation layers will be configurated
+ by this dict. If act_cfg is a sequence of dicts, the first
+ activation layer will be configurated by the first dict and the
+ second activation layer will be configurated by the second dict.
+ Default: (dict(type='ReLU'), dict(type='Sigmoid'))
+ """
+
+ def __init__(self,
+ channels,
+ ratio=16,
+ conv_cfg=None,
+ act_cfg=(dict(type='ReLU'), dict(type='Sigmoid'))):
+ super().__init__()
+ if isinstance(act_cfg, dict):
+ act_cfg = (act_cfg, act_cfg)
+ assert len(act_cfg) == 2
+ assert mmcv.is_tuple_of(act_cfg, dict)
+ self.global_avgpool = nn.AdaptiveAvgPool2d(1)
+ self.conv1 = ConvModule(
+ in_channels=channels,
+ out_channels=int(channels / ratio),
+ kernel_size=1,
+ stride=1,
+ conv_cfg=conv_cfg,
+ act_cfg=act_cfg[0])
+ self.conv2 = ConvModule(
+ in_channels=int(channels / ratio),
+ out_channels=channels,
+ kernel_size=1,
+ stride=1,
+ conv_cfg=conv_cfg,
+ act_cfg=act_cfg[1])
+
+ def forward(self, x):
+ out = self.global_avgpool(x)
+ out = self.conv1(out)
+ out = self.conv2(out)
+ return x * out
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/utils.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..a9ac948653adeb849e0f510bc1014664741fe6f9
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/utils/utils.py
@@ -0,0 +1,87 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+from collections import OrderedDict
+
+from mmcv.runner.checkpoint import _load_checkpoint, load_state_dict
+
+
+def load_checkpoint(model,
+ filename,
+ map_location='cpu',
+ strict=False,
+ logger=None):
+ """Load checkpoint from a file or URI.
+
+ Args:
+ model (Module): Module to load checkpoint.
+ filename (str): Accept local filepath, URL, ``torchvision://xxx``,
+ ``open-mmlab://xxx``.
+ map_location (str): Same as :func:`torch.load`.
+ strict (bool): Whether to allow different params for the model and
+ checkpoint.
+ logger (:mod:`logging.Logger` or None): The logger for error message.
+
+ Returns:
+ dict or OrderedDict: The loaded checkpoint.
+ """
+ checkpoint = _load_checkpoint(filename, map_location)
+ # OrderedDict is a subclass of dict
+ if not isinstance(checkpoint, dict):
+ raise RuntimeError(
+ f'No state_dict found in checkpoint file {filename}')
+ # get state_dict from checkpoint
+ if 'state_dict' in checkpoint:
+ state_dict_tmp = checkpoint['state_dict']
+ else:
+ state_dict_tmp = checkpoint
+
+ state_dict = OrderedDict()
+ # strip prefix of state_dict
+ for k, v in state_dict_tmp.items():
+ if k.startswith('module.backbone.'):
+ state_dict[k[16:]] = v
+ elif k.startswith('module.'):
+ state_dict[k[7:]] = v
+ elif k.startswith('backbone.'):
+ state_dict[k[9:]] = v
+ else:
+ state_dict[k] = v
+ # load state_dict
+ load_state_dict(model, state_dict, strict, logger)
+ return checkpoint
+
+
+def get_state_dict(filename, map_location='cpu'):
+ """Get state_dict from a file or URI.
+
+ Args:
+ filename (str): Accept local filepath, URL, ``torchvision://xxx``,
+ ``open-mmlab://xxx``.
+ map_location (str): Same as :func:`torch.load`.
+
+ Returns:
+ OrderedDict: The state_dict.
+ """
+ checkpoint = _load_checkpoint(filename, map_location)
+ # OrderedDict is a subclass of dict
+ if not isinstance(checkpoint, dict):
+ raise RuntimeError(
+ f'No state_dict found in checkpoint file {filename}')
+ # get state_dict from checkpoint
+ if 'state_dict' in checkpoint:
+ state_dict_tmp = checkpoint['state_dict']
+ else:
+ state_dict_tmp = checkpoint
+
+ state_dict = OrderedDict()
+ # strip prefix of state_dict
+ for k, v in state_dict_tmp.items():
+ if k.startswith('module.backbone.'):
+ state_dict[k[16:]] = v
+ elif k.startswith('module.'):
+ state_dict[k[7:]] = v
+ elif k.startswith('backbone.'):
+ state_dict[k[9:]] = v
+ else:
+ state_dict[k] = v
+
+ return state_dict
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vgg.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vgg.py
new file mode 100644
index 0000000000000000000000000000000000000000..f7d467017a5520f399c84b1235ec64c99b805b42
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vgg.py
@@ -0,0 +1,193 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import torch.nn as nn
+from mmcv.cnn import ConvModule, constant_init, kaiming_init, normal_init
+from mmcv.utils.parrots_wrapper import _BatchNorm
+
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+
+
+def make_vgg_layer(in_channels,
+ out_channels,
+ num_blocks,
+ conv_cfg=None,
+ norm_cfg=None,
+ act_cfg=dict(type='ReLU'),
+ dilation=1,
+ with_norm=False,
+ ceil_mode=False):
+ layers = []
+ for _ in range(num_blocks):
+ layer = ConvModule(
+ in_channels=in_channels,
+ out_channels=out_channels,
+ kernel_size=3,
+ dilation=dilation,
+ padding=dilation,
+ bias=True,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg)
+ layers.append(layer)
+ in_channels = out_channels
+ layers.append(nn.MaxPool2d(kernel_size=2, stride=2, ceil_mode=ceil_mode))
+
+ return layers
+
+
+@BACKBONES.register_module()
+class VGG(BaseBackbone):
+ """VGG backbone.
+
+ Args:
+ depth (int): Depth of vgg, from {11, 13, 16, 19}.
+ with_norm (bool): Use BatchNorm or not.
+ num_classes (int): number of classes for classification.
+ num_stages (int): VGG stages, normally 5.
+ dilations (Sequence[int]): Dilation of each stage.
+ out_indices (Sequence[int]): Output from which stages. If only one
+ stage is specified, a single tensor (feature map) is returned,
+ otherwise multiple stages are specified, a tuple of tensors will
+ be returned. When it is None, the default behavior depends on
+ whether num_classes is specified. If num_classes <= 0, the default
+ value is (4, ), outputting the last feature map before classifier.
+ If num_classes > 0, the default value is (5, ), outputting the
+ classification score. Default: None.
+ frozen_stages (int): Stages to be frozen (all param fixed). -1 means
+ not freezing any parameters.
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ ceil_mode (bool): Whether to use ceil_mode of MaxPool. Default: False.
+ with_last_pool (bool): Whether to keep the last pooling before
+ classifier. Default: True.
+ """
+
+ # Parameters to build layers. Each element specifies the number of conv in
+ # each stage. For example, VGG11 contains 11 layers with learnable
+ # parameters. 11 is computed as 11 = (1 + 1 + 2 + 2 + 2) + 3,
+ # where 3 indicates the last three fully-connected layers.
+ arch_settings = {
+ 11: (1, 1, 2, 2, 2),
+ 13: (2, 2, 2, 2, 2),
+ 16: (2, 2, 3, 3, 3),
+ 19: (2, 2, 4, 4, 4)
+ }
+
+ def __init__(self,
+ depth,
+ num_classes=-1,
+ num_stages=5,
+ dilations=(1, 1, 1, 1, 1),
+ out_indices=None,
+ frozen_stages=-1,
+ conv_cfg=None,
+ norm_cfg=None,
+ act_cfg=dict(type='ReLU'),
+ norm_eval=False,
+ ceil_mode=False,
+ with_last_pool=True):
+ super().__init__()
+ if depth not in self.arch_settings:
+ raise KeyError(f'invalid depth {depth} for vgg')
+ assert num_stages >= 1 and num_stages <= 5
+ stage_blocks = self.arch_settings[depth]
+ self.stage_blocks = stage_blocks[:num_stages]
+ assert len(dilations) == num_stages
+
+ self.num_classes = num_classes
+ self.frozen_stages = frozen_stages
+ self.norm_eval = norm_eval
+ with_norm = norm_cfg is not None
+
+ if out_indices is None:
+ out_indices = (5, ) if num_classes > 0 else (4, )
+ assert max(out_indices) <= num_stages
+ self.out_indices = out_indices
+
+ self.in_channels = 3
+ start_idx = 0
+ vgg_layers = []
+ self.range_sub_modules = []
+ for i, num_blocks in enumerate(self.stage_blocks):
+ num_modules = num_blocks + 1
+ end_idx = start_idx + num_modules
+ dilation = dilations[i]
+ out_channels = 64 * 2**i if i < 4 else 512
+ vgg_layer = make_vgg_layer(
+ self.in_channels,
+ out_channels,
+ num_blocks,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=act_cfg,
+ dilation=dilation,
+ with_norm=with_norm,
+ ceil_mode=ceil_mode)
+ vgg_layers.extend(vgg_layer)
+ self.in_channels = out_channels
+ self.range_sub_modules.append([start_idx, end_idx])
+ start_idx = end_idx
+ if not with_last_pool:
+ vgg_layers.pop(-1)
+ self.range_sub_modules[-1][1] -= 1
+ self.module_name = 'features'
+ self.add_module(self.module_name, nn.Sequential(*vgg_layers))
+
+ if self.num_classes > 0:
+ self.classifier = nn.Sequential(
+ nn.Linear(512 * 7 * 7, 4096),
+ nn.ReLU(True),
+ nn.Dropout(),
+ nn.Linear(4096, 4096),
+ nn.ReLU(True),
+ nn.Dropout(),
+ nn.Linear(4096, num_classes),
+ )
+
+ def init_weights(self, pretrained=None):
+ super().init_weights(pretrained)
+ if pretrained is None:
+ for m in self.modules():
+ if isinstance(m, nn.Conv2d):
+ kaiming_init(m)
+ elif isinstance(m, _BatchNorm):
+ constant_init(m, 1)
+ elif isinstance(m, nn.Linear):
+ normal_init(m, std=0.01)
+
+ def forward(self, x):
+ outs = []
+ vgg_layers = getattr(self, self.module_name)
+ for i in range(len(self.stage_blocks)):
+ for j in range(*self.range_sub_modules[i]):
+ vgg_layer = vgg_layers[j]
+ x = vgg_layer(x)
+ if i in self.out_indices:
+ outs.append(x)
+ if self.num_classes > 0:
+ x = x.view(x.size(0), -1)
+ x = self.classifier(x)
+ outs.append(x)
+ if len(outs) == 1:
+ return outs[0]
+ else:
+ return tuple(outs)
+
+ def _freeze_stages(self):
+ vgg_layers = getattr(self, self.module_name)
+ for i in range(self.frozen_stages):
+ for j in range(*self.range_sub_modules[i]):
+ m = vgg_layers[j]
+ m.eval()
+ for param in m.parameters():
+ param.requires_grad = False
+
+ def train(self, mode=True):
+ super().train(mode)
+ self._freeze_stages()
+ if mode and self.norm_eval:
+ for m in self.modules():
+ # trick: eval have effect on BatchNorm only
+ if isinstance(m, _BatchNorm):
+ m.eval()
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vipnas_mbv3.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vipnas_mbv3.py
new file mode 100644
index 0000000000000000000000000000000000000000..ed990e3966b27301dbaf081e3ec0e908704dfc8b
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vipnas_mbv3.py
@@ -0,0 +1,179 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+import logging
+
+import torch.nn as nn
+from mmcv.cnn import ConvModule
+from torch.nn.modules.batchnorm import _BatchNorm
+
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+from .utils import InvertedResidual, load_checkpoint
+
+
+@BACKBONES.register_module()
+class ViPNAS_MobileNetV3(BaseBackbone):
+ """ViPNAS_MobileNetV3 backbone.
+
+ "ViPNAS: Efficient Video Pose Estimation via Neural Architecture Search"
+ More details can be found in the `paper
+ `__ .
+
+ Args:
+ wid (list(int)): Searched width config for each stage.
+ expan (list(int)): Searched expansion ratio config for each stage.
+ dep (list(int)): Searched depth config for each stage.
+ ks (list(int)): Searched kernel size config for each stage.
+ group (list(int)): Searched group number config for each stage.
+ att (list(bool)): Searched attention config for each stage.
+ stride (list(int)): Stride config for each stage.
+ act (list(dict)): Activation config for each stage.
+ conv_cfg (dict): Config dict for convolution layer.
+ Default: None, which means using conv2d.
+ norm_cfg (dict): Config dict for normalization layer.
+ Default: dict(type='BN').
+ frozen_stages (int): Stages to be frozen (all param fixed).
+ Default: -1, which means not freezing any parameters.
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save
+ some memory while slowing down the training speed.
+ Default: False.
+ """
+
+ def __init__(self,
+ wid=[16, 16, 24, 40, 80, 112, 160],
+ expan=[None, 1, 5, 4, 5, 5, 6],
+ dep=[None, 1, 4, 4, 4, 4, 4],
+ ks=[3, 3, 7, 7, 5, 7, 5],
+ group=[None, 8, 120, 20, 100, 280, 240],
+ att=[None, True, True, False, True, True, True],
+ stride=[2, 1, 2, 2, 2, 1, 2],
+ act=[
+ 'HSwish', 'ReLU', 'ReLU', 'ReLU', 'HSwish', 'HSwish',
+ 'HSwish'
+ ],
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ frozen_stages=-1,
+ norm_eval=False,
+ with_cp=False):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+ self.wid = wid
+ self.expan = expan
+ self.dep = dep
+ self.ks = ks
+ self.group = group
+ self.att = att
+ self.stride = stride
+ self.act = act
+ self.conv_cfg = conv_cfg
+ self.norm_cfg = norm_cfg
+ self.frozen_stages = frozen_stages
+ self.norm_eval = norm_eval
+ self.with_cp = with_cp
+
+ self.conv1 = ConvModule(
+ in_channels=3,
+ out_channels=self.wid[0],
+ kernel_size=self.ks[0],
+ stride=self.stride[0],
+ padding=self.ks[0] // 2,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ act_cfg=dict(type=self.act[0]))
+
+ self.layers = self._make_layer()
+
+ def _make_layer(self):
+ layers = []
+ layer_index = 0
+ for i, dep in enumerate(self.dep[1:]):
+ mid_channels = self.wid[i + 1] * self.expan[i + 1]
+
+ if self.att[i + 1]:
+ se_cfg = dict(
+ channels=mid_channels,
+ ratio=4,
+ act_cfg=(dict(type='ReLU'), dict(type='HSigmoid')))
+ else:
+ se_cfg = None
+
+ if self.expan[i + 1] == 1:
+ with_expand_conv = False
+ else:
+ with_expand_conv = True
+
+ for j in range(dep):
+ if j == 0:
+ stride = self.stride[i + 1]
+ in_channels = self.wid[i]
+ else:
+ stride = 1
+ in_channels = self.wid[i + 1]
+
+ layer = InvertedResidual(
+ in_channels=in_channels,
+ out_channels=self.wid[i + 1],
+ mid_channels=mid_channels,
+ kernel_size=self.ks[i + 1],
+ groups=self.group[i + 1],
+ stride=stride,
+ se_cfg=se_cfg,
+ with_expand_conv=with_expand_conv,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ act_cfg=dict(type=self.act[i + 1]),
+ with_cp=self.with_cp)
+ layer_index += 1
+ layer_name = f'layer{layer_index}'
+ self.add_module(layer_name, layer)
+ layers.append(layer_name)
+ return layers
+
+ def init_weights(self, pretrained=None):
+ if isinstance(pretrained, str):
+ logger = logging.getLogger()
+ load_checkpoint(self, pretrained, strict=False, logger=logger)
+ elif pretrained is None:
+ for m in self.modules():
+ if isinstance(m, nn.Conv2d):
+ nn.init.normal_(m.weight, std=0.001)
+ for name, _ in m.named_parameters():
+ if name in ['bias']:
+ nn.init.constant_(m.bias, 0)
+ elif isinstance(m, nn.BatchNorm2d):
+ nn.init.constant_(m.weight, 1)
+ nn.init.constant_(m.bias, 0)
+ else:
+ raise TypeError('pretrained must be a str or None')
+
+ def forward(self, x):
+ x = self.conv1(x)
+
+ for i, layer_name in enumerate(self.layers):
+ layer = getattr(self, layer_name)
+ x = layer(x)
+
+ return x
+
+ def _freeze_stages(self):
+ if self.frozen_stages >= 0:
+ for param in self.conv1.parameters():
+ param.requires_grad = False
+ for i in range(1, self.frozen_stages + 1):
+ layer = getattr(self, f'layer{i}')
+ layer.eval()
+ for param in layer.parameters():
+ param.requires_grad = False
+
+ def train(self, mode=True):
+ super().train(mode)
+ self._freeze_stages()
+ if mode and self.norm_eval:
+ for m in self.modules():
+ if isinstance(m, _BatchNorm):
+ m.eval()
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vipnas_resnet.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vipnas_resnet.py
new file mode 100644
index 0000000000000000000000000000000000000000..81b028ed5f5caad5f59c68b7f82c1a4661cf4d6f
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vipnas_resnet.py
@@ -0,0 +1,589 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy
+
+import torch.nn as nn
+import torch.utils.checkpoint as cp
+from mmcv.cnn import ConvModule, build_conv_layer, build_norm_layer
+from mmcv.cnn.bricks import ContextBlock
+from mmcv.utils.parrots_wrapper import _BatchNorm
+
+from ..builder import BACKBONES
+from .base_backbone import BaseBackbone
+
+
+class ViPNAS_Bottleneck(nn.Module):
+ """Bottleneck block for ViPNAS_ResNet.
+
+ Args:
+ in_channels (int): Input channels of this block.
+ out_channels (int): Output channels of this block.
+ expansion (int): The ratio of ``out_channels/mid_channels`` where
+ ``mid_channels`` is the input/output channels of conv2. Default: 4.
+ stride (int): stride of the block. Default: 1
+ dilation (int): dilation of convolution. Default: 1
+ downsample (nn.Module): downsample operation on identity branch.
+ Default: None.
+ style (str): ``"pytorch"`` or ``"caffe"``. If set to "pytorch", the
+ stride-two layer is the 3x3 conv layer, otherwise the stride-two
+ layer is the first 1x1 conv layer. Default: "pytorch".
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed.
+ conv_cfg (dict): dictionary to construct and config conv layer.
+ Default: None
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ kernel_size (int): kernel size of conv2 searched in ViPANS.
+ groups (int): group number of conv2 searched in ViPNAS.
+ attention (bool): whether to use attention module in the end of
+ the block.
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ expansion=4,
+ stride=1,
+ dilation=1,
+ downsample=None,
+ style='pytorch',
+ with_cp=False,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ kernel_size=3,
+ groups=1,
+ attention=False):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+ assert style in ['pytorch', 'caffe']
+
+ self.in_channels = in_channels
+ self.out_channels = out_channels
+ self.expansion = expansion
+ assert out_channels % expansion == 0
+ self.mid_channels = out_channels // expansion
+ self.stride = stride
+ self.dilation = dilation
+ self.style = style
+ self.with_cp = with_cp
+ self.conv_cfg = conv_cfg
+ self.norm_cfg = norm_cfg
+
+ if self.style == 'pytorch':
+ self.conv1_stride = 1
+ self.conv2_stride = stride
+ else:
+ self.conv1_stride = stride
+ self.conv2_stride = 1
+
+ self.norm1_name, norm1 = build_norm_layer(
+ norm_cfg, self.mid_channels, postfix=1)
+ self.norm2_name, norm2 = build_norm_layer(
+ norm_cfg, self.mid_channels, postfix=2)
+ self.norm3_name, norm3 = build_norm_layer(
+ norm_cfg, out_channels, postfix=3)
+
+ self.conv1 = build_conv_layer(
+ conv_cfg,
+ in_channels,
+ self.mid_channels,
+ kernel_size=1,
+ stride=self.conv1_stride,
+ bias=False)
+ self.add_module(self.norm1_name, norm1)
+ self.conv2 = build_conv_layer(
+ conv_cfg,
+ self.mid_channels,
+ self.mid_channels,
+ kernel_size=kernel_size,
+ stride=self.conv2_stride,
+ padding=kernel_size // 2,
+ groups=groups,
+ dilation=dilation,
+ bias=False)
+
+ self.add_module(self.norm2_name, norm2)
+ self.conv3 = build_conv_layer(
+ conv_cfg,
+ self.mid_channels,
+ out_channels,
+ kernel_size=1,
+ bias=False)
+ self.add_module(self.norm3_name, norm3)
+
+ if attention:
+ self.attention = ContextBlock(out_channels,
+ max(1.0 / 16, 16.0 / out_channels))
+ else:
+ self.attention = None
+
+ self.relu = nn.ReLU(inplace=True)
+ self.downsample = downsample
+
+ @property
+ def norm1(self):
+ """nn.Module: the normalization layer named "norm1" """
+ return getattr(self, self.norm1_name)
+
+ @property
+ def norm2(self):
+ """nn.Module: the normalization layer named "norm2" """
+ return getattr(self, self.norm2_name)
+
+ @property
+ def norm3(self):
+ """nn.Module: the normalization layer named "norm3" """
+ return getattr(self, self.norm3_name)
+
+ def forward(self, x):
+ """Forward function."""
+
+ def _inner_forward(x):
+ identity = x
+
+ out = self.conv1(x)
+ out = self.norm1(out)
+ out = self.relu(out)
+
+ out = self.conv2(out)
+ out = self.norm2(out)
+ out = self.relu(out)
+
+ out = self.conv3(out)
+ out = self.norm3(out)
+
+ if self.attention is not None:
+ out = self.attention(out)
+
+ if self.downsample is not None:
+ identity = self.downsample(x)
+
+ out += identity
+
+ return out
+
+ if self.with_cp and x.requires_grad:
+ out = cp.checkpoint(_inner_forward, x)
+ else:
+ out = _inner_forward(x)
+
+ out = self.relu(out)
+
+ return out
+
+
+def get_expansion(block, expansion=None):
+ """Get the expansion of a residual block.
+
+ The block expansion will be obtained by the following order:
+
+ 1. If ``expansion`` is given, just return it.
+ 2. If ``block`` has the attribute ``expansion``, then return
+ ``block.expansion``.
+ 3. Return the default value according the the block type:
+ 4 for ``ViPNAS_Bottleneck``.
+
+ Args:
+ block (class): The block class.
+ expansion (int | None): The given expansion ratio.
+
+ Returns:
+ int: The expansion of the block.
+ """
+ if isinstance(expansion, int):
+ assert expansion > 0
+ elif expansion is None:
+ if hasattr(block, 'expansion'):
+ expansion = block.expansion
+ elif issubclass(block, ViPNAS_Bottleneck):
+ expansion = 1
+ else:
+ raise TypeError(f'expansion is not specified for {block.__name__}')
+ else:
+ raise TypeError('expansion must be an integer or None')
+
+ return expansion
+
+
+class ViPNAS_ResLayer(nn.Sequential):
+ """ViPNAS_ResLayer to build ResNet style backbone.
+
+ Args:
+ block (nn.Module): Residual block used to build ViPNAS ResLayer.
+ num_blocks (int): Number of blocks.
+ in_channels (int): Input channels of this block.
+ out_channels (int): Output channels of this block.
+ expansion (int, optional): The expansion for BasicBlock/Bottleneck.
+ If not specified, it will firstly be obtained via
+ ``block.expansion``. If the block has no attribute "expansion",
+ the following default values will be used: 1 for BasicBlock and
+ 4 for Bottleneck. Default: None.
+ stride (int): stride of the first block. Default: 1.
+ avg_down (bool): Use AvgPool instead of stride conv when
+ downsampling in the bottleneck. Default: False
+ conv_cfg (dict): dictionary to construct and config conv layer.
+ Default: None
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ downsample_first (bool): Downsample at the first block or last block.
+ False for Hourglass, True for ResNet. Default: True
+ kernel_size (int): Kernel Size of the corresponding convolution layer
+ searched in the block.
+ groups (int): Group number of the corresponding convolution layer
+ searched in the block.
+ attention (bool): Whether to use attention module in the end of the
+ block.
+ """
+
+ def __init__(self,
+ block,
+ num_blocks,
+ in_channels,
+ out_channels,
+ expansion=None,
+ stride=1,
+ avg_down=False,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN'),
+ downsample_first=True,
+ kernel_size=3,
+ groups=1,
+ attention=False,
+ **kwargs):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ self.block = block
+ self.expansion = get_expansion(block, expansion)
+
+ downsample = None
+ if stride != 1 or in_channels != out_channels:
+ downsample = []
+ conv_stride = stride
+ if avg_down and stride != 1:
+ conv_stride = 1
+ downsample.append(
+ nn.AvgPool2d(
+ kernel_size=stride,
+ stride=stride,
+ ceil_mode=True,
+ count_include_pad=False))
+ downsample.extend([
+ build_conv_layer(
+ conv_cfg,
+ in_channels,
+ out_channels,
+ kernel_size=1,
+ stride=conv_stride,
+ bias=False),
+ build_norm_layer(norm_cfg, out_channels)[1]
+ ])
+ downsample = nn.Sequential(*downsample)
+
+ layers = []
+ if downsample_first:
+ layers.append(
+ block(
+ in_channels=in_channels,
+ out_channels=out_channels,
+ expansion=self.expansion,
+ stride=stride,
+ downsample=downsample,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ kernel_size=kernel_size,
+ groups=groups,
+ attention=attention,
+ **kwargs))
+ in_channels = out_channels
+ for _ in range(1, num_blocks):
+ layers.append(
+ block(
+ in_channels=in_channels,
+ out_channels=out_channels,
+ expansion=self.expansion,
+ stride=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ kernel_size=kernel_size,
+ groups=groups,
+ attention=attention,
+ **kwargs))
+ else: # downsample_first=False is for HourglassModule
+ for i in range(0, num_blocks - 1):
+ layers.append(
+ block(
+ in_channels=in_channels,
+ out_channels=in_channels,
+ expansion=self.expansion,
+ stride=1,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ kernel_size=kernel_size,
+ groups=groups,
+ attention=attention,
+ **kwargs))
+ layers.append(
+ block(
+ in_channels=in_channels,
+ out_channels=out_channels,
+ expansion=self.expansion,
+ stride=stride,
+ downsample=downsample,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ kernel_size=kernel_size,
+ groups=groups,
+ attention=attention,
+ **kwargs))
+
+ super().__init__(*layers)
+
+
+@BACKBONES.register_module()
+class ViPNAS_ResNet(BaseBackbone):
+ """ViPNAS_ResNet backbone.
+
+ "ViPNAS: Efficient Video Pose Estimation via Neural Architecture Search"
+ More details can be found in the `paper
+ `__ .
+
+ Args:
+ depth (int): Network depth, from {18, 34, 50, 101, 152}.
+ in_channels (int): Number of input image channels. Default: 3.
+ num_stages (int): Stages of the network. Default: 4.
+ strides (Sequence[int]): Strides of the first block of each stage.
+ Default: ``(1, 2, 2, 2)``.
+ dilations (Sequence[int]): Dilation of each stage.
+ Default: ``(1, 1, 1, 1)``.
+ out_indices (Sequence[int]): Output from which stages. If only one
+ stage is specified, a single tensor (feature map) is returned,
+ otherwise multiple stages are specified, a tuple of tensors will
+ be returned. Default: ``(3, )``.
+ style (str): `pytorch` or `caffe`. If set to "pytorch", the stride-two
+ layer is the 3x3 conv layer, otherwise the stride-two layer is
+ the first 1x1 conv layer.
+ deep_stem (bool): Replace 7x7 conv in input stem with 3 3x3 conv.
+ Default: False.
+ avg_down (bool): Use AvgPool instead of stride conv when
+ downsampling in the bottleneck. Default: False.
+ frozen_stages (int): Stages to be frozen (stop grad and set eval mode).
+ -1 means not freezing any parameters. Default: -1.
+ conv_cfg (dict | None): The config dict for conv layers. Default: None.
+ norm_cfg (dict): The config dict for norm layers.
+ norm_eval (bool): Whether to set norm layers to eval mode, namely,
+ freeze running stats (mean and var). Note: Effect on Batch Norm
+ and its variants only. Default: False.
+ with_cp (bool): Use checkpoint or not. Using checkpoint will save some
+ memory while slowing down the training speed. Default: False.
+ zero_init_residual (bool): Whether to use zero init for last norm layer
+ in resblocks to let them behave as identity. Default: True.
+ wid (list(int)): Searched width config for each stage.
+ expan (list(int)): Searched expansion ratio config for each stage.
+ dep (list(int)): Searched depth config for each stage.
+ ks (list(int)): Searched kernel size config for each stage.
+ group (list(int)): Searched group number config for each stage.
+ att (list(bool)): Searched attention config for each stage.
+ """
+
+ arch_settings = {
+ 50: ViPNAS_Bottleneck,
+ }
+
+ def __init__(self,
+ depth,
+ in_channels=3,
+ num_stages=4,
+ strides=(1, 2, 2, 2),
+ dilations=(1, 1, 1, 1),
+ out_indices=(3, ),
+ style='pytorch',
+ deep_stem=False,
+ avg_down=False,
+ frozen_stages=-1,
+ conv_cfg=None,
+ norm_cfg=dict(type='BN', requires_grad=True),
+ norm_eval=False,
+ with_cp=False,
+ zero_init_residual=True,
+ wid=[48, 80, 160, 304, 608],
+ expan=[None, 1, 1, 1, 1],
+ dep=[None, 4, 6, 7, 3],
+ ks=[7, 3, 5, 5, 5],
+ group=[None, 16, 16, 16, 16],
+ att=[None, True, False, True, True]):
+ # Protect mutable default arguments
+ norm_cfg = copy.deepcopy(norm_cfg)
+ super().__init__()
+ if depth not in self.arch_settings:
+ raise KeyError(f'invalid depth {depth} for resnet')
+ self.depth = depth
+ self.stem_channels = dep[0]
+ self.num_stages = num_stages
+ assert 1 <= num_stages <= 4
+ self.strides = strides
+ self.dilations = dilations
+ assert len(strides) == len(dilations) == num_stages
+ self.out_indices = out_indices
+ assert max(out_indices) < num_stages
+ self.style = style
+ self.deep_stem = deep_stem
+ self.avg_down = avg_down
+ self.frozen_stages = frozen_stages
+ self.conv_cfg = conv_cfg
+ self.norm_cfg = norm_cfg
+ self.with_cp = with_cp
+ self.norm_eval = norm_eval
+ self.zero_init_residual = zero_init_residual
+ self.block = self.arch_settings[depth]
+ self.stage_blocks = dep[1:1 + num_stages]
+
+ self._make_stem_layer(in_channels, wid[0], ks[0])
+
+ self.res_layers = []
+ _in_channels = wid[0]
+ for i, num_blocks in enumerate(self.stage_blocks):
+ expansion = get_expansion(self.block, expan[i + 1])
+ _out_channels = wid[i + 1] * expansion
+ stride = strides[i]
+ dilation = dilations[i]
+ res_layer = self.make_res_layer(
+ block=self.block,
+ num_blocks=num_blocks,
+ in_channels=_in_channels,
+ out_channels=_out_channels,
+ expansion=expansion,
+ stride=stride,
+ dilation=dilation,
+ style=self.style,
+ avg_down=self.avg_down,
+ with_cp=with_cp,
+ conv_cfg=conv_cfg,
+ norm_cfg=norm_cfg,
+ kernel_size=ks[i + 1],
+ groups=group[i + 1],
+ attention=att[i + 1])
+ _in_channels = _out_channels
+ layer_name = f'layer{i + 1}'
+ self.add_module(layer_name, res_layer)
+ self.res_layers.append(layer_name)
+
+ self._freeze_stages()
+
+ self.feat_dim = res_layer[-1].out_channels
+
+ def make_res_layer(self, **kwargs):
+ """Make a ViPNAS ResLayer."""
+ return ViPNAS_ResLayer(**kwargs)
+
+ @property
+ def norm1(self):
+ """nn.Module: the normalization layer named "norm1" """
+ return getattr(self, self.norm1_name)
+
+ def _make_stem_layer(self, in_channels, stem_channels, kernel_size):
+ """Make stem layer."""
+ if self.deep_stem:
+ self.stem = nn.Sequential(
+ ConvModule(
+ in_channels,
+ stem_channels // 2,
+ kernel_size=3,
+ stride=2,
+ padding=1,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ inplace=True),
+ ConvModule(
+ stem_channels // 2,
+ stem_channels // 2,
+ kernel_size=3,
+ stride=1,
+ padding=1,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ inplace=True),
+ ConvModule(
+ stem_channels // 2,
+ stem_channels,
+ kernel_size=3,
+ stride=1,
+ padding=1,
+ conv_cfg=self.conv_cfg,
+ norm_cfg=self.norm_cfg,
+ inplace=True))
+ else:
+ self.conv1 = build_conv_layer(
+ self.conv_cfg,
+ in_channels,
+ stem_channels,
+ kernel_size=kernel_size,
+ stride=2,
+ padding=kernel_size // 2,
+ bias=False)
+ self.norm1_name, norm1 = build_norm_layer(
+ self.norm_cfg, stem_channels, postfix=1)
+ self.add_module(self.norm1_name, norm1)
+ self.relu = nn.ReLU(inplace=True)
+ self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)
+
+ def _freeze_stages(self):
+ """Freeze parameters."""
+ if self.frozen_stages >= 0:
+ if self.deep_stem:
+ self.stem.eval()
+ for param in self.stem.parameters():
+ param.requires_grad = False
+ else:
+ self.norm1.eval()
+ for m in [self.conv1, self.norm1]:
+ for param in m.parameters():
+ param.requires_grad = False
+
+ for i in range(1, self.frozen_stages + 1):
+ m = getattr(self, f'layer{i}')
+ m.eval()
+ for param in m.parameters():
+ param.requires_grad = False
+
+ def init_weights(self, pretrained=None):
+ """Initialize model weights."""
+ super().init_weights(pretrained)
+ if pretrained is None:
+ for m in self.modules():
+ if isinstance(m, nn.Conv2d):
+ nn.init.normal_(m.weight, std=0.001)
+ for name, _ in m.named_parameters():
+ if name in ['bias']:
+ nn.init.constant_(m.bias, 0)
+ elif isinstance(m, nn.BatchNorm2d):
+ nn.init.constant_(m.weight, 1)
+ nn.init.constant_(m.bias, 0)
+
+ def forward(self, x):
+ """Forward function."""
+ if self.deep_stem:
+ x = self.stem(x)
+ else:
+ x = self.conv1(x)
+ x = self.norm1(x)
+ x = self.relu(x)
+ x = self.maxpool(x)
+ outs = []
+ for i, layer_name in enumerate(self.res_layers):
+ res_layer = getattr(self, layer_name)
+ x = res_layer(x)
+ if i in self.out_indices:
+ outs.append(x)
+ if len(outs) == 1:
+ return outs[0]
+ return tuple(outs)
+
+ def train(self, mode=True):
+ """Convert the model into training mode."""
+ super().train(mode)
+ self._freeze_stages()
+ if mode and self.norm_eval:
+ for m in self.modules():
+ # trick: eval have effect on BatchNorm only
+ if isinstance(m, _BatchNorm):
+ m.eval()
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vit.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vit.py
new file mode 100644
index 0000000000000000000000000000000000000000..465dfad7c614145ed18c6f5e8095b799aef2755d
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/backbones/vit.py
@@ -0,0 +1,308 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import math
+
+import torch
+from functools import partial
+import torch.nn as nn
+import torch.nn.functional as F
+import torch.utils.checkpoint as checkpoint
+
+from timm.models.layers import drop_path, to_2tuple, trunc_normal_
+
+# from ..builder import BACKBONES
+# from .base_backbone import BaseBackbone
+
+class DropPath(nn.Module):
+ """Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks).
+ """
+ def __init__(self, drop_prob=None):
+ super(DropPath, self).__init__()
+ self.drop_prob = drop_prob
+
+ def forward(self, x):
+ return drop_path(x, self.drop_prob, self.training)
+
+ def extra_repr(self):
+ return 'p={}'.format(self.drop_prob)
+
+class Mlp(nn.Module):
+ def __init__(self, in_features, hidden_features=None, out_features=None, act_layer=nn.GELU, drop=0.):
+ super().__init__()
+ out_features = out_features or in_features
+ hidden_features = hidden_features or in_features
+ self.fc1 = nn.Linear(in_features, hidden_features)
+ self.act = act_layer()
+ self.fc2 = nn.Linear(hidden_features, out_features)
+ self.drop = nn.Dropout(drop)
+
+ def forward(self, x):
+ x = self.fc1(x)
+ x = self.act(x)
+ x = self.fc2(x)
+ x = self.drop(x)
+ return x
+
+class Attention(nn.Module):
+ def __init__(
+ self, dim, num_heads=8, qkv_bias=False, qk_scale=None, attn_drop=0.,
+ proj_drop=0., attn_head_dim=None,):
+ super().__init__()
+ self.num_heads = num_heads
+ head_dim = dim // num_heads
+ self.dim = dim
+
+ if attn_head_dim is not None:
+ head_dim = attn_head_dim
+ all_head_dim = head_dim * self.num_heads
+
+ self.scale = qk_scale or head_dim ** -0.5
+
+ self.qkv = nn.Linear(dim, all_head_dim * 3, bias=qkv_bias)
+
+ self.attn_drop = nn.Dropout(attn_drop)
+ self.proj = nn.Linear(all_head_dim, dim)
+ self.proj_drop = nn.Dropout(proj_drop)
+
+ def forward(self, x):
+ B, N, C = x.shape
+ qkv = self.qkv(x)
+ qkv = qkv.reshape(B, N, 3, self.num_heads, -1).permute(2, 0, 3, 1, 4)
+ q, k, v = qkv[0], qkv[1], qkv[2] # make torchscript happy (cannot use tensor as tuple)
+
+ q = q * self.scale
+ attn = (q @ k.transpose(-2, -1))
+
+ attn = attn.softmax(dim=-1)
+ attn = self.attn_drop(attn)
+
+ x = (attn @ v).transpose(1, 2).reshape(B, N, -1)
+ x = self.proj(x)
+ x = self.proj_drop(x)
+
+ return x
+
+class Block(nn.Module):
+
+ def __init__(self, dim, num_heads, mlp_ratio=4., qkv_bias=False, qk_scale=None,
+ drop=0., attn_drop=0., drop_path=0., act_layer=nn.GELU,
+ norm_layer=nn.LayerNorm, attn_head_dim=None
+ ):
+ super().__init__()
+
+ self.norm1 = norm_layer(dim)
+ self.attn = Attention(
+ dim, num_heads=num_heads, qkv_bias=qkv_bias, qk_scale=qk_scale,
+ attn_drop=attn_drop, proj_drop=drop, attn_head_dim=attn_head_dim
+ )
+
+ # NOTE: drop path for stochastic depth, we shall see if this is better than dropout here
+ self.drop_path = DropPath(drop_path) if drop_path > 0. else nn.Identity()
+ self.norm2 = norm_layer(dim)
+ mlp_hidden_dim = int(dim * mlp_ratio)
+ self.mlp = Mlp(in_features=dim, hidden_features=mlp_hidden_dim, act_layer=act_layer, drop=drop)
+
+ def forward(self, x):
+ x = x + self.drop_path(self.attn(self.norm1(x)))
+ x = x + self.drop_path(self.mlp(self.norm2(x)))
+ return x
+
+
+class PatchEmbed(nn.Module):
+ """ Image to Patch Embedding
+ """
+ def __init__(self, img_size=224, patch_size=16, in_chans=3, embed_dim=768, ratio=1):
+ super().__init__()
+ img_size = to_2tuple(img_size)
+ patch_size = to_2tuple(patch_size)
+ num_patches = (img_size[1] // patch_size[1]) * (img_size[0] // patch_size[0]) * (ratio ** 2)
+ self.patch_shape = (int(img_size[0] // patch_size[0] * ratio), int(img_size[1] // patch_size[1] * ratio))
+ self.origin_patch_shape = (int(img_size[0] // patch_size[0]), int(img_size[1] // patch_size[1]))
+ self.img_size = img_size
+ self.patch_size = patch_size
+ self.num_patches = num_patches
+
+ self.proj = nn.Conv2d(in_chans, embed_dim, kernel_size=patch_size, stride=(patch_size[0] // ratio), padding=4 + 2 * (ratio//2-1))
+
+ def forward(self, x, **kwargs):
+ B, C, H, W = x.shape
+ x = self.proj(x)
+ Hp, Wp = x.shape[2], x.shape[3]
+
+ x = x.flatten(2).transpose(1, 2)
+ return x, (Hp, Wp)
+
+
+class HybridEmbed(nn.Module):
+ """ CNN Feature Map Embedding
+ Extract feature map from CNN, flatten, project to embedding dim.
+ """
+ def __init__(self, backbone, img_size=224, feature_size=None, in_chans=3, embed_dim=768):
+ super().__init__()
+ assert isinstance(backbone, nn.Module)
+ img_size = to_2tuple(img_size)
+ self.img_size = img_size
+ self.backbone = backbone
+ if feature_size is None:
+ with torch.no_grad():
+ training = backbone.training
+ if training:
+ backbone.eval()
+ o = self.backbone(torch.zeros(1, in_chans, img_size[0], img_size[1]))[-1]
+ feature_size = o.shape[-2:]
+ feature_dim = o.shape[1]
+ backbone.train(training)
+ else:
+ feature_size = to_2tuple(feature_size)
+ feature_dim = self.backbone.feature_info.channels()[-1]
+ self.num_patches = feature_size[0] * feature_size[1]
+ self.proj = nn.Linear(feature_dim, embed_dim)
+
+ def forward(self, x):
+ x = self.backbone(x)[-1]
+ x = x.flatten(2).transpose(1, 2)
+ x = self.proj(x)
+ return x
+
+
+# @BACKBONES.register_module()
+class ViT(nn.Module):
+
+ def __init__(self,
+ img_size=224, patch_size=16, in_chans=3, num_classes=80, embed_dim=768, depth=12,
+ num_heads=12, mlp_ratio=4., qkv_bias=False, qk_scale=None, drop_rate=0., attn_drop_rate=0.,
+ drop_path_rate=0., hybrid_backbone=None, norm_layer=None, use_checkpoint=False,
+ frozen_stages=-1, ratio=1, last_norm=True,
+ patch_padding='pad', freeze_attn=False, freeze_ffn=False,
+ ):
+ # Protect mutable default arguments
+ super(ViT, self).__init__()
+ norm_layer = norm_layer or partial(nn.LayerNorm, eps=1e-6)
+ self.num_classes = num_classes
+ self.num_features = self.embed_dim = embed_dim # num_features for consistency with other models
+ self.frozen_stages = frozen_stages
+ self.use_checkpoint = use_checkpoint
+ self.patch_padding = patch_padding
+ self.freeze_attn = freeze_attn
+ self.freeze_ffn = freeze_ffn
+ self.depth = depth
+
+ if hybrid_backbone is not None:
+ self.patch_embed = HybridEmbed(
+ hybrid_backbone, img_size=img_size, in_chans=in_chans, embed_dim=embed_dim)
+ else:
+ self.patch_embed = PatchEmbed(
+ img_size=img_size, patch_size=patch_size, in_chans=in_chans, embed_dim=embed_dim, ratio=ratio)
+ num_patches = self.patch_embed.num_patches
+
+ # since the pretraining model has class token
+ self.pos_embed = nn.Parameter(torch.zeros(1, num_patches + 1, embed_dim))
+
+ dpr = [x.item() for x in torch.linspace(0, drop_path_rate, depth)] # stochastic depth decay rule
+
+ self.blocks = nn.ModuleList([
+ Block(
+ dim=embed_dim, num_heads=num_heads, mlp_ratio=mlp_ratio, qkv_bias=qkv_bias, qk_scale=qk_scale,
+ drop=drop_rate, attn_drop=attn_drop_rate, drop_path=dpr[i], norm_layer=norm_layer,
+ )
+ for i in range(depth)])
+
+ self.last_norm = norm_layer(embed_dim) if last_norm else nn.Identity()
+
+ if self.pos_embed is not None:
+ trunc_normal_(self.pos_embed, std=.02)
+
+ self._freeze_stages()
+
+ def _freeze_stages(self):
+ """Freeze parameters."""
+ if self.frozen_stages >= 0:
+ self.patch_embed.eval()
+ for param in self.patch_embed.parameters():
+ param.requires_grad = False
+
+ for i in range(1, self.frozen_stages + 1):
+ m = self.blocks[i]
+ m.eval()
+ for param in m.parameters():
+ param.requires_grad = False
+
+ if self.freeze_attn:
+ for i in range(0, self.depth):
+ m = self.blocks[i]
+ m.attn.eval()
+ m.norm1.eval()
+ for param in m.attn.parameters():
+ param.requires_grad = False
+ for param in m.norm1.parameters():
+ param.requires_grad = False
+
+ if self.freeze_ffn:
+ self.pos_embed.requires_grad = False
+ self.patch_embed.eval()
+ for param in self.patch_embed.parameters():
+ param.requires_grad = False
+ for i in range(0, self.depth):
+ m = self.blocks[i]
+ m.mlp.eval()
+ m.norm2.eval()
+ for param in m.mlp.parameters():
+ param.requires_grad = False
+ for param in m.norm2.parameters():
+ param.requires_grad = False
+
+ def init_weights(self, pretrained=None):
+ """Initialize the weights in backbone.
+ Args:
+ pretrained (str, optional): Path to pre-trained weights.
+ Defaults to None.
+ """
+ super().init_weights(pretrained, patch_padding=self.patch_padding)
+
+ if pretrained is None:
+ def _init_weights(m):
+ if isinstance(m, nn.Linear):
+ trunc_normal_(m.weight, std=.02)
+ if isinstance(m, nn.Linear) and m.bias is not None:
+ nn.init.constant_(m.bias, 0)
+ elif isinstance(m, nn.LayerNorm):
+ nn.init.constant_(m.bias, 0)
+ nn.init.constant_(m.weight, 1.0)
+
+ self.apply(_init_weights)
+
+ def get_num_layers(self):
+ return len(self.blocks)
+
+ @torch.jit.ignore
+ def no_weight_decay(self):
+ return {'pos_embed', 'cls_token'}
+
+ def forward_features(self, x):
+ B, C, H, W = x.shape
+ x, (Hp, Wp) = self.patch_embed(x)
+
+ if self.pos_embed is not None:
+ # fit for multiple GPU training
+ # since the first element for pos embed (sin-cos manner) is zero, it will cause no difference
+ x = x + self.pos_embed[:, 1:] + self.pos_embed[:, :1]
+
+ for blk in self.blocks:
+ if self.use_checkpoint:
+ x = checkpoint.checkpoint(blk, x)
+ else:
+ x = blk(x)
+
+ x = self.last_norm(x)
+
+ xp = x.permute(0, 2, 1).reshape(B, -1, Hp, Wp).contiguous()
+
+ return xp
+
+ def forward(self, x):
+ x = self.forward_features(x)
+ return x
+
+ def train(self, mode=True):
+ """Convert the model into training mode."""
+ super().train(mode)
+ self._freeze_stages()
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_base_coco_256x192.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_base_coco_256x192.py
new file mode 100644
index 0000000000000000000000000000000000000000..de927db3841c2779f99252c9436b139284ede6c6
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_base_coco_256x192.py
@@ -0,0 +1,168 @@
+_base_ = [
+ '../../../../_base_/default_runtime.py',
+ '../../../../_base_/datasets/coco.py'
+]
+evaluation = dict(interval=10, metric='mAP', save_best='AP')
+
+optimizer = dict(type='AdamW', lr=5e-4, betas=(0.9, 0.999), weight_decay=0.1,
+ constructor='LayerDecayOptimizerConstructor',
+ paramwise_cfg=dict(
+ num_layers=12,
+ layer_decay_rate=0.75,
+ custom_keys={
+ 'bias': dict(decay_multi=0.),
+ 'pos_embed': dict(decay_mult=0.),
+ 'relative_position_bias_table': dict(decay_mult=0.),
+ 'norm': dict(decay_mult=0.)
+ }
+ )
+ )
+
+optimizer_config = dict(grad_clip=dict(max_norm=1., norm_type=2))
+
+# learning policy
+lr_config = dict(
+ policy='step',
+ warmup='linear',
+ warmup_iters=500,
+ warmup_ratio=0.001,
+ step=[170, 200])
+total_epochs = 210
+target_type = 'GaussianHeatmap'
+channel_cfg = dict(
+ num_output_channels=17,
+ dataset_joints=17,
+ dataset_channel=[
+ [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16],
+ ],
+ inference_channel=[
+ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16
+ ])
+
+# model settings
+model = dict(
+ type='TopDown',
+ pretrained=None,
+ backbone=dict(
+ type='ViT',
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=768,
+ depth=12,
+ num_heads=12,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.3,
+ ),
+ keypoint_head=dict(
+ type='TopdownHeatmapSimpleHead',
+ in_channels=768,
+ num_deconv_layers=2,
+ num_deconv_filters=(256, 256),
+ num_deconv_kernels=(4, 4),
+ extra=dict(final_conv_kernel=1, ),
+ out_channels=channel_cfg['num_output_channels'],
+ loss_keypoint=dict(type='JointsMSELoss', use_target_weight=True)),
+ train_cfg=dict(),
+ test_cfg=dict())
+
+data_cfg = dict(
+ image_size=[192, 256],
+ heatmap_size=[48, 64],
+ num_output_channels=channel_cfg['num_output_channels'],
+ num_joints=channel_cfg['dataset_joints'],
+ dataset_channel=channel_cfg['dataset_channel'],
+ inference_channel=channel_cfg['inference_channel'],
+ soft_nms=False,
+ nms_thr=1.0,
+ oks_thr=0.9,
+ vis_thr=0.2,
+ use_gt_bbox=False,
+ det_bbox_thr=0.0,
+ bbox_file='data/coco/person_detection_results/'
+ 'COCO_val2017_detections_AP_H_56_person.json',
+)
+
+train_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownRandomFlip', flip_prob=0.5),
+ dict(
+ type='TopDownHalfBodyTransform',
+ num_joints_half_body=8,
+ prob_half_body=0.3),
+ dict(
+ type='TopDownGetRandomScaleRotation', rot_factor=40, scale_factor=0.5),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='TopDownGenerateTarget',
+ sigma=2,
+ encoding='UDP',
+ target_type=target_type),
+ dict(
+ type='Collect',
+ keys=['img', 'target', 'target_weight'],
+ meta_keys=[
+ 'image_file', 'joints_3d', 'joints_3d_visible', 'center', 'scale',
+ 'rotation', 'bbox_score', 'flip_pairs'
+ ]),
+]
+
+val_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='Collect',
+ keys=['img'],
+ meta_keys=[
+ 'image_file', 'center', 'scale', 'rotation', 'bbox_score',
+ 'flip_pairs'
+ ]),
+]
+
+test_pipeline = val_pipeline
+
+data_root = 'data/coco'
+# data = dict(
+# samples_per_gpu=64,
+# workers_per_gpu=4,
+# val_dataloader=dict(samples_per_gpu=32),
+# test_dataloader=dict(samples_per_gpu=32),
+# train=dict(
+# type='TopDownCocoDataset',
+# ann_file=f'{data_root}/annotations/person_keypoints_train2017.json',
+# img_prefix=f'{data_root}/train2017/',
+# data_cfg=data_cfg,
+# pipeline=train_pipeline,
+# dataset_info={{_base_.dataset_info}}),
+# val=dict(
+# type='TopDownCocoDataset',
+# ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+# img_prefix=f'{data_root}/val2017/',
+# data_cfg=data_cfg,
+# pipeline=val_pipeline,
+# dataset_info={{_base_.dataset_info}}),
+# test=dict(
+# type='TopDownCocoDataset',
+# ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+# img_prefix=f'{data_root}/val2017/',
+# data_cfg=data_cfg,
+# pipeline=test_pipeline,
+# dataset_info={{_base_.dataset_info}}),
+# )
+
+def make_cfg(model=model,data_cfg=data_cfg):
+ cfg={}
+ cfg['model'] = model
+ cfg['data_cfg'] = data_cfg
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_base_simple_coco_256x192.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_base_simple_coco_256x192.py
new file mode 100644
index 0000000000000000000000000000000000000000..d410a1534f35d0bcd1f9d01f408748081576a2b5
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_base_simple_coco_256x192.py
@@ -0,0 +1,171 @@
+_base_ = [
+ '../../../../_base_/default_runtime.py',
+ '../../../../_base_/datasets/coco.py'
+]
+evaluation = dict(interval=10, metric='mAP', save_best='AP')
+
+optimizer = dict(type='AdamW', lr=5e-4, betas=(0.9, 0.999), weight_decay=0.1,
+ constructor='LayerDecayOptimizerConstructor',
+ paramwise_cfg=dict(
+ num_layers=12,
+ layer_decay_rate=0.75,
+ custom_keys={
+ 'bias': dict(decay_multi=0.),
+ 'pos_embed': dict(decay_mult=0.),
+ 'relative_position_bias_table': dict(decay_mult=0.),
+ 'norm': dict(decay_mult=0.)
+ }
+ )
+ )
+
+optimizer_config = dict(grad_clip=dict(max_norm=1., norm_type=2))
+
+# learning policy
+lr_config = dict(
+ policy='step',
+ warmup='linear',
+ warmup_iters=500,
+ warmup_ratio=0.001,
+ step=[170, 200])
+total_epochs = 210
+target_type = 'GaussianHeatmap'
+channel_cfg = dict(
+ num_output_channels=17,
+ dataset_joints=17,
+ dataset_channel=[
+ [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16],
+ ],
+ inference_channel=[
+ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16
+ ])
+
+# model settings
+model = dict(
+ type='TopDown',
+ pretrained=None,
+ backbone=dict(
+ type='ViT',
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=768,
+ depth=12,
+ num_heads=12,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.3,
+ ),
+ keypoint_head=dict(
+ type='TopdownHeatmapSimpleHead',
+ in_channels=768,
+ num_deconv_layers=0,
+ num_deconv_filters=[],
+ num_deconv_kernels=[],
+ upsample=4,
+ extra=dict(final_conv_kernel=3, ),
+ out_channels=channel_cfg['num_output_channels'],
+ loss_keypoint=dict(type='JointsMSELoss', use_target_weight=True)),
+ train_cfg=dict(),
+ test_cfg=dict(
+ flip_test=True,
+ post_process='default',
+ shift_heatmap=False,
+ target_type=target_type,
+ modulate_kernel=11,
+ use_udp=True))
+
+data_cfg = dict(
+ image_size=[192, 256],
+ heatmap_size=[48, 64],
+ num_output_channels=channel_cfg['num_output_channels'],
+ num_joints=channel_cfg['dataset_joints'],
+ dataset_channel=channel_cfg['dataset_channel'],
+ inference_channel=channel_cfg['inference_channel'],
+ soft_nms=False,
+ nms_thr=1.0,
+ oks_thr=0.9,
+ vis_thr=0.2,
+ use_gt_bbox=False,
+ det_bbox_thr=0.0,
+ bbox_file='data/coco/person_detection_results/'
+ 'COCO_val2017_detections_AP_H_56_person.json',
+)
+
+train_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownRandomFlip', flip_prob=0.5),
+ dict(
+ type='TopDownHalfBodyTransform',
+ num_joints_half_body=8,
+ prob_half_body=0.3),
+ dict(
+ type='TopDownGetRandomScaleRotation', rot_factor=40, scale_factor=0.5),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='TopDownGenerateTarget',
+ sigma=2,
+ encoding='UDP',
+ target_type=target_type),
+ dict(
+ type='Collect',
+ keys=['img', 'target', 'target_weight'],
+ meta_keys=[
+ 'image_file', 'joints_3d', 'joints_3d_visible', 'center', 'scale',
+ 'rotation', 'bbox_score', 'flip_pairs'
+ ]),
+]
+
+val_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='Collect',
+ keys=['img'],
+ meta_keys=[
+ 'image_file', 'center', 'scale', 'rotation', 'bbox_score',
+ 'flip_pairs'
+ ]),
+]
+
+test_pipeline = val_pipeline
+
+data_root = 'data/coco'
+data = dict(
+ samples_per_gpu=64,
+ workers_per_gpu=4,
+ val_dataloader=dict(samples_per_gpu=32),
+ test_dataloader=dict(samples_per_gpu=32),
+ train=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_train2017.json',
+ img_prefix=f'{data_root}/train2017/',
+ data_cfg=data_cfg,
+ pipeline=train_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+ val=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+ img_prefix=f'{data_root}/val2017/',
+ data_cfg=data_cfg,
+ pipeline=val_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+ test=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+ img_prefix=f'{data_root}/val2017/',
+ data_cfg=data_cfg,
+ pipeline=test_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+)
+
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_huge_coco_256x192.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_huge_coco_256x192.py
new file mode 100644
index 0000000000000000000000000000000000000000..298b2b59ef8310c73d481e95eb9fa39a8d0a7fef
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_huge_coco_256x192.py
@@ -0,0 +1,170 @@
+_base_ = [
+ '../../../../_base_/default_runtime.py',
+ '../../../../_base_/datasets/coco.py'
+]
+evaluation = dict(interval=10, metric='mAP', save_best='AP')
+
+optimizer = dict(type='AdamW', lr=5e-4, betas=(0.9, 0.999), weight_decay=0.1,
+ constructor='LayerDecayOptimizerConstructor',
+ paramwise_cfg=dict(
+ num_layers=32,
+ layer_decay_rate=0.85,
+ custom_keys={
+ 'bias': dict(decay_multi=0.),
+ 'pos_embed': dict(decay_mult=0.),
+ 'relative_position_bias_table': dict(decay_mult=0.),
+ 'norm': dict(decay_mult=0.)
+ }
+ )
+ )
+
+optimizer_config = dict(grad_clip=dict(max_norm=1., norm_type=2))
+
+# learning policy
+lr_config = dict(
+ policy='step',
+ warmup='linear',
+ warmup_iters=500,
+ warmup_ratio=0.001,
+ step=[170, 200])
+total_epochs = 210
+target_type = 'GaussianHeatmap'
+channel_cfg = dict(
+ num_output_channels=17,
+ dataset_joints=17,
+ dataset_channel=[
+ [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16],
+ ],
+ inference_channel=[
+ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16
+ ])
+
+# model settings
+model = dict(
+ type='TopDown',
+ pretrained=None,
+ backbone=dict(
+ type='ViT',
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=1280,
+ depth=32,
+ num_heads=16,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.55,
+ ),
+ keypoint_head=dict(
+ type='TopdownHeatmapSimpleHead',
+ in_channels=1280,
+ num_deconv_layers=2,
+ num_deconv_filters=(256, 256),
+ num_deconv_kernels=(4, 4),
+ extra=dict(final_conv_kernel=1, ),
+ out_channels=channel_cfg['num_output_channels'],
+ loss_keypoint=dict(type='JointsMSELoss', use_target_weight=True)),
+ train_cfg=dict(),
+ test_cfg=dict(
+ flip_test=True,
+ post_process='default',
+ shift_heatmap=False,
+ target_type=target_type,
+ modulate_kernel=11,
+ use_udp=True))
+
+data_cfg = dict(
+ image_size=[192, 256],
+ heatmap_size=[48, 64],
+ num_output_channels=channel_cfg['num_output_channels'],
+ num_joints=channel_cfg['dataset_joints'],
+ dataset_channel=channel_cfg['dataset_channel'],
+ inference_channel=channel_cfg['inference_channel'],
+ soft_nms=False,
+ nms_thr=1.0,
+ oks_thr=0.9,
+ vis_thr=0.2,
+ use_gt_bbox=False,
+ det_bbox_thr=0.0,
+ bbox_file='data/coco/person_detection_results/'
+ 'COCO_val2017_detections_AP_H_56_person.json',
+)
+
+train_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownRandomFlip', flip_prob=0.5),
+ dict(
+ type='TopDownHalfBodyTransform',
+ num_joints_half_body=8,
+ prob_half_body=0.3),
+ dict(
+ type='TopDownGetRandomScaleRotation', rot_factor=40, scale_factor=0.5),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='TopDownGenerateTarget',
+ sigma=2,
+ encoding='UDP',
+ target_type=target_type),
+ dict(
+ type='Collect',
+ keys=['img', 'target', 'target_weight'],
+ meta_keys=[
+ 'image_file', 'joints_3d', 'joints_3d_visible', 'center', 'scale',
+ 'rotation', 'bbox_score', 'flip_pairs'
+ ]),
+]
+
+val_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='Collect',
+ keys=['img'],
+ meta_keys=[
+ 'image_file', 'center', 'scale', 'rotation', 'bbox_score',
+ 'flip_pairs'
+ ]),
+]
+
+test_pipeline = val_pipeline
+
+data_root = 'data/coco'
+data = dict(
+ samples_per_gpu=64,
+ workers_per_gpu=4,
+ val_dataloader=dict(samples_per_gpu=32),
+ test_dataloader=dict(samples_per_gpu=32),
+ train=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_train2017.json',
+ img_prefix=f'{data_root}/train2017/',
+ data_cfg=data_cfg,
+ pipeline=train_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+ val=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+ img_prefix=f'{data_root}/val2017/',
+ data_cfg=data_cfg,
+ pipeline=val_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+ test=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+ img_prefix=f'{data_root}/val2017/',
+ data_cfg=data_cfg,
+ pipeline=test_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+)
+
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_huge_simple_coco_256x192.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_huge_simple_coco_256x192.py
new file mode 100644
index 0000000000000000000000000000000000000000..f9a86f0290d6be4046ac5e5dc3fc22288d373775
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_huge_simple_coco_256x192.py
@@ -0,0 +1,171 @@
+_base_ = [
+ '../../../../_base_/default_runtime.py',
+ '../../../../_base_/datasets/coco.py'
+]
+evaluation = dict(interval=10, metric='mAP', save_best='AP')
+
+optimizer = dict(type='AdamW', lr=5e-4, betas=(0.9, 0.999), weight_decay=0.1,
+ constructor='LayerDecayOptimizerConstructor',
+ paramwise_cfg=dict(
+ num_layers=32,
+ layer_decay_rate=0.85,
+ custom_keys={
+ 'bias': dict(decay_multi=0.),
+ 'pos_embed': dict(decay_mult=0.),
+ 'relative_position_bias_table': dict(decay_mult=0.),
+ 'norm': dict(decay_mult=0.)
+ }
+ )
+ )
+
+optimizer_config = dict(grad_clip=dict(max_norm=1., norm_type=2))
+
+# learning policy
+lr_config = dict(
+ policy='step',
+ warmup='linear',
+ warmup_iters=500,
+ warmup_ratio=0.001,
+ step=[170, 200])
+total_epochs = 210
+target_type = 'GaussianHeatmap'
+channel_cfg = dict(
+ num_output_channels=17,
+ dataset_joints=17,
+ dataset_channel=[
+ [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16],
+ ],
+ inference_channel=[
+ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16
+ ])
+
+# model settings
+model = dict(
+ type='TopDown',
+ pretrained=None,
+ backbone=dict(
+ type='ViT',
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=1280,
+ depth=32,
+ num_heads=16,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.55,
+ ),
+ keypoint_head=dict(
+ type='TopdownHeatmapSimpleHead',
+ in_channels=1280,
+ num_deconv_layers=0,
+ num_deconv_filters=[],
+ num_deconv_kernels=[],
+ upsample=4,
+ extra=dict(final_conv_kernel=3, ),
+ out_channels=channel_cfg['num_output_channels'],
+ loss_keypoint=dict(type='JointsMSELoss', use_target_weight=True)),
+ train_cfg=dict(),
+ test_cfg=dict(
+ flip_test=True,
+ post_process='default',
+ shift_heatmap=False,
+ target_type=target_type,
+ modulate_kernel=11,
+ use_udp=True))
+
+data_cfg = dict(
+ image_size=[192, 256],
+ heatmap_size=[48, 64],
+ num_output_channels=channel_cfg['num_output_channels'],
+ num_joints=channel_cfg['dataset_joints'],
+ dataset_channel=channel_cfg['dataset_channel'],
+ inference_channel=channel_cfg['inference_channel'],
+ soft_nms=False,
+ nms_thr=1.0,
+ oks_thr=0.9,
+ vis_thr=0.2,
+ use_gt_bbox=False,
+ det_bbox_thr=0.0,
+ bbox_file='data/coco/person_detection_results/'
+ 'COCO_val2017_detections_AP_H_56_person.json',
+)
+
+train_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownRandomFlip', flip_prob=0.5),
+ dict(
+ type='TopDownHalfBodyTransform',
+ num_joints_half_body=8,
+ prob_half_body=0.3),
+ dict(
+ type='TopDownGetRandomScaleRotation', rot_factor=40, scale_factor=0.5),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='TopDownGenerateTarget',
+ sigma=2,
+ encoding='UDP',
+ target_type=target_type),
+ dict(
+ type='Collect',
+ keys=['img', 'target', 'target_weight'],
+ meta_keys=[
+ 'image_file', 'joints_3d', 'joints_3d_visible', 'center', 'scale',
+ 'rotation', 'bbox_score', 'flip_pairs'
+ ]),
+]
+
+val_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='Collect',
+ keys=['img'],
+ meta_keys=[
+ 'image_file', 'center', 'scale', 'rotation', 'bbox_score',
+ 'flip_pairs'
+ ]),
+]
+
+test_pipeline = val_pipeline
+
+data_root = 'data/coco'
+data = dict(
+ samples_per_gpu=64,
+ workers_per_gpu=4,
+ val_dataloader=dict(samples_per_gpu=32),
+ test_dataloader=dict(samples_per_gpu=32),
+ train=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_train2017.json',
+ img_prefix=f'{data_root}/train2017/',
+ data_cfg=data_cfg,
+ pipeline=train_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+ val=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+ img_prefix=f'{data_root}/val2017/',
+ data_cfg=data_cfg,
+ pipeline=val_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+ test=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+ img_prefix=f'{data_root}/val2017/',
+ data_cfg=data_cfg,
+ pipeline=test_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+)
+
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_large_coco_256x192.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_large_coco_256x192.py
new file mode 100644
index 0000000000000000000000000000000000000000..0753a3cb2d48f8a8bc50f37bc25e90399494fdea
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_large_coco_256x192.py
@@ -0,0 +1,170 @@
+_base_ = [
+ '../../../../_base_/default_runtime.py',
+ '../../../../_base_/datasets/coco.py'
+]
+evaluation = dict(interval=10, metric='mAP', save_best='AP')
+
+optimizer = dict(type='AdamW', lr=5e-4, betas=(0.9, 0.999), weight_decay=0.1,
+ constructor='LayerDecayOptimizerConstructor',
+ paramwise_cfg=dict(
+ num_layers=16,
+ layer_decay_rate=0.8,
+ custom_keys={
+ 'bias': dict(decay_multi=0.),
+ 'pos_embed': dict(decay_mult=0.),
+ 'relative_position_bias_table': dict(decay_mult=0.),
+ 'norm': dict(decay_mult=0.)
+ }
+ )
+ )
+
+optimizer_config = dict(grad_clip=dict(max_norm=1., norm_type=2))
+
+# learning policy
+lr_config = dict(
+ policy='step',
+ warmup='linear',
+ warmup_iters=500,
+ warmup_ratio=0.001,
+ step=[170, 200])
+total_epochs = 210
+target_type = 'GaussianHeatmap'
+channel_cfg = dict(
+ num_output_channels=17,
+ dataset_joints=17,
+ dataset_channel=[
+ [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16],
+ ],
+ inference_channel=[
+ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16
+ ])
+
+# model settings
+model = dict(
+ type='TopDown',
+ pretrained=None,
+ backbone=dict(
+ type='ViT',
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=1024,
+ depth=24,
+ num_heads=16,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.5,
+ ),
+ keypoint_head=dict(
+ type='TopdownHeatmapSimpleHead',
+ in_channels=1024,
+ num_deconv_layers=2,
+ num_deconv_filters=(256, 256),
+ num_deconv_kernels=(4, 4),
+ extra=dict(final_conv_kernel=1, ),
+ out_channels=channel_cfg['num_output_channels'],
+ loss_keypoint=dict(type='JointsMSELoss', use_target_weight=True)),
+ train_cfg=dict(),
+ test_cfg=dict(
+ flip_test=True,
+ post_process='default',
+ shift_heatmap=False,
+ target_type=target_type,
+ modulate_kernel=11,
+ use_udp=True))
+
+data_cfg = dict(
+ image_size=[192, 256],
+ heatmap_size=[48, 64],
+ num_output_channels=channel_cfg['num_output_channels'],
+ num_joints=channel_cfg['dataset_joints'],
+ dataset_channel=channel_cfg['dataset_channel'],
+ inference_channel=channel_cfg['inference_channel'],
+ soft_nms=False,
+ nms_thr=1.0,
+ oks_thr=0.9,
+ vis_thr=0.2,
+ use_gt_bbox=False,
+ det_bbox_thr=0.0,
+ bbox_file='data/coco/person_detection_results/'
+ 'COCO_val2017_detections_AP_H_56_person.json',
+)
+
+train_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownRandomFlip', flip_prob=0.5),
+ dict(
+ type='TopDownHalfBodyTransform',
+ num_joints_half_body=8,
+ prob_half_body=0.3),
+ dict(
+ type='TopDownGetRandomScaleRotation', rot_factor=40, scale_factor=0.5),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='TopDownGenerateTarget',
+ sigma=2,
+ encoding='UDP',
+ target_type=target_type),
+ dict(
+ type='Collect',
+ keys=['img', 'target', 'target_weight'],
+ meta_keys=[
+ 'image_file', 'joints_3d', 'joints_3d_visible', 'center', 'scale',
+ 'rotation', 'bbox_score', 'flip_pairs'
+ ]),
+]
+
+val_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='Collect',
+ keys=['img'],
+ meta_keys=[
+ 'image_file', 'center', 'scale', 'rotation', 'bbox_score',
+ 'flip_pairs'
+ ]),
+]
+
+test_pipeline = val_pipeline
+
+data_root = 'data/coco'
+data = dict(
+ samples_per_gpu=64,
+ workers_per_gpu=4,
+ val_dataloader=dict(samples_per_gpu=32),
+ test_dataloader=dict(samples_per_gpu=32),
+ train=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_train2017.json',
+ img_prefix=f'{data_root}/train2017/',
+ data_cfg=data_cfg,
+ pipeline=train_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+ val=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+ img_prefix=f'{data_root}/val2017/',
+ data_cfg=data_cfg,
+ pipeline=val_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+ test=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+ img_prefix=f'{data_root}/val2017/',
+ data_cfg=data_cfg,
+ pipeline=test_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+)
+
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_large_simple_coco_256x192.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_large_simple_coco_256x192.py
new file mode 100644
index 0000000000000000000000000000000000000000..63c794940805fbc67836f3b0a3ff3d029a7991ac
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/ViTPose_large_simple_coco_256x192.py
@@ -0,0 +1,171 @@
+_base_ = [
+ '../../../../_base_/default_runtime.py',
+ '../../../../_base_/datasets/coco.py'
+]
+evaluation = dict(interval=10, metric='mAP', save_best='AP')
+
+optimizer = dict(type='AdamW', lr=5e-4, betas=(0.9, 0.999), weight_decay=0.1,
+ constructor='LayerDecayOptimizerConstructor',
+ paramwise_cfg=dict(
+ num_layers=24,
+ layer_decay_rate=0.8,
+ custom_keys={
+ 'bias': dict(decay_multi=0.),
+ 'pos_embed': dict(decay_mult=0.),
+ 'relative_position_bias_table': dict(decay_mult=0.),
+ 'norm': dict(decay_mult=0.)
+ }
+ )
+ )
+
+optimizer_config = dict(grad_clip=dict(max_norm=1., norm_type=2))
+
+# learning policy
+lr_config = dict(
+ policy='step',
+ warmup='linear',
+ warmup_iters=500,
+ warmup_ratio=0.001,
+ step=[170, 200])
+total_epochs = 210
+target_type = 'GaussianHeatmap'
+channel_cfg = dict(
+ num_output_channels=17,
+ dataset_joints=17,
+ dataset_channel=[
+ [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16],
+ ],
+ inference_channel=[
+ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16
+ ])
+
+# model settings
+model = dict(
+ type='TopDown',
+ pretrained=None,
+ backbone=dict(
+ type='ViT',
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=1024,
+ depth=24,
+ num_heads=16,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.5,
+ ),
+ keypoint_head=dict(
+ type='TopdownHeatmapSimpleHead',
+ in_channels=1024,
+ num_deconv_layers=0,
+ num_deconv_filters=[],
+ num_deconv_kernels=[],
+ upsample=4,
+ extra=dict(final_conv_kernel=3, ),
+ out_channels=channel_cfg['num_output_channels'],
+ loss_keypoint=dict(type='JointsMSELoss', use_target_weight=True)),
+ train_cfg=dict(),
+ test_cfg=dict(
+ flip_test=True,
+ post_process='default',
+ shift_heatmap=False,
+ target_type=target_type,
+ modulate_kernel=11,
+ use_udp=True))
+
+data_cfg = dict(
+ image_size=[192, 256],
+ heatmap_size=[48, 64],
+ num_output_channels=channel_cfg['num_output_channels'],
+ num_joints=channel_cfg['dataset_joints'],
+ dataset_channel=channel_cfg['dataset_channel'],
+ inference_channel=channel_cfg['inference_channel'],
+ soft_nms=False,
+ nms_thr=1.0,
+ oks_thr=0.9,
+ vis_thr=0.2,
+ use_gt_bbox=False,
+ det_bbox_thr=0.0,
+ bbox_file='data/coco/person_detection_results/'
+ 'COCO_val2017_detections_AP_H_56_person.json',
+)
+
+train_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownRandomFlip', flip_prob=0.5),
+ dict(
+ type='TopDownHalfBodyTransform',
+ num_joints_half_body=8,
+ prob_half_body=0.3),
+ dict(
+ type='TopDownGetRandomScaleRotation', rot_factor=40, scale_factor=0.5),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='TopDownGenerateTarget',
+ sigma=2,
+ encoding='UDP',
+ target_type=target_type),
+ dict(
+ type='Collect',
+ keys=['img', 'target', 'target_weight'],
+ meta_keys=[
+ 'image_file', 'joints_3d', 'joints_3d_visible', 'center', 'scale',
+ 'rotation', 'bbox_score', 'flip_pairs'
+ ]),
+]
+
+val_pipeline = [
+ dict(type='LoadImageFromFile'),
+ dict(type='TopDownAffine', use_udp=True),
+ dict(type='ToTensor'),
+ dict(
+ type='NormalizeTensor',
+ mean=[0.485, 0.456, 0.406],
+ std=[0.229, 0.224, 0.225]),
+ dict(
+ type='Collect',
+ keys=['img'],
+ meta_keys=[
+ 'image_file', 'center', 'scale', 'rotation', 'bbox_score',
+ 'flip_pairs'
+ ]),
+]
+
+test_pipeline = val_pipeline
+
+data_root = 'data/coco'
+data = dict(
+ samples_per_gpu=64,
+ workers_per_gpu=4,
+ val_dataloader=dict(samples_per_gpu=32),
+ test_dataloader=dict(samples_per_gpu=32),
+ train=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_train2017.json',
+ img_prefix=f'{data_root}/train2017/',
+ data_cfg=data_cfg,
+ pipeline=train_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+ val=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+ img_prefix=f'{data_root}/val2017/',
+ data_cfg=data_cfg,
+ pipeline=val_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+ test=dict(
+ type='TopDownCocoDataset',
+ ann_file=f'{data_root}/annotations/person_keypoints_val2017.json',
+ img_prefix=f'{data_root}/val2017/',
+ data_cfg=data_cfg,
+ pipeline=test_pipeline,
+ dataset_info={{_base_.dataset_info}}),
+)
+
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/__init__.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/configs/coco/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/__init__.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..bccac7519fc35e9a7564c2e826b5f4215be23728
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/__init__.py
@@ -0,0 +1,24 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+# from .ae_higher_resolution_head import AEHigherResolutionHead
+# from .ae_multi_stage_head import AEMultiStageHead
+# from .ae_simple_head import AESimpleHead
+# from .deconv_head import DeconvHead
+# from .deeppose_regression_head import DeepposeRegressionHead
+# from .hmr_head import HMRMeshHead
+# from .interhand_3d_head import Interhand3DHead
+# from .temporal_regression_head import TemporalRegressionHead
+from .topdown_heatmap_base_head import TopdownHeatmapBaseHead
+# from .topdown_heatmap_multi_stage_head import (TopdownHeatmapMSMUHead,
+# TopdownHeatmapMultiStageHead)
+from .topdown_heatmap_simple_head import TopdownHeatmapSimpleHead
+# from .vipnas_heatmap_simple_head import ViPNASHeatmapSimpleHead
+# from .voxelpose_head import CuboidCenterHead, CuboidPoseHead
+
+# __all__ = [
+# 'TopdownHeatmapSimpleHead', 'TopdownHeatmapMultiStageHead',
+# 'TopdownHeatmapMSMUHead', 'TopdownHeatmapBaseHead',
+# 'AEHigherResolutionHead', 'AESimpleHead', 'AEMultiStageHead',
+# 'DeepposeRegressionHead', 'TemporalRegressionHead', 'Interhand3DHead',
+# 'HMRMeshHead', 'DeconvHead', 'ViPNASHeatmapSimpleHead', 'CuboidCenterHead',
+# 'CuboidPoseHead'
+# ]
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/deconv_head.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/deconv_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..90846d27af46d65091f4ad7e0e6687377ebd86e1
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/deconv_head.py
@@ -0,0 +1,295 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import torch
+import torch.nn as nn
+from mmcv.cnn import (build_conv_layer, build_norm_layer, build_upsample_layer,
+ constant_init, normal_init)
+
+from mmpose.models.builder import HEADS, build_loss
+from mmpose.models.utils.ops import resize
+
+
+@HEADS.register_module()
+class DeconvHead(nn.Module):
+ """Simple deconv head.
+
+ Args:
+ in_channels (int): Number of input channels.
+ out_channels (int): Number of output channels.
+ num_deconv_layers (int): Number of deconv layers.
+ num_deconv_layers should >= 0. Note that 0 means
+ no deconv layers.
+ num_deconv_filters (list|tuple): Number of filters.
+ If num_deconv_layers > 0, the length of
+ num_deconv_kernels (list|tuple): Kernel sizes.
+ in_index (int|Sequence[int]): Input feature index. Default: 0
+ input_transform (str|None): Transformation type of input features.
+ Options: 'resize_concat', 'multiple_select', None.
+ Default: None.
+
+ - 'resize_concat': Multiple feature maps will be resized to the
+ same size as the first one and then concat together.
+ Usually used in FCN head of HRNet.
+ - 'multiple_select': Multiple feature maps will be bundle into
+ a list and passed into decode head.
+ - None: Only one select feature map is allowed.
+ align_corners (bool): align_corners argument of F.interpolate.
+ Default: False.
+ loss_keypoint (dict): Config for loss. Default: None.
+ """
+
+ def __init__(self,
+ in_channels=3,
+ out_channels=17,
+ num_deconv_layers=3,
+ num_deconv_filters=(256, 256, 256),
+ num_deconv_kernels=(4, 4, 4),
+ extra=None,
+ in_index=0,
+ input_transform=None,
+ align_corners=False,
+ loss_keypoint=None):
+ super().__init__()
+
+ self.in_channels = in_channels
+ self.loss = build_loss(loss_keypoint)
+
+ self._init_inputs(in_channels, in_index, input_transform)
+ self.in_index = in_index
+ self.align_corners = align_corners
+
+ if extra is not None and not isinstance(extra, dict):
+ raise TypeError('extra should be dict or None.')
+
+ if num_deconv_layers > 0:
+ self.deconv_layers = self._make_deconv_layer(
+ num_deconv_layers,
+ num_deconv_filters,
+ num_deconv_kernels,
+ )
+ elif num_deconv_layers == 0:
+ self.deconv_layers = nn.Identity()
+ else:
+ raise ValueError(
+ f'num_deconv_layers ({num_deconv_layers}) should >= 0.')
+
+ identity_final_layer = False
+ if extra is not None and 'final_conv_kernel' in extra:
+ assert extra['final_conv_kernel'] in [0, 1, 3]
+ if extra['final_conv_kernel'] == 3:
+ padding = 1
+ elif extra['final_conv_kernel'] == 1:
+ padding = 0
+ else:
+ # 0 for Identity mapping.
+ identity_final_layer = True
+ kernel_size = extra['final_conv_kernel']
+ else:
+ kernel_size = 1
+ padding = 0
+
+ if identity_final_layer:
+ self.final_layer = nn.Identity()
+ else:
+ conv_channels = num_deconv_filters[
+ -1] if num_deconv_layers > 0 else self.in_channels
+
+ layers = []
+ if extra is not None:
+ num_conv_layers = extra.get('num_conv_layers', 0)
+ num_conv_kernels = extra.get('num_conv_kernels',
+ [1] * num_conv_layers)
+
+ for i in range(num_conv_layers):
+ layers.append(
+ build_conv_layer(
+ dict(type='Conv2d'),
+ in_channels=conv_channels,
+ out_channels=conv_channels,
+ kernel_size=num_conv_kernels[i],
+ stride=1,
+ padding=(num_conv_kernels[i] - 1) // 2))
+ layers.append(
+ build_norm_layer(dict(type='BN'), conv_channels)[1])
+ layers.append(nn.ReLU(inplace=True))
+
+ layers.append(
+ build_conv_layer(
+ cfg=dict(type='Conv2d'),
+ in_channels=conv_channels,
+ out_channels=out_channels,
+ kernel_size=kernel_size,
+ stride=1,
+ padding=padding))
+
+ if len(layers) > 1:
+ self.final_layer = nn.Sequential(*layers)
+ else:
+ self.final_layer = layers[0]
+
+ def _init_inputs(self, in_channels, in_index, input_transform):
+ """Check and initialize input transforms.
+
+ The in_channels, in_index and input_transform must match.
+ Specifically, when input_transform is None, only single feature map
+ will be selected. So in_channels and in_index must be of type int.
+ When input_transform is not None, in_channels and in_index must be
+ list or tuple, with the same length.
+
+ Args:
+ in_channels (int|Sequence[int]): Input channels.
+ in_index (int|Sequence[int]): Input feature index.
+ input_transform (str|None): Transformation type of input features.
+ Options: 'resize_concat', 'multiple_select', None.
+
+ - 'resize_concat': Multiple feature maps will be resize to the
+ same size as first one and than concat together.
+ Usually used in FCN head of HRNet.
+ - 'multiple_select': Multiple feature maps will be bundle into
+ a list and passed into decode head.
+ - None: Only one select feature map is allowed.
+ """
+
+ if input_transform is not None:
+ assert input_transform in ['resize_concat', 'multiple_select']
+ self.input_transform = input_transform
+ self.in_index = in_index
+ if input_transform is not None:
+ assert isinstance(in_channels, (list, tuple))
+ assert isinstance(in_index, (list, tuple))
+ assert len(in_channels) == len(in_index)
+ if input_transform == 'resize_concat':
+ self.in_channels = sum(in_channels)
+ else:
+ self.in_channels = in_channels
+ else:
+ assert isinstance(in_channels, int)
+ assert isinstance(in_index, int)
+ self.in_channels = in_channels
+
+ def _transform_inputs(self, inputs):
+ """Transform inputs for decoder.
+
+ Args:
+ inputs (list[Tensor] | Tensor): multi-level img features.
+
+ Returns:
+ Tensor: The transformed inputs
+ """
+ if not isinstance(inputs, list):
+ return inputs
+
+ if self.input_transform == 'resize_concat':
+ inputs = [inputs[i] for i in self.in_index]
+ upsampled_inputs = [
+ resize(
+ input=x,
+ size=inputs[0].shape[2:],
+ mode='bilinear',
+ align_corners=self.align_corners) for x in inputs
+ ]
+ inputs = torch.cat(upsampled_inputs, dim=1)
+ elif self.input_transform == 'multiple_select':
+ inputs = [inputs[i] for i in self.in_index]
+ else:
+ inputs = inputs[self.in_index]
+
+ return inputs
+
+ def _make_deconv_layer(self, num_layers, num_filters, num_kernels):
+ """Make deconv layers."""
+ if num_layers != len(num_filters):
+ error_msg = f'num_layers({num_layers}) ' \
+ f'!= length of num_filters({len(num_filters)})'
+ raise ValueError(error_msg)
+ if num_layers != len(num_kernels):
+ error_msg = f'num_layers({num_layers}) ' \
+ f'!= length of num_kernels({len(num_kernels)})'
+ raise ValueError(error_msg)
+
+ layers = []
+ for i in range(num_layers):
+ kernel, padding, output_padding = \
+ self._get_deconv_cfg(num_kernels[i])
+
+ planes = num_filters[i]
+ layers.append(
+ build_upsample_layer(
+ dict(type='deconv'),
+ in_channels=self.in_channels,
+ out_channels=planes,
+ kernel_size=kernel,
+ stride=2,
+ padding=padding,
+ output_padding=output_padding,
+ bias=False))
+ layers.append(nn.BatchNorm2d(planes))
+ layers.append(nn.ReLU(inplace=True))
+ self.in_channels = planes
+
+ return nn.Sequential(*layers)
+
+ @staticmethod
+ def _get_deconv_cfg(deconv_kernel):
+ """Get configurations for deconv layers."""
+ if deconv_kernel == 4:
+ padding = 1
+ output_padding = 0
+ elif deconv_kernel == 3:
+ padding = 1
+ output_padding = 1
+ elif deconv_kernel == 2:
+ padding = 0
+ output_padding = 0
+ else:
+ raise ValueError(f'Not supported num_kernels ({deconv_kernel}).')
+
+ return deconv_kernel, padding, output_padding
+
+ def get_loss(self, outputs, targets, masks):
+ """Calculate bottom-up masked mse loss.
+
+ Note:
+ - batch_size: N
+ - num_channels: C
+ - heatmaps height: H
+ - heatmaps weight: W
+
+ Args:
+ outputs (List(torch.Tensor[N,C,H,W])): Multi-scale outputs.
+ targets (List(torch.Tensor[N,C,H,W])): Multi-scale targets.
+ masks (List(torch.Tensor[N,H,W])): Masks of multi-scale targets.
+ """
+
+ losses = dict()
+
+ for idx in range(len(targets)):
+ if 'loss' not in losses:
+ losses['loss'] = self.loss(outputs[idx], targets[idx],
+ masks[idx])
+ else:
+ losses['loss'] += self.loss(outputs[idx], targets[idx],
+ masks[idx])
+
+ return losses
+
+ def forward(self, x):
+ """Forward function."""
+ x = self._transform_inputs(x)
+ final_outputs = []
+ x = self.deconv_layers(x)
+ y = self.final_layer(x)
+ final_outputs.append(y)
+ return final_outputs
+
+ def init_weights(self):
+ """Initialize model weights."""
+ for _, m in self.deconv_layers.named_modules():
+ if isinstance(m, nn.ConvTranspose2d):
+ normal_init(m, std=0.001)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
+ for m in self.final_layer.modules():
+ if isinstance(m, nn.Conv2d):
+ normal_init(m, std=0.001, bias=0)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/deeppose_regression_head.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/deeppose_regression_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..f326e26fa624bd99e9603ad28ff71dccb29b5638
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/deeppose_regression_head.py
@@ -0,0 +1,176 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import numpy as np
+import torch.nn as nn
+from mmcv.cnn import normal_init
+
+from mmpose.core.evaluation import (keypoint_pck_accuracy,
+ keypoints_from_regression)
+from mmpose.core.post_processing import fliplr_regression
+from mmpose.models.builder import HEADS, build_loss
+
+
+@HEADS.register_module()
+class DeepposeRegressionHead(nn.Module):
+ """Deeppose regression head with fully connected layers.
+
+ "DeepPose: Human Pose Estimation via Deep Neural Networks".
+
+ Args:
+ in_channels (int): Number of input channels
+ num_joints (int): Number of joints
+ loss_keypoint (dict): Config for keypoint loss. Default: None.
+ """
+
+ def __init__(self,
+ in_channels,
+ num_joints,
+ loss_keypoint=None,
+ train_cfg=None,
+ test_cfg=None):
+ super().__init__()
+
+ self.in_channels = in_channels
+ self.num_joints = num_joints
+
+ self.loss = build_loss(loss_keypoint)
+
+ self.train_cfg = {} if train_cfg is None else train_cfg
+ self.test_cfg = {} if test_cfg is None else test_cfg
+
+ self.fc = nn.Linear(self.in_channels, self.num_joints * 2)
+
+ def forward(self, x):
+ """Forward function."""
+ output = self.fc(x)
+ N, C = output.shape
+ return output.reshape([N, C // 2, 2])
+
+ def get_loss(self, output, target, target_weight):
+ """Calculate top-down keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+
+ Args:
+ output (torch.Tensor[N, K, 2]): Output keypoints.
+ target (torch.Tensor[N, K, 2]): Target keypoints.
+ target_weight (torch.Tensor[N, K, 2]):
+ Weights across different joint types.
+ """
+
+ losses = dict()
+ assert not isinstance(self.loss, nn.Sequential)
+ assert target.dim() == 3 and target_weight.dim() == 3
+ losses['reg_loss'] = self.loss(output, target, target_weight)
+
+ return losses
+
+ def get_accuracy(self, output, target, target_weight):
+ """Calculate accuracy for top-down keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+
+ Args:
+ output (torch.Tensor[N, K, 2]): Output keypoints.
+ target (torch.Tensor[N, K, 2]): Target keypoints.
+ target_weight (torch.Tensor[N, K, 2]):
+ Weights across different joint types.
+ """
+
+ accuracy = dict()
+
+ N = output.shape[0]
+
+ _, avg_acc, cnt = keypoint_pck_accuracy(
+ output.detach().cpu().numpy(),
+ target.detach().cpu().numpy(),
+ target_weight[:, :, 0].detach().cpu().numpy() > 0,
+ thr=0.05,
+ normalize=np.ones((N, 2), dtype=np.float32))
+ accuracy['acc_pose'] = avg_acc
+
+ return accuracy
+
+ def inference_model(self, x, flip_pairs=None):
+ """Inference function.
+
+ Returns:
+ output_regression (np.ndarray): Output regression.
+
+ Args:
+ x (torch.Tensor[N, K, 2]): Input features.
+ flip_pairs (None | list[tuple()):
+ Pairs of keypoints which are mirrored.
+ """
+ output = self.forward(x)
+
+ if flip_pairs is not None:
+ output_regression = fliplr_regression(
+ output.detach().cpu().numpy(), flip_pairs)
+ else:
+ output_regression = output.detach().cpu().numpy()
+ return output_regression
+
+ def decode(self, img_metas, output, **kwargs):
+ """Decode the keypoints from output regression.
+
+ Args:
+ img_metas (list(dict)): Information about data augmentation
+ By default this includes:
+
+ - "image_file: path to the image file
+ - "center": center of the bbox
+ - "scale": scale of the bbox
+ - "rotation": rotation of the bbox
+ - "bbox_score": score of bbox
+ output (np.ndarray[N, K, 2]): predicted regression vector.
+ kwargs: dict contains 'img_size'.
+ img_size (tuple(img_width, img_height)): input image size.
+ """
+ batch_size = len(img_metas)
+
+ if 'bbox_id' in img_metas[0]:
+ bbox_ids = []
+ else:
+ bbox_ids = None
+
+ c = np.zeros((batch_size, 2), dtype=np.float32)
+ s = np.zeros((batch_size, 2), dtype=np.float32)
+ image_paths = []
+ score = np.ones(batch_size)
+ for i in range(batch_size):
+ c[i, :] = img_metas[i]['center']
+ s[i, :] = img_metas[i]['scale']
+ image_paths.append(img_metas[i]['image_file'])
+
+ if 'bbox_score' in img_metas[i]:
+ score[i] = np.array(img_metas[i]['bbox_score']).reshape(-1)
+ if bbox_ids is not None:
+ bbox_ids.append(img_metas[i]['bbox_id'])
+
+ preds, maxvals = keypoints_from_regression(output, c, s,
+ kwargs['img_size'])
+
+ all_preds = np.zeros((batch_size, preds.shape[1], 3), dtype=np.float32)
+ all_boxes = np.zeros((batch_size, 6), dtype=np.float32)
+ all_preds[:, :, 0:2] = preds[:, :, 0:2]
+ all_preds[:, :, 2:3] = maxvals
+ all_boxes[:, 0:2] = c[:, 0:2]
+ all_boxes[:, 2:4] = s[:, 0:2]
+ all_boxes[:, 4] = np.prod(s * 200.0, axis=1)
+ all_boxes[:, 5] = score
+
+ result = {}
+
+ result['preds'] = all_preds
+ result['boxes'] = all_boxes
+ result['image_paths'] = image_paths
+ result['bbox_ids'] = bbox_ids
+
+ return result
+
+ def init_weights(self):
+ normal_init(self.fc, mean=0, std=0.01, bias=0)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/hmr_head.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/hmr_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..015a3076bcba53d1590de226fab39444708cb3f9
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/hmr_head.py
@@ -0,0 +1,94 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import numpy as np
+import torch
+import torch.nn as nn
+from mmcv.cnn import xavier_init
+
+from ..builder import HEADS
+from ..utils.geometry import rot6d_to_rotmat
+
+
+@HEADS.register_module()
+class HMRMeshHead(nn.Module):
+ """SMPL parameters regressor head of simple baseline. "End-to-end Recovery
+ of Human Shape and Pose", CVPR'2018.
+
+ Args:
+ in_channels (int): Number of input channels
+ smpl_mean_params (str): The file name of the mean SMPL parameters
+ n_iter (int): The iterations of estimating delta parameters
+ """
+
+ def __init__(self, in_channels, smpl_mean_params=None, n_iter=3):
+ super().__init__()
+
+ self.in_channels = in_channels
+ self.n_iter = n_iter
+
+ npose = 24 * 6
+ nbeta = 10
+ ncam = 3
+ hidden_dim = 1024
+
+ self.fc1 = nn.Linear(in_channels + npose + nbeta + ncam, hidden_dim)
+ self.drop1 = nn.Dropout()
+ self.fc2 = nn.Linear(hidden_dim, hidden_dim)
+ self.drop2 = nn.Dropout()
+ self.decpose = nn.Linear(hidden_dim, npose)
+ self.decshape = nn.Linear(hidden_dim, nbeta)
+ self.deccam = nn.Linear(hidden_dim, ncam)
+
+ # Load mean SMPL parameters
+ if smpl_mean_params is None:
+ init_pose = torch.zeros([1, npose])
+ init_shape = torch.zeros([1, nbeta])
+ init_cam = torch.FloatTensor([[1, 0, 0]])
+ else:
+ mean_params = np.load(smpl_mean_params)
+ init_pose = torch.from_numpy(
+ mean_params['pose'][:]).unsqueeze(0).float()
+ init_shape = torch.from_numpy(
+ mean_params['shape'][:]).unsqueeze(0).float()
+ init_cam = torch.from_numpy(
+ mean_params['cam']).unsqueeze(0).float()
+ self.register_buffer('init_pose', init_pose)
+ self.register_buffer('init_shape', init_shape)
+ self.register_buffer('init_cam', init_cam)
+
+ def forward(self, x):
+ """Forward function.
+
+ x is the image feature map and is expected to be in shape (batch size x
+ channel number x height x width)
+ """
+ batch_size = x.shape[0]
+ # extract the global feature vector by average along
+ # spatial dimension.
+ x = x.mean(dim=-1).mean(dim=-1)
+
+ init_pose = self.init_pose.expand(batch_size, -1)
+ init_shape = self.init_shape.expand(batch_size, -1)
+ init_cam = self.init_cam.expand(batch_size, -1)
+
+ pred_pose = init_pose
+ pred_shape = init_shape
+ pred_cam = init_cam
+ for _ in range(self.n_iter):
+ xc = torch.cat([x, pred_pose, pred_shape, pred_cam], 1)
+ xc = self.fc1(xc)
+ xc = self.drop1(xc)
+ xc = self.fc2(xc)
+ xc = self.drop2(xc)
+ pred_pose = self.decpose(xc) + pred_pose
+ pred_shape = self.decshape(xc) + pred_shape
+ pred_cam = self.deccam(xc) + pred_cam
+
+ pred_rotmat = rot6d_to_rotmat(pred_pose).view(batch_size, 24, 3, 3)
+ out = (pred_rotmat, pred_shape, pred_cam)
+ return out
+
+ def init_weights(self):
+ """Initialize model weights."""
+ xavier_init(self.decpose, gain=0.01)
+ xavier_init(self.decshape, gain=0.01)
+ xavier_init(self.deccam, gain=0.01)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/interhand_3d_head.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/interhand_3d_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..aebe4a5f61e5fd1dcd5ecfb64962f88da94d5664
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/interhand_3d_head.py
@@ -0,0 +1,521 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import numpy as np
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from mmcv.cnn import (build_conv_layer, build_norm_layer, build_upsample_layer,
+ constant_init, normal_init)
+
+from mmpose.core.evaluation.top_down_eval import (
+ keypoints_from_heatmaps3d, multilabel_classification_accuracy)
+from mmpose.core.post_processing import flip_back
+from mmpose.models.builder import build_loss
+from mmpose.models.necks import GlobalAveragePooling
+from ..builder import HEADS
+
+
+class Heatmap3DHead(nn.Module):
+ """Heatmap3DHead is a sub-module of Interhand3DHead, and outputs 3D
+ heatmaps. Heatmap3DHead is composed of (>=0) number of deconv layers and a
+ simple conv2d layer.
+
+ Args:
+ in_channels (int): Number of input channels
+ out_channels (int): Number of output channels
+ depth_size (int): Number of depth discretization size
+ num_deconv_layers (int): Number of deconv layers.
+ num_deconv_layers should >= 0. Note that 0 means no deconv layers.
+ num_deconv_filters (list|tuple): Number of filters.
+ num_deconv_kernels (list|tuple): Kernel sizes.
+ extra (dict): Configs for extra conv layers. Default: None
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ depth_size=64,
+ num_deconv_layers=3,
+ num_deconv_filters=(256, 256, 256),
+ num_deconv_kernels=(4, 4, 4),
+ extra=None):
+
+ super().__init__()
+
+ assert out_channels % depth_size == 0
+ self.depth_size = depth_size
+ self.in_channels = in_channels
+
+ if extra is not None and not isinstance(extra, dict):
+ raise TypeError('extra should be dict or None.')
+
+ if num_deconv_layers > 0:
+ self.deconv_layers = self._make_deconv_layer(
+ num_deconv_layers,
+ num_deconv_filters,
+ num_deconv_kernels,
+ )
+ elif num_deconv_layers == 0:
+ self.deconv_layers = nn.Identity()
+ else:
+ raise ValueError(
+ f'num_deconv_layers ({num_deconv_layers}) should >= 0.')
+
+ identity_final_layer = False
+ if extra is not None and 'final_conv_kernel' in extra:
+ assert extra['final_conv_kernel'] in [0, 1, 3]
+ if extra['final_conv_kernel'] == 3:
+ padding = 1
+ elif extra['final_conv_kernel'] == 1:
+ padding = 0
+ else:
+ # 0 for Identity mapping.
+ identity_final_layer = True
+ kernel_size = extra['final_conv_kernel']
+ else:
+ kernel_size = 1
+ padding = 0
+
+ if identity_final_layer:
+ self.final_layer = nn.Identity()
+ else:
+ conv_channels = num_deconv_filters[
+ -1] if num_deconv_layers > 0 else self.in_channels
+
+ layers = []
+ if extra is not None:
+ num_conv_layers = extra.get('num_conv_layers', 0)
+ num_conv_kernels = extra.get('num_conv_kernels',
+ [1] * num_conv_layers)
+
+ for i in range(num_conv_layers):
+ layers.append(
+ build_conv_layer(
+ dict(type='Conv2d'),
+ in_channels=conv_channels,
+ out_channels=conv_channels,
+ kernel_size=num_conv_kernels[i],
+ stride=1,
+ padding=(num_conv_kernels[i] - 1) // 2))
+ layers.append(
+ build_norm_layer(dict(type='BN'), conv_channels)[1])
+ layers.append(nn.ReLU(inplace=True))
+
+ layers.append(
+ build_conv_layer(
+ cfg=dict(type='Conv2d'),
+ in_channels=conv_channels,
+ out_channels=out_channels,
+ kernel_size=kernel_size,
+ stride=1,
+ padding=padding))
+
+ if len(layers) > 1:
+ self.final_layer = nn.Sequential(*layers)
+ else:
+ self.final_layer = layers[0]
+
+ def _make_deconv_layer(self, num_layers, num_filters, num_kernels):
+ """Make deconv layers."""
+ if num_layers != len(num_filters):
+ error_msg = f'num_layers({num_layers}) ' \
+ f'!= length of num_filters({len(num_filters)})'
+ raise ValueError(error_msg)
+ if num_layers != len(num_kernels):
+ error_msg = f'num_layers({num_layers}) ' \
+ f'!= length of num_kernels({len(num_kernels)})'
+ raise ValueError(error_msg)
+
+ layers = []
+ for i in range(num_layers):
+ kernel, padding, output_padding = \
+ self._get_deconv_cfg(num_kernels[i])
+
+ planes = num_filters[i]
+ layers.append(
+ build_upsample_layer(
+ dict(type='deconv'),
+ in_channels=self.in_channels,
+ out_channels=planes,
+ kernel_size=kernel,
+ stride=2,
+ padding=padding,
+ output_padding=output_padding,
+ bias=False))
+ layers.append(nn.BatchNorm2d(planes))
+ layers.append(nn.ReLU(inplace=True))
+ self.in_channels = planes
+
+ return nn.Sequential(*layers)
+
+ @staticmethod
+ def _get_deconv_cfg(deconv_kernel):
+ """Get configurations for deconv layers."""
+ if deconv_kernel == 4:
+ padding = 1
+ output_padding = 0
+ elif deconv_kernel == 3:
+ padding = 1
+ output_padding = 1
+ elif deconv_kernel == 2:
+ padding = 0
+ output_padding = 0
+ else:
+ raise ValueError(f'Not supported num_kernels ({deconv_kernel}).')
+
+ return deconv_kernel, padding, output_padding
+
+ def forward(self, x):
+ """Forward function."""
+ x = self.deconv_layers(x)
+ x = self.final_layer(x)
+ N, C, H, W = x.shape
+ # reshape the 2D heatmap to 3D heatmap
+ x = x.reshape(N, C // self.depth_size, self.depth_size, H, W)
+ return x
+
+ def init_weights(self):
+ """Initialize model weights."""
+ for _, m in self.deconv_layers.named_modules():
+ if isinstance(m, nn.ConvTranspose2d):
+ normal_init(m, std=0.001)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
+ for m in self.final_layer.modules():
+ if isinstance(m, nn.Conv2d):
+ normal_init(m, std=0.001, bias=0)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
+
+
+class Heatmap1DHead(nn.Module):
+ """Heatmap1DHead is a sub-module of Interhand3DHead, and outputs 1D
+ heatmaps.
+
+ Args:
+ in_channels (int): Number of input channels
+ heatmap_size (int): Heatmap size
+ hidden_dims (list|tuple): Number of feature dimension of FC layers.
+ """
+
+ def __init__(self, in_channels=2048, heatmap_size=64, hidden_dims=(512, )):
+ super().__init__()
+
+ self.in_channels = in_channels
+ self.heatmap_size = heatmap_size
+
+ feature_dims = [in_channels, *hidden_dims, heatmap_size]
+ self.fc = self._make_linear_layers(feature_dims, relu_final=False)
+
+ def soft_argmax_1d(self, heatmap1d):
+ heatmap1d = F.softmax(heatmap1d, 1)
+ accu = heatmap1d * torch.arange(
+ self.heatmap_size, dtype=heatmap1d.dtype,
+ device=heatmap1d.device)[None, :]
+ coord = accu.sum(dim=1)
+ return coord
+
+ def _make_linear_layers(self, feat_dims, relu_final=False):
+ """Make linear layers."""
+ layers = []
+ for i in range(len(feat_dims) - 1):
+ layers.append(nn.Linear(feat_dims[i], feat_dims[i + 1]))
+ if i < len(feat_dims) - 2 or \
+ (i == len(feat_dims) - 2 and relu_final):
+ layers.append(nn.ReLU(inplace=True))
+ return nn.Sequential(*layers)
+
+ def forward(self, x):
+ """Forward function."""
+ heatmap1d = self.fc(x)
+ value = self.soft_argmax_1d(heatmap1d).view(-1, 1)
+ return value
+
+ def init_weights(self):
+ """Initialize model weights."""
+ for m in self.fc.modules():
+ if isinstance(m, nn.Linear):
+ normal_init(m, mean=0, std=0.01, bias=0)
+
+
+class MultilabelClassificationHead(nn.Module):
+ """MultilabelClassificationHead is a sub-module of Interhand3DHead, and
+ outputs hand type classification.
+
+ Args:
+ in_channels (int): Number of input channels
+ num_labels (int): Number of labels
+ hidden_dims (list|tuple): Number of hidden dimension of FC layers.
+ """
+
+ def __init__(self, in_channels=2048, num_labels=2, hidden_dims=(512, )):
+ super().__init__()
+
+ self.in_channels = in_channels
+ self.num_labesl = num_labels
+
+ feature_dims = [in_channels, *hidden_dims, num_labels]
+ self.fc = self._make_linear_layers(feature_dims, relu_final=False)
+
+ def _make_linear_layers(self, feat_dims, relu_final=False):
+ """Make linear layers."""
+ layers = []
+ for i in range(len(feat_dims) - 1):
+ layers.append(nn.Linear(feat_dims[i], feat_dims[i + 1]))
+ if i < len(feat_dims) - 2 or \
+ (i == len(feat_dims) - 2 and relu_final):
+ layers.append(nn.ReLU(inplace=True))
+ return nn.Sequential(*layers)
+
+ def forward(self, x):
+ """Forward function."""
+ labels = torch.sigmoid(self.fc(x))
+ return labels
+
+ def init_weights(self):
+ for m in self.fc.modules():
+ if isinstance(m, nn.Linear):
+ normal_init(m, mean=0, std=0.01, bias=0)
+
+
+@HEADS.register_module()
+class Interhand3DHead(nn.Module):
+ """Interhand 3D head of paper ref: Gyeongsik Moon. "InterHand2.6M: A
+ Dataset and Baseline for 3D Interacting Hand Pose Estimation from a Single
+ RGB Image".
+
+ Args:
+ keypoint_head_cfg (dict): Configs of Heatmap3DHead for hand
+ keypoint estimation.
+ root_head_cfg (dict): Configs of Heatmap1DHead for relative
+ hand root depth estimation.
+ hand_type_head_cfg (dict): Configs of MultilabelClassificationHead
+ for hand type classification.
+ loss_keypoint (dict): Config for keypoint loss. Default: None.
+ loss_root_depth (dict): Config for relative root depth loss.
+ Default: None.
+ loss_hand_type (dict): Config for hand type classification
+ loss. Default: None.
+ """
+
+ def __init__(self,
+ keypoint_head_cfg,
+ root_head_cfg,
+ hand_type_head_cfg,
+ loss_keypoint=None,
+ loss_root_depth=None,
+ loss_hand_type=None,
+ train_cfg=None,
+ test_cfg=None):
+ super().__init__()
+
+ # build sub-module heads
+ self.right_hand_head = Heatmap3DHead(**keypoint_head_cfg)
+ self.left_hand_head = Heatmap3DHead(**keypoint_head_cfg)
+ self.root_head = Heatmap1DHead(**root_head_cfg)
+ self.hand_type_head = MultilabelClassificationHead(
+ **hand_type_head_cfg)
+ self.neck = GlobalAveragePooling()
+
+ # build losses
+ self.keypoint_loss = build_loss(loss_keypoint)
+ self.root_depth_loss = build_loss(loss_root_depth)
+ self.hand_type_loss = build_loss(loss_hand_type)
+ self.train_cfg = {} if train_cfg is None else train_cfg
+ self.test_cfg = {} if test_cfg is None else test_cfg
+ self.target_type = self.test_cfg.get('target_type', 'GaussianHeatmap')
+
+ def init_weights(self):
+ self.left_hand_head.init_weights()
+ self.right_hand_head.init_weights()
+ self.root_head.init_weights()
+ self.hand_type_head.init_weights()
+
+ def get_loss(self, output, target, target_weight):
+ """Calculate loss for hand keypoint heatmaps, relative root depth and
+ hand type.
+
+ Args:
+ output (list[Tensor]): a list of outputs from multiple heads.
+ target (list[Tensor]): a list of targets for multiple heads.
+ target_weight (list[Tensor]): a list of targets weight for
+ multiple heads.
+ """
+ losses = dict()
+
+ # hand keypoint loss
+ assert not isinstance(self.keypoint_loss, nn.Sequential)
+ out, tar, tar_weight = output[0], target[0], target_weight[0]
+ assert tar.dim() == 5 and tar_weight.dim() == 3
+ losses['hand_loss'] = self.keypoint_loss(out, tar, tar_weight)
+
+ # relative root depth loss
+ assert not isinstance(self.root_depth_loss, nn.Sequential)
+ out, tar, tar_weight = output[1], target[1], target_weight[1]
+ assert tar.dim() == 2 and tar_weight.dim() == 2
+ losses['rel_root_loss'] = self.root_depth_loss(out, tar, tar_weight)
+
+ # hand type loss
+ assert not isinstance(self.hand_type_loss, nn.Sequential)
+ out, tar, tar_weight = output[2], target[2], target_weight[2]
+ assert tar.dim() == 2 and tar_weight.dim() in [1, 2]
+ losses['hand_type_loss'] = self.hand_type_loss(out, tar, tar_weight)
+
+ return losses
+
+ def get_accuracy(self, output, target, target_weight):
+ """Calculate accuracy for hand type.
+
+ Args:
+ output (list[Tensor]): a list of outputs from multiple heads.
+ target (list[Tensor]): a list of targets for multiple heads.
+ target_weight (list[Tensor]): a list of targets weight for
+ multiple heads.
+ """
+ accuracy = dict()
+ avg_acc = multilabel_classification_accuracy(
+ output[2].detach().cpu().numpy(),
+ target[2].detach().cpu().numpy(),
+ target_weight[2].detach().cpu().numpy(),
+ )
+ accuracy['acc_classification'] = float(avg_acc)
+ return accuracy
+
+ def forward(self, x):
+ """Forward function."""
+ outputs = []
+ outputs.append(
+ torch.cat([self.right_hand_head(x),
+ self.left_hand_head(x)], dim=1))
+ x = self.neck(x)
+ outputs.append(self.root_head(x))
+ outputs.append(self.hand_type_head(x))
+ return outputs
+
+ def inference_model(self, x, flip_pairs=None):
+ """Inference function.
+
+ Returns:
+ output (list[np.ndarray]): list of output hand keypoint
+ heatmaps, relative root depth and hand type.
+
+ Args:
+ x (torch.Tensor[N,K,H,W]): Input features.
+ flip_pairs (None | list[tuple()):
+ Pairs of keypoints which are mirrored.
+ """
+
+ output = self.forward(x)
+
+ if flip_pairs is not None:
+ # flip 3D heatmap
+ heatmap_3d = output[0]
+ N, K, D, H, W = heatmap_3d.shape
+ # reshape 3D heatmap to 2D heatmap
+ heatmap_3d = heatmap_3d.reshape(N, K * D, H, W)
+ # 2D heatmap flip
+ heatmap_3d_flipped_back = flip_back(
+ heatmap_3d.detach().cpu().numpy(),
+ flip_pairs,
+ target_type=self.target_type)
+ # reshape back to 3D heatmap
+ heatmap_3d_flipped_back = heatmap_3d_flipped_back.reshape(
+ N, K, D, H, W)
+ # feature is not aligned, shift flipped heatmap for higher accuracy
+ if self.test_cfg.get('shift_heatmap', False):
+ heatmap_3d_flipped_back[...,
+ 1:] = heatmap_3d_flipped_back[..., :-1]
+ output[0] = heatmap_3d_flipped_back
+
+ # flip relative hand root depth
+ output[1] = -output[1].detach().cpu().numpy()
+
+ # flip hand type
+ hand_type = output[2].detach().cpu().numpy()
+ hand_type_flipped_back = hand_type.copy()
+ hand_type_flipped_back[:, 0] = hand_type[:, 1]
+ hand_type_flipped_back[:, 1] = hand_type[:, 0]
+ output[2] = hand_type_flipped_back
+ else:
+ output = [out.detach().cpu().numpy() for out in output]
+
+ return output
+
+ def decode(self, img_metas, output, **kwargs):
+ """Decode hand keypoint, relative root depth and hand type.
+
+ Args:
+ img_metas (list(dict)): Information about data augmentation
+ By default this includes:
+
+ - "image_file: path to the image file
+ - "center": center of the bbox
+ - "scale": scale of the bbox
+ - "rotation": rotation of the bbox
+ - "bbox_score": score of bbox
+ - "heatmap3d_depth_bound": depth bound of hand keypoint
+ 3D heatmap
+ - "root_depth_bound": depth bound of relative root depth
+ 1D heatmap
+ output (list[np.ndarray]): model predicted 3D heatmaps, relative
+ root depth and hand type.
+ """
+
+ batch_size = len(img_metas)
+ result = {}
+
+ heatmap3d_depth_bound = np.ones(batch_size, dtype=np.float32)
+ root_depth_bound = np.ones(batch_size, dtype=np.float32)
+ center = np.zeros((batch_size, 2), dtype=np.float32)
+ scale = np.zeros((batch_size, 2), dtype=np.float32)
+ image_paths = []
+ score = np.ones(batch_size, dtype=np.float32)
+ if 'bbox_id' in img_metas[0]:
+ bbox_ids = []
+ else:
+ bbox_ids = None
+
+ for i in range(batch_size):
+ heatmap3d_depth_bound[i] = img_metas[i]['heatmap3d_depth_bound']
+ root_depth_bound[i] = img_metas[i]['root_depth_bound']
+ center[i, :] = img_metas[i]['center']
+ scale[i, :] = img_metas[i]['scale']
+ image_paths.append(img_metas[i]['image_file'])
+
+ if 'bbox_score' in img_metas[i]:
+ score[i] = np.array(img_metas[i]['bbox_score']).reshape(-1)
+ if bbox_ids is not None:
+ bbox_ids.append(img_metas[i]['bbox_id'])
+
+ all_boxes = np.zeros((batch_size, 6), dtype=np.float32)
+ all_boxes[:, 0:2] = center[:, 0:2]
+ all_boxes[:, 2:4] = scale[:, 0:2]
+ # scale is defined as: bbox_size / 200.0, so we
+ # need multiply 200.0 to get bbox size
+ all_boxes[:, 4] = np.prod(scale * 200.0, axis=1)
+ all_boxes[:, 5] = score
+ result['boxes'] = all_boxes
+ result['image_paths'] = image_paths
+ result['bbox_ids'] = bbox_ids
+
+ # decode 3D heatmaps of hand keypoints
+ heatmap3d = output[0]
+ preds, maxvals = keypoints_from_heatmaps3d(heatmap3d, center, scale)
+ keypoints_3d = np.zeros((batch_size, preds.shape[1], 4),
+ dtype=np.float32)
+ keypoints_3d[:, :, 0:3] = preds[:, :, 0:3]
+ keypoints_3d[:, :, 3:4] = maxvals
+ # transform keypoint depth to camera space
+ keypoints_3d[:, :, 2] = \
+ (keypoints_3d[:, :, 2] / self.right_hand_head.depth_size - 0.5) \
+ * heatmap3d_depth_bound[:, np.newaxis]
+
+ result['preds'] = keypoints_3d
+
+ # decode relative hand root depth
+ # transform relative root depth to camera space
+ result['rel_root_depth'] = (output[1] / self.root_head.heatmap_size -
+ 0.5) * root_depth_bound
+
+ # decode hand type
+ result['hand_type'] = output[2] > 0.5
+ return result
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/temporal_regression_head.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/temporal_regression_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..97a07f9cf2c9ef0497380ca5c602142b206f3b52
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/temporal_regression_head.py
@@ -0,0 +1,319 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import numpy as np
+import torch.nn as nn
+from mmcv.cnn import build_conv_layer, constant_init, kaiming_init
+from mmcv.utils.parrots_wrapper import _BatchNorm
+
+from mmpose.core import (WeightNormClipHook, compute_similarity_transform,
+ fliplr_regression)
+from mmpose.models.builder import HEADS, build_loss
+
+
+@HEADS.register_module()
+class TemporalRegressionHead(nn.Module):
+ """Regression head of VideoPose3D.
+
+ "3D human pose estimation in video with temporal convolutions and
+ semi-supervised training", CVPR'2019.
+
+ Args:
+ in_channels (int): Number of input channels
+ num_joints (int): Number of joints
+ loss_keypoint (dict): Config for keypoint loss. Default: None.
+ max_norm (float|None): if not None, the weight of convolution layers
+ will be clipped to have a maximum norm of max_norm.
+ is_trajectory (bool): If the model only predicts root joint
+ position, then this arg should be set to True. In this case,
+ traj_loss will be calculated. Otherwise, it should be set to
+ False. Default: False.
+ """
+
+ def __init__(self,
+ in_channels,
+ num_joints,
+ max_norm=None,
+ loss_keypoint=None,
+ is_trajectory=False,
+ train_cfg=None,
+ test_cfg=None):
+ super().__init__()
+
+ self.in_channels = in_channels
+ self.num_joints = num_joints
+ self.max_norm = max_norm
+ self.loss = build_loss(loss_keypoint)
+ self.is_trajectory = is_trajectory
+ if self.is_trajectory:
+ assert self.num_joints == 1
+
+ self.train_cfg = {} if train_cfg is None else train_cfg
+ self.test_cfg = {} if test_cfg is None else test_cfg
+
+ self.conv = build_conv_layer(
+ dict(type='Conv1d'), in_channels, num_joints * 3, 1)
+
+ if self.max_norm is not None:
+ # Apply weight norm clip to conv layers
+ weight_clip = WeightNormClipHook(self.max_norm)
+ for module in self.modules():
+ if isinstance(module, nn.modules.conv._ConvNd):
+ weight_clip.register(module)
+
+ @staticmethod
+ def _transform_inputs(x):
+ """Transform inputs for decoder.
+
+ Args:
+ inputs (tuple or list of Tensor | Tensor): multi-level features.
+
+ Returns:
+ Tensor: The transformed inputs
+ """
+ if not isinstance(x, (list, tuple)):
+ return x
+
+ assert len(x) > 0
+
+ # return the top-level feature of the 1D feature pyramid
+ return x[-1]
+
+ def forward(self, x):
+ """Forward function."""
+ x = self._transform_inputs(x)
+
+ assert x.ndim == 3 and x.shape[2] == 1, f'Invalid shape {x.shape}'
+ output = self.conv(x)
+ N = output.shape[0]
+ return output.reshape(N, self.num_joints, 3)
+
+ def get_loss(self, output, target, target_weight):
+ """Calculate keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+
+ Args:
+ output (torch.Tensor[N, K, 3]): Output keypoints.
+ target (torch.Tensor[N, K, 3]): Target keypoints.
+ target_weight (torch.Tensor[N, K, 3]):
+ Weights across different joint types.
+ If self.is_trajectory is True and target_weight is None,
+ target_weight will be set inversely proportional to joint
+ depth.
+ """
+ losses = dict()
+ assert not isinstance(self.loss, nn.Sequential)
+
+ # trajectory model
+ if self.is_trajectory:
+ if target.dim() == 2:
+ target.unsqueeze_(1)
+
+ if target_weight is None:
+ target_weight = (1 / target[:, :, 2:]).expand(target.shape)
+ assert target.dim() == 3 and target_weight.dim() == 3
+
+ losses['traj_loss'] = self.loss(output, target, target_weight)
+
+ # pose model
+ else:
+ if target_weight is None:
+ target_weight = target.new_ones(target.shape)
+ assert target.dim() == 3 and target_weight.dim() == 3
+ losses['reg_loss'] = self.loss(output, target, target_weight)
+
+ return losses
+
+ def get_accuracy(self, output, target, target_weight, metas):
+ """Calculate accuracy for keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+
+ Args:
+ output (torch.Tensor[N, K, 3]): Output keypoints.
+ target (torch.Tensor[N, K, 3]): Target keypoints.
+ target_weight (torch.Tensor[N, K, 3]):
+ Weights across different joint types.
+ metas (list(dict)): Information about data augmentation including:
+
+ - target_image_path (str): Optional, path to the image file
+ - target_mean (float): Optional, normalization parameter of
+ the target pose.
+ - target_std (float): Optional, normalization parameter of the
+ target pose.
+ - root_position (np.ndarray[3,1]): Optional, global
+ position of the root joint.
+ - root_index (torch.ndarray[1,]): Optional, original index of
+ the root joint before root-centering.
+ """
+
+ accuracy = dict()
+
+ N = output.shape[0]
+ output_ = output.detach().cpu().numpy()
+ target_ = target.detach().cpu().numpy()
+ # Denormalize the predicted pose
+ if 'target_mean' in metas[0] and 'target_std' in metas[0]:
+ target_mean = np.stack([m['target_mean'] for m in metas])
+ target_std = np.stack([m['target_std'] for m in metas])
+ output_ = self._denormalize_joints(output_, target_mean,
+ target_std)
+ target_ = self._denormalize_joints(target_, target_mean,
+ target_std)
+
+ # Restore global position
+ if self.test_cfg.get('restore_global_position', False):
+ root_pos = np.stack([m['root_position'] for m in metas])
+ root_idx = metas[0].get('root_position_index', None)
+ output_ = self._restore_global_position(output_, root_pos,
+ root_idx)
+ target_ = self._restore_global_position(target_, root_pos,
+ root_idx)
+ # Get target weight
+ if target_weight is None:
+ target_weight_ = np.ones_like(target_)
+ else:
+ target_weight_ = target_weight.detach().cpu().numpy()
+ if self.test_cfg.get('restore_global_position', False):
+ root_idx = metas[0].get('root_position_index', None)
+ root_weight = metas[0].get('root_joint_weight', 1.0)
+ target_weight_ = self._restore_root_target_weight(
+ target_weight_, root_weight, root_idx)
+
+ mpjpe = np.mean(
+ np.linalg.norm((output_ - target_) * target_weight_, axis=-1))
+
+ transformed_output = np.zeros_like(output_)
+ for i in range(N):
+ transformed_output[i, :, :] = compute_similarity_transform(
+ output_[i, :, :], target_[i, :, :])
+ p_mpjpe = np.mean(
+ np.linalg.norm(
+ (transformed_output - target_) * target_weight_, axis=-1))
+
+ accuracy['mpjpe'] = output.new_tensor(mpjpe)
+ accuracy['p_mpjpe'] = output.new_tensor(p_mpjpe)
+
+ return accuracy
+
+ def inference_model(self, x, flip_pairs=None):
+ """Inference function.
+
+ Returns:
+ output_regression (np.ndarray): Output regression.
+
+ Args:
+ x (torch.Tensor[N, K, 2]): Input features.
+ flip_pairs (None | list[tuple()):
+ Pairs of keypoints which are mirrored.
+ """
+ output = self.forward(x)
+
+ if flip_pairs is not None:
+ output_regression = fliplr_regression(
+ output.detach().cpu().numpy(),
+ flip_pairs,
+ center_mode='static',
+ center_x=0)
+ else:
+ output_regression = output.detach().cpu().numpy()
+ return output_regression
+
+ def decode(self, metas, output):
+ """Decode the keypoints from output regression.
+
+ Args:
+ metas (list(dict)): Information about data augmentation.
+ By default this includes:
+
+ - "target_image_path": path to the image file
+ output (np.ndarray[N, K, 3]): predicted regression vector.
+ metas (list(dict)): Information about data augmentation including:
+
+ - target_image_path (str): Optional, path to the image file
+ - target_mean (float): Optional, normalization parameter of
+ the target pose.
+ - target_std (float): Optional, normalization parameter of the
+ target pose.
+ - root_position (np.ndarray[3,1]): Optional, global
+ position of the root joint.
+ - root_index (torch.ndarray[1,]): Optional, original index of
+ the root joint before root-centering.
+ """
+
+ # Denormalize the predicted pose
+ if 'target_mean' in metas[0] and 'target_std' in metas[0]:
+ target_mean = np.stack([m['target_mean'] for m in metas])
+ target_std = np.stack([m['target_std'] for m in metas])
+ output = self._denormalize_joints(output, target_mean, target_std)
+
+ # Restore global position
+ if self.test_cfg.get('restore_global_position', False):
+ root_pos = np.stack([m['root_position'] for m in metas])
+ root_idx = metas[0].get('root_position_index', None)
+ output = self._restore_global_position(output, root_pos, root_idx)
+
+ target_image_paths = [m.get('target_image_path', None) for m in metas]
+ result = {'preds': output, 'target_image_paths': target_image_paths}
+
+ return result
+
+ @staticmethod
+ def _denormalize_joints(x, mean, std):
+ """Denormalize joint coordinates with given statistics mean and std.
+
+ Args:
+ x (np.ndarray[N, K, 3]): Normalized joint coordinates.
+ mean (np.ndarray[K, 3]): Mean value.
+ std (np.ndarray[K, 3]): Std value.
+ """
+ assert x.ndim == 3
+ assert x.shape == mean.shape == std.shape
+
+ return x * std + mean
+
+ @staticmethod
+ def _restore_global_position(x, root_pos, root_idx=None):
+ """Restore global position of the root-centered joints.
+
+ Args:
+ x (np.ndarray[N, K, 3]): root-centered joint coordinates
+ root_pos (np.ndarray[N,1,3]): The global position of the
+ root joint.
+ root_idx (int|None): If not none, the root joint will be inserted
+ back to the pose at the given index.
+ """
+ x = x + root_pos
+ if root_idx is not None:
+ x = np.insert(x, root_idx, root_pos.squeeze(1), axis=1)
+ return x
+
+ @staticmethod
+ def _restore_root_target_weight(target_weight, root_weight, root_idx=None):
+ """Restore the target weight of the root joint after the restoration of
+ the global position.
+
+ Args:
+ target_weight (np.ndarray[N, K, 1]): Target weight of relativized
+ joints.
+ root_weight (float): The target weight value of the root joint.
+ root_idx (int|None): If not none, the root joint weight will be
+ inserted back to the target weight at the given index.
+ """
+ if root_idx is not None:
+ root_weight = np.full(
+ target_weight.shape[0], root_weight, dtype=target_weight.dtype)
+ target_weight = np.insert(
+ target_weight, root_idx, root_weight[:, None], axis=1)
+ return target_weight
+
+ def init_weights(self):
+ """Initialize the weights."""
+ for m in self.modules():
+ if isinstance(m, nn.modules.conv._ConvNd):
+ kaiming_init(m, mode='fan_in', nonlinearity='relu')
+ elif isinstance(m, _BatchNorm):
+ constant_init(m, 1)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/topdown_heatmap_base_head.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/topdown_heatmap_base_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..08d483c652210855e244ea525f7e59515b79a192
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/topdown_heatmap_base_head.py
@@ -0,0 +1,120 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+from abc import ABCMeta, abstractmethod
+
+import numpy as np
+import torch.nn as nn
+
+# from mmpose.core.evaluation.top_down_eval import keypoints_from_heatmaps
+
+
+class TopdownHeatmapBaseHead(nn.Module):
+ """Base class for top-down heatmap heads.
+
+ All top-down heatmap heads should subclass it.
+ All subclass should overwrite:
+
+ Methods:`get_loss`, supporting to calculate loss.
+ Methods:`get_accuracy`, supporting to calculate accuracy.
+ Methods:`forward`, supporting to forward model.
+ Methods:`inference_model`, supporting to inference model.
+ """
+
+ __metaclass__ = ABCMeta
+
+ @abstractmethod
+ def get_loss(self, **kwargs):
+ """Gets the loss."""
+
+ @abstractmethod
+ def get_accuracy(self, **kwargs):
+ """Gets the accuracy."""
+
+ @abstractmethod
+ def forward(self, **kwargs):
+ """Forward function."""
+
+ @abstractmethod
+ def inference_model(self, **kwargs):
+ """Inference function."""
+
+ def decode(self, img_metas, output, **kwargs):
+ """Decode keypoints from heatmaps.
+
+ Args:
+ img_metas (list(dict)): Information about data augmentation
+ By default this includes:
+
+ - "image_file: path to the image file
+ - "center": center of the bbox
+ - "scale": scale of the bbox
+ - "rotation": rotation of the bbox
+ - "bbox_score": score of bbox
+ output (np.ndarray[N, K, H, W]): model predicted heatmaps.
+ """
+ # batch_size = len(img_metas)
+
+ # if 'bbox_id' in img_metas[0]:
+ # bbox_ids = []
+ # else:
+ # bbox_ids = None
+
+ # c = np.zeros((batch_size, 2), dtype=np.float32)
+ # s = np.zeros((batch_size, 2), dtype=np.float32)
+ # image_paths = []
+ # score = np.ones(batch_size)
+ # for i in range(batch_size):
+ # c[i, :] = img_metas[i]['center']
+ # s[i, :] = img_metas[i]['scale']
+ # image_paths.append(img_metas[i]['image_file'])
+
+ # if 'bbox_score' in img_metas[i]:
+ # score[i] = np.array(img_metas[i]['bbox_score']).reshape(-1)
+ # if bbox_ids is not None:
+ # bbox_ids.append(img_metas[i]['bbox_id'])
+
+ # preds, maxvals = keypoints_from_heatmaps(
+ # output,
+ # c,
+ # s,
+ # unbiased=self.test_cfg.get('unbiased_decoding', False),
+ # post_process=self.test_cfg.get('post_process', 'default'),
+ # kernel=self.test_cfg.get('modulate_kernel', 11),
+ # valid_radius_factor=self.test_cfg.get('valid_radius_factor',
+ # 0.0546875),
+ # use_udp=self.test_cfg.get('use_udp', False),
+ # target_type=self.test_cfg.get('target_type', 'GaussianHeatmap'))
+
+ # all_preds = np.zeros((batch_size, preds.shape[1], 3), dtype=np.float32)
+ # all_boxes = np.zeros((batch_size, 6), dtype=np.float32)
+ # all_preds[:, :, 0:2] = preds[:, :, 0:2]
+ # all_preds[:, :, 2:3] = maxvals
+ # all_boxes[:, 0:2] = c[:, 0:2]
+ # all_boxes[:, 2:4] = s[:, 0:2]
+ # all_boxes[:, 4] = np.prod(s * 200.0, axis=1)
+ # all_boxes[:, 5] = score
+
+ # result = {}
+
+ # result['preds'] = all_preds
+ # result['boxes'] = all_boxes
+ # result['image_paths'] = image_paths
+ # result['bbox_ids'] = bbox_ids
+
+ return None
+
+ @staticmethod
+ def _get_deconv_cfg(deconv_kernel):
+ """Get configurations for deconv layers."""
+ if deconv_kernel == 4:
+ padding = 1
+ output_padding = 0
+ elif deconv_kernel == 3:
+ padding = 1
+ output_padding = 1
+ elif deconv_kernel == 2:
+ padding = 0
+ output_padding = 0
+ else:
+ raise ValueError(f'Not supported num_kernels ({deconv_kernel}).')
+
+ return deconv_kernel, padding, output_padding
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/topdown_heatmap_multi_stage_head.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/topdown_heatmap_multi_stage_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..c439f5b6332d72a66db75bf599035411c4e1e0d1
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/topdown_heatmap_multi_stage_head.py
@@ -0,0 +1,572 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import copy as cp
+
+import torch.nn as nn
+from mmcv.cnn import (ConvModule, DepthwiseSeparableConvModule, Linear,
+ build_activation_layer, build_conv_layer,
+ build_norm_layer, build_upsample_layer, constant_init,
+ kaiming_init, normal_init)
+
+from mmpose.core.evaluation import pose_pck_accuracy
+from mmpose.core.post_processing import flip_back
+from mmpose.models.builder import build_loss
+from ..builder import HEADS
+from .topdown_heatmap_base_head import TopdownHeatmapBaseHead
+
+
+@HEADS.register_module()
+class TopdownHeatmapMultiStageHead(TopdownHeatmapBaseHead):
+ """Top-down heatmap multi-stage head.
+
+ TopdownHeatmapMultiStageHead is consisted of multiple branches,
+ each of which has num_deconv_layers(>=0) number of deconv layers
+ and a simple conv2d layer.
+
+ Args:
+ in_channels (int): Number of input channels.
+ out_channels (int): Number of output channels.
+ num_stages (int): Number of stages.
+ num_deconv_layers (int): Number of deconv layers.
+ num_deconv_layers should >= 0. Note that 0 means
+ no deconv layers.
+ num_deconv_filters (list|tuple): Number of filters.
+ If num_deconv_layers > 0, the length of
+ num_deconv_kernels (list|tuple): Kernel sizes.
+ loss_keypoint (dict): Config for keypoint loss. Default: None.
+ """
+
+ def __init__(self,
+ in_channels=512,
+ out_channels=17,
+ num_stages=1,
+ num_deconv_layers=3,
+ num_deconv_filters=(256, 256, 256),
+ num_deconv_kernels=(4, 4, 4),
+ extra=None,
+ loss_keypoint=None,
+ train_cfg=None,
+ test_cfg=None):
+ super().__init__()
+
+ self.in_channels = in_channels
+ self.num_stages = num_stages
+ self.loss = build_loss(loss_keypoint)
+
+ self.train_cfg = {} if train_cfg is None else train_cfg
+ self.test_cfg = {} if test_cfg is None else test_cfg
+ self.target_type = self.test_cfg.get('target_type', 'GaussianHeatmap')
+
+ if extra is not None and not isinstance(extra, dict):
+ raise TypeError('extra should be dict or None.')
+
+ # build multi-stage deconv layers
+ self.multi_deconv_layers = nn.ModuleList([])
+ for _ in range(self.num_stages):
+ if num_deconv_layers > 0:
+ deconv_layers = self._make_deconv_layer(
+ num_deconv_layers,
+ num_deconv_filters,
+ num_deconv_kernels,
+ )
+ elif num_deconv_layers == 0:
+ deconv_layers = nn.Identity()
+ else:
+ raise ValueError(
+ f'num_deconv_layers ({num_deconv_layers}) should >= 0.')
+ self.multi_deconv_layers.append(deconv_layers)
+
+ identity_final_layer = False
+ if extra is not None and 'final_conv_kernel' in extra:
+ assert extra['final_conv_kernel'] in [0, 1, 3]
+ if extra['final_conv_kernel'] == 3:
+ padding = 1
+ elif extra['final_conv_kernel'] == 1:
+ padding = 0
+ else:
+ # 0 for Identity mapping.
+ identity_final_layer = True
+ kernel_size = extra['final_conv_kernel']
+ else:
+ kernel_size = 1
+ padding = 0
+
+ # build multi-stage final layers
+ self.multi_final_layers = nn.ModuleList([])
+ for i in range(self.num_stages):
+ if identity_final_layer:
+ final_layer = nn.Identity()
+ else:
+ final_layer = build_conv_layer(
+ cfg=dict(type='Conv2d'),
+ in_channels=num_deconv_filters[-1]
+ if num_deconv_layers > 0 else in_channels,
+ out_channels=out_channels,
+ kernel_size=kernel_size,
+ stride=1,
+ padding=padding)
+ self.multi_final_layers.append(final_layer)
+
+ def get_loss(self, output, target, target_weight):
+ """Calculate top-down keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+ - num_outputs: O
+ - heatmaps height: H
+ - heatmaps weight: W
+
+ Args:
+ output (torch.Tensor[N,K,H,W]):
+ Output heatmaps.
+ target (torch.Tensor[N,K,H,W]):
+ Target heatmaps.
+ target_weight (torch.Tensor[N,K,1]):
+ Weights across different joint types.
+ """
+
+ losses = dict()
+
+ assert isinstance(output, list)
+ assert target.dim() == 4 and target_weight.dim() == 3
+
+ if isinstance(self.loss, nn.Sequential):
+ assert len(self.loss) == len(output)
+ for i in range(len(output)):
+ target_i = target
+ target_weight_i = target_weight
+ if isinstance(self.loss, nn.Sequential):
+ loss_func = self.loss[i]
+ else:
+ loss_func = self.loss
+ loss_i = loss_func(output[i], target_i, target_weight_i)
+ if 'heatmap_loss' not in losses:
+ losses['heatmap_loss'] = loss_i
+ else:
+ losses['heatmap_loss'] += loss_i
+
+ return losses
+
+ def get_accuracy(self, output, target, target_weight):
+ """Calculate accuracy for top-down keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+ - heatmaps height: H
+ - heatmaps weight: W
+
+ Args:
+ output (torch.Tensor[N,K,H,W]): Output heatmaps.
+ target (torch.Tensor[N,K,H,W]): Target heatmaps.
+ target_weight (torch.Tensor[N,K,1]):
+ Weights across different joint types.
+ """
+
+ accuracy = dict()
+
+ if self.target_type == 'GaussianHeatmap':
+ _, avg_acc, _ = pose_pck_accuracy(
+ output[-1].detach().cpu().numpy(),
+ target.detach().cpu().numpy(),
+ target_weight.detach().cpu().numpy().squeeze(-1) > 0)
+ accuracy['acc_pose'] = float(avg_acc)
+
+ return accuracy
+
+ def forward(self, x):
+ """Forward function.
+
+ Returns:
+ out (list[Tensor]): a list of heatmaps from multiple stages.
+ """
+ out = []
+ assert isinstance(x, list)
+ for i in range(self.num_stages):
+ y = self.multi_deconv_layers[i](x[i])
+ y = self.multi_final_layers[i](y)
+ out.append(y)
+ return out
+
+ def inference_model(self, x, flip_pairs=None):
+ """Inference function.
+
+ Returns:
+ output_heatmap (np.ndarray): Output heatmaps.
+
+ Args:
+ x (List[torch.Tensor[NxKxHxW]]): Input features.
+ flip_pairs (None | list[tuple()):
+ Pairs of keypoints which are mirrored.
+ """
+ output = self.forward(x)
+ assert isinstance(output, list)
+ output = output[-1]
+
+ if flip_pairs is not None:
+ # perform flip
+ output_heatmap = flip_back(
+ output.detach().cpu().numpy(),
+ flip_pairs,
+ target_type=self.target_type)
+ # feature is not aligned, shift flipped heatmap for higher accuracy
+ if self.test_cfg.get('shift_heatmap', False):
+ output_heatmap[:, :, :, 1:] = output_heatmap[:, :, :, :-1]
+ else:
+ output_heatmap = output.detach().cpu().numpy()
+
+ return output_heatmap
+
+ def _make_deconv_layer(self, num_layers, num_filters, num_kernels):
+ """Make deconv layers."""
+ if num_layers != len(num_filters):
+ error_msg = f'num_layers({num_layers}) ' \
+ f'!= length of num_filters({len(num_filters)})'
+ raise ValueError(error_msg)
+ if num_layers != len(num_kernels):
+ error_msg = f'num_layers({num_layers}) ' \
+ f'!= length of num_kernels({len(num_kernels)})'
+ raise ValueError(error_msg)
+
+ layers = []
+ for i in range(num_layers):
+ kernel, padding, output_padding = \
+ self._get_deconv_cfg(num_kernels[i])
+
+ planes = num_filters[i]
+ layers.append(
+ build_upsample_layer(
+ dict(type='deconv'),
+ in_channels=self.in_channels,
+ out_channels=planes,
+ kernel_size=kernel,
+ stride=2,
+ padding=padding,
+ output_padding=output_padding,
+ bias=False))
+ layers.append(nn.BatchNorm2d(planes))
+ layers.append(nn.ReLU(inplace=True))
+ self.in_channels = planes
+
+ return nn.Sequential(*layers)
+
+ def init_weights(self):
+ """Initialize model weights."""
+ for _, m in self.multi_deconv_layers.named_modules():
+ if isinstance(m, nn.ConvTranspose2d):
+ normal_init(m, std=0.001)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
+ for m in self.multi_final_layers.modules():
+ if isinstance(m, nn.Conv2d):
+ normal_init(m, std=0.001, bias=0)
+
+
+class PredictHeatmap(nn.Module):
+ """Predict the heat map for an input feature.
+
+ Args:
+ unit_channels (int): Number of input channels.
+ out_channels (int): Number of output channels.
+ out_shape (tuple): Shape of the output heatmap.
+ use_prm (bool): Whether to use pose refine machine. Default: False.
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ """
+
+ def __init__(self,
+ unit_channels,
+ out_channels,
+ out_shape,
+ use_prm=False,
+ norm_cfg=dict(type='BN')):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+ self.unit_channels = unit_channels
+ self.out_channels = out_channels
+ self.out_shape = out_shape
+ self.use_prm = use_prm
+ if use_prm:
+ self.prm = PRM(out_channels, norm_cfg=norm_cfg)
+ self.conv_layers = nn.Sequential(
+ ConvModule(
+ unit_channels,
+ unit_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=norm_cfg,
+ inplace=False),
+ ConvModule(
+ unit_channels,
+ out_channels,
+ kernel_size=3,
+ stride=1,
+ padding=1,
+ norm_cfg=norm_cfg,
+ act_cfg=None,
+ inplace=False))
+
+ def forward(self, feature):
+ feature = self.conv_layers(feature)
+ output = nn.functional.interpolate(
+ feature, size=self.out_shape, mode='bilinear', align_corners=True)
+ if self.use_prm:
+ output = self.prm(output)
+ return output
+
+
+class PRM(nn.Module):
+ """Pose Refine Machine.
+
+ Please refer to "Learning Delicate Local Representations
+ for Multi-Person Pose Estimation" (ECCV 2020).
+
+ Args:
+ out_channels (int): Channel number of the output. Equals to
+ the number of key points.
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ """
+
+ def __init__(self, out_channels, norm_cfg=dict(type='BN')):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+ self.out_channels = out_channels
+ self.global_pooling = nn.AdaptiveAvgPool2d((1, 1))
+ self.middle_path = nn.Sequential(
+ Linear(self.out_channels, self.out_channels),
+ build_norm_layer(dict(type='BN1d'), out_channels)[1],
+ build_activation_layer(dict(type='ReLU')),
+ Linear(self.out_channels, self.out_channels),
+ build_norm_layer(dict(type='BN1d'), out_channels)[1],
+ build_activation_layer(dict(type='ReLU')),
+ build_activation_layer(dict(type='Sigmoid')))
+
+ self.bottom_path = nn.Sequential(
+ ConvModule(
+ self.out_channels,
+ self.out_channels,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ norm_cfg=norm_cfg,
+ inplace=False),
+ DepthwiseSeparableConvModule(
+ self.out_channels,
+ 1,
+ kernel_size=9,
+ stride=1,
+ padding=4,
+ norm_cfg=norm_cfg,
+ inplace=False), build_activation_layer(dict(type='Sigmoid')))
+ self.conv_bn_relu_prm_1 = ConvModule(
+ self.out_channels,
+ self.out_channels,
+ kernel_size=3,
+ stride=1,
+ padding=1,
+ norm_cfg=norm_cfg,
+ inplace=False)
+
+ def forward(self, x):
+ out = self.conv_bn_relu_prm_1(x)
+ out_1 = out
+
+ out_2 = self.global_pooling(out_1)
+ out_2 = out_2.view(out_2.size(0), -1)
+ out_2 = self.middle_path(out_2)
+ out_2 = out_2.unsqueeze(2)
+ out_2 = out_2.unsqueeze(3)
+
+ out_3 = self.bottom_path(out_1)
+ out = out_1 * (1 + out_2 * out_3)
+
+ return out
+
+
+@HEADS.register_module()
+class TopdownHeatmapMSMUHead(TopdownHeatmapBaseHead):
+ """Heads for multi-stage multi-unit heads used in Multi-Stage Pose
+ estimation Network (MSPN), and Residual Steps Networks (RSN).
+
+ Args:
+ unit_channels (int): Number of input channels.
+ out_channels (int): Number of output channels.
+ out_shape (tuple): Shape of the output heatmap.
+ num_stages (int): Number of stages.
+ num_units (int): Number of units in each stage.
+ use_prm (bool): Whether to use pose refine machine (PRM).
+ Default: False.
+ norm_cfg (dict): dictionary to construct and config norm layer.
+ Default: dict(type='BN')
+ loss_keypoint (dict): Config for keypoint loss. Default: None.
+ """
+
+ def __init__(self,
+ out_shape,
+ unit_channels=256,
+ out_channels=17,
+ num_stages=4,
+ num_units=4,
+ use_prm=False,
+ norm_cfg=dict(type='BN'),
+ loss_keypoint=None,
+ train_cfg=None,
+ test_cfg=None):
+ # Protect mutable default arguments
+ norm_cfg = cp.deepcopy(norm_cfg)
+ super().__init__()
+
+ self.train_cfg = {} if train_cfg is None else train_cfg
+ self.test_cfg = {} if test_cfg is None else test_cfg
+ self.target_type = self.test_cfg.get('target_type', 'GaussianHeatmap')
+
+ self.out_shape = out_shape
+ self.unit_channels = unit_channels
+ self.out_channels = out_channels
+ self.num_stages = num_stages
+ self.num_units = num_units
+
+ self.loss = build_loss(loss_keypoint)
+
+ self.predict_layers = nn.ModuleList([])
+ for i in range(self.num_stages):
+ for j in range(self.num_units):
+ self.predict_layers.append(
+ PredictHeatmap(
+ unit_channels,
+ out_channels,
+ out_shape,
+ use_prm,
+ norm_cfg=norm_cfg))
+
+ def get_loss(self, output, target, target_weight):
+ """Calculate top-down keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+ - num_outputs: O
+ - heatmaps height: H
+ - heatmaps weight: W
+
+ Args:
+ output (torch.Tensor[N,O,K,H,W]): Output heatmaps.
+ target (torch.Tensor[N,O,K,H,W]): Target heatmaps.
+ target_weight (torch.Tensor[N,O,K,1]):
+ Weights across different joint types.
+ """
+
+ losses = dict()
+
+ assert isinstance(output, list)
+ assert target.dim() == 5 and target_weight.dim() == 4
+ assert target.size(1) == len(output)
+
+ if isinstance(self.loss, nn.Sequential):
+ assert len(self.loss) == len(output)
+ for i in range(len(output)):
+ target_i = target[:, i, :, :, :]
+ target_weight_i = target_weight[:, i, :, :]
+
+ if isinstance(self.loss, nn.Sequential):
+ loss_func = self.loss[i]
+ else:
+ loss_func = self.loss
+
+ loss_i = loss_func(output[i], target_i, target_weight_i)
+ if 'heatmap_loss' not in losses:
+ losses['heatmap_loss'] = loss_i
+ else:
+ losses['heatmap_loss'] += loss_i
+
+ return losses
+
+ def get_accuracy(self, output, target, target_weight):
+ """Calculate accuracy for top-down keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+ - heatmaps height: H
+ - heatmaps weight: W
+
+ Args:
+ output (torch.Tensor[N,K,H,W]): Output heatmaps.
+ target (torch.Tensor[N,K,H,W]): Target heatmaps.
+ target_weight (torch.Tensor[N,K,1]):
+ Weights across different joint types.
+ """
+
+ accuracy = dict()
+
+ if self.target_type == 'GaussianHeatmap':
+ assert isinstance(output, list)
+ assert target.dim() == 5 and target_weight.dim() == 4
+ _, avg_acc, _ = pose_pck_accuracy(
+ output[-1].detach().cpu().numpy(),
+ target[:, -1, ...].detach().cpu().numpy(),
+ target_weight[:, -1,
+ ...].detach().cpu().numpy().squeeze(-1) > 0)
+ accuracy['acc_pose'] = float(avg_acc)
+
+ return accuracy
+
+ def forward(self, x):
+ """Forward function.
+
+ Returns:
+ out (list[Tensor]): a list of heatmaps from multiple stages
+ and units.
+ """
+ out = []
+ assert isinstance(x, list)
+ assert len(x) == self.num_stages
+ assert isinstance(x[0], list)
+ assert len(x[0]) == self.num_units
+ assert x[0][0].shape[1] == self.unit_channels
+ for i in range(self.num_stages):
+ for j in range(self.num_units):
+ y = self.predict_layers[i * self.num_units + j](x[i][j])
+ out.append(y)
+
+ return out
+
+ def inference_model(self, x, flip_pairs=None):
+ """Inference function.
+
+ Returns:
+ output_heatmap (np.ndarray): Output heatmaps.
+
+ Args:
+ x (list[torch.Tensor[N,K,H,W]]): Input features.
+ flip_pairs (None | list[tuple]):
+ Pairs of keypoints which are mirrored.
+ """
+ output = self.forward(x)
+ assert isinstance(output, list)
+ output = output[-1]
+ if flip_pairs is not None:
+ output_heatmap = flip_back(
+ output.detach().cpu().numpy(),
+ flip_pairs,
+ target_type=self.target_type)
+ # feature is not aligned, shift flipped heatmap for higher accuracy
+ if self.test_cfg.get('shift_heatmap', False):
+ output_heatmap[:, :, :, 1:] = output_heatmap[:, :, :, :-1]
+ else:
+ output_heatmap = output.detach().cpu().numpy()
+ return output_heatmap
+
+ def init_weights(self):
+ """Initialize model weights."""
+ for m in self.predict_layers.modules():
+ if isinstance(m, nn.Conv2d):
+ kaiming_init(m)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
+ elif isinstance(m, nn.Linear):
+ normal_init(m, std=0.01)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/topdown_heatmap_simple_head.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/topdown_heatmap_simple_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..9725ab4a84002b06c45d354415fc1bce1edeb263
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/topdown_heatmap_simple_head.py
@@ -0,0 +1,392 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import torch
+import torch.nn as nn
+# from mmcv.cnn import (build_conv_layer, build_norm_layer, build_upsample_layer,
+# constant_init, normal_init)
+
+# from mmpose.core.evaluation import pose_pck_accuracy
+# from mmpose.core.post_processing import flip_back
+# from mmpose.models.builder import build_loss
+# from mmpose.models.utils.ops import resize
+# from ..builder import HEADS
+import torch.nn.functional as F
+from .topdown_heatmap_base_head import TopdownHeatmapBaseHead
+
+def build_conv_layer(cfg, *args, **kwargs) -> nn.Module:
+ """LICENSE"""
+
+ if cfg is None:
+ cfg_ = dict(type='Conv2d')
+ else:
+ if not isinstance(cfg, dict):
+ raise TypeError('cfg must be a dict')
+ if 'type' not in cfg:
+ raise KeyError('the cfg dict must contain the key "type"')
+ cfg_ = cfg.copy()
+
+ layer_type = cfg_.pop('type')
+ if layer_type !='Conv2d':
+ raise KeyError(f'Unrecognized layer type {layer_type}')
+ else:
+ conv_layer = nn.Conv2d
+
+ layer = conv_layer(*args, **kwargs, **cfg_)
+
+ return layer
+
+def build_upsample_layer(cfg, *args, **kwargs) -> nn.Module:
+
+ if not isinstance(cfg, dict):
+ raise TypeError(f'cfg must be a dict, but got {type(cfg)}')
+ if 'type' not in cfg:
+ raise KeyError(
+ f'the cfg dict must contain the key "type", but got {cfg}')
+ cfg_ = cfg.copy()
+
+ layer_type = cfg_.pop('type')
+ if layer_type !='deconv':
+ raise KeyError(f'Unrecognized upsample type {layer_type}')
+ else:
+ upsample = nn.ConvTranspose2d
+
+ if upsample is nn.Upsample:
+ cfg_['mode'] = layer_type
+ layer = upsample(*args, **kwargs, **cfg_)
+ return layer
+
+# @HEADS.register_module()
+class TopdownHeatmapSimpleHead(TopdownHeatmapBaseHead):
+ """Top-down heatmap simple head. paper ref: Bin Xiao et al. ``Simple
+ Baselines for Human Pose Estimation and Tracking``.
+
+ TopdownHeatmapSimpleHead is consisted of (>=0) number of deconv layers
+ and a simple conv2d layer.
+
+ Args:
+ in_channels (int): Number of input channels
+ out_channels (int): Number of output channels
+ num_deconv_layers (int): Number of deconv layers.
+ num_deconv_layers should >= 0. Note that 0 means
+ no deconv layers.
+ num_deconv_filters (list|tuple): Number of filters.
+ If num_deconv_layers > 0, the length of
+ num_deconv_kernels (list|tuple): Kernel sizes.
+ in_index (int|Sequence[int]): Input feature index. Default: 0
+ input_transform (str|None): Transformation type of input features.
+ Options: 'resize_concat', 'multiple_select', None.
+ Default: None.
+
+ - 'resize_concat': Multiple feature maps will be resized to the
+ same size as the first one and then concat together.
+ Usually used in FCN head of HRNet.
+ - 'multiple_select': Multiple feature maps will be bundle into
+ a list and passed into decode head.
+ - None: Only one select feature map is allowed.
+ align_corners (bool): align_corners argument of F.interpolate.
+ Default: False.
+ loss_keypoint (dict): Config for keypoint loss. Default: None.
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ num_deconv_layers=3,
+ num_deconv_filters=(256, 256, 256),
+ num_deconv_kernels=(4, 4, 4),
+ extra=None,
+ in_index=0,
+ input_transform=None,
+ align_corners=False,
+ loss_keypoint=None,
+ train_cfg=None,
+ test_cfg=None,
+ upsample=0,):
+ super().__init__()
+
+ self.in_channels = in_channels
+ self.loss = None
+ self.upsample = upsample
+
+ self.train_cfg = {} if train_cfg is None else train_cfg
+ self.test_cfg = {} if test_cfg is None else test_cfg
+ self.target_type = self.test_cfg.get('target_type', 'GaussianHeatmap')
+
+ self._init_inputs(in_channels, in_index, input_transform)
+ self.in_index = in_index
+ self.align_corners = align_corners
+
+ if extra is not None and not isinstance(extra, dict):
+ raise TypeError('extra should be dict or None.')
+
+ if num_deconv_layers > 0:
+ self.deconv_layers = self._make_deconv_layer(
+ num_deconv_layers,
+ num_deconv_filters,
+ num_deconv_kernels,
+ )
+ elif num_deconv_layers == 0:
+ self.deconv_layers = nn.Identity()
+ else:
+ raise ValueError(
+ f'num_deconv_layers ({num_deconv_layers}) should >= 0.')
+
+ identity_final_layer = False
+ if extra is not None and 'final_conv_kernel' in extra:
+ assert extra['final_conv_kernel'] in [0, 1, 3]
+ if extra['final_conv_kernel'] == 3:
+ padding = 1
+ elif extra['final_conv_kernel'] == 1:
+ padding = 0
+ else:
+ # 0 for Identity mapping.
+ identity_final_layer = True
+ kernel_size = extra['final_conv_kernel']
+ else:
+ kernel_size = 1
+ padding = 0
+
+ if identity_final_layer:
+ self.final_layer = nn.Identity()
+ else:
+ conv_channels = num_deconv_filters[
+ -1] if num_deconv_layers > 0 else self.in_channels
+
+ layers = []
+ if extra is not None:
+ num_conv_layers = extra.get('num_conv_layers', 0)
+ num_conv_kernels = extra.get('num_conv_kernels',
+ [1] * num_conv_layers)
+
+ for i in range(num_conv_layers):
+ layers.append(
+ build_conv_layer(
+ dict(type='Conv2d'),
+ in_channels=conv_channels,
+ out_channels=conv_channels,
+ kernel_size=num_conv_kernels[i],
+ stride=1,
+ padding=(num_conv_kernels[i] - 1) // 2))
+ layers.append(
+ nn.BatchNorm2d(conv_channels)
+)
+ layers.append(nn.ReLU(inplace=True))
+
+ layers.append(
+ build_conv_layer(
+ cfg=dict(type='Conv2d'),
+ in_channels=conv_channels,
+ out_channels=out_channels,
+ kernel_size=kernel_size,
+ stride=1,
+ padding=padding))
+
+ if len(layers) > 1:
+ self.final_layer = nn.Sequential(*layers)
+ else:
+ self.final_layer = layers[0]
+
+ def get_loss(self, output, target, target_weight):
+ """Calculate top-down keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+ - heatmaps height: H
+ - heatmaps weight: W
+
+ Args:
+ output (torch.Tensor[N,K,H,W]): Output heatmaps.
+ target (torch.Tensor[N,K,H,W]): Target heatmaps.
+ target_weight (torch.Tensor[N,K,1]):
+ Weights across different joint types.
+ """
+
+ losses = dict()
+
+ assert not isinstance(self.loss, nn.Sequential)
+ assert target.dim() == 4 and target_weight.dim() == 3
+ losses['heatmap_loss'] = self.loss(output, target, target_weight)
+
+ return losses
+
+ def get_accuracy(self, output, target, target_weight):
+ """Calculate accuracy for top-down keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+ - heatmaps height: H
+ - heatmaps weight: W
+
+ Args:
+ output (torch.Tensor[N,K,H,W]): Output heatmaps.
+ target (torch.Tensor[N,K,H,W]): Target heatmaps.
+ target_weight (torch.Tensor[N,K,1]):
+ Weights across different joint types.
+ """
+
+ accuracy = dict()
+
+ if self.target_type == 'GaussianHeatmap':
+ _, avg_acc, _ = pose_pck_accuracy(
+ output.detach().cpu().numpy(),
+ target.detach().cpu().numpy(),
+ target_weight.detach().cpu().numpy().squeeze(-1) > 0)
+ accuracy['acc_pose'] = float(avg_acc)
+
+ return accuracy
+
+ def forward(self, x):
+ """Forward function."""
+ x = self._transform_inputs(x)
+ x = self.deconv_layers(x)
+ x = self.final_layer(x)
+ return x
+
+ def inference_model(self, x, flip_pairs=None):
+ """Inference function.
+
+ Returns:
+ output_heatmap (np.ndarray): Output heatmaps.
+
+ Args:
+ x (torch.Tensor[N,K,H,W]): Input features.
+ flip_pairs (None | list[tuple]):
+ Pairs of keypoints which are mirrored.
+ """
+ output = self.forward(x)
+
+ if flip_pairs is not None:
+ output_heatmap = flip_back(
+ output.detach().cpu().numpy(),
+ flip_pairs,
+ target_type=self.target_type)
+ # feature is not aligned, shift flipped heatmap for higher accuracy
+ if self.test_cfg.get('shift_heatmap', False):
+ output_heatmap[:, :, :, 1:] = output_heatmap[:, :, :, :-1]
+ else:
+ output_heatmap = output.detach().cpu().numpy()
+ return output_heatmap
+
+ def _init_inputs(self, in_channels, in_index, input_transform):
+ """Check and initialize input transforms.
+
+ The in_channels, in_index and input_transform must match.
+ Specifically, when input_transform is None, only single feature map
+ will be selected. So in_channels and in_index must be of type int.
+ When input_transform is not None, in_channels and in_index must be
+ list or tuple, with the same length.
+
+ Args:
+ in_channels (int|Sequence[int]): Input channels.
+ in_index (int|Sequence[int]): Input feature index.
+ input_transform (str|None): Transformation type of input features.
+ Options: 'resize_concat', 'multiple_select', None.
+
+ - 'resize_concat': Multiple feature maps will be resize to the
+ same size as first one and than concat together.
+ Usually used in FCN head of HRNet.
+ - 'multiple_select': Multiple feature maps will be bundle into
+ a list and passed into decode head.
+ - None: Only one select feature map is allowed.
+ """
+
+ if input_transform is not None:
+ assert input_transform in ['resize_concat', 'multiple_select']
+ self.input_transform = input_transform
+ self.in_index = in_index
+ if input_transform is not None:
+ assert isinstance(in_channels, (list, tuple))
+ assert isinstance(in_index, (list, tuple))
+ assert len(in_channels) == len(in_index)
+ if input_transform == 'resize_concat':
+ self.in_channels = sum(in_channels)
+ else:
+ self.in_channels = in_channels
+ else:
+ assert isinstance(in_channels, int)
+ assert isinstance(in_index, int)
+ self.in_channels = in_channels
+
+ def _transform_inputs(self, inputs):
+ """Transform inputs for decoder.
+
+ Args:
+ inputs (list[Tensor] | Tensor): multi-level img features.
+
+ Returns:
+ Tensor: The transformed inputs
+ """
+ if not isinstance(inputs, list):
+ if not isinstance(inputs, list):
+ if self.upsample > 0:
+ inputs = resize(
+ input=F.relu(inputs),
+ scale_factor=self.upsample,
+ mode='bilinear',
+ align_corners=self.align_corners
+ )
+ return inputs
+
+ if self.input_transform == 'resize_concat':
+ inputs = [inputs[i] for i in self.in_index]
+ upsampled_inputs = [
+ resize(
+ input=x,
+ size=inputs[0].shape[2:],
+ mode='bilinear',
+ align_corners=self.align_corners) for x in inputs
+ ]
+ inputs = torch.cat(upsampled_inputs, dim=1)
+ elif self.input_transform == 'multiple_select':
+ inputs = [inputs[i] for i in self.in_index]
+ else:
+ inputs = inputs[self.in_index]
+
+ return inputs
+
+ def _make_deconv_layer(self, num_layers, num_filters, num_kernels):
+ """Make deconv layers."""
+ if num_layers != len(num_filters):
+ error_msg = f'num_layers({num_layers}) ' \
+ f'!= length of num_filters({len(num_filters)})'
+ raise ValueError(error_msg)
+ if num_layers != len(num_kernels):
+ error_msg = f'num_layers({num_layers}) ' \
+ f'!= length of num_kernels({len(num_kernels)})'
+ raise ValueError(error_msg)
+
+ layers = []
+ for i in range(num_layers):
+ kernel, padding, output_padding = \
+ self._get_deconv_cfg(num_kernels[i])
+
+ planes = num_filters[i]
+ layers.append(
+ build_upsample_layer(
+ dict(type='deconv'),
+ in_channels=self.in_channels,
+ out_channels=planes,
+ kernel_size=kernel,
+ stride=2,
+ padding=padding,
+ output_padding=output_padding,
+ bias=False))
+ layers.append(nn.BatchNorm2d(planes))
+ layers.append(nn.ReLU(inplace=True))
+ self.in_channels = planes
+
+ return nn.Sequential(*layers)
+
+ def init_weights(self):
+ """Initialize model weights."""
+ for _, m in self.deconv_layers.named_modules():
+ if isinstance(m, nn.ConvTranspose2d):
+ normal_init(m, std=0.001)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
+ for m in self.final_layer.modules():
+ if isinstance(m, nn.Conv2d):
+ normal_init(m, std=0.001, bias=0)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/vipnas_heatmap_simple_head.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/vipnas_heatmap_simple_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..5844fd5583af338a161fcc3a68253d1e052dd75d
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/vipnas_heatmap_simple_head.py
@@ -0,0 +1,349 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import torch
+import torch.nn as nn
+# from mmcv.cnn import (build_conv_layer, build_norm_layer, build_upsample_layer,
+# constant_init, normal_init)
+
+# from mmpose.core.evaluation import pose_pck_accuracy
+# from mmpose.core.post_processing import flip_back
+# from mmpose.models.builder import build_loss
+# from mmpose.models.utils.ops import resize
+# from ..builder import HEADS
+# from .topdown_heatmap_base_head import TopdownHeatmapBaseHead
+
+
+# @HEADS.register_module()
+class ViPNASHeatmapSimpleHead(TopdownHeatmapBaseHead):
+ """ViPNAS heatmap simple head.
+
+ ViPNAS: Efficient Video Pose Estimation via Neural Architecture Search.
+ More details can be found in the `paper
+ `__ .
+
+ TopdownHeatmapSimpleHead is consisted of (>=0) number of deconv layers
+ and a simple conv2d layer.
+
+ Args:
+ in_channels (int): Number of input channels
+ out_channels (int): Number of output channels
+ num_deconv_layers (int): Number of deconv layers.
+ num_deconv_layers should >= 0. Note that 0 means
+ no deconv layers.
+ num_deconv_filters (list|tuple): Number of filters.
+ If num_deconv_layers > 0, the length of
+ num_deconv_kernels (list|tuple): Kernel sizes.
+ num_deconv_groups (list|tuple): Group number.
+ in_index (int|Sequence[int]): Input feature index. Default: -1
+ input_transform (str|None): Transformation type of input features.
+ Options: 'resize_concat', 'multiple_select', None.
+ Default: None.
+
+ - 'resize_concat': Multiple feature maps will be resize to the
+ same size as first one and than concat together.
+ Usually used in FCN head of HRNet.
+ - 'multiple_select': Multiple feature maps will be bundle into
+ a list and passed into decode head.
+ - None: Only one select feature map is allowed.
+ align_corners (bool): align_corners argument of F.interpolate.
+ Default: False.
+ loss_keypoint (dict): Config for keypoint loss. Default: None.
+ """
+
+ def __init__(self,
+ in_channels,
+ out_channels,
+ num_deconv_layers=3,
+ num_deconv_filters=(144, 144, 144),
+ num_deconv_kernels=(4, 4, 4),
+ num_deconv_groups=(16, 16, 16),
+ extra=None,
+ in_index=0,
+ input_transform=None,
+ align_corners=False,
+ loss_keypoint=None,
+ train_cfg=None,
+ test_cfg=None):
+ super().__init__()
+
+ self.in_channels = in_channels
+ self.loss = build_loss(loss_keypoint)
+
+ self.train_cfg = {} if train_cfg is None else train_cfg
+ self.test_cfg = {} if test_cfg is None else test_cfg
+ self.target_type = self.test_cfg.get('target_type', 'GaussianHeatmap')
+
+ self._init_inputs(in_channels, in_index, input_transform)
+ self.in_index = in_index
+ self.align_corners = align_corners
+
+ if extra is not None and not isinstance(extra, dict):
+ raise TypeError('extra should be dict or None.')
+
+ if num_deconv_layers > 0:
+ self.deconv_layers = self._make_deconv_layer(
+ num_deconv_layers, num_deconv_filters, num_deconv_kernels,
+ num_deconv_groups)
+ elif num_deconv_layers == 0:
+ self.deconv_layers = nn.Identity()
+ else:
+ raise ValueError(
+ f'num_deconv_layers ({num_deconv_layers}) should >= 0.')
+
+ identity_final_layer = False
+ if extra is not None and 'final_conv_kernel' in extra:
+ assert extra['final_conv_kernel'] in [0, 1, 3]
+ if extra['final_conv_kernel'] == 3:
+ padding = 1
+ elif extra['final_conv_kernel'] == 1:
+ padding = 0
+ else:
+ # 0 for Identity mapping.
+ identity_final_layer = True
+ kernel_size = extra['final_conv_kernel']
+ else:
+ kernel_size = 1
+ padding = 0
+
+ if identity_final_layer:
+ self.final_layer = nn.Identity()
+ else:
+ conv_channels = num_deconv_filters[
+ -1] if num_deconv_layers > 0 else self.in_channels
+
+ layers = []
+ if extra is not None:
+ num_conv_layers = extra.get('num_conv_layers', 0)
+ num_conv_kernels = extra.get('num_conv_kernels',
+ [1] * num_conv_layers)
+
+ for i in range(num_conv_layers):
+ layers.append(
+ build_conv_layer(
+ dict(type='Conv2d'),
+ in_channels=conv_channels,
+ out_channels=conv_channels,
+ kernel_size=num_conv_kernels[i],
+ stride=1,
+ padding=(num_conv_kernels[i] - 1) // 2))
+ layers.append(
+ build_norm_layer(dict(type='BN'), conv_channels)[1])
+ layers.append(nn.ReLU(inplace=True))
+
+ layers.append(
+ build_conv_layer(
+ cfg=dict(type='Conv2d'),
+ in_channels=conv_channels,
+ out_channels=out_channels,
+ kernel_size=kernel_size,
+ stride=1,
+ padding=padding))
+
+ if len(layers) > 1:
+ self.final_layer = nn.Sequential(*layers)
+ else:
+ self.final_layer = layers[0]
+
+ def get_loss(self, output, target, target_weight):
+ """Calculate top-down keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+ - heatmaps height: H
+ - heatmaps weight: W
+
+ Args:
+ output (torch.Tensor[N,K,H,W]): Output heatmaps.
+ target (torch.Tensor[N,K,H,W]): Target heatmaps.
+ target_weight (torch.Tensor[N,K,1]):
+ Weights across different joint types.
+ """
+
+ losses = dict()
+
+ assert not isinstance(self.loss, nn.Sequential)
+ assert target.dim() == 4 and target_weight.dim() == 3
+ losses['heatmap_loss'] = self.loss(output, target, target_weight)
+
+ return losses
+
+ def get_accuracy(self, output, target, target_weight):
+ """Calculate accuracy for top-down keypoint loss.
+
+ Note:
+ - batch_size: N
+ - num_keypoints: K
+ - heatmaps height: H
+ - heatmaps weight: W
+
+ Args:
+ output (torch.Tensor[N,K,H,W]): Output heatmaps.
+ target (torch.Tensor[N,K,H,W]): Target heatmaps.
+ target_weight (torch.Tensor[N,K,1]):
+ Weights across different joint types.
+ """
+
+ accuracy = dict()
+
+ if self.target_type.lower() == 'GaussianHeatmap'.lower():
+ _, avg_acc, _ = pose_pck_accuracy(
+ output.detach().cpu().numpy(),
+ target.detach().cpu().numpy(),
+ target_weight.detach().cpu().numpy().squeeze(-1) > 0)
+ accuracy['acc_pose'] = float(avg_acc)
+
+ return accuracy
+
+ def forward(self, x):
+ """Forward function."""
+ x = self._transform_inputs(x)
+ x = self.deconv_layers(x)
+ x = self.final_layer(x)
+ return x
+
+ def inference_model(self, x, flip_pairs=None):
+ """Inference function.
+
+ Returns:
+ output_heatmap (np.ndarray): Output heatmaps.
+
+ Args:
+ x (torch.Tensor[N,K,H,W]): Input features.
+ flip_pairs (None | list[tuple]):
+ Pairs of keypoints which are mirrored.
+ """
+ output = self.forward(x)
+
+ if flip_pairs is not None:
+ output_heatmap = flip_back(
+ output.detach().cpu().numpy(),
+ flip_pairs,
+ target_type=self.target_type)
+ # feature is not aligned, shift flipped heatmap for higher accuracy
+ if self.test_cfg.get('shift_heatmap', False):
+ output_heatmap[:, :, :, 1:] = output_heatmap[:, :, :, :-1]
+ else:
+ output_heatmap = output.detach().cpu().numpy()
+ return output_heatmap
+
+ def _init_inputs(self, in_channels, in_index, input_transform):
+ """Check and initialize input transforms.
+
+ The in_channels, in_index and input_transform must match.
+ Specifically, when input_transform is None, only single feature map
+ will be selected. So in_channels and in_index must be of type int.
+ When input_transform is not None, in_channels and in_index must be
+ list or tuple, with the same length.
+
+ Args:
+ in_channels (int|Sequence[int]): Input channels.
+ in_index (int|Sequence[int]): Input feature index.
+ input_transform (str|None): Transformation type of input features.
+ Options: 'resize_concat', 'multiple_select', None.
+
+ - 'resize_concat': Multiple feature maps will be resize to the
+ same size as first one and than concat together.
+ Usually used in FCN head of HRNet.
+ - 'multiple_select': Multiple feature maps will be bundle into
+ a list and passed into decode head.
+ - None: Only one select feature map is allowed.
+ """
+
+ if input_transform is not None:
+ assert input_transform in ['resize_concat', 'multiple_select']
+ self.input_transform = input_transform
+ self.in_index = in_index
+ if input_transform is not None:
+ assert isinstance(in_channels, (list, tuple))
+ assert isinstance(in_index, (list, tuple))
+ assert len(in_channels) == len(in_index)
+ if input_transform == 'resize_concat':
+ self.in_channels = sum(in_channels)
+ else:
+ self.in_channels = in_channels
+ else:
+ assert isinstance(in_channels, int)
+ assert isinstance(in_index, int)
+ self.in_channels = in_channels
+
+ def _transform_inputs(self, inputs):
+ """Transform inputs for decoder.
+
+ Args:
+ inputs (list[Tensor] | Tensor): multi-level img features.
+
+ Returns:
+ Tensor: The transformed inputs
+ """
+ if not isinstance(inputs, list):
+ return inputs
+
+ if self.input_transform == 'resize_concat':
+ inputs = [inputs[i] for i in self.in_index]
+ upsampled_inputs = [
+ resize(
+ input=x,
+ size=inputs[0].shape[2:],
+ mode='bilinear',
+ align_corners=self.align_corners) for x in inputs
+ ]
+ inputs = torch.cat(upsampled_inputs, dim=1)
+ elif self.input_transform == 'multiple_select':
+ inputs = [inputs[i] for i in self.in_index]
+ else:
+ inputs = inputs[self.in_index]
+
+ return inputs
+
+ def _make_deconv_layer(self, num_layers, num_filters, num_kernels,
+ num_groups):
+ """Make deconv layers."""
+ if num_layers != len(num_filters):
+ error_msg = f'num_layers({num_layers}) ' \
+ f'!= length of num_filters({len(num_filters)})'
+ raise ValueError(error_msg)
+ if num_layers != len(num_kernels):
+ error_msg = f'num_layers({num_layers}) ' \
+ f'!= length of num_kernels({len(num_kernels)})'
+ raise ValueError(error_msg)
+ if num_layers != len(num_groups):
+ error_msg = f'num_layers({num_layers}) ' \
+ f'!= length of num_groups({len(num_groups)})'
+ raise ValueError(error_msg)
+
+ layers = []
+ for i in range(num_layers):
+ kernel, padding, output_padding = \
+ self._get_deconv_cfg(num_kernels[i])
+
+ planes = num_filters[i]
+ groups = num_groups[i]
+ layers.append(
+ build_upsample_layer(
+ dict(type='deconv'),
+ in_channels=self.in_channels,
+ out_channels=planes,
+ kernel_size=kernel,
+ groups=groups,
+ stride=2,
+ padding=padding,
+ output_padding=output_padding,
+ bias=False))
+ layers.append(nn.BatchNorm2d(planes))
+ layers.append(nn.ReLU(inplace=True))
+ self.in_channels = planes
+
+ return nn.Sequential(*layers)
+
+ def init_weights(self):
+ """Initialize model weights."""
+ for _, m in self.deconv_layers.named_modules():
+ if isinstance(m, nn.ConvTranspose2d):
+ normal_init(m, std=0.001)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
+ for m in self.final_layer.modules():
+ if isinstance(m, nn.Conv2d):
+ normal_init(m, std=0.001, bias=0)
+ elif isinstance(m, nn.BatchNorm2d):
+ constant_init(m, 1)
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/voxelpose_head.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/voxelpose_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..8799bdc2c0a888973f6cf98f3da00c60a891e699
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/heads/voxelpose_head.py
@@ -0,0 +1,167 @@
+# ------------------------------------------------------------------------------
+# Copyright and License Information
+# https://github.com/microsoft/voxelpose-pytorch/blob/main/lib/models
+# Original Licence: MIT License
+# ------------------------------------------------------------------------------
+
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+
+from ..builder import HEADS
+
+
+@HEADS.register_module()
+class CuboidCenterHead(nn.Module):
+ """Get results from the 3D human center heatmap. In this module, human 3D
+ centers are local maximums obtained from the 3D heatmap via NMS (max-
+ pooling).
+
+ Args:
+ space_size (list[3]): The size of the 3D space.
+ cube_size (list[3]): The size of the heatmap volume.
+ space_center (list[3]): The coordinate of space center.
+ max_num (int): Maximum of human center detections.
+ max_pool_kernel (int): Kernel size of the max-pool kernel in nms.
+ """
+
+ def __init__(self,
+ space_size,
+ space_center,
+ cube_size,
+ max_num=10,
+ max_pool_kernel=3):
+ super(CuboidCenterHead, self).__init__()
+ # use register_buffer
+ self.register_buffer('grid_size', torch.tensor(space_size))
+ self.register_buffer('cube_size', torch.tensor(cube_size))
+ self.register_buffer('grid_center', torch.tensor(space_center))
+
+ self.num_candidates = max_num
+ self.max_pool_kernel = max_pool_kernel
+ self.loss = nn.MSELoss()
+
+ def _get_real_locations(self, indices):
+ """
+ Args:
+ indices (torch.Tensor(NXP)): Indices of points in the 3D tensor
+
+ Returns:
+ real_locations (torch.Tensor(NXPx3)): Locations of points
+ in the world coordinate system
+ """
+ real_locations = indices.float() / (
+ self.cube_size - 1) * self.grid_size + \
+ self.grid_center - self.grid_size / 2.0
+ return real_locations
+
+ def _nms_by_max_pool(self, heatmap_volumes):
+ max_num = self.num_candidates
+ batch_size = heatmap_volumes.shape[0]
+ root_cubes_nms = self._max_pool(heatmap_volumes)
+ root_cubes_nms_reshape = root_cubes_nms.reshape(batch_size, -1)
+ topk_values, topk_index = root_cubes_nms_reshape.topk(max_num)
+ topk_unravel_index = self._get_3d_indices(topk_index,
+ heatmap_volumes[0].shape)
+
+ return topk_values, topk_unravel_index
+
+ def _max_pool(self, inputs):
+ kernel = self.max_pool_kernel
+ padding = (kernel - 1) // 2
+ max = F.max_pool3d(
+ inputs, kernel_size=kernel, stride=1, padding=padding)
+ keep = (inputs == max).float()
+ return keep * inputs
+
+ @staticmethod
+ def _get_3d_indices(indices, shape):
+ """Get indices in the 3-D tensor.
+
+ Args:
+ indices (torch.Tensor(NXp)): Indices of points in the 1D tensor
+ shape (torch.Size(3)): The shape of the original 3D tensor
+
+ Returns:
+ indices: Indices of points in the original 3D tensor
+ """
+ batch_size = indices.shape[0]
+ num_people = indices.shape[1]
+ indices_x = (indices //
+ (shape[1] * shape[2])).reshape(batch_size, num_people, -1)
+ indices_y = ((indices % (shape[1] * shape[2])) //
+ shape[2]).reshape(batch_size, num_people, -1)
+ indices_z = (indices % shape[2]).reshape(batch_size, num_people, -1)
+ indices = torch.cat([indices_x, indices_y, indices_z], dim=2)
+ return indices
+
+ def forward(self, heatmap_volumes):
+ """
+
+ Args:
+ heatmap_volumes (torch.Tensor(NXLXWXH)):
+ 3D human center heatmaps predicted by the network.
+ Returns:
+ human_centers (torch.Tensor(NXPX5)):
+ Coordinates of human centers.
+ """
+ batch_size = heatmap_volumes.shape[0]
+
+ topk_values, topk_unravel_index = self._nms_by_max_pool(
+ heatmap_volumes.detach())
+
+ topk_unravel_index = self._get_real_locations(topk_unravel_index)
+
+ human_centers = torch.zeros(
+ batch_size, self.num_candidates, 5, device=heatmap_volumes.device)
+ human_centers[:, :, 0:3] = topk_unravel_index
+ human_centers[:, :, 4] = topk_values
+
+ return human_centers
+
+ def get_loss(self, pred_cubes, gt):
+
+ return dict(loss_center=self.loss(pred_cubes, gt))
+
+
+@HEADS.register_module()
+class CuboidPoseHead(nn.Module):
+
+ def __init__(self, beta):
+ """Get results from the 3D human pose heatmap. Instead of obtaining
+ maximums on the heatmap, this module regresses the coordinates of
+ keypoints via integral pose regression. Refer to `paper.
+
+ ` for more details.
+
+ Args:
+ beta: Constant to adjust the magnification of soft-maxed heatmap.
+ """
+ super(CuboidPoseHead, self).__init__()
+ self.beta = beta
+ self.loss = nn.L1Loss()
+
+ def forward(self, heatmap_volumes, grid_coordinates):
+ """
+
+ Args:
+ heatmap_volumes (torch.Tensor(NxKxLxWxH)):
+ 3D human pose heatmaps predicted by the network.
+ grid_coordinates (torch.Tensor(Nx(LxWxH)x3)):
+ Coordinates of the grids in the heatmap volumes.
+ Returns:
+ human_poses (torch.Tensor(NxKx3)): Coordinates of human poses.
+ """
+ batch_size = heatmap_volumes.size(0)
+ channel = heatmap_volumes.size(1)
+ x = heatmap_volumes.reshape(batch_size, channel, -1, 1)
+ x = F.softmax(self.beta * x, dim=2)
+ grid_coordinates = grid_coordinates.unsqueeze(1)
+ x = torch.mul(x, grid_coordinates)
+ human_poses = torch.sum(x, dim=2)
+
+ return human_poses
+
+ def get_loss(self, preds, targets, weights):
+
+ return dict(loss_pose=self.loss(preds * weights, targets * weights))
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/model_builder.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/model_builder.py
new file mode 100644
index 0000000000000000000000000000000000000000..724cfcbc5db237cfc8bedf956490b05852deb54f
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/builder/model_builder.py
@@ -0,0 +1,67 @@
+import torch
+
+# from configs.coco.ViTPose_base_coco_256x192 import model
+from .heads.topdown_heatmap_simple_head import TopdownHeatmapSimpleHead
+
+# import TopdownHeatmapSimpleHead
+from .backbones import ViT
+
+# print(model)
+import torch
+from functools import partial
+import torch.nn as nn
+import torch.nn.functional as F
+from importlib import import_module
+
+
+def build_model(model_name, checkpoint=None):
+ try:
+ path = ".configs.coco." + model_name
+ mod = import_module(path, package="src.vitpose_infer")
+
+ model = getattr(mod, "model")
+ # from path import model
+ except:
+ raise ValueError("not a correct config")
+
+ head = TopdownHeatmapSimpleHead(
+ in_channels=model["keypoint_head"]["in_channels"],
+ out_channels=model["keypoint_head"]["out_channels"],
+ num_deconv_filters=model["keypoint_head"]["num_deconv_filters"],
+ num_deconv_kernels=model["keypoint_head"]["num_deconv_kernels"],
+ num_deconv_layers=model["keypoint_head"]["num_deconv_layers"],
+ extra=model["keypoint_head"]["extra"],
+ )
+ # print(head)
+ backbone = ViT(
+ img_size=model["backbone"]["img_size"],
+ patch_size=model["backbone"]["patch_size"],
+ embed_dim=model["backbone"]["embed_dim"],
+ depth=model["backbone"]["depth"],
+ num_heads=model["backbone"]["num_heads"],
+ ratio=model["backbone"]["ratio"],
+ mlp_ratio=model["backbone"]["mlp_ratio"],
+ qkv_bias=model["backbone"]["qkv_bias"],
+ drop_path_rate=model["backbone"]["drop_path_rate"],
+ )
+
+ class VitPoseModel(nn.Module):
+ def __init__(self, backbone, keypoint_head):
+ super(VitPoseModel, self).__init__()
+ self.backbone = backbone
+ self.keypoint_head = keypoint_head
+
+ def forward(self, x):
+ x = self.backbone(x)
+ x = self.keypoint_head(x)
+ return x
+
+ pose = VitPoseModel(backbone, head)
+ if checkpoint is not None:
+ check = torch.load(checkpoint)
+
+ pose.load_state_dict(check["state_dict"])
+ return pose
+
+
+# pose = build_model('ViTPose_base_coco_256x192','./models/vitpose-b-multi-coco.pth')
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/model_builder.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/model_builder.py
new file mode 100644
index 0000000000000000000000000000000000000000..1fcb014d8c67bfc3e3eebcab35e92e8649c084eb
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/model_builder.py
@@ -0,0 +1,167 @@
+import torch
+
+# from configs.coco.ViTPose_base_coco_256x192 import model
+from .builder.heads.topdown_heatmap_simple_head import TopdownHeatmapSimpleHead
+
+# import TopdownHeatmapSimpleHead
+from .builder.backbones import ViT
+
+# print(model)
+import torch
+from functools import partial
+import torch.nn as nn
+import torch.nn.functional as F
+from importlib import import_module
+
+models = {
+ "ViTPose_huge_coco_256x192": dict(
+ type="TopDown",
+ pretrained=None,
+ backbone=dict(
+ type="ViT",
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=1280,
+ depth=32,
+ num_heads=16,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.55,
+ ),
+ keypoint_head=dict(
+ type="TopdownHeatmapSimpleHead",
+ in_channels=1280,
+ num_deconv_layers=2,
+ num_deconv_filters=(256, 256),
+ num_deconv_kernels=(4, 4),
+ extra=dict(
+ final_conv_kernel=1,
+ ),
+ out_channels=17,
+ loss_keypoint=dict(type="JointsMSELoss", use_target_weight=True),
+ ),
+ train_cfg=dict(),
+ test_cfg=dict(),
+ ),
+ "ViTPose_base_coco_256x192": dict(
+ type="TopDown",
+ pretrained=None,
+ backbone=dict(
+ type="ViT",
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=768,
+ depth=12,
+ num_heads=12,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.3,
+ ),
+ keypoint_head=dict(
+ type="TopdownHeatmapSimpleHead",
+ in_channels=768,
+ num_deconv_layers=2,
+ num_deconv_filters=(256, 256),
+ num_deconv_kernels=(4, 4),
+ extra=dict(
+ final_conv_kernel=1,
+ ),
+ out_channels=17,
+ loss_keypoint=dict(type="JointsMSELoss", use_target_weight=True),
+ ),
+ train_cfg=dict(),
+ test_cfg=dict(),
+ ),
+ "ViTPose_base_simple_coco_256x192": dict(
+ type="TopDown",
+ pretrained=None,
+ backbone=dict(
+ type="ViT",
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=768,
+ depth=12,
+ num_heads=12,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.3,
+ ),
+ keypoint_head=dict(
+ type="TopdownHeatmapSimpleHead",
+ in_channels=768,
+ num_deconv_layers=0,
+ num_deconv_filters=[],
+ num_deconv_kernels=[],
+ upsample=4,
+ extra=dict(
+ final_conv_kernel=3,
+ ),
+ out_channels=17,
+ loss_keypoint=dict(type="JointsMSELoss", use_target_weight=True),
+ ),
+ train_cfg=dict(),
+ test_cfg=dict(
+ flip_test=True,
+ post_process="default",
+ shift_heatmap=False,
+ target_type="GaussianHeatmap",
+ modulate_kernel=11,
+ use_udp=True,
+ ),
+ ),
+}
+
+
+def build_model(model_name, checkpoint=None):
+ try:
+ model = models[model_name]
+ except:
+ raise ValueError("not a correct config")
+
+ head = TopdownHeatmapSimpleHead(
+ in_channels=model["keypoint_head"]["in_channels"],
+ out_channels=model["keypoint_head"]["out_channels"],
+ num_deconv_filters=model["keypoint_head"]["num_deconv_filters"],
+ num_deconv_kernels=model["keypoint_head"]["num_deconv_kernels"],
+ num_deconv_layers=model["keypoint_head"]["num_deconv_layers"],
+ extra=model["keypoint_head"]["extra"],
+ )
+ # print(head)
+ backbone = ViT(
+ img_size=model["backbone"]["img_size"],
+ patch_size=model["backbone"]["patch_size"],
+ embed_dim=model["backbone"]["embed_dim"],
+ depth=model["backbone"]["depth"],
+ num_heads=model["backbone"]["num_heads"],
+ ratio=model["backbone"]["ratio"],
+ mlp_ratio=model["backbone"]["mlp_ratio"],
+ qkv_bias=model["backbone"]["qkv_bias"],
+ drop_path_rate=model["backbone"]["drop_path_rate"],
+ )
+
+ class VitPoseModel(nn.Module):
+ def __init__(self, backbone, keypoint_head):
+ super(VitPoseModel, self).__init__()
+ self.backbone = backbone
+ self.keypoint_head = keypoint_head
+
+ def forward(self, x):
+ x = self.backbone(x)
+ x = self.keypoint_head(x)
+ return x
+
+ pose = VitPoseModel(backbone, head)
+ if checkpoint is not None:
+ check = torch.load(checkpoint)
+
+ pose.load_state_dict(check["state_dict"])
+ return pose
+
+
+# pose = build_model('ViTPose_base_coco_256x192','./models/vitpose-b-multi-coco.pth')
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/ViTPose_trt.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/ViTPose_trt.py
new file mode 100644
index 0000000000000000000000000000000000000000..c0dd6d652c39d1bd301ccb1e57e8332aa6c576ec
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/ViTPose_trt.py
@@ -0,0 +1,102 @@
+import tensorrt as trt
+import torch.nn
+from collections import OrderedDict, namedtuple
+import numpy as np
+
+def torch_device_from_trt(device):
+ if device == trt.TensorLocation.DEVICE:
+ return torch.device("cuda")
+ elif device == trt.TensorLocation.HOST:
+ return torch.device("cpu")
+ else:
+ return TypeError("%s is not supported by torch" % device)
+def torch_dtype_from_trt(dtype):
+ if dtype == trt.int8:
+ return torch.int8
+ elif trt.__version__ >= '7.0' and dtype == trt.bool:
+ return torch.bool
+ elif dtype == trt.int32:
+ return torch.int32
+ elif dtype == trt.float16:
+ return torch.float16
+ elif dtype == trt.float32:
+ return torch.float32
+ else:
+ raise TypeError("%s is not supported by torch" % dtype)
+class TRTModule_ViTPose(torch.nn.Module):
+ def __init__(self, engine=None, input_names=None, output_names=None, input_flattener=None, output_flattener=None,path=None,device=None):
+ super(TRTModule_ViTPose, self).__init__()
+ # self._register_state_dict_hook(TRTModule._on_state_dict)
+ # self.engine = engine
+ logger = trt.Logger(trt.Logger.INFO)
+ with open(path, 'rb') as f, trt.Runtime(logger) as runtime:
+ self.engine = runtime.deserialize_cuda_engine(f.read())
+ if self.engine is not None:
+ self.context = self.engine.create_execution_context()
+ self.input_names = ['images']
+ self.output_names = []
+ self.input_flattener = input_flattener
+ self.output_flattener = output_flattener
+ Binding = namedtuple('Binding', ('name', 'dtype', 'shape', 'data', 'ptr'))
+
+ # with open(path, 'rb') as f, trt.Runtime(logger) as runtime:
+ # self.model = runtime.deserialize_cuda_engine(f.read())
+ # self.context = self.model.create_execution_context()
+ self.bindings = OrderedDict()
+ # self.output_names = []
+ fp16 = False # default updated below
+ dynamic = False
+ for i in range(self.engine.num_bindings):
+ name = self.engine.get_binding_name(i)
+ dtype = trt.nptype(self.engine.get_binding_dtype(i))
+ if self.engine.binding_is_input(i):
+ if -1 in tuple(self.engine.get_binding_shape(i)): # dynamic
+ dynamic = True
+ self.context.set_binding_shape(i, tuple(self.engine.get_profile_shape(0, i)[2]))
+ if dtype == np.float16:
+ fp16 = True
+ else: # output
+ self.output_names.append(name)
+ shape = tuple(self.context.get_binding_shape(i))
+ im = torch.from_numpy(np.empty(shape, dtype=dtype)).to(device)
+ self.bindings[name] = Binding(name, dtype, shape, im, int(im.data_ptr()))
+ self.binding_addrs = OrderedDict((n, d.ptr) for n, d in self.bindings.items())
+ self.batch_size = self.bindings['images'].shape[0]
+
+
+
+ def forward(self, *inputs):
+ bindings = [None] * (len(self.input_names) + len(self.output_names))
+
+ if self.input_flattener is not None:
+ inputs = self.input_flattener.flatten(inputs)
+
+ for i, input_name in enumerate(self.input_names):
+ idx = self.engine.get_binding_index(input_name)
+ shape = tuple(inputs[i].shape)
+ bindings[idx] = inputs[i].contiguous().data_ptr()
+ self.context.set_binding_shape(idx, shape)
+
+ # create output tensors
+ outputs = [None] * len(self.output_names)
+ for i, output_name in enumerate(self.output_names):
+ idx = self.engine.get_binding_index(output_name)
+ dtype = torch_dtype_from_trt(self.engine.get_binding_dtype(idx))
+ shape = tuple(self.context.get_binding_shape(idx))
+ device = torch_device_from_trt(self.engine.get_location(idx))
+ output = torch.empty(size=shape, dtype=dtype, device=device)
+ outputs[i] = output
+ bindings[idx] = output.data_ptr()
+
+ self.context.execute_async_v2(
+ bindings, torch.cuda.current_stream().cuda_stream
+ )
+
+ if self.output_flattener is not None:
+ outputs = self.output_flattener.unflatten(outputs)
+ else:
+ outputs = tuple(outputs)
+ if len(outputs) == 1:
+ outputs = outputs[0]
+
+ return outputs
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/__init__.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/convert_to_trt.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/convert_to_trt.py
new file mode 100644
index 0000000000000000000000000000000000000000..8af8bd39190d1622b7ef124a8f5ca531bb1beb3c
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/convert_to_trt.py
@@ -0,0 +1,9 @@
+from torch2trt import TRTModule,torch2trt
+from builder import build_model
+import torch
+pose = build_model('ViTPose_base_coco_256x192','./models/vitpose-b.pth')
+pose.cuda().eval()
+
+x = torch.ones(1,3,256,192).cuda()
+net_trt = torch2trt(pose, [x],max_batch_size=10, fp16_mode=True)
+torch.save(net_trt.state_dict(), 'vitpose_trt.pth')
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/general_utils.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/general_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..2dba4ac8095e9c7e7129b19af9a5374afbd515f8
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/general_utils.py
@@ -0,0 +1,111 @@
+#!/usr/bin/env python3
+# -*- coding: utf-8 -*-
+"""
+Created on Wed Jun 15 15:49:22 2022
+
+@author: gpastal
+"""
+import numpy as np
+import numpy.ma as ma
+# import pika
+import json
+from collections import OrderedDict
+from collections.abc import Iterable
+from itertools import chain
+import argparse
+
+def make_parser():
+ parser = argparse.ArgumentParser("ByteTrack Demo!")
+ # exp file
+ # tracking args
+ parser.add_argument("--track_thresh", type=float, default=0.2, help="tracking confidence threshold")
+ parser.add_argument("--track_buffer", type=int, default=240, help="the frames for keep lost tracks")
+ parser.add_argument("--match_thresh", type=float, default=0.8, help="matching threshold for tracking")
+ parser.add_argument(
+ "--aspect_ratio_thresh", type=float, default=1.6,
+ help="threshold for filtering out boxes of which aspect ratio are above the given value."
+ )
+ parser.add_argument('--min_box_area', type=float, default=10, help='filter out tiny boxes')
+ parser.add_argument("--mot20", dest="mot20", default=False, action="store_true", help="test mot20.")
+ return parser
+
+def jitter(tracking,temp,id1):
+ pass
+def jitter2(tracking,temp,id1) :
+ pass
+
+def create_json_rabbitmq( FRAME_ID,pose):
+ pass
+
+def producer_rabbitmq():
+ pass
+def fix_head(xyz):
+ pass
+
+def flatten_lst(x):
+ if isinstance(x, Iterable):
+ return [a for i in x for a in flatten_lst(i)]
+ else:
+ return [x]
+
+def polys_from_pose(pts):
+ seg=[]
+ for ind, i in enumerate(pts):
+ list_=[]
+ list_sc=[]
+ # list1 = [i[0][1],i[0][0]]
+
+ # list2 = [i[0][1],i[0][0]]
+ # print(i)
+ for j in i:
+
+ temp_ = [j[1],j[0]]
+
+ if j[2]>0.4:
+
+ temp2_ = [1]
+ else:
+ temp2_ =[0]
+ list_.append(temp_)
+ list_sc.append(temp2_)
+ # print(list_sc)
+ # list2 = [i[6][1],i[6][0]]
+ # list3 = [i[11][1],i[11][0]]
+ # list4 = [i[12][1],i[12][0]]
+
+ # list_ = flatten_lst(list_)
+ # print(list_)
+ list_=fix_list_order(list_,list_sc)
+ # print(list_)
+ # list_=list(list_)
+ # list_ = list_.to_list()
+ # print(list_)
+ # temp__=list(chain(*list_))
+ seg.append(list_) # temp_ = list(chain(list1,list2,list3,list4,list1))
+ return seg
+def fix_list_order(list_,list2):
+ # for index,values in enumerate(list_):
+ myorder = [0, 2, 4, 6, 8,10,12,14,16,15,13,11,9,7,5,3,1]
+ cor_list = [list_[i] for i in myorder]
+ cor_list2 = [list2[i] for i in myorder]
+ # print(cor_list)
+ # result = list(set(map(tuple,cor_list)) & set(map(tuple,cor_list2)))
+ # arr = np.array([x for x in cor_list])
+ # print(cor_list)
+ data = np.asarray(cor_list)
+ # print(data)
+ mask = np.column_stack((cor_list2, cor_list2))
+ # masked = ma.masked_array(data, mask=np.column_stack((cor_list2, cor_list2)))#[cor_list2,cor_list2])
+ # result = list(set(masked[~masked.mask]))
+ # print(result)
+ # print(data)
+ # print(mask)
+ result2 = []
+ for inde,i in enumerate(data):
+ # print(mask[inde])
+ if mask[inde].all()==1:
+ result2.append(i[0])
+ result2.append(i[1])
+ # result = [int(result[i] for i in result)]
+ return result2
+
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/inference_test.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/inference_test.py
new file mode 100644
index 0000000000000000000000000000000000000000..fbb6220f693b9e9cc43c8366d588831a1bdc28f3
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/inference_test.py
@@ -0,0 +1,29 @@
+from builder import build_model
+import torch
+from ViTPose_trt import TRTModule_ViTPose
+# pose = TRTModule_ViTPose(path='pose_higher_hrnet_w32_512.engine',device='cuda:0')
+pose = build_model('ViTPose_base_coco_256x192','./models/vitpose-b.pth')
+pose.cuda().eval()
+if pose.training:
+ print('train')
+else:
+ print('eval')
+device = torch.device("cuda")
+# pose.to(device)
+dummy_input = torch.randn(10, 3,256,192, dtype=torch.float).to(device)
+repetitions=100
+total_time = 0
+starter, ender = torch.cuda.Event(enable_timing=True), torch.cuda.Event(enable_timing=True)
+with torch.no_grad():
+ for rep in range(repetitions):
+ # starter, ender = torch.cuda.Event(enable_timing=True), torch.cuda.Event(enable_timing=True)
+ starter.record()
+ # for k in range(10):
+ _ = pose(dummy_input)
+ ender.record()
+ torch.cuda.synchronize()
+ curr_time = starter.elapsed_time(ender)/1000
+ total_time += curr_time
+Throughput = repetitions*10/total_time
+print('Final Throughput:',Throughput)
+print('Total time',total_time)
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/logger_helper.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/logger_helper.py
new file mode 100644
index 0000000000000000000000000000000000000000..e58d24005ea4b71a975bc48ca0b1860897d16901
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/logger_helper.py
@@ -0,0 +1,23 @@
+import logging
+
+class CustomFormatter(logging.Formatter):
+
+ grey = "\x1b[38;20m"
+ yellow = "\x1b[33;20m"
+ red = "\x1b[31;20m"
+ bold_red = "\x1b[31;1m"
+ reset = "\x1b[0m"
+ format = "%(asctime)s - %(name)s - %(levelname)s - %(message)s (%(filename)s:%(lineno)d)"
+
+ FORMATS = {
+ logging.DEBUG: grey + format + reset,
+ logging.INFO: grey + format + reset,
+ logging.WARNING: yellow + format + reset,
+ logging.ERROR: red + format + reset,
+ logging.CRITICAL: bold_red + format + reset
+ }
+
+ def format(self, record):
+ log_fmt = self.FORMATS.get(record.levelno)
+ formatter = logging.Formatter(log_fmt)
+ return formatter.format(record)
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/pose_utils.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/pose_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..0d6cb06fe8909f50f70389c2838352927ce697ec
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/pose_utils.py
@@ -0,0 +1,183 @@
+#!/usr/bin/env python3
+# -*- coding: utf-8 -*-
+"""
+Created on Wed Jun 15 15:45:33 2022
+
+@author: gpastal
+"""
+import torch
+import torchvision
+import torch.nn.functional as F
+from torchvision import transforms as TR
+import numpy as np
+import cv2
+import logging
+# from simpleHRNet.models_.hrnet import HRNet
+# from torch2trt import torch2trt,TRTModule
+logger = logging.getLogger("Tracker !")
+from .timerr import Timer
+from pathlib import Path
+# import gdown
+timer_det = Timer()
+timer_track = Timer()
+timer_pose = Timer()
+
+def pose_points_yolo5(detector,image,pose,tracker,tensorrt):
+ timer_det.tic()
+ # starter, ender = torch.cuda.Event(enable_timing=True), torch.cuda.Event(enable_timing=True)
+
+ transform = TR.Compose([
+ TR.ToPILImage(),
+ # Padd(),
+ TR.Resize((256, 192)), # (height, width)
+ TR.ToTensor(),
+ TR.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),
+ ])
+ # image = cv2.cvtColor(image, cv2.COLOR_BGRA2BGR)
+
+ detections = detector(image)
+ timer_det.toc()
+ logger.info('DET FPS -- %s',1./timer_det.average_time)
+ # print(detections.shape)
+ dets = detections.xyxy[0]
+ dets = dets[dets[:,5] == 0.]
+ # dets = dets[dets[:,4] > 0.3]
+ # logger.warning(len(dets))
+
+ # if len(dets)>0:
+ # image_gpu = torch.tensor(image).cuda()/255.
+ # print(image_gpu.size())
+ timer_track.tic()
+ online_targets=tracker.update(dets,[image.shape[0],image.shape[1]],image.shape)
+
+ online_tlwhs = []
+ online_ids = []
+ online_scores = []
+ for t in online_targets:
+ tlwh = t.tlwh
+ tid = t.track_id
+ # vertical = tlwh[2] / tlwh[3] > args.aspect_ratio_threshs
+ if tlwh[2] * tlwh[3] > 10 :#and not vertical:
+ online_tlwhs.append(tlwh)
+ online_ids.append(tid)
+ online_scores.append(t.score)
+ # tracker.update()
+ timer_track.toc()
+ logger.info('TRACKING FPS --%s',1./timer_track.average_time)
+ device='cuda'
+ nof_people = len(online_ids) if online_ids is not None else 0
+ # nof_people=1
+ # print(dets)
+ # print(nof_people)
+ boxes = torch.empty((nof_people, 4), dtype=torch.int32,device= 'cuda')
+ # boxes = []
+ images = torch.empty((nof_people, 3, 256, 192)) # (height, width)
+ heatmaps = np.zeros((nof_people, 17, 64, 48),
+ dtype=np.float32)
+ # starter.record()
+ # print(online_tlwhs)
+ if len(online_tlwhs):
+ for i, (x1, y1, x2, y2) in enumerate(online_tlwhs):
+ # for i, (x1, y1, x2, y2) in enumerate(np.array([[55,399,424-55,479-399]])):
+ # if i<1:
+ x1 = x1.astype(np.int32)
+ x2 = x1+x2.astype(np.int32)
+ y1 = y1.astype(np.int32)
+ y2 = y1+ y2.astype(np.int32)
+ if x2>image.shape[1]:x2=image.shape[1]-1
+ if y2>image.shape[0]:y2=image.shape[0]-1
+ if y1<0: y1=0
+ if x1<0 : x1=0
+ # print([x1,x2,y1,y2])
+ # image = cv2.rectangle(image, (x1,y1), (x2,y2), (0,0,0), 1)
+ # cv2.imwrite('saved.png',image)
+ # # Adapt detections to match HRNet input aspect ratio (as suggested by xtyDoge in issue #14)
+ correction_factor = 256 / 192 * (x2 - x1) / (y2 - y1)
+ if correction_factor > 1:
+ # increase y side
+ center = y1 + (y2 - y1) // 2
+ length = int(round((y2 - y1) * correction_factor))
+ y1_new = int( center - length // 2)
+ y2_new = int( center + length // 2)
+ image_crop = image[y1:y2, x1:x2, ::-1]
+ # print(y1,y2,x1,x2)
+ pad = (int(abs(y1_new-y1))), int(abs(y2_new-y2))
+ image_crop = np.pad(image_crop,((pad), (0, 0), (0, 0)))
+ images[i] = transform(image_crop)
+ boxes[i]= torch.tensor([x1, y1_new, x2, y2_new])
+
+ elif correction_factor < 1:
+ # increase x side
+ center = x1 + (x2 - x1) // 2
+ length = int(round((x2 - x1) * 1 / correction_factor))
+ x1_new = int( center - length // 2)
+ x2_new = int( center + length // 2)
+ # images[i] = transform(image[y1:y2, x1:x2, ::-1])
+ image_crop = image[y1:y2, x1:x2, ::-1]
+ pad = (abs(x1_new-x1)), int(abs(x2_new-x2))
+ image_crop = np.pad(image_crop,((0, 0), (pad), (0, 0)))
+ images[i] = transform(image_crop)
+ boxes[i]= torch.tensor([x1_new, y1, x2_new, y2])
+
+
+ if images.shape[0] > 0:
+ images = images.to(device)
+
+ if tensorrt:
+ out = torch.zeros((images.shape[0],17,64,48),device=device)
+ with torch.no_grad():
+ timer_pose.tic()
+
+ for i in range(images.shape[0]):
+ # timer_pose.tic()
+ # print(images[i].size())
+
+ out[i] = pose(images[i].unsqueeze(0))
+ timer_pose.toc()
+ logger.info('POSE FPS -- %s',1./timer_pose.average_time)
+ else:
+ with torch.no_grad():
+
+ timer_pose.tic()
+
+
+
+ out = pose(images)
+ timer_pose.toc()
+ logger.info('POSE FPS -- %s',1./timer_pose.average_time)
+
+
+ pts = torch.empty((out.shape[0], out.shape[1], 3), dtype=torch.float32,device=device)
+ pts2 = np.empty((out.shape[0], out.shape[1], 3), dtype=np.float32)
+
+ (b,indices)=torch.max(out,dim=2)
+ (b,indices)=torch.max(b,dim=2)
+
+ (c,indicesc)=torch.max(out,dim=3)
+ (c,indicesc)=torch.max(c,dim=2)
+ dim1= torch.tensor(1. / 64,device=device)
+ dim2= torch.tensor(1. / 48,device=device)
+
+ for i in range(0,out.shape[0]):
+
+ pts[i, :, 0] = indicesc[i,:] * dim1 * (boxes[i][3] - boxes[i][1]) + boxes[i][1]
+ pts[i, :, 1] = indices[i,:] *dim2* (boxes[i][2] - boxes[i][0]) + boxes[i][0]
+ pts[i, :, 2] = c[i,:]
+
+ pts=pts.cpu().numpy()
+ # print(pts)
+ else:
+ pts = np.empty((0, 0, 3), dtype=np.float32)
+ online_tlwhs = []
+ online_ids = []
+ online_scores=[]
+ res = list()
+
+ res.append(pts)
+
+
+ if len(res) > 1:
+ return res,online_tlwhs,online_ids,online_scores#,pts2
+ else:
+ return res[0],online_tlwhs,online_ids,online_scores#,pts2
+
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/pose_viz.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/pose_viz.py
new file mode 100644
index 0000000000000000000000000000000000000000..ba575816a91e8d30d81e2ed04037d3f021177cb5
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/pose_viz.py
@@ -0,0 +1,293 @@
+import cv2
+import matplotlib.pyplot as plt
+import numpy as np
+import torch
+import torchvision
+import ffmpeg
+
+
+def joints_dict():
+ joints = {
+ "coco": {
+ "keypoints": {
+ 0: "nose",
+ 1: "left_eye",
+ 2: "right_eye",
+ 3: "left_ear",
+ 4: "right_ear",
+ 5: "left_shoulder",
+ 6: "right_shoulder",
+ 7: "left_elbow",
+ 8: "right_elbow",
+ 9: "left_wrist",
+ 10: "right_wrist",
+ 11: "left_hip",
+ 12: "right_hip",
+ 13: "left_knee",
+ 14: "right_knee",
+ 15: "left_ankle",
+ 16: "right_ankle"
+ },
+ "skeleton": [
+ # # [16, 14], [14, 12], [17, 15], [15, 13], [12, 13], [6, 12], [7, 13], [6, 7], [6, 8],
+ # # [7, 9], [8, 10], [9, 11], [2, 3], [1, 2], [1, 3], [2, 4], [3, 5], [4, 6], [5, 7]
+ # [15, 13], [13, 11], [16, 14], [14, 12], [11, 12], [5, 11], [6, 12], [5, 6], [5, 7],
+ # [6, 8], [7, 9], [8, 10], [1, 2], [0, 1], [0, 2], [1, 3], [2, 4], [3, 5], [4, 6]
+ [15, 13], [13, 11], [16, 14], [14, 12], [11, 12], [5, 11], [6, 12], [5, 6], [5, 7],
+ [6, 8], [7, 9], [8, 10], [1, 2], [0, 1], [0, 2], [1, 3], [2, 4], # [3, 5], [4, 6]
+ [0, 5], [0, 6]
+ ]
+ },
+ "mpii": {
+ "keypoints": {
+ 0: "right_ankle",
+ 1: "right_knee",
+ 2: "right_hip",
+ 3: "left_hip",
+ 4: "left_knee",
+ 5: "left_ankle",
+ 6: "pelvis",
+ 7: "thorax",
+ 8: "upper_neck",
+ 9: "head top",
+ 10: "right_wrist",
+ 11: "right_elbow",
+ 12: "right_shoulder",
+ 13: "left_shoulder",
+ 14: "left_elbow",
+ 15: "left_wrist"
+ },
+ "skeleton": [
+ # [5, 4], [4, 3], [0, 1], [1, 2], [3, 2], [13, 3], [12, 2], [13, 12], [13, 14],
+ # [12, 11], [14, 15], [11, 10], # [2, 3], [1, 2], [1, 3], [2, 4], [3, 5], [4, 6], [5, 7]
+ [5, 4], [4, 3], [0, 1], [1, 2], [3, 2], [3, 6], [2, 6], [6, 7], [7, 8], [8, 9],
+ [13, 7], [12, 7], [13, 14], [12, 11], [14, 15], [11, 10],
+ ]
+ },
+ }
+ return joints
+
+
+def draw_points(image, points, color_palette='tab20', palette_samples=16, confidence_threshold=0.5):
+ """
+ Draws `points` on `image`.
+
+ Args:
+ image: image in opencv format
+ points: list of points to be drawn.
+ Shape: (nof_points, 3)
+ Format: each point should contain (y, x, confidence)
+ color_palette: name of a matplotlib color palette
+ Default: 'tab20'
+ palette_samples: number of different colors sampled from the `color_palette`
+ Default: 16
+ confidence_threshold: only points with a confidence higher than this threshold will be drawn. Range: [0, 1]
+ Default: 0.5
+
+ Returns:
+ A new image with overlaid points
+
+ """
+ try:
+ colors = np.round(
+ np.array(plt.get_cmap(color_palette).colors) * 255
+ ).astype(np.uint8)[:, ::-1].tolist()
+ except AttributeError: # if palette has not pre-defined colors
+ colors = np.round(
+ np.array(plt.get_cmap(color_palette)(np.linspace(0, 1, palette_samples))) * 255
+ ).astype(np.uint8)[:, -2::-1].tolist()
+
+ circle_size = max(1, min(image.shape[:2]) // 160) # ToDo Shape it taking into account the size of the detection
+ # circle_size = max(2, int(np.sqrt(np.max(np.max(points, axis=0) - np.min(points, axis=0)) // 16)))
+
+ for i, pt in enumerate(points):
+ if pt[2] > confidence_threshold:
+ image = cv2.circle(image, (int(pt[1]), int(pt[0])), circle_size, tuple(colors[i % len(colors)]), -1)
+
+ return image
+
+
+def draw_skeleton(image, points, skeleton, color_palette='Set2', palette_samples=8, person_index=0,
+ confidence_threshold=0.5):
+ """
+ Draws a `skeleton` on `image`.
+
+ Args:
+ image: image in opencv format
+ points: list of points to be drawn.
+ Shape: (nof_points, 3)
+ Format: each point should contain (y, x, confidence)
+ skeleton: list of joints to be drawn
+ Shape: (nof_joints, 2)
+ Format: each joint should contain (point_a, point_b) where `point_a` and `point_b` are an index in `points`
+ color_palette: name of a matplotlib color palette
+ Default: 'Set2'
+ palette_samples: number of different colors sampled from the `color_palette`
+ Default: 8
+ person_index: index of the person in `image`
+ Default: 0
+ confidence_threshold: only points with a confidence higher than this threshold will be drawn. Range: [0, 1]
+ Default: 0.5
+
+ Returns:
+ A new image with overlaid joints
+
+ """
+ try:
+ colors = np.round(
+ np.array(plt.get_cmap(color_palette).colors) * 255
+ ).astype(np.uint8)[:, ::-1].tolist()
+ except AttributeError: # if palette has not pre-defined colors
+ colors = np.round(
+ np.array(plt.get_cmap(color_palette)(np.linspace(0, 1, palette_samples))) * 255
+ ).astype(np.uint8)[:, -2::-1].tolist()
+
+ for i, joint in enumerate(skeleton):
+ pt1, pt2 = points[joint]
+ if pt1[2] > confidence_threshold and pt2[2] > confidence_threshold:
+ image = cv2.line(
+ image, (int(pt1[1]), int(pt1[0])), (int(pt2[1]), int(pt2[0])),
+ tuple(colors[person_index % len(colors)]), 2
+ )
+
+ return image
+
+
+def draw_points_and_skeleton(image, points, skeleton, points_color_palette='tab20', points_palette_samples=16,
+ skeleton_color_palette='Set2', skeleton_palette_samples=8, person_index=0,
+ confidence_threshold=0.5):
+ """
+ Draws `points` and `skeleton` on `image`.
+
+ Args:
+ image: image in opencv format
+ points: list of points to be drawn.
+ Shape: (nof_points, 3)
+ Format: each point should contain (y, x, confidence)
+ skeleton: list of joints to be drawn
+ Shape: (nof_joints, 2)
+ Format: each joint should contain (point_a, point_b) where `point_a` and `point_b` are an index in `points`
+ points_color_palette: name of a matplotlib color palette
+ Default: 'tab20'
+ points_palette_samples: number of different colors sampled from the `color_palette`
+ Default: 16
+ skeleton_color_palette: name of a matplotlib color palette
+ Default: 'Set2'
+ skeleton_palette_samples: number of different colors sampled from the `color_palette`
+ Default: 8
+ person_index: index of the person in `image`
+ Default: 0
+ confidence_threshold: only points with a confidence higher than this threshold will be drawn. Range: [0, 1]
+ Default: 0.5
+
+ Returns:
+ A new image with overlaid joints
+
+ """
+ image = draw_skeleton(image, points, skeleton, color_palette=skeleton_color_palette,
+ palette_samples=skeleton_palette_samples, person_index=person_index,
+ confidence_threshold=confidence_threshold)
+ image = draw_points(image, points, color_palette=points_color_palette, palette_samples=points_palette_samples,
+ confidence_threshold=confidence_threshold)
+ return image
+
+
+def save_images(images, target, joint_target, output, joint_output, joint_visibility, summary_writer=None, step=0,
+ prefix=''):
+ """
+ Creates a grid of images with gt joints and a grid with predicted joints.
+ This is a basic function for debugging purposes only.
+
+ If summary_writer is not None, the grid will be written in that SummaryWriter with name "{prefix}_images" and
+ "{prefix}_predictions".
+
+ Args:
+ images (torch.Tensor): a tensor of images with shape (batch x channels x height x width).
+ target (torch.Tensor): a tensor of gt heatmaps with shape (batch x channels x height x width).
+ joint_target (torch.Tensor): a tensor of gt joints with shape (batch x joints x 2).
+ output (torch.Tensor): a tensor of predicted heatmaps with shape (batch x channels x height x width).
+ joint_output (torch.Tensor): a tensor of predicted joints with shape (batch x joints x 2).
+ joint_visibility (torch.Tensor): a tensor of joint visibility with shape (batch x joints).
+ summary_writer (tb.SummaryWriter): a SummaryWriter where write the grids.
+ Default: None
+ step (int): summary_writer step.
+ Default: 0
+ prefix (str): summary_writer name prefix.
+ Default: ""
+
+ Returns:
+ A pair of images which are built from torchvision.utils.make_grid
+ """
+ # Input images with gt
+ images_ok = images.detach().clone()
+ images_ok[:, 0].mul_(0.229).add_(0.485)
+ images_ok[:, 1].mul_(0.224).add_(0.456)
+ images_ok[:, 2].mul_(0.225).add_(0.406)
+ for i in range(images.shape[0]):
+ joints = joint_target[i] * 4.
+ joints_vis = joint_visibility[i]
+
+ for joint, joint_vis in zip(joints, joints_vis):
+ if joint_vis[0]:
+ a = int(joint[1].item())
+ b = int(joint[0].item())
+ # images_ok[i][:, a-1:a+1, b-1:b+1] = torch.tensor([1, 0, 0])
+ images_ok[i][0, a - 1:a + 1, b - 1:b + 1] = 1
+ images_ok[i][1:, a - 1:a + 1, b - 1:b + 1] = 0
+ grid_gt = torchvision.utils.make_grid(images_ok, nrow=int(images_ok.shape[0] ** 0.5), padding=2, normalize=False)
+ if summary_writer is not None:
+ summary_writer.add_image(prefix + 'images', grid_gt, global_step=step)
+
+ # Input images with prediction
+ images_ok = images.detach().clone()
+ images_ok[:, 0].mul_(0.229).add_(0.485)
+ images_ok[:, 1].mul_(0.224).add_(0.456)
+ images_ok[:, 2].mul_(0.225).add_(0.406)
+ for i in range(images.shape[0]):
+ joints = joint_output[i] * 4.
+ joints_vis = joint_visibility[i]
+
+ for joint, joint_vis in zip(joints, joints_vis):
+ if joint_vis[0]:
+ a = int(joint[1].item())
+ b = int(joint[0].item())
+ # images_ok[i][:, a-1:a+1, b-1:b+1] = torch.tensor([1, 0, 0])
+ images_ok[i][0, a - 1:a + 1, b - 1:b + 1] = 1
+ images_ok[i][1:, a - 1:a + 1, b - 1:b + 1] = 0
+ grid_pred = torchvision.utils.make_grid(images_ok, nrow=int(images_ok.shape[0] ** 0.5), padding=2, normalize=False)
+ if summary_writer is not None:
+ summary_writer.add_image(prefix + 'predictions', grid_pred, global_step=step)
+
+ # Heatmaps
+ # ToDo
+ # for h in range(0,17):
+ # heatmap = torchvision.utils.make_grid(output[h].detach(), nrow=int(np.sqrt(output.shape[0])),
+ # padding=2, normalize=True, range=(0, 1))
+ # summary_writer.add_image('train_heatmap_%d' % h, heatmap, global_step=step + epoch*len_dl_train)
+
+ return grid_gt, grid_pred
+
+
+def check_video_rotation(filename):
+ # thanks to
+ # https://stackoverflow.com/questions/53097092/frame-from-video-is-upside-down-after-extracting/55747773#55747773
+
+ # this returns meta-data of the video file in form of a dictionary
+ meta_dict = ffmpeg.probe(filename)
+
+ # from the dictionary, meta_dict['streams'][0]['tags']['rotate'] is the key
+ # we are looking for
+ rotation_code = None
+ try:
+ if int(meta_dict['streams'][0]['tags']['rotate']) == 90:
+ rotation_code = cv2.ROTATE_90_CLOCKWISE
+ elif int(meta_dict['streams'][0]['tags']['rotate']) == 180:
+ rotation_code = cv2.ROTATE_180
+ elif int(meta_dict['streams'][0]['tags']['rotate']) == 270:
+ rotation_code = cv2.ROTATE_90_COUNTERCLOCKWISE
+ else:
+ raise ValueError
+ except KeyError:
+ pass
+
+ return rotation_code
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/timerr.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/timerr.py
new file mode 100644
index 0000000000000000000000000000000000000000..c9b15fb969bce7b31a1613a6401141dcc9cf180a
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/timerr.py
@@ -0,0 +1,37 @@
+import time
+
+
+class Timer(object):
+ """A simple timer."""
+ def __init__(self):
+ self.total_time = 0.
+ self.calls = 0
+ self.start_time = 0.
+ self.diff = 0.
+ self.average_time = 0.
+
+ self.duration = 0.
+
+ def tic(self):
+ # using time.time instead of time.clock because time time.clock
+ # does not normalize for multithreading
+ self.start_time = time.time()
+
+ def toc(self, average=True):
+ self.diff = time.time() - self.start_time
+ self.total_time += self.diff
+ self.calls += 1
+ self.average_time = self.total_time / self.calls
+ if average:
+ self.duration = self.average_time
+ else:
+ self.duration = self.diff
+ return self.duration
+
+ def clear(self):
+ self.total_time = 0.
+ self.calls = 0
+ self.start_time = 0.
+ self.diff = 0.
+ self.average_time = 0.
+ self.duration = 0.
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/visualizer.py b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/visualizer.py
new file mode 100644
index 0000000000000000000000000000000000000000..d2d2305f0d5246c92a7b782c51db2099b1164586
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/preproc/vitpose_pytorch/src/vitpose_infer/pose_utils/visualizer.py
@@ -0,0 +1,162 @@
+import cv2
+import numpy as np
+
+__all__ = ["vis"]
+
+
+def vis(img, boxes, scores, cls_ids, conf=0.5, class_names=None):
+
+ for i in range(len(boxes)):
+ box = boxes[i]
+ cls_id = int(cls_ids[i])
+ score = scores[i]
+ if score < conf:
+ continue
+ x0 = int(box[0])
+ y0 = int(box[1])
+ x1 = int(box[2])
+ y1 = int(box[3])
+
+ color = (_COLORS[cls_id] * 255).astype(np.uint8).tolist()
+ text = '{}:{:.1f}%'.format(class_names[cls_id], score * 100)
+ txt_color = (0, 0, 0) if np.mean(_COLORS[cls_id]) > 0.5 else (255, 255, 255)
+ font = cv2.FONT_HERSHEY_SIMPLEX
+
+ txt_size = cv2.getTextSize(text, font, 0.4, 1)[0]
+ cv2.rectangle(img, (x0, y0), (x1, y1), color, 2)
+
+ txt_bk_color = (_COLORS[cls_id] * 255 * 0.7).astype(np.uint8).tolist()
+ cv2.rectangle(
+ img,
+ (x0, y0 + 1),
+ (x0 + txt_size[0] + 1, y0 + int(1.5*txt_size[1])),
+ txt_bk_color,
+ -1
+ )
+ cv2.putText(img, text, (x0, y0 + txt_size[1]), font, 0.4, txt_color, thickness=1)
+
+ return img
+
+
+def get_color(idx):
+ idx = idx * 3
+ color = ((37 * idx) % 255, (17 * idx) % 255, (29 * idx) % 255)
+
+ return color
+
+
+def plot_tracking(image, tlwhs, obj_ids, scores=None, frame_id=0, fps=0., ids2=None):
+ im = np.ascontiguousarray(np.copy(image))
+ im_h, im_w = im.shape[:2]
+
+ top_view = np.zeros([im_w, im_w, 3], dtype=np.uint8) + 255
+
+ #text_scale = max(1, image.shape[1] / 1600.)
+ #text_thickness = 2
+ #line_thickness = max(1, int(image.shape[1] / 500.))
+ text_scale = 2
+ text_thickness = 2
+ line_thickness = 3
+
+ radius = max(5, int(im_w/140.))
+ cv2.putText(im, 'frame: %d fps: %.2f num: %d' % (frame_id, fps, len(tlwhs)),
+ (0, int(15 * text_scale)), cv2.FONT_HERSHEY_PLAIN, 2, (0, 0, 255), thickness=2)
+
+ for i, tlwh in enumerate(tlwhs):
+ x1, y1, w, h = tlwh
+ intbox = tuple(map(int, (x1, y1, x1 + w, y1 + h)))
+ obj_id = int(obj_ids[i])
+ id_text = '{}'.format(int(obj_id))
+ if ids2 is not None:
+ id_text = id_text + ', {}'.format(int(ids2[i]))
+ color = get_color(abs(obj_id))
+ cv2.rectangle(im, intbox[0:2], intbox[2:4], color=color, thickness=line_thickness)
+ cv2.putText(im, id_text, (intbox[0], intbox[1]), cv2.FONT_HERSHEY_PLAIN, text_scale, (0, 0, 255),
+ thickness=text_thickness)
+ return im
+
+
+_COLORS = np.array(
+ [
+ 0.000, 0.447, 0.741,
+ 0.850, 0.325, 0.098,
+ 0.929, 0.694, 0.125,
+ 0.494, 0.184, 0.556,
+ 0.466, 0.674, 0.188,
+ 0.301, 0.745, 0.933,
+ 0.635, 0.078, 0.184,
+ 0.300, 0.300, 0.300,
+ 0.600, 0.600, 0.600,
+ 1.000, 0.000, 0.000,
+ 1.000, 0.500, 0.000,
+ 0.749, 0.749, 0.000,
+ 0.000, 1.000, 0.000,
+ 0.000, 0.000, 1.000,
+ 0.667, 0.000, 1.000,
+ 0.333, 0.333, 0.000,
+ 0.333, 0.667, 0.000,
+ 0.333, 1.000, 0.000,
+ 0.667, 0.333, 0.000,
+ 0.667, 0.667, 0.000,
+ 0.667, 1.000, 0.000,
+ 1.000, 0.333, 0.000,
+ 1.000, 0.667, 0.000,
+ 1.000, 1.000, 0.000,
+ 0.000, 0.333, 0.500,
+ 0.000, 0.667, 0.500,
+ 0.000, 1.000, 0.500,
+ 0.333, 0.000, 0.500,
+ 0.333, 0.333, 0.500,
+ 0.333, 0.667, 0.500,
+ 0.333, 1.000, 0.500,
+ 0.667, 0.000, 0.500,
+ 0.667, 0.333, 0.500,
+ 0.667, 0.667, 0.500,
+ 0.667, 1.000, 0.500,
+ 1.000, 0.000, 0.500,
+ 1.000, 0.333, 0.500,
+ 1.000, 0.667, 0.500,
+ 1.000, 1.000, 0.500,
+ 0.000, 0.333, 1.000,
+ 0.000, 0.667, 1.000,
+ 0.000, 1.000, 1.000,
+ 0.333, 0.000, 1.000,
+ 0.333, 0.333, 1.000,
+ 0.333, 0.667, 1.000,
+ 0.333, 1.000, 1.000,
+ 0.667, 0.000, 1.000,
+ 0.667, 0.333, 1.000,
+ 0.667, 0.667, 1.000,
+ 0.667, 1.000, 1.000,
+ 1.000, 0.000, 1.000,
+ 1.000, 0.333, 1.000,
+ 1.000, 0.667, 1.000,
+ 0.333, 0.000, 0.000,
+ 0.500, 0.000, 0.000,
+ 0.667, 0.000, 0.000,
+ 0.833, 0.000, 0.000,
+ 1.000, 0.000, 0.000,
+ 0.000, 0.167, 0.000,
+ 0.000, 0.333, 0.000,
+ 0.000, 0.500, 0.000,
+ 0.000, 0.667, 0.000,
+ 0.000, 0.833, 0.000,
+ 0.000, 1.000, 0.000,
+ 0.000, 0.000, 0.167,
+ 0.000, 0.000, 0.333,
+ 0.000, 0.000, 0.500,
+ 0.000, 0.000, 0.667,
+ 0.000, 0.000, 0.833,
+ 0.000, 0.000, 1.000,
+ 0.000, 0.000, 0.000,
+ 0.143, 0.143, 0.143,
+ 0.286, 0.286, 0.286,
+ 0.429, 0.429, 0.429,
+ 0.571, 0.571, 0.571,
+ 0.714, 0.714, 0.714,
+ 0.857, 0.857, 0.857,
+ 0.000, 0.447, 0.741,
+ 0.314, 0.717, 0.741,
+ 0.50, 0.5, 0
+ ]
+).astype(np.float32).reshape(-1, 3)
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/utils/pylogger.py b/third_party/GVHMR/hmr4d/utils/pylogger.py
new file mode 100644
index 0000000000000000000000000000000000000000..b827475115cfd67a66bdefc72bf56db544956139
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/pylogger.py
@@ -0,0 +1,76 @@
+from time import time
+import logging
+import torch
+from colorlog import ColoredFormatter
+
+
+def sync_time():
+ torch.cuda.synchronize()
+ return time()
+
+
+Log = logging.getLogger()
+Log.time = time
+Log.sync_time = sync_time
+
+# Set default
+Log.setLevel(logging.INFO)
+ch = logging.StreamHandler()
+ch.setLevel(logging.INFO)
+# Use colorlog
+formatstring = "[%(cyan)s%(asctime)s%(reset)s][%(log_color)s%(levelname)s%(reset)s] %(message)s"
+datefmt = "%m/%d %H:%M:%S"
+ch.setFormatter(ColoredFormatter(formatstring, datefmt=datefmt))
+
+Log.addHandler(ch)
+# Log.info("Init-Logger")
+
+
+def timer(sync_cuda=False, mem=False, loop=1):
+ """
+ Args:
+ func: function
+ sync_cuda: bool, whether to synchronize cuda
+ mem: bool, whether to log memory
+ """
+
+ def decorator(func):
+ def wrapper(*args, **kwargs):
+ if mem:
+ start_mem = torch.cuda.memory_allocated() / 1024**2
+ if sync_cuda:
+ torch.cuda.synchronize()
+
+ start = Log.time()
+ for _ in range(loop):
+ result = func(*args, **kwargs)
+
+ if sync_cuda:
+ torch.cuda.synchronize()
+ if loop == 1:
+ message = f"{func.__name__} took {Log.time() - start:.3f} s."
+ else:
+ message = f"{func.__name__} took {((Log.time() - start))/loop:.3f} s. (loop={loop})"
+
+ if mem:
+ end_mem = torch.cuda.memory_allocated() / 1024**2
+ end_max_mem = torch.cuda.max_memory_allocated() / 1024**2
+ message += f" Start_Mem {start_mem:.1f} Max {end_max_mem:.1f} MB"
+ Log.info(message)
+
+ return result
+
+ return wrapper
+
+ return decorator
+
+
+def timed(fn):
+ """example usage: timed(lambda: model(inp))"""
+ start = torch.cuda.Event(enable_timing=True)
+ end = torch.cuda.Event(enable_timing=True)
+ start.record()
+ result = fn()
+ end.record()
+ torch.cuda.synchronize()
+ return result, start.elapsed_time(end) / 1000
diff --git a/third_party/GVHMR/hmr4d/utils/seq_utils.py b/third_party/GVHMR/hmr4d/utils/seq_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..790cf3bc33df48fb3d4259b18c22cee3cc9c6eb7
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/seq_utils.py
@@ -0,0 +1,184 @@
+import torch
+import numpy as np
+
+# def get_frame_id_list_from_mask(mask):
+# """
+# Args:
+# mask (F,), bool.
+# Return:
+# frame_id_list: List of frame_ids.
+# """
+# frame_id_list = []
+# i = 0
+# while i < len(mask):
+# if not mask[i]:
+# i += 1
+# else:
+# j = i
+# while j < len(mask) and mask[j]:
+# j += 1
+# frame_id_list.append(torch.arange(i, j))
+# i = j
+
+# return frame_id_list
+
+
+# From GPT
+def get_frame_id_list_from_mask(mask):
+ # batch=64, 0.13s
+ """
+ Vectorized approach to get frame id list from a boolean mask.
+
+ Args:
+ mask (F,), bool tensor: Mask array where `True` indicates a frame to be processed.
+
+ Returns:
+ frame_id_list: List of torch.Tensors, each tensor containing continuous indices where mask is True.
+ """
+ # Find the indices where the mask changes from False to True and vice versa
+ padded_mask = torch.cat(
+ [torch.tensor([False], device=mask.device), mask, torch.tensor([False], device=mask.device)]
+ )
+ diffs = torch.diff(padded_mask.int())
+ starts = (diffs == 1).nonzero(as_tuple=False).squeeze()
+ ends = (diffs == -1).nonzero(as_tuple=False).squeeze()
+ if starts.numel() == 0:
+ return []
+ if starts.numel() == 1:
+ starts = starts.reshape(-1)
+ ends = ends.reshape(-1)
+
+ # Create list of ranges
+ frame_id_list = [torch.arange(start, end) for start, end in zip(starts, ends)]
+ return frame_id_list
+
+
+def get_batch_frame_id_lists_from_mask_BLC(masks):
+ # batch=64, 0.10s
+ """
+ 处理三维掩码数组,为每个批次和通道提取连续True区段的索引列表。
+
+ 参数:
+ masks (B, L, C), 布尔张量:每个元素代表一个掩码,True表示需要处理的帧。
+
+ 返回:
+ batch_frame_id_lists: 对应于每个批次和每个通道的帧id列表的嵌套列表。
+ """
+ B, L, C = masks.size()
+ # 在序列长度两端添加一个False
+ padded_masks = torch.cat(
+ [
+ torch.zeros((B, 1, C), dtype=torch.bool, device=masks.device),
+ masks,
+ torch.zeros((B, 1, C), dtype=torch.bool, device=masks.device),
+ ],
+ dim=1,
+ )
+ # 计算差分来找到True区段的起始和结束点
+ diffs = torch.diff(padded_masks.int(), dim=1)
+ starts = (diffs == 1).nonzero(as_tuple=True)
+ ends = (diffs == -1).nonzero(as_tuple=True)
+
+ # 初始化返回列表
+ batch_frame_id_lists = [[[] for _ in range(C)] for _ in range(B)]
+ for b in range(B):
+ for c in range(C):
+ batch_start = starts[0][(starts[0] == b) & (starts[2] == c)]
+ batch_end = ends[0][(ends[0] == b) & (ends[2] == c)]
+ # 确保start和end都是1维张量
+ batch_frame_id_lists[b][c] = [
+ torch.arange(start.item(), end.item()) for start, end in zip(batch_start, batch_end)
+ ]
+
+ return batch_frame_id_lists
+
+
+def get_frame_id_list_from_frame_id(frame_id):
+ mask = torch.zeros(frame_id[-1] + 1, dtype=torch.bool)
+ mask[frame_id] = True
+ frame_id_list = get_frame_id_list_from_mask(mask)
+ return frame_id_list
+
+
+def rearrange_by_mask(x, mask):
+ """
+ x (L, *)
+ mask (M,), M >= L
+ """
+ M = mask.size(0)
+ L = x.size(0)
+ if M == L:
+ return x
+ assert M > L
+ assert mask.sum() == L
+ x_rearranged = torch.zeros((M, *x.size()[1:]), dtype=x.dtype, device=x.device)
+ x_rearranged[mask] = x
+ return x_rearranged
+
+
+def frame_id_to_mask(frame_id, max_len):
+ mask = torch.zeros(max_len, dtype=torch.bool)
+ mask[frame_id] = True
+ return mask
+
+
+def mask_to_frame_id(mask):
+ frame_id = torch.where(mask)[0]
+ return frame_id
+
+
+def linear_interpolate_frame_ids(data, frame_id_list):
+ data = data.clone()
+ for i, invalid_frame_ids in enumerate(frame_id_list):
+ # interplate between prev, next
+ # if at beginning or end, use the same value
+ if invalid_frame_ids[0] - 1 < 0 or invalid_frame_ids[-1] + 1 >= len(data):
+ if invalid_frame_ids[0] - 1 < 0:
+ data[invalid_frame_ids] = data[invalid_frame_ids[-1] + 1].clone()
+ else:
+ data[invalid_frame_ids] = data[invalid_frame_ids[0] - 1].clone()
+ else:
+ prev = data[invalid_frame_ids[0] - 1]
+ next = data[invalid_frame_ids[-1] + 1]
+ data[invalid_frame_ids] = (
+ torch.linspace(0, 1, len(invalid_frame_ids) + 2)[1:-1][:, None] * (next - prev)[None] + prev[None]
+ )
+ return data
+
+
+def linear_interpolate(data, N_middle_frames):
+ """
+ Args:
+ data: (2, C)
+ Returns:
+ data_interpolated: (1+N+1, C)
+ """
+ prev = data[0]
+ next = data[1]
+ middle = torch.linspace(0, 1, N_middle_frames + 2)[1:-1][:, None] * (next - prev)[None] + prev[None] # (N, C)
+ data_interpolated = torch.cat([data[0][None], middle, data[1][None]], dim=0) # (1+N+1, C)
+ return data_interpolated
+
+
+def find_top_k_span(mask, k=3):
+ """
+ Args:
+ mask: (L,)
+ Return:
+ topk_span: List of tuple, usage: [start, end)
+ """
+ if isinstance(mask, np.ndarray):
+ mask = torch.from_numpy(mask)
+ if mask.sum() == 0:
+ return []
+ mask = mask.clone().float()
+ mask = torch.cat([mask.new([0]), mask, mask.new([0])])
+ diff = mask[1:] - mask[:-1]
+ start = torch.where(diff == 1)[0]
+ end = torch.where(diff == -1)[0]
+ assert len(start) == len(end)
+ span_lengths = end - start
+ span_lengths, idx = span_lengths.sort(descending=True)
+ start = start[idx]
+ end = end[idx]
+ return list(zip(start.tolist(), end.tolist()))[:k]
diff --git a/third_party/GVHMR/hmr4d/utils/smplx_utils.py b/third_party/GVHMR/hmr4d/utils/smplx_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..00477fd21a035f34fa9f2d836fb70f275751ed16
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/smplx_utils.py
@@ -0,0 +1,442 @@
+import torch
+import torch.nn.functional as F
+import numpy as np
+import smplx
+import pickle
+from smplx import SMPL, SMPLX, SMPLXLayer
+from hmr4d.utils.body_model import BodyModelSMPLH, BodyModelSMPLX
+from hmr4d.utils.body_model.smplx_lite import SmplxLiteCoco17, SmplxLiteV437Coco17, SmplxLiteSmplN24
+from hmr4d import PROJ_ROOT
+
+# fmt: off
+SMPLH_PARENTS = torch.tensor([-1, 0, 0, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 9, 9, 12, 13, 14,
+ 16, 17, 18, 19, 20, 22, 23, 20, 25, 26, 20, 28, 29, 20, 31, 32, 20, 34,
+ 35, 21, 37, 38, 21, 40, 41, 21, 43, 44, 21, 46, 47, 21, 49, 50])
+# fmt: on
+
+
+def make_smplx(type="neu_fullpose", **kwargs):
+ if type == "neu_fullpose":
+ model = smplx.create(
+ model_path="inputs/models/smplx/SMPLX_NEUTRAL.npz", use_pca=False, flat_hand_mean=True, **kwargs
+ )
+ elif type == "supermotion":
+ # SuperMotion is trained on BEDLAM dataset, the smplx config is the same except only 10 betas are used
+ bm_kwargs = {
+ "model_type": "smplx",
+ "gender": "neutral",
+ "num_pca_comps": 12,
+ "flat_hand_mean": False,
+ }
+ bm_kwargs.update(kwargs)
+ model = BodyModelSMPLX(model_path=PROJ_ROOT / "inputs/checkpoints/body_models", **bm_kwargs)
+ elif type == "supermotion_EVAL3DPW":
+ # SuperMotion is trained on BEDLAM dataset, the smplx config is the same except only 10 betas are used
+ bm_kwargs = {
+ "model_type": "smplx",
+ "gender": "neutral",
+ "num_pca_comps": 12,
+ "flat_hand_mean": True,
+ }
+ bm_kwargs.update(kwargs)
+ model = BodyModelSMPLX(model_path="third_party/GVHMR/inputs/checkpoints/body_models", **bm_kwargs)
+ elif type == "supermotion_coco17":
+ # Fast but only predicts 17 joints
+ model = SmplxLiteCoco17()
+ elif type == "supermotion_v437coco17":
+ # Predicts 437 verts and 17 joints
+ model = SmplxLiteV437Coco17()
+ elif type == "supermotion_smpl24":
+ model = SmplxLiteSmplN24()
+ elif type == "rich-smplx":
+ # https://github.com/paulchhuang/rich_toolkit/blob/main/smplx2images.py
+ bm_kwargs = {
+ "model_type": "smplx",
+ "gender": kwargs.get("gender", "male"),
+ "num_pca_comps": 12,
+ "flat_hand_mean": False,
+ # create_expression=True, create_jaw_pose=Ture
+ }
+ # A /smplx folder should exist under the model_path
+ model = BodyModelSMPLX(model_path="third_party/GVHMR/inputs/checkpoints/body_models", **bm_kwargs)
+ elif type == "rich-smplh":
+ bm_kwargs = {
+ "model_type": "smplh",
+ "gender": kwargs.get("gender", "male"),
+ "use_pca": False,
+ "flat_hand_mean": True,
+ }
+ model = BodyModelSMPLH(model_path="third_party/GVHMR/inputs/checkpoints/body_models", **bm_kwargs)
+
+ elif type in ["smplx-circle", "smplx-groundlink"]:
+ # don't use hand
+ bm_kwargs = {
+ "model_path": "third_party/GVHMR/inputs/checkpoints/body_models",
+ "model_type": "smplx",
+ "gender": kwargs.get("gender"),
+ "num_betas": 16,
+ "num_expression": 0,
+ }
+ model = BodyModelSMPLX(**bm_kwargs)
+
+ elif type == "smplx-motionx":
+ layer_args = {
+ "create_global_orient": False,
+ "create_body_pose": False,
+ "create_left_hand_pose": False,
+ "create_right_hand_pose": False,
+ "create_jaw_pose": False,
+ "create_leye_pose": False,
+ "create_reye_pose": False,
+ "create_betas": False,
+ "create_expression": False,
+ "create_transl": False,
+ }
+
+ bm_kwargs = {
+ "model_type": "smplx",
+ "model_path": "third_party/GVHMR/inputs/checkpoints/body_models",
+ "gender": "neutral",
+ "use_pca": False,
+ "use_face_contour": True,
+ **layer_args,
+ }
+ model = smplx.create(**bm_kwargs)
+
+ elif type == "smplx-samp":
+ # don't use hand
+ bm_kwargs = {
+ "model_path": "third_party/GVHMR/inputs/checkpoints/body_models",
+ "model_type": "smplx",
+ "gender": kwargs.get("gender"),
+ "num_betas": 10,
+ "num_expression": 0,
+ }
+ model = BodyModelSMPLX(**bm_kwargs)
+
+ elif type == "smplx-bedlam":
+ # don't use hand
+ bm_kwargs = {
+ "model_path": "third_party/GVHMR/inputs/checkpoints/body_models",
+ "model_type": "smplx",
+ "gender": kwargs.get("gender"),
+ "num_betas": 11,
+ "num_expression": 0,
+ }
+ model = BodyModelSMPLX(**bm_kwargs)
+
+ elif type in ["smplx-layer", "smplx-fit3d"]:
+ # Use layer
+ if type == "smplx-fit3d":
+ assert (
+ kwargs.get("gender") == "neutral"
+ ), "smplx-fit3d use neutral model: https://github.com/sminchisescu-research/imar_vision_datasets_tools/blob/e8c8f83ffac23cc36adf8ec8d0fd1c55679484ef/util/smplx_util.py#L15C34-L15C34"
+
+ bm_kwargs = {
+ "model_path": "third_party/GVHMR/inputs/checkpoints/body_models/smplx",
+ "gender": kwargs.get("gender"),
+ "num_betas": 10,
+ "num_expression": 10,
+ }
+ model = SMPLXLayer(**bm_kwargs)
+
+ elif type == "smpl":
+ bm_kwargs = {
+ "model_path": PROJ_ROOT / "inputs/checkpoints/body_models",
+ "model_type": "smpl",
+ "gender": "neutral",
+ "num_betas": 10,
+ "create_body_pose": False,
+ "create_betas": False,
+ "create_global_orient": False,
+ "create_transl": False,
+ }
+ bm_kwargs.update(kwargs)
+ # model = SMPL(**bm_kwargs)
+ model = BodyModelSMPLH(**bm_kwargs)
+ elif type == "smplh":
+ bm_kwargs = {
+ "model_type": "smplh",
+ "gender": kwargs.get("gender", "male"),
+ "use_pca": False,
+ "flat_hand_mean": False,
+ }
+ model = BodyModelSMPLH(model_path="third_party/GVHMR/inputs/checkpoints/body_models", **bm_kwargs)
+
+ else:
+ raise NotImplementedError
+
+ return model
+
+
+def load_parents(npz_path="models/smplx/SMPLX_NEUTRAL.npz"):
+ smplx_struct = np.load("models/smplx/SMPLX_NEUTRAL.npz", allow_pickle=True)
+ parents = smplx_struct["kintree_table"][0].astype(np.long)
+ parents[0] = -1
+ return parents
+
+
+def load_smpl_faces(npz_path="models/smplh/SMPLH_FEMALE.pkl"):
+ with open(npz_path, "rb") as f:
+ smpl_model = pickle.load(f, encoding="latin1")
+ faces = np.array(smpl_model["f"].astype(np.int64))
+ return faces
+
+
+def decompose_fullpose(fullpose, model_type="smplx"):
+ assert model_type == "smplx"
+
+ fullpose_dict = {
+ "global_orient": fullpose[..., :3],
+ "body_pose": fullpose[..., 3:66],
+ "jaw_pose": fullpose[..., 66:69],
+ "leye_pose": fullpose[..., 69:72],
+ "reye_pose": fullpose[..., 72:75],
+ "left_hand_pose": fullpose[..., 75:120],
+ "right_hand_pose": fullpose[..., 120:165],
+ }
+
+ return fullpose_dict
+
+
+def compose_fullpose(fullpose_dict, model_type="smplx"):
+ assert model_type == "smplx"
+ fullpose = torch.cat(
+ [
+ fullpose_dict[k]
+ for k in [
+ "global_orient",
+ "body_pose",
+ "jaw_pose",
+ "leye_pose",
+ "reye_pose",
+ "left_hand_pose",
+ "right_hand_pose",
+ ]
+ ],
+ dim=-1,
+ )
+ return fullpose
+
+
+def compute_R_from_kinetree(rot_mats, parents):
+ """operation of lbs/batch_rigid_transform, focus on 3x3 R only
+ Parameters
+ ----------
+ rot_mats: torch.tensor BxNx3x3
+ Tensor of rotation matrices
+ parents : torch.tensor BxN
+ The kinematic tree of each object
+
+ Returns
+ -------
+ R : torch.tensor BxNx3x3
+ Tensor of rotation matrices
+ """
+ rot_mat_chain = [rot_mats[:, 0]]
+ for i in range(1, parents.shape[0]):
+ curr_res = torch.matmul(rot_mat_chain[parents[i]], rot_mats[:, i])
+ rot_mat_chain.append(curr_res)
+
+ R = torch.stack(rot_mat_chain, dim=1)
+ return R
+
+
+def compute_relR_from_kinetree(R, parents):
+ """Inverse operation of lbs/batch_rigid_transform, focus on 3x3 R only
+ Parameters
+ ----------
+ R : torch.tensor BxNx4x4 or BxNx3x3
+ Tensor of rotation matrices
+ parents : torch.tensor BxN
+ The kinematic tree of each object
+
+ Returns
+ -------
+ rot_mats: torch.tensor BxNx3x3
+ Tensor of rotation matrices
+ """
+ R = R[:, :, :3, :3]
+
+ Rp = R[:, parents] # Rp[:, 0] is invalid
+ rot_mats = Rp.transpose(2, 3) @ R
+ rot_mats[:, 0] = R[:, 0]
+
+ return rot_mats
+
+
+def quat_mul(x, y):
+ """
+ Performs quaternion multiplication on arrays of quaternions
+
+ :param x: tensor of quaternions of shape (..., Nb of joints, 4)
+ :param y: tensor of quaternions of shape (..., Nb of joints, 4)
+ :return: The resulting quaternions
+ """
+ x0, x1, x2, x3 = x[..., 0:1], x[..., 1:2], x[..., 2:3], x[..., 3:4]
+ y0, y1, y2, y3 = y[..., 0:1], y[..., 1:2], y[..., 2:3], y[..., 3:4]
+
+ # res = np.concatenate(
+ # [
+ # y0 * x0 - y1 * x1 - y2 * x2 - y3 * x3,
+ # y0 * x1 + y1 * x0 - y2 * x3 + y3 * x2,
+ # y0 * x2 + y1 * x3 + y2 * x0 - y3 * x1,
+ # y0 * x3 - y1 * x2 + y2 * x1 + y3 * x0,
+ # ],
+ # axis=-1,
+ # )
+ res = torch.cat(
+ [
+ y0 * x0 - y1 * x1 - y2 * x2 - y3 * x3,
+ y0 * x1 + y1 * x0 - y2 * x3 + y3 * x2,
+ y0 * x2 + y1 * x3 + y2 * x0 - y3 * x1,
+ y0 * x3 - y1 * x2 + y2 * x1 + y3 * x0,
+ ],
+ axis=-1,
+ )
+
+ return res
+
+
+def quat_inv(q):
+ """
+ Inverts a tensor of quaternions
+
+ :param q: quaternion tensor
+ :return: tensor of inverted quaternions
+ """
+ # res = np.asarray([1, -1, -1, -1], dtype=np.float32) * q
+ res = torch.tensor([1, -1, -1, -1], device=q.device).float() * q
+ return res
+
+
+def quat_mul_vec(q, x):
+ """
+ Performs multiplication of an array of 3D vectors by an array of quaternions (rotation).
+
+ :param q: tensor of quaternions of shape (..., Nb of joints, 4)
+ :param x: tensor of vectors of shape (..., Nb of joints, 3)
+ :return: the resulting array of rotated vectors
+ """
+ # t = 2.0 * np.cross(q[..., 1:], x)
+ t = 2.0 * torch.cross(q[..., 1:], x)
+ # res = x + q[..., 0][..., np.newaxis] * t + np.cross(q[..., 1:], t)
+ res = x + q[..., 0][..., None] * t + torch.cross(q[..., 1:], t)
+
+ return res
+
+
+def inverse_kinematics_motion(
+ global_pos,
+ global_rot,
+ parents=SMPLH_PARENTS,
+):
+ """
+ Args:
+ global_pos : (B, T, J-1, 3)
+ global_rot (q) : (B, T, J-1, 4)
+ parents : SMPLH_PARENTS
+ Returns:
+ local_pos : (B, T, J-1, 3)
+ local_rot (q) : (B, T, J-1, 4)
+ """
+ J = 22
+ local_pos = quat_mul_vec(
+ quat_inv(global_rot[..., parents[1:J], :]),
+ global_pos - global_pos[..., parents[1:J], :],
+ )
+ local_rot = (quat_mul(quat_inv(global_rot[..., parents[1:J], :]), global_rot),)
+ return local_pos, local_rot
+
+
+def transform_mat(R, t):
+ """Creates a batch of transformation matrices
+ Args:
+ - R: Bx3x3 array of a batch of rotation matrices
+ - t: Bx3x1 array of a batch of translation vectors
+ Returns:
+ - T: Bx4x4 Transformation matrix
+ """
+ # No padding left or right, only add an extra row
+ return torch.cat([F.pad(R, [0, 0, 0, 1]), F.pad(t, [0, 0, 0, 1], value=1)], dim=2)
+
+
+def normalize_joints(joints):
+ """
+ Args:
+ joints: (B, *, J, 3)
+ """
+ LR_hips_xy = joints[..., 2, [0, 1]] - joints[..., 1, [0, 1]]
+ LR_shoulders_xy = joints[..., 17, [0, 1]] - joints[..., 16, [0, 1]]
+ LR_xy = (LR_hips_xy + LR_shoulders_xy) / 2 # (B, *, J, 2)
+
+ x_dir = F.pad(F.normalize(LR_xy, 2, -1), (0, 1), "constant", 0) # (B, *, 3)
+ z_dir = torch.zeros_like(x_dir) # (B, *, 3)
+ z_dir[..., 2] = 1
+ y_dir = torch.cross(z_dir, x_dir, dim=-1)
+
+ joints_normalized = (joints - joints[..., [0], :]) @ torch.stack([x_dir, y_dir, z_dir], dim=-1)
+ return joints_normalized
+
+
+@torch.no_grad()
+def compute_Rt_af2az(joints, inverse=False):
+ """Assume z coord is upward
+ Args:
+ joints: (B, J, 3), in the start-frame
+ Returns:
+ R_af2az: (B, 3, 3)
+ t_af2az: (B, 3)
+ """
+ t_af2az = joints[:, 0, :].detach().clone()
+ t_af2az[:, 2] = 0 # do not modify z
+
+ LR_xy = joints[:, 2, [0, 1]] - joints[:, 1, [0, 1]] # (B, 2)
+ I_mask = LR_xy.pow(2).sum(-1) < 1e-4 # do not rotate, when can't decided the face direction
+ x_dir = F.pad(F.normalize(LR_xy, 2, -1), (0, 1), "constant", 0) # (B, 3)
+ z_dir = torch.zeros_like(x_dir)
+ z_dir[..., 2] = 1
+ y_dir = torch.cross(z_dir, x_dir, dim=-1)
+ R_af2az = torch.stack([x_dir, y_dir, z_dir], dim=-1) # (B, 3, 3)
+ R_af2az[I_mask] = torch.eye(3).to(R_af2az)
+
+ if inverse:
+ R_az2af = R_af2az.transpose(1, 2)
+ t_az2af = -(R_az2af @ t_af2az.unsqueeze(2)).squeeze(2)
+ return R_az2af, t_az2af
+ else:
+ return R_af2az, t_af2az
+
+
+def finite_difference_forward(x, dim_t=1, dup_last=True):
+ if dim_t == 1:
+ v = x[:, 1:] - x[:, :-1]
+ if dup_last:
+ v = torch.cat([v, v[:, [-1]]], dim=1)
+ else:
+ raise NotImplementedError
+
+ return v
+
+
+def compute_joints_zero(betas, gender):
+ """
+ Args:
+ betas: (16)
+ gender: 'male' or 'female'
+ Returns:
+ joints_zero: (22, 3)
+ """
+ body_model = {
+ "male": make_smplx(type="humor", gender="male"),
+ "female": make_smplx(type="humor", gender="female"),
+ }
+
+ smpl_params = {
+ "root_orient": torch.zeros((1, 3)),
+ "pose_body": torch.zeros((1, 63)),
+ "betas": betas[None],
+ "trans": torch.zeros(1, 3),
+ }
+ joints_zero = body_model[gender](**smpl_params).Jtr[0, :22]
+ return joints_zero
diff --git a/third_party/GVHMR/hmr4d/utils/video_io_utils.py b/third_party/GVHMR/hmr4d/utils/video_io_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..9ccfeaf811af7f6e4eb4d4809f295995c3239437
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/video_io_utils.py
@@ -0,0 +1,113 @@
+import imageio.v3 as iio
+import numpy as np
+import torch
+from pathlib import Path
+import shutil
+import ffmpeg
+from tqdm import tqdm
+import cv2
+
+
+def get_video_lwh(video_path):
+ L, H, W, _ = iio.improps(video_path, plugin="pyav").shape
+ return L, W, H
+
+
+def read_video_np(video_path, start_frame=0, end_frame=-1, scale=1.0):
+ """
+ Args:
+ video_path: str
+ Returns:
+ frames: np.array, (N, H, W, 3) RGB, uint8
+ """
+ # If video path not exists, an error will be raised by ffmpegs
+ filter_args = []
+ should_check_length = False
+
+ # 1. Trim
+ if not (start_frame == 0 and end_frame == -1):
+ if end_frame == -1:
+ filter_args.append(("trim", f"start_frame={start_frame}"))
+ else:
+ should_check_length = True
+ filter_args.append(("trim", f"start_frame={start_frame}:end_frame={end_frame}"))
+
+ # 2. Scale
+ if scale != 1.0:
+ filter_args.append(("scale", f"iw*{scale}:ih*{scale}"))
+
+ # Excute then check
+ frames = iio.imread(video_path, plugin="pyav", filter_sequence=filter_args)
+ if should_check_length:
+ assert len(frames) == end_frame - start_frame
+
+ return frames
+
+
+def get_video_reader(video_path):
+ return iio.imiter(video_path, plugin="pyav")
+
+
+def read_images_np(image_paths, verbose=False):
+ """
+ Args:
+ image_paths: list of str
+ Returns:
+ images: np.array, (N, H, W, 3) RGB, uint8
+ """
+ if verbose:
+ images = [cv2.imread(str(img_path))[..., ::-1] for img_path in tqdm(image_paths)]
+ else:
+ images = [cv2.imread(str(img_path))[..., ::-1] for img_path in image_paths]
+ images = np.stack(images, axis=0)
+ return images
+
+
+def save_video(images, video_path, fps=30, crf=17):
+ """
+ Args:
+ images: (N, H, W, 3) RGB, uint8
+ crf: 17 is visually lossless, 23 is default, +6 results in half the bitrate
+ 0 is lossless, https://trac.ffmpeg.org/wiki/Encode/H.264#crf
+ """
+ if isinstance(images, torch.Tensor):
+ images = images.cpu().numpy().astype(np.uint8)
+ elif isinstance(images, list):
+ images = np.array(images).astype(np.uint8)
+
+ with iio.imopen(video_path, "w", plugin="pyav") as writer:
+ writer.init_video_stream("libx264", fps=fps)
+ writer._video_stream.options = {"crf": str(crf)}
+ writer.write(images)
+
+
+def get_writer(video_path, fps=30, crf=17):
+ """remember to .close()"""
+ writer = iio.imopen(video_path, "w", plugin="pyav")
+ writer.init_video_stream("libx264", fps=fps)
+ writer._video_stream.options = {"crf": str(crf)}
+ return writer
+
+
+def copy_file(video_path, out_video_path, overwrite=True):
+ if not overwrite and Path(out_video_path).exists():
+ return
+ shutil.copy(video_path, out_video_path)
+
+
+def merge_videos_horizontal(in_video_paths: list, out_video_path: str):
+ if len(in_video_paths) < 2:
+ raise ValueError("At least two video paths are required for merging.")
+ inputs = [ffmpeg.input(path) for path in in_video_paths]
+ merged_video = ffmpeg.filter(inputs, "hstack", inputs=len(inputs))
+ output = ffmpeg.output(merged_video, out_video_path)
+ ffmpeg.run(output, overwrite_output=True, quiet=True)
+
+
+def merge_videos_vertical(in_video_paths: list, out_video_path: str):
+ if len(in_video_paths) < 2:
+ raise ValueError("At least two video paths are required for merging.")
+ inputs = [ffmpeg.input(path) for path in in_video_paths]
+ merged_video = ffmpeg.filter(inputs, "vstack", inputs=len(inputs))
+ output = ffmpeg.output(merged_video, out_video_path)
+ ffmpeg.run(output, overwrite_output=True, quiet=True)
diff --git a/third_party/GVHMR/hmr4d/utils/vis/README.md b/third_party/GVHMR/hmr4d/utils/vis/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..9582abc7ca64409209ae252ac53b361f12ed6ecc
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/vis/README.md
@@ -0,0 +1,20 @@
+## Pytorch3D Renderer
+
+Example:
+```python
+from hmr4d.utils.vis.renderer import Renderer
+import imageio
+
+fps = 30
+focal_length = data["cam_int"][0][0, 0]
+width, height = img_hw
+faces = smplh[data["gender"]].bm.faces
+renderer = Renderer(width, height, focal_length, "cuda", faces)
+writer = imageio.get_writer("tmp_debug.mp4", fps=fps, mode="I", format="FFMPEG", macro_block_size=1)
+
+for i in tqdm(range(length)):
+ img = np.zeros((height, width, 3), dtype=np.uint8)
+ img = renderer.render_mesh(smplh_out.vertices[i].cuda(), img)
+ writer.append_data(img)
+writer.close()
+```
\ No newline at end of file
diff --git a/third_party/GVHMR/hmr4d/utils/vis/cv2_utils.py b/third_party/GVHMR/hmr4d/utils/vis/cv2_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..77ca33a4339921d0c86b8141b05258e4b059404d
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/vis/cv2_utils.py
@@ -0,0 +1,144 @@
+import torch
+import cv2
+import numpy as np
+from hmr4d.utils.wis3d_utils import get_colors_by_conf
+
+
+def to_numpy(x):
+ if isinstance(x, np.ndarray):
+ return x.copy()
+ elif isinstance(x, list):
+ return np.array(x)
+ return x.clone().cpu().numpy()
+
+
+def draw_bbx_xys_on_image(bbx_xys, image, conf=True):
+ assert isinstance(bbx_xys, np.ndarray)
+ assert isinstance(image, np.ndarray)
+ image = image.copy()
+ lu_point = (bbx_xys[:2] - bbx_xys[2:] / 2).astype(int)
+ rd_point = (bbx_xys[:2] + bbx_xys[2:] / 2).astype(int)
+ color = (255, 178, 102) if conf == True else (128, 128, 128) # orange or gray
+ image = cv2.rectangle(image, lu_point, rd_point, color, 2)
+ return image
+
+
+def draw_bbx_xys_on_image_batch(bbx_xys_batch, image_batch, conf=None):
+ """conf: if provided, list of bool"""
+ use_conf = conf is not None
+ bbx_xys_batch = to_numpy(bbx_xys_batch)
+ assert len(bbx_xys_batch) == len(image_batch)
+ image_batch_out = []
+ for i in range(len(bbx_xys_batch)):
+ if use_conf:
+ image_batch_out.append(draw_bbx_xys_on_image(bbx_xys_batch[i], image_batch[i], conf[i]))
+ else:
+ image_batch_out.append(draw_bbx_xys_on_image(bbx_xys_batch[i], image_batch[i]))
+ return image_batch_out
+
+
+def draw_bbx_xyxy_on_image(bbx_xys, image, conf=True):
+ bbx_xys = to_numpy(bbx_xys)
+ image = to_numpy(image)
+ color = (255, 178, 102) if conf == True else (128, 128, 128) # orange or gray
+ image = cv2.rectangle(image, (int(bbx_xys[0]), int(bbx_xys[1])), (int(bbx_xys[2]), int(bbx_xys[3])), color, 2)
+ return image
+
+
+def draw_bbx_xyxy_on_image_batch(bbx_xyxy_batch, image_batch, mask=None, conf=None):
+ """
+ Args:
+ conf: if provided, list of bool, mutually exclusive with mask
+ mask: whether to draw, historically used
+ """
+ if mask is not None:
+ assert conf is None
+ if conf is not None:
+ assert mask is None
+ use_conf = conf is not None
+ bbx_xyxy_batch = to_numpy(bbx_xyxy_batch)
+ image_batch = to_numpy(image_batch)
+ assert len(bbx_xyxy_batch) == len(image_batch)
+ image_batch_out = []
+ for i in range(len(bbx_xyxy_batch)):
+ if use_conf:
+ image_batch_out.append(draw_bbx_xyxy_on_image(bbx_xyxy_batch[i], image_batch[i], conf[i]))
+ else:
+ if mask is None or mask[i]:
+ image_batch_out.append(draw_bbx_xyxy_on_image(bbx_xyxy_batch[i], image_batch[i]))
+ else:
+ image_batch_out.append(image_batch[i])
+ return image_batch_out
+
+
+def draw_kpts(frame, keypoints, color=(0, 255, 0), thickness=2):
+ frame_ = frame.copy()
+ for x, y in keypoints:
+ cv2.circle(frame_, (int(x), int(y)), thickness, color, -1)
+ return frame_
+
+
+def draw_kpts_with_conf(frame, kp2d, conf, thickness=2):
+ """
+ Args:
+ kp2d: (J, 2),
+ conf: (J,)
+ """
+ frame_ = frame.copy()
+ conf = conf.reshape(-1)
+ colors = get_colors_by_conf(conf) # (J, 3)
+ colors = colors[:, [2, 1, 0]].int().numpy().tolist()
+ for j in range(kp2d.shape[0]):
+ x, y = kp2d[j, :2]
+ c = colors[j]
+ cv2.circle(frame_, (int(x), int(y)), thickness, c, -1)
+ return frame_
+
+
+def draw_kpts_with_conf_batch(frames, kp2d_batch, conf_batch, thickness=2):
+ """
+ Args:
+ kp2d_batch: (B, J, 2),
+ conf_batch: (B, J)
+ """
+ assert len(frames) == len(kp2d_batch)
+ assert len(frames) == len(conf_batch)
+ frames_ = []
+ for i in range(len(frames)):
+ frames_.append(draw_kpts_with_conf(frames[i], kp2d_batch[i], conf_batch[i], thickness))
+ return frames_
+
+
+def draw_coco17_skeleton(img, keypoints, conf_thr=0):
+ use_conf_thr = True if keypoints.shape[1] == 3 else False
+ img = img.copy()
+ # fmt:off
+ coco_skel = [[15, 13], [13, 11], [16, 14], [14, 12], [11, 12], [5, 11], [6, 12], [5, 6], [5, 7], [6, 8], [7, 9], [8, 10], [1, 2], [0, 1], [0, 2], [1, 3], [2, 4], [3, 5], [4, 6]]
+ # fmt:on
+ for bone in coco_skel:
+ if use_conf_thr:
+ kp1 = keypoints[bone[0]][:2].astype(int)
+ kp2 = keypoints[bone[1]][:2].astype(int)
+ kp1_c = keypoints[bone[0]][2]
+ kp2_c = keypoints[bone[1]][2]
+ if kp1_c > conf_thr and kp2_c > conf_thr:
+ img = cv2.line(img, (kp1[0], kp1[1]), (kp2[0], kp2[1]), (0, 255, 0), 4)
+ if kp1_c > conf_thr:
+ img = cv2.circle(img, (kp1[0], kp1[1]), 6, (0, 255, 0), -1)
+ if kp2_c > conf_thr:
+ img = cv2.circle(img, (kp2[0], kp2[1]), 6, (0, 255, 0), -1)
+
+ else:
+ kp1 = keypoints[bone[0]][:2].astype(int)
+ kp2 = keypoints[bone[1]][:2].astype(int)
+ img = cv2.line(img, (kp1[0], kp1[1]), (kp2[0], kp2[1]), (0, 255, 0), 4)
+ return img
+
+
+def draw_coco17_skeleton_batch(imgs, keypoints_batch, conf_thr=0):
+ assert len(imgs) == len(keypoints_batch)
+ keypoints_batch = to_numpy(keypoints_batch)
+ imgs_out = []
+ for i in range(len(imgs)):
+ imgs_out.append(draw_coco17_skeleton(imgs[i], keypoints_batch[i], conf_thr))
+ return imgs_out
diff --git a/third_party/GVHMR/hmr4d/utils/vis/renderer.py b/third_party/GVHMR/hmr4d/utils/vis/renderer.py
new file mode 100644
index 0000000000000000000000000000000000000000..8f26bbeb05e5df9fe14aa445d10c365545d3d3bb
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/vis/renderer.py
@@ -0,0 +1,368 @@
+import cv2
+import torch
+import numpy as np
+
+from pytorch3d.renderer import (
+ PerspectiveCameras,
+ TexturesVertex,
+ PointLights,
+ Materials,
+ RasterizationSettings,
+ MeshRenderer,
+ MeshRasterizer,
+ SoftPhongShader,
+)
+from pytorch3d.structures import Meshes
+from pytorch3d.structures.meshes import join_meshes_as_scene
+from pytorch3d.renderer.cameras import look_at_rotation
+from pytorch3d.transforms import axis_angle_to_matrix
+
+from .renderer_tools import get_colors, checkerboard_geometry
+
+
+colors_str_map = {
+ "gray": [0.8, 0.8, 0.8],
+ "green": [39, 194, 128],
+}
+
+
+def overlay_image_onto_background(image, mask, bbox, background, alpha=1.0):
+ if isinstance(image, torch.Tensor):
+ image = image.detach().cpu().numpy()
+ if isinstance(mask, torch.Tensor):
+ mask = mask.detach().cpu().numpy()
+
+ out_image = background.copy()
+ bbox = bbox[0].int().cpu().numpy().copy()
+ roi_image = out_image[bbox[1] : bbox[3], bbox[0] : bbox[2]]
+
+ if alpha >= 1.0:
+ roi_image[mask] = image[mask]
+ else:
+ a = float(np.clip(alpha, 0.0, 1.0))
+ roi_f = roi_image.astype(np.float32)
+ img_f = image.astype(np.float32)
+ roi_f[mask] = roi_f[mask] * (1.0 - a) + img_f[mask] * a
+ roi_image = roi_f.astype(background.dtype, copy=False)
+ out_image[bbox[1] : bbox[3], bbox[0] : bbox[2]] = roi_image
+
+ return out_image
+
+
+def update_intrinsics_from_bbox(K_org, bbox):
+ device, dtype = K_org.device, K_org.dtype
+
+ K = torch.zeros((K_org.shape[0], 4, 4)).to(device=device, dtype=dtype)
+ K[:, :3, :3] = K_org.clone()
+ K[:, 2, 2] = 0
+ K[:, 2, -1] = 1
+ K[:, -1, 2] = 1
+
+ image_sizes = []
+ for idx, bbox in enumerate(bbox):
+ left, upper, right, lower = bbox
+ cx, cy = K[idx, 0, 2], K[idx, 1, 2]
+
+ new_cx = cx - left
+ new_cy = cy - upper
+ new_height = max(lower - upper, 1)
+ new_width = max(right - left, 1)
+ new_cx = new_width - new_cx
+ new_cy = new_height - new_cy
+
+ K[idx, 0, 2] = new_cx
+ K[idx, 1, 2] = new_cy
+ image_sizes.append((int(new_height), int(new_width)))
+
+ return K, image_sizes
+
+
+def perspective_projection(x3d, K, R=None, T=None):
+ if R != None:
+ x3d = torch.matmul(R, x3d.transpose(1, 2)).transpose(1, 2)
+ if T != None:
+ x3d = x3d + T.transpose(1, 2)
+
+ x2d = torch.div(x3d, x3d[..., 2:])
+ x2d = torch.matmul(K, x2d.transpose(-1, -2)).transpose(-1, -2)[..., :2]
+ return x2d
+
+
+def compute_bbox_from_points(X, img_w, img_h, scaleFactor=1.2):
+ left = torch.clamp(X.min(1)[0][:, 0], min=0, max=img_w)
+ right = torch.clamp(X.max(1)[0][:, 0], min=0, max=img_w)
+ top = torch.clamp(X.min(1)[0][:, 1], min=0, max=img_h)
+ bottom = torch.clamp(X.max(1)[0][:, 1], min=0, max=img_h)
+
+ cx = (left + right) / 2
+ cy = (top + bottom) / 2
+ width = right - left
+ height = bottom - top
+
+ new_left = torch.clamp(cx - width / 2 * scaleFactor, min=0, max=img_w - 1)
+ new_right = torch.clamp(cx + width / 2 * scaleFactor, min=1, max=img_w)
+ new_top = torch.clamp(cy - height / 2 * scaleFactor, min=0, max=img_h - 1)
+ new_bottom = torch.clamp(cy + height / 2 * scaleFactor, min=1, max=img_h)
+
+ bbox = torch.stack((new_left.detach(), new_top.detach(), new_right.detach(), new_bottom.detach())).int().float().T
+
+ return bbox
+
+
+class Renderer:
+ def __init__(self, width, height, focal_length=None, device="cuda", faces=None, K=None, bin_size=None):
+ """set bin_size to 0 for no binning"""
+ self.width = width
+ self.height = height
+ self.bin_size = bin_size
+ assert (focal_length is not None) ^ (K is not None), "focal_length and K are mutually exclusive"
+
+ self.device = device
+ if faces is not None:
+ if isinstance(faces, np.ndarray):
+ faces = torch.from_numpy((faces).astype("int"))
+ self.faces = faces.unsqueeze(0).to(self.device)
+
+ self.initialize_camera_params(focal_length, K)
+ self.lights = PointLights(device=device, location=[[0.0, 0.0, -10.0]])
+ self.create_renderer()
+
+ def create_renderer(self):
+ self.renderer = MeshRenderer(
+ rasterizer=MeshRasterizer(
+ raster_settings=RasterizationSettings(
+ image_size=self.image_sizes[0], blur_radius=1e-5, bin_size=self.bin_size
+ ),
+ ),
+ shader=SoftPhongShader(
+ device=self.device,
+ lights=self.lights,
+ ),
+ )
+
+ def create_camera(self, R=None, T=None):
+ if R is not None:
+ self.R = R.clone().view(1, 3, 3).to(self.device)
+ if T is not None:
+ self.T = T.clone().view(1, 3).to(self.device)
+
+ return PerspectiveCameras(
+ device=self.device, R=self.R.mT, T=self.T, K=self.K_full, image_size=self.image_sizes, in_ndc=False
+ )
+
+ def initialize_camera_params(self, focal_length, K):
+ # Extrinsics
+ self.R = torch.diag(torch.tensor([1, 1, 1])).float().to(self.device).unsqueeze(0)
+
+ self.T = torch.tensor([0, 0, 0]).unsqueeze(0).float().to(self.device)
+
+ # Intrinsics
+ if K is not None:
+ self.K = K.float().reshape(1, 3, 3).to(self.device)
+ else:
+ assert focal_length is not None, "focal_length or K should be provided"
+ self.K = (
+ torch.tensor([[focal_length, 0, self.width / 2], [0, focal_length, self.height / 2], [0, 0, 1]])
+ .float()
+ .reshape(1, 3, 3)
+ .to(self.device)
+ )
+ self.bboxes = torch.tensor([[0, 0, self.width, self.height]]).float()
+ self.K_full, self.image_sizes = update_intrinsics_from_bbox(self.K, self.bboxes)
+ self.cameras = self.create_camera()
+
+ def set_intrinsic(self, K):
+ self.K = K.reshape(1, 3, 3)
+
+ def set_ground(self, length, center_x, center_z):
+ device = self.device
+ length, center_x, center_z = map(float, (length, center_x, center_z))
+ v, f, vc, fc = map(torch.from_numpy, checkerboard_geometry(length=length, c1=center_x, c2=center_z, up="y"))
+ v, f, vc = v.to(device), f.to(device), vc.to(device)
+ self.ground_geometry = [v, f, vc]
+
+ def update_bbox(self, x3d, scale=2.0, mask=None):
+ """Update bbox of cameras from the given 3d points
+
+ x3d: input 3D keypoints (or vertices), (num_frames, num_points, 3)
+ """
+
+ if x3d.size(-1) != 3:
+ x2d = x3d.unsqueeze(0)
+ else:
+ x2d = perspective_projection(x3d.unsqueeze(0), self.K, self.R, self.T.reshape(1, 3, 1))
+
+ if mask is not None:
+ x2d = x2d[:, ~mask]
+
+ bbox = compute_bbox_from_points(x2d, self.width, self.height, scale)
+ self.bboxes = bbox
+
+ self.K_full, self.image_sizes = update_intrinsics_from_bbox(self.K, bbox)
+ self.cameras = self.create_camera()
+ self.create_renderer()
+
+ def reset_bbox(
+ self,
+ ):
+ bbox = torch.zeros((1, 4)).float().to(self.device)
+ bbox[0, 2] = self.width
+ bbox[0, 3] = self.height
+ self.bboxes = bbox
+
+ self.K_full, self.image_sizes = update_intrinsics_from_bbox(self.K, bbox)
+ self.cameras = self.create_camera()
+ self.create_renderer()
+
+ def render_mesh(self, vertices, background=None, colors=[0.8, 0.8, 0.8], VI=50, alpha=1.0):
+ self.update_bbox(vertices[::VI], scale=1.2)
+ vertices = vertices.unsqueeze(0)
+
+ if isinstance(colors, torch.Tensor):
+ # per-vertex color
+ verts_features = colors.to(device=vertices.device, dtype=vertices.dtype)
+ colors = [0.8, 0.8, 0.8]
+ else:
+ # Accept either [0..1] floats or [0..255] uint8-like colors.
+ # Don't key off `colors[0]` because valid RGB like green [0,255,0] would fail.
+ try:
+ if max(colors) > 1:
+ colors = [c / 255.0 for c in colors]
+ except Exception:
+ pass
+ verts_features = torch.tensor(colors).reshape(1, 1, 3).to(device=vertices.device, dtype=vertices.dtype)
+ verts_features = verts_features.repeat(1, vertices.shape[1], 1)
+ textures = TexturesVertex(verts_features=verts_features)
+
+ mesh = Meshes(
+ verts=vertices,
+ faces=self.faces,
+ textures=textures,
+ )
+
+ materials = Materials(device=self.device, specular_color=(colors,), shininess=0)
+
+ results = torch.flip(self.renderer(mesh, materials=materials, cameras=self.cameras, lights=self.lights), [1, 2])
+ image = results[0, ..., :3] * 255
+ mask = results[0, ..., -1] > 1e-3
+
+ if background is None:
+ background = np.ones((self.height, self.width, 3)).astype(np.uint8) * 255
+
+ image = overlay_image_onto_background(image, mask, self.bboxes, background.copy(), alpha=alpha)
+ self.reset_bbox()
+ return image
+
+ def render_with_ground(self, verts, colors, cameras, lights, faces=None):
+ """
+ :param verts (N, V, 3), potential multiple people
+ :param colors (N, 3) or (N, V, 3)
+ :param faces (N, F, 3), optional, otherwise self.faces is used will be used
+ """
+ # Sanity check of input verts, colors and faces: (B, V, 3), (B, F, 3), (B, V, 3)
+ N, V, _ = verts.shape
+ if faces is None:
+ faces = self.faces.clone().expand(N, -1, -1)
+ else:
+ assert len(faces.shape) == 3, "faces should have shape of (N, F, 3)"
+
+ assert len(colors.shape) in [2, 3]
+ if len(colors.shape) == 2:
+ assert len(colors) == N, "colors of shape 2 should be (N, 3)"
+ colors = colors[:, None]
+ colors = colors.expand(N, V, -1)[..., :3]
+
+ # (V, 3), (F, 3), (V, 3)
+ gv, gf, gc = self.ground_geometry
+ verts = list(torch.unbind(verts, dim=0)) + [gv]
+ faces = list(torch.unbind(faces, dim=0)) + [gf]
+ colors = list(torch.unbind(colors, dim=0)) + [gc[..., :3]]
+ mesh = create_meshes(verts, faces, colors)
+
+ materials = Materials(device=self.device, shininess=0)
+
+ results = self.renderer(mesh, cameras=cameras, lights=lights, materials=materials)
+ image = (results[0, ..., :3].cpu().numpy() * 255).astype(np.uint8)
+
+ return image
+
+
+def create_meshes(verts, faces, colors):
+ """
+ :param verts (B, V, 3)
+ :param faces (B, F, 3)
+ :param colors (B, V, 3)
+ """
+ textures = TexturesVertex(verts_features=colors)
+ meshes = Meshes(verts=verts, faces=faces, textures=textures)
+ return join_meshes_as_scene(meshes)
+
+
+def get_global_cameras(verts, device="cuda", distance=5, position=(-5.0, 5.0, 0.0)):
+ """This always put object at the center of view"""
+ positions = torch.tensor([position]).repeat(len(verts), 1)
+ targets = verts.mean(1)
+
+ directions = targets - positions
+ directions = directions / torch.norm(directions, dim=-1).unsqueeze(-1) * distance
+ positions = targets - directions
+
+ rotation = look_at_rotation(positions, targets).mT
+ translation = -(rotation @ positions.unsqueeze(-1)).squeeze(-1)
+
+ lights = PointLights(device=device, location=[position])
+ return rotation, translation, lights
+
+
+def get_global_cameras_static(
+ verts, beta=4.0, cam_height_degree=30, target_center_height=1.0, use_long_axis=False, vec_rot=45, device="cuda"
+):
+ L, V, _ = verts.shape
+
+ # Compute target trajectory, denote as center + scale
+ targets = verts.mean(1) # (L, 3)
+ targets[:, 1] = 0 # project to xz-plane
+ target_center = targets.mean(0) # (3,)
+ target_scale, target_idx = torch.norm(targets - target_center, dim=-1).max(0)
+
+ # a 45 degree vec from longest axis
+ if use_long_axis:
+ long_vec = targets[target_idx] - target_center # (x, 0, z)
+ long_vec = long_vec / torch.norm(long_vec)
+ R = axis_angle_to_matrix(torch.tensor([0, np.pi / 4, 0])).to(long_vec)
+ vec = R @ long_vec
+ else:
+ vec_rad = vec_rot / 180 * np.pi
+ vec = torch.tensor([np.sin(vec_rad), 0, np.cos(vec_rad)]).float()
+ vec = vec / torch.norm(vec)
+
+ # Compute camera position (center + scale * vec * beta) + y=4
+ target_scale = max(target_scale, 1.0) * beta
+ position = target_center + vec * target_scale
+ position[1] = target_scale * np.tan(np.pi * cam_height_degree / 180) + target_center_height
+
+ # Compute camera rotation and translation
+ positions = position.unsqueeze(0).repeat(L, 1)
+ target_centers = target_center.unsqueeze(0).repeat(L, 1)
+ target_centers[:, 1] = target_center_height
+ rotation = look_at_rotation(positions, target_centers).mT
+ translation = -(rotation @ positions.unsqueeze(-1)).squeeze(-1)
+
+ lights = PointLights(device=device, location=[position.tolist()])
+ return rotation, translation, lights
+
+
+def get_ground_params_from_points(root_points, vert_points):
+ """xz-plane is the ground plane
+ Args:
+ root_points: (L, 3), to decide center
+ vert_points: (L, V, 3), to decide scale
+ """
+ root_max = root_points.max(0)[0] # (3,)
+ root_min = root_points.min(0)[0] # (3,)
+ cx, _, cz = (root_max + root_min) / 2.0
+
+ vert_max = vert_points.reshape(-1, 3).max(0)[0] # (L, 3)
+ vert_min = vert_points.reshape(-1, 3).min(0)[0] # (L, 3)
+ scale = (vert_max - vert_min)[[0, 2]].max()
+ return float(scale), float(cx), float(cz)
diff --git a/third_party/GVHMR/hmr4d/utils/vis/renderer_tools.py b/third_party/GVHMR/hmr4d/utils/vis/renderer_tools.py
new file mode 100644
index 0000000000000000000000000000000000000000..68107fc8299938d9e46970601a2a33bd4f2c7fd3
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/vis/renderer_tools.py
@@ -0,0 +1,804 @@
+import os
+import cv2
+import numpy as np
+import torch
+from PIL import Image
+
+
+def read_image(path, scale=1):
+ im = Image.open(path)
+ if scale == 1:
+ return np.array(im)
+ W, H = im.size
+ w, h = int(scale * W), int(scale * H)
+ return np.array(im.resize((w, h), Image.ANTIALIAS))
+
+
+def transform_torch3d(T_c2w):
+ """
+ :param T_c2w (*, 4, 4)
+ returns (*, 3, 3), (*, 3)
+ """
+ R1 = torch.tensor(
+ [
+ [-1.0, 0.0, 0.0],
+ [0.0, -1.0, 0.0],
+ [0.0, 0.0, 1.0],
+ ],
+ device=T_c2w.device,
+ )
+ R2 = torch.tensor(
+ [
+ [1.0, 0.0, 0.0],
+ [0.0, -1.0, 0.0],
+ [0.0, 0.0, -1.0],
+ ],
+ device=T_c2w.device,
+ )
+ cam_R, cam_t = T_c2w[..., :3, :3], T_c2w[..., :3, 3]
+ cam_R = torch.einsum("...ij,jk->...ik", cam_R, R1)
+ cam_t = torch.einsum("ij,...j->...i", R2, cam_t)
+ return cam_R, cam_t
+
+
+def transform_pyrender(T_c2w):
+ """
+ :param T_c2w (*, 4, 4)
+ """
+ T_vis = torch.tensor(
+ [
+ [1.0, 0.0, 0.0, 0.0],
+ [0.0, -1.0, 0.0, 0.0],
+ [0.0, 0.0, -1.0, 0.0],
+ [0.0, 0.0, 0.0, 1.0],
+ ],
+ device=T_c2w.device,
+ )
+ return torch.einsum("...ij,jk->...ik", torch.einsum("ij,...jk->...ik", T_vis, T_c2w), T_vis)
+
+
+def smpl_to_geometry(verts, faces, vis_mask=None, track_ids=None):
+ """
+ :param verts (B, T, V, 3)
+ :param faces (F, 3)
+ :param vis_mask (optional) (B, T) visibility of each person
+ :param track_ids (optional) (B,)
+ returns list of T verts (B, V, 3), faces (F, 3), colors (B, 3)
+ where B is different depending on the visibility of the people
+ """
+ B, T = verts.shape[:2]
+ device = verts.device
+
+ # (B, 3)
+ colors = track_to_colors(track_ids) if track_ids is not None else torch.ones(B, 3, device) * 0.5
+
+ # list T (B, V, 3), T (B, 3), T (F, 3)
+ return filter_visible_meshes(verts, colors, faces, vis_mask)
+
+
+def filter_visible_meshes(verts, colors, faces, vis_mask=None, vis_opacity=False):
+ """
+ :param verts (B, T, V, 3)
+ :param colors (B, 3)
+ :param faces (F, 3)
+ :param vis_mask (optional tensor, default None) (B, T) ternary mask
+ -1 if not in frame
+ 0 if temporarily occluded
+ 1 if visible
+ :param vis_opacity (optional bool, default False)
+ if True, make occluded people alpha=0.5, otherwise alpha=1
+ returns a list of T lists verts (Bi, V, 3), colors (Bi, 4), faces (F, 3)
+ """
+ # import ipdb; ipdb.set_trace()
+ B, T = verts.shape[:2]
+ faces = [faces for t in range(T)]
+ if vis_mask is None:
+ verts = [verts[:, t] for t in range(T)]
+ colors = [colors for t in range(T)]
+ return verts, colors, faces
+
+ # render occluded and visible, but not removed
+ vis_mask = vis_mask >= 0
+ if vis_opacity:
+ alpha = 0.5 * (vis_mask[..., None] + 1)
+ else:
+ alpha = (vis_mask[..., None] >= 0).float()
+ vert_list = [verts[vis_mask[:, t], t] for t in range(T)]
+ colors = [torch.cat([colors[vis_mask[:, t]], alpha[vis_mask[:, t], t]], dim=-1) for t in range(T)]
+ bounds = get_bboxes(verts, vis_mask)
+ return vert_list, colors, faces, bounds
+
+
+def get_bboxes(verts, vis_mask):
+ """
+ return bb_min, bb_max, and mean for each track (B, 3) over entire trajectory
+ :param verts (B, T, V, 3)
+ :param vis_mask (B, T)
+ """
+ B, T, *_ = verts.shape
+ bb_min, bb_max, mean = [], [], []
+ for b in range(B):
+ v = verts[b, vis_mask[b, :T]] # (Tb, V, 3)
+ bb_min.append(v.amin(dim=(0, 1)))
+ bb_max.append(v.amax(dim=(0, 1)))
+ mean.append(v.mean(dim=(0, 1)))
+ bb_min = torch.stack(bb_min, dim=0)
+ bb_max = torch.stack(bb_max, dim=0)
+ mean = torch.stack(mean, dim=0)
+ # point to a track that's long and close to the camera
+ zs = mean[:, 2]
+ counts = vis_mask[:, :T].sum(dim=-1) # (B,)
+ mask = counts < 0.8 * T
+ zs[mask] = torch.inf
+ sel = torch.argmin(zs)
+ return bb_min.amin(dim=0), bb_max.amax(dim=0), mean[sel]
+
+
+def track_to_colors(track_ids):
+ """
+ :param track_ids (B)
+ """
+ color_map = torch.from_numpy(get_colors()).to(track_ids)
+ return color_map[track_ids] / 255 # (B, 3)
+
+
+def get_colors():
+ # color_file = os.path.abspath(os.path.join(__file__, "../colors_phalp.txt"))
+ color_file = os.path.abspath(os.path.join(__file__, "../colors.txt"))
+ RGB_tuples = np.vstack(
+ [
+ np.loadtxt(color_file, skiprows=0),
+ # np.loadtxt(color_file, skiprows=1),
+ np.random.uniform(0, 255, size=(10000, 3)),
+ [[0, 0, 0]],
+ ]
+ )
+ b = np.where(RGB_tuples == 0)
+ RGB_tuples[b] = 1
+ return RGB_tuples.astype(np.float32)
+
+
+def checkerboard_geometry(
+ length=12.0,
+ color0=[0.8, 0.9, 0.9],
+ color1=[0.6, 0.7, 0.7],
+ tile_width=0.5,
+ alpha=1.0,
+ up="y",
+ c1=0.0,
+ c2=0.0,
+):
+ assert up == "y" or up == "z"
+ color0 = np.array(color0 + [alpha])
+ color1 = np.array(color1 + [alpha])
+ num_rows = num_cols = max(2, int(length / tile_width))
+ radius = float(num_rows * tile_width) / 2.0
+ vertices = []
+ vert_colors = []
+ faces = []
+ face_colors = []
+ for i in range(num_rows):
+ for j in range(num_cols):
+ u0, v0 = j * tile_width - radius, i * tile_width - radius
+ us = np.array([u0, u0, u0 + tile_width, u0 + tile_width])
+ vs = np.array([v0, v0 + tile_width, v0 + tile_width, v0])
+ zs = np.zeros(4)
+ if up == "y":
+ cur_verts = np.stack([us, zs, vs], axis=-1) # (4, 3)
+ cur_verts[:, 0] += c1
+ cur_verts[:, 2] += c2
+ else:
+ cur_verts = np.stack([us, vs, zs], axis=-1) # (4, 3)
+ cur_verts[:, 0] += c1
+ cur_verts[:, 1] += c2
+
+ cur_faces = np.array([[0, 1, 3], [1, 2, 3], [0, 3, 1], [1, 3, 2]], dtype=np.int64)
+ cur_faces += 4 * (i * num_cols + j) # the number of previously added verts
+ use_color0 = (i % 2 == 0 and j % 2 == 0) or (i % 2 == 1 and j % 2 == 1)
+ cur_color = color0 if use_color0 else color1
+ cur_colors = np.array([cur_color, cur_color, cur_color, cur_color])
+
+ vertices.append(cur_verts)
+ faces.append(cur_faces)
+ vert_colors.append(cur_colors)
+ face_colors.append(cur_colors)
+
+ vertices = np.concatenate(vertices, axis=0).astype(np.float32)
+ vert_colors = np.concatenate(vert_colors, axis=0).astype(np.float32)
+ faces = np.concatenate(faces, axis=0).astype(np.float32)
+ face_colors = np.concatenate(face_colors, axis=0).astype(np.float32)
+
+ return vertices, faces, vert_colors, face_colors
+
+
+def camera_marker_geometry(radius, height, up):
+ assert up == "y" or up == "z"
+ if up == "y":
+ vertices = np.array(
+ [
+ [-radius, -radius, 0],
+ [radius, -radius, 0],
+ [radius, radius, 0],
+ [-radius, radius, 0],
+ [0, 0, height],
+ ]
+ )
+ else:
+ vertices = np.array(
+ [
+ [-radius, 0, -radius],
+ [radius, 0, -radius],
+ [radius, 0, radius],
+ [-radius, 0, radius],
+ [0, -height, 0],
+ ]
+ )
+
+ faces = np.array(
+ [
+ [0, 3, 1],
+ [1, 3, 2],
+ [0, 1, 4],
+ [1, 2, 4],
+ [2, 3, 4],
+ [3, 0, 4],
+ ]
+ )
+
+ face_colors = np.array(
+ [
+ [1.0, 1.0, 1.0, 1.0],
+ [1.0, 1.0, 1.0, 1.0],
+ [0.0, 1.0, 0.0, 1.0],
+ [1.0, 0.0, 0.0, 1.0],
+ [0.0, 1.0, 0.0, 1.0],
+ [1.0, 0.0, 0.0, 1.0],
+ ]
+ )
+ return vertices, faces, face_colors
+
+
+def vis_keypoints(
+ keypts_list,
+ img_size,
+ radius=6,
+ thickness=3,
+ kpt_score_thr=0.3,
+ dataset="TopDownCocoDataset",
+):
+ """
+ Visualize keypoints
+ From ViTPose/mmpose/apis/inference.py
+ """
+ palette = np.array(
+ [
+ [255, 128, 0],
+ [255, 153, 51],
+ [255, 178, 102],
+ [230, 230, 0],
+ [255, 153, 255],
+ [153, 204, 255],
+ [255, 102, 255],
+ [255, 51, 255],
+ [102, 178, 255],
+ [51, 153, 255],
+ [255, 153, 153],
+ [255, 102, 102],
+ [255, 51, 51],
+ [153, 255, 153],
+ [102, 255, 102],
+ [51, 255, 51],
+ [0, 255, 0],
+ [0, 0, 255],
+ [255, 0, 0],
+ [255, 255, 255],
+ ]
+ )
+
+ if dataset in (
+ "TopDownCocoDataset",
+ "BottomUpCocoDataset",
+ "TopDownOCHumanDataset",
+ "AnimalMacaqueDataset",
+ ):
+ # show the results
+ skeleton = [
+ [15, 13],
+ [13, 11],
+ [16, 14],
+ [14, 12],
+ [11, 12],
+ [5, 11],
+ [6, 12],
+ [5, 6],
+ [5, 7],
+ [6, 8],
+ [7, 9],
+ [8, 10],
+ [1, 2],
+ [0, 1],
+ [0, 2],
+ [1, 3],
+ [2, 4],
+ [3, 5],
+ [4, 6],
+ ]
+
+ pose_link_color = palette[[0, 0, 0, 0, 7, 7, 7, 9, 9, 9, 9, 9, 16, 16, 16, 16, 16, 16, 16]]
+ pose_kpt_color = palette[[16, 16, 16, 16, 16, 9, 9, 9, 9, 9, 9, 0, 0, 0, 0, 0, 0]]
+
+ elif dataset == "TopDownCocoWholeBodyDataset":
+ # show the results
+ skeleton = [
+ [15, 13],
+ [13, 11],
+ [16, 14],
+ [14, 12],
+ [11, 12],
+ [5, 11],
+ [6, 12],
+ [5, 6],
+ [5, 7],
+ [6, 8],
+ [7, 9],
+ [8, 10],
+ [1, 2],
+ [0, 1],
+ [0, 2],
+ [1, 3],
+ [2, 4],
+ [3, 5],
+ [4, 6],
+ [15, 17],
+ [15, 18],
+ [15, 19],
+ [16, 20],
+ [16, 21],
+ [16, 22],
+ [91, 92],
+ [92, 93],
+ [93, 94],
+ [94, 95],
+ [91, 96],
+ [96, 97],
+ [97, 98],
+ [98, 99],
+ [91, 100],
+ [100, 101],
+ [101, 102],
+ [102, 103],
+ [91, 104],
+ [104, 105],
+ [105, 106],
+ [106, 107],
+ [91, 108],
+ [108, 109],
+ [109, 110],
+ [110, 111],
+ [112, 113],
+ [113, 114],
+ [114, 115],
+ [115, 116],
+ [112, 117],
+ [117, 118],
+ [118, 119],
+ [119, 120],
+ [112, 121],
+ [121, 122],
+ [122, 123],
+ [123, 124],
+ [112, 125],
+ [125, 126],
+ [126, 127],
+ [127, 128],
+ [112, 129],
+ [129, 130],
+ [130, 131],
+ [131, 132],
+ ]
+
+ pose_link_color = palette[
+ [0, 0, 0, 0, 7, 7, 7, 9, 9, 9, 9, 9, 16, 16, 16, 16, 16, 16, 16]
+ + [16, 16, 16, 16, 16, 16]
+ + [0, 0, 0, 0, 4, 4, 4, 4, 8, 8, 8, 8, 12, 12, 12, 12, 16, 16, 16, 16]
+ + [0, 0, 0, 0, 4, 4, 4, 4, 8, 8, 8, 8, 12, 12, 12, 12, 16, 16, 16, 16]
+ ]
+ pose_kpt_color = palette[
+ [16, 16, 16, 16, 16, 9, 9, 9, 9, 9, 9, 0, 0, 0, 0, 0, 0] + [0, 0, 0, 0, 0, 0] + [19] * (68 + 42)
+ ]
+
+ elif dataset == "TopDownAicDataset":
+ skeleton = [
+ [2, 1],
+ [1, 0],
+ [0, 13],
+ [13, 3],
+ [3, 4],
+ [4, 5],
+ [8, 7],
+ [7, 6],
+ [6, 9],
+ [9, 10],
+ [10, 11],
+ [12, 13],
+ [0, 6],
+ [3, 9],
+ ]
+
+ pose_link_color = palette[[9, 9, 9, 9, 9, 9, 16, 16, 16, 16, 16, 0, 7, 7]]
+ pose_kpt_color = palette[[9, 9, 9, 9, 9, 9, 16, 16, 16, 16, 16, 16, 0, 0]]
+
+ elif dataset == "TopDownMpiiDataset":
+ skeleton = [
+ [0, 1],
+ [1, 2],
+ [2, 6],
+ [6, 3],
+ [3, 4],
+ [4, 5],
+ [6, 7],
+ [7, 8],
+ [8, 9],
+ [8, 12],
+ [12, 11],
+ [11, 10],
+ [8, 13],
+ [13, 14],
+ [14, 15],
+ ]
+
+ pose_link_color = palette[[16, 16, 16, 16, 16, 16, 7, 7, 0, 9, 9, 9, 9, 9, 9]]
+ pose_kpt_color = palette[[16, 16, 16, 16, 16, 16, 7, 7, 0, 0, 9, 9, 9, 9, 9, 9]]
+
+ elif dataset == "TopDownMpiiTrbDataset":
+ skeleton = [
+ [12, 13],
+ [13, 0],
+ [13, 1],
+ [0, 2],
+ [1, 3],
+ [2, 4],
+ [3, 5],
+ [0, 6],
+ [1, 7],
+ [6, 7],
+ [6, 8],
+ [7, 9],
+ [8, 10],
+ [9, 11],
+ [14, 15],
+ [16, 17],
+ [18, 19],
+ [20, 21],
+ [22, 23],
+ [24, 25],
+ [26, 27],
+ [28, 29],
+ [30, 31],
+ [32, 33],
+ [34, 35],
+ [36, 37],
+ [38, 39],
+ ]
+
+ pose_link_color = palette[[16] * 14 + [19] * 13]
+ pose_kpt_color = palette[[16] * 14 + [0] * 26]
+
+ elif dataset in ("OneHand10KDataset", "FreiHandDataset", "PanopticDataset"):
+ skeleton = [
+ [0, 1],
+ [1, 2],
+ [2, 3],
+ [3, 4],
+ [0, 5],
+ [5, 6],
+ [6, 7],
+ [7, 8],
+ [0, 9],
+ [9, 10],
+ [10, 11],
+ [11, 12],
+ [0, 13],
+ [13, 14],
+ [14, 15],
+ [15, 16],
+ [0, 17],
+ [17, 18],
+ [18, 19],
+ [19, 20],
+ ]
+
+ pose_link_color = palette[[0, 0, 0, 0, 4, 4, 4, 4, 8, 8, 8, 8, 12, 12, 12, 12, 16, 16, 16, 16]]
+ pose_kpt_color = palette[[0, 0, 0, 0, 0, 4, 4, 4, 4, 8, 8, 8, 8, 12, 12, 12, 12, 16, 16, 16, 16]]
+
+ elif dataset == "InterHand2DDataset":
+ skeleton = [
+ [0, 1],
+ [1, 2],
+ [2, 3],
+ [4, 5],
+ [5, 6],
+ [6, 7],
+ [8, 9],
+ [9, 10],
+ [10, 11],
+ [12, 13],
+ [13, 14],
+ [14, 15],
+ [16, 17],
+ [17, 18],
+ [18, 19],
+ [3, 20],
+ [7, 20],
+ [11, 20],
+ [15, 20],
+ [19, 20],
+ ]
+
+ pose_link_color = palette[[0, 0, 0, 4, 4, 4, 8, 8, 8, 12, 12, 12, 16, 16, 16, 0, 4, 8, 12, 16]]
+ pose_kpt_color = palette[[0, 0, 0, 0, 4, 4, 4, 4, 8, 8, 8, 8, 12, 12, 12, 12, 16, 16, 16, 16, 0]]
+
+ elif dataset == "Face300WDataset":
+ # show the results
+ skeleton = []
+
+ pose_link_color = palette[[]]
+ pose_kpt_color = palette[[19] * 68]
+ kpt_score_thr = 0
+
+ elif dataset == "FaceAFLWDataset":
+ # show the results
+ skeleton = []
+
+ pose_link_color = palette[[]]
+ pose_kpt_color = palette[[19] * 19]
+ kpt_score_thr = 0
+
+ elif dataset == "FaceCOFWDataset":
+ # show the results
+ skeleton = []
+
+ pose_link_color = palette[[]]
+ pose_kpt_color = palette[[19] * 29]
+ kpt_score_thr = 0
+
+ elif dataset == "FaceWFLWDataset":
+ # show the results
+ skeleton = []
+
+ pose_link_color = palette[[]]
+ pose_kpt_color = palette[[19] * 98]
+ kpt_score_thr = 0
+
+ elif dataset == "AnimalHorse10Dataset":
+ skeleton = [
+ [0, 1],
+ [1, 12],
+ [12, 16],
+ [16, 21],
+ [21, 17],
+ [17, 11],
+ [11, 10],
+ [10, 8],
+ [8, 9],
+ [9, 12],
+ [2, 3],
+ [3, 4],
+ [5, 6],
+ [6, 7],
+ [13, 14],
+ [14, 15],
+ [18, 19],
+ [19, 20],
+ ]
+
+ pose_link_color = palette[[4] * 10 + [6] * 2 + [6] * 2 + [7] * 2 + [7] * 2]
+ pose_kpt_color = palette[[4, 4, 6, 6, 6, 6, 6, 6, 4, 4, 4, 4, 4, 7, 7, 7, 4, 4, 7, 7, 7, 4]]
+
+ elif dataset == "AnimalFlyDataset":
+ skeleton = [
+ [1, 0],
+ [2, 0],
+ [3, 0],
+ [4, 3],
+ [5, 4],
+ [7, 6],
+ [8, 7],
+ [9, 8],
+ [11, 10],
+ [12, 11],
+ [13, 12],
+ [15, 14],
+ [16, 15],
+ [17, 16],
+ [19, 18],
+ [20, 19],
+ [21, 20],
+ [23, 22],
+ [24, 23],
+ [25, 24],
+ [27, 26],
+ [28, 27],
+ [29, 28],
+ [30, 3],
+ [31, 3],
+ ]
+
+ pose_link_color = palette[[0] * 25]
+ pose_kpt_color = palette[[0] * 32]
+
+ elif dataset == "AnimalLocustDataset":
+ skeleton = [
+ [1, 0],
+ [2, 1],
+ [3, 2],
+ [4, 3],
+ [6, 5],
+ [7, 6],
+ [9, 8],
+ [10, 9],
+ [11, 10],
+ [13, 12],
+ [14, 13],
+ [15, 14],
+ [17, 16],
+ [18, 17],
+ [19, 18],
+ [21, 20],
+ [22, 21],
+ [24, 23],
+ [25, 24],
+ [26, 25],
+ [28, 27],
+ [29, 28],
+ [30, 29],
+ [32, 31],
+ [33, 32],
+ [34, 33],
+ ]
+
+ pose_link_color = palette[[0] * 26]
+ pose_kpt_color = palette[[0] * 35]
+
+ elif dataset == "AnimalZebraDataset":
+ skeleton = [[1, 0], [2, 1], [3, 2], [4, 2], [5, 7], [6, 7], [7, 2], [8, 7]]
+
+ pose_link_color = palette[[0] * 8]
+ pose_kpt_color = palette[[0] * 9]
+
+ elif dataset in "AnimalPoseDataset":
+ skeleton = [
+ [0, 1],
+ [0, 2],
+ [1, 3],
+ [0, 4],
+ [1, 4],
+ [4, 5],
+ [5, 7],
+ [6, 7],
+ [5, 8],
+ [8, 12],
+ [12, 16],
+ [5, 9],
+ [9, 13],
+ [13, 17],
+ [6, 10],
+ [10, 14],
+ [14, 18],
+ [6, 11],
+ [11, 15],
+ [15, 19],
+ ]
+
+ pose_link_color = palette[[0] * 20]
+ pose_kpt_color = palette[[0] * 20]
+ else:
+ NotImplementedError()
+
+ img_w, img_h = img_size
+ img = 255 * np.ones((img_h, img_w, 3), dtype=np.uint8)
+ img = imshow_keypoints(
+ img,
+ keypts_list,
+ skeleton,
+ kpt_score_thr,
+ pose_kpt_color,
+ pose_link_color,
+ radius,
+ thickness,
+ )
+ alpha = 255 * (img != 255).any(axis=-1, keepdims=True).astype(np.uint8)
+ return np.concatenate([img, alpha], axis=-1)
+
+
+def imshow_keypoints(
+ img,
+ pose_result,
+ skeleton=None,
+ kpt_score_thr=0.3,
+ pose_kpt_color=None,
+ pose_link_color=None,
+ radius=4,
+ thickness=1,
+ show_keypoint_weight=False,
+):
+ """Draw keypoints and links on an image.
+ From ViTPose/mmpose/core/visualization/image.py
+
+ Args:
+ img (H, W, 3) array
+ pose_result (list[kpts]): The poses to draw. Each element kpts is
+ a set of K keypoints as an Kx3 numpy.ndarray, where each
+ keypoint is represented as x, y, score.
+ kpt_score_thr (float, optional): Minimum score of keypoints
+ to be shown. Default: 0.3.
+ pose_kpt_color (np.array[Nx3]`): Color of N keypoints. If None,
+ the keypoint will not be drawn.
+ pose_link_color (np.array[Mx3]): Color of M links. If None, the
+ links will not be drawn.
+ thickness (int): Thickness of lines.
+ show_keypoint_weight (bool): If True, opacity indicates keypoint score
+ """
+ img_h, img_w, _ = img.shape
+ idcs = [0, 16, 15, 18, 17, 5, 2, 6, 3, 7, 4, 12, 9, 13, 10, 14, 11]
+ for kpts in pose_result:
+ kpts = np.array(kpts, copy=False)[idcs]
+
+ # draw each point on image
+ if pose_kpt_color is not None:
+ assert len(pose_kpt_color) == len(kpts)
+ for kid, kpt in enumerate(kpts):
+ x_coord, y_coord, kpt_score = int(kpt[0]), int(kpt[1]), kpt[2]
+ if kpt_score > kpt_score_thr:
+ color = tuple(int(c) for c in pose_kpt_color[kid])
+ if show_keypoint_weight:
+ img_copy = img.copy()
+ cv2.circle(img_copy, (int(x_coord), int(y_coord)), radius, color, -1)
+ transparency = max(0, min(1, kpt_score))
+ cv2.addWeighted(img_copy, transparency, img, 1 - transparency, 0, dst=img)
+ else:
+ cv2.circle(img, (int(x_coord), int(y_coord)), radius, color, -1)
+
+ # draw links
+ if skeleton is not None and pose_link_color is not None:
+ assert len(pose_link_color) == len(skeleton)
+ for sk_id, sk in enumerate(skeleton):
+ pos1 = (int(kpts[sk[0], 0]), int(kpts[sk[0], 1]))
+ pos2 = (int(kpts[sk[1], 0]), int(kpts[sk[1], 1]))
+ if (
+ pos1[0] > 0
+ and pos1[0] < img_w
+ and pos1[1] > 0
+ and pos1[1] < img_h
+ and pos2[0] > 0
+ and pos2[0] < img_w
+ and pos2[1] > 0
+ and pos2[1] < img_h
+ and kpts[sk[0], 2] > kpt_score_thr
+ and kpts[sk[1], 2] > kpt_score_thr
+ ):
+ color = tuple(int(c) for c in pose_link_color[sk_id])
+ if show_keypoint_weight:
+ img_copy = img.copy()
+ X = (pos1[0], pos2[0])
+ Y = (pos1[1], pos2[1])
+ mX = np.mean(X)
+ mY = np.mean(Y)
+ length = ((Y[0] - Y[1]) ** 2 + (X[0] - X[1]) ** 2) ** 0.5
+ angle = math.degrees(math.atan2(Y[0] - Y[1], X[0] - X[1]))
+ stickwidth = 2
+ polygon = cv2.ellipse2Poly(
+ (int(mX), int(mY)),
+ (int(length / 2), int(stickwidth)),
+ int(angle),
+ 0,
+ 360,
+ 1,
+ )
+ cv2.fillConvexPoly(img_copy, polygon, color)
+ transparency = max(0, min(1, 0.5 * (kpts[sk[0], 2] + kpts[sk[1], 2])))
+ cv2.addWeighted(img_copy, transparency, img, 1 - transparency, 0, dst=img)
+ else:
+ cv2.line(img, pos1, pos2, color, thickness=thickness)
+
+ return img
diff --git a/third_party/GVHMR/hmr4d/utils/vis/renderer_utils.py b/third_party/GVHMR/hmr4d/utils/vis/renderer_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..634c84bd2d7e6b88df9427a99f11b53589ed4238
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/vis/renderer_utils.py
@@ -0,0 +1,38 @@
+from hmr4d.utils.vis.renderer import Renderer
+from tqdm import tqdm
+import numpy as np
+
+
+def simple_render_mesh(render_dict):
+ """Render an camera-space mesh, blank background"""
+ width, height, focal_length = render_dict["whf"]
+ faces = render_dict["faces"]
+ verts = render_dict["verts"]
+
+ renderer = Renderer(width, height, focal_length, device="cuda", faces=faces)
+ outputs = []
+ for i in tqdm(range(len(verts)), desc=f"Rendering"):
+ img = renderer.render_mesh(verts[i].cuda(), colors=[0.8, 0.8, 0.8])
+ outputs.append(img)
+ outputs = np.stack(outputs, axis=0)
+ return outputs
+
+
+def simple_render_mesh_background(render_dict, VI=50, colors=[0.8, 0.8, 0.8]):
+ """Render an camera-space mesh, blank background"""
+ K = render_dict["K"]
+ faces = render_dict["faces"]
+ verts = render_dict["verts"]
+ background = render_dict["background"]
+ N_frames = len(verts)
+ if len(background.shape) == 3:
+ background = [background] * N_frames
+ height, width = background[0].shape[:2]
+
+ renderer = Renderer(width, height, device="cuda", faces=faces, K=K)
+ outputs = []
+ for i in tqdm(range(len(verts)), desc=f"Rendering"):
+ img = renderer.render_mesh(verts[i].cuda(), colors=colors, background=background[i], VI=VI)
+ outputs.append(img)
+ outputs = np.stack(outputs, axis=0)
+ return outputs
diff --git a/third_party/GVHMR/hmr4d/utils/vis/rich_logger.py b/third_party/GVHMR/hmr4d/utils/vis/rich_logger.py
new file mode 100644
index 0000000000000000000000000000000000000000..80b551319b13282c0ca573a18aad78a4902b4d15
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/vis/rich_logger.py
@@ -0,0 +1,36 @@
+from pytorch_lightning.utilities import rank_zero_only
+from omegaconf import DictConfig, OmegaConf
+import rich
+import rich.tree
+import rich.syntax
+from hmr4d.utils.pylogger import Log
+
+
+@rank_zero_only
+def print_cfg(cfg: DictConfig, use_rich: bool = False):
+ if use_rich:
+ print_order = ("data", "model", "callbacks", "logger", "pl_trainer")
+ style = "dim"
+ tree = rich.tree.Tree("CONFIG", style=style, guide_style=style)
+
+ # add fields from `print_order` to queue
+ # add all the other fields to queue (not specified in `print_order`)
+ queue = []
+ for field in print_order:
+ queue.append(field) if field in cfg else Log.warn(f"Field '{field}' not found in config. Skipping.")
+ for field in cfg:
+ if field not in queue:
+ queue.append(field)
+
+ # generate config tree from queue
+ for field in queue:
+ branch = tree.add(field, style=style, guide_style=style)
+ config_group = cfg[field]
+ if isinstance(config_group, DictConfig):
+ branch_content = OmegaConf.to_yaml(config_group, resolve=False)
+ else:
+ branch_content = str(config_group)
+ branch.add(rich.syntax.Syntax(branch_content, "yaml"))
+ rich.print(tree)
+ else:
+ Log.info(OmegaConf.to_yaml(cfg, resolve=False))
diff --git a/third_party/GVHMR/hmr4d/utils/wis3d_utils.py b/third_party/GVHMR/hmr4d/utils/wis3d_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..df54d8b1854a389081a0817a60c26b1ab780ae34
--- /dev/null
+++ b/third_party/GVHMR/hmr4d/utils/wis3d_utils.py
@@ -0,0 +1,403 @@
+from wis3d import Wis3D
+from pathlib import Path
+from datetime import datetime
+import torch
+import numpy as np
+from einops import einsum
+from pytorch3d.transforms import axis_angle_to_matrix
+
+
+def make_wis3d(output_dir="outputs/wis3d", name="debug", time_postfix=False):
+ """
+ Make a Wis3D instance. e.g.:
+ from hmr4d.utils.wis3d_utils import make_wis3d
+ wis3d = make_wis3d(time_postfix=True)
+ """
+ output_dir = Path(output_dir)
+ output_dir.mkdir(parents=True, exist_ok=True)
+ if time_postfix:
+ time_str = datetime.now().strftime("%m%d-%H%M-%S")
+ name = f"{name}_{time_str}"
+ print(f"Creating Wis3D {name}")
+ wis3d = Wis3D(output_dir.absolute(), name)
+ return wis3d
+
+
+color_schemes = {
+ "red": ([255, 168, 154], [153, 17, 1]),
+ "green": ([183, 255, 191], [0, 171, 8]),
+ "blue": ([183, 255, 255], [0, 0, 255]),
+ "cyan": ([183, 255, 255], [0, 255, 255]),
+ "magenta": ([255, 183, 255], [255, 0, 255]),
+ "black": ([0, 0, 0], [0, 0, 0]),
+ "orange": ([255, 183, 0], [255, 128, 0]),
+ "grey": ([203, 203, 203], [203, 203, 203]),
+}
+
+
+def get_gradient_colors(scheme="red", num_points=120, alpha=1.0):
+ """
+ Return a list of colors that are gradient from start to end.
+ """
+ start_rgba = torch.tensor(color_schemes[scheme][0] + [255 * alpha]) / 255
+ end_rgba = torch.tensor(color_schemes[scheme][1] + [255 * alpha]) / 255
+ colors = torch.stack([torch.linspace(s, e, steps=num_points) for s, e in zip(start_rgba, end_rgba)], dim=-1)
+ return colors
+
+
+def get_const_colors(name="red", partial_shape=(120, 5), alpha=1.0):
+ """
+ Return colors (partial_shape, 4)
+ """
+ rgba = torch.tensor(color_schemes[name][1] + [255 * alpha]) / 255
+ partial_shape = tuple(partial_shape)
+ colors = rgba[None].repeat(*partial_shape, 1)
+ return colors
+
+
+def get_colors_by_conf(conf, low="red", high="green"):
+ colors = torch.stack([conf] * 3, dim=-1)
+ colors = colors * torch.tensor(color_schemes[high][1]) + (1 - colors) * torch.tensor(color_schemes[low][1])
+ return colors
+
+
+# ================== Colored Motion Sequence ================== #
+
+
+KINEMATIC_CHAINS = {
+ "smpl22": [
+ [0, 2, 5, 8, 11], # right-leg
+ [0, 1, 4, 7, 10], # left-leg
+ [0, 3, 6, 9, 12, 15], # spine
+ [9, 14, 17, 19, 21], # right-arm
+ [9, 13, 16, 18, 20], # left-arm
+ ],
+ "h36m17": [
+ [0, 1, 2, 3], # right-leg
+ [0, 4, 5, 6], # left-leg
+ [0, 7, 8, 9, 10], # spine
+ [8, 14, 15, 16], # right-arm
+ [8, 11, 12, 13], # left-arm
+ ],
+ "coco17": [
+ [12, 14, 16], # right-leg
+ [11, 13, 15], # left-leg
+ [4, 2, 0, 1, 3], # replace spine with head
+ [6, 8, 10], # right-arm
+ [5, 7, 9], # left-arm
+ ],
+}
+
+
+def convert_motion_as_line_mesh(motion, skeleton_type="smpl22", const_color=None):
+ if isinstance(motion, np.ndarray):
+ motion = torch.from_numpy(motion)
+ motion = motion.detach().cpu()
+ kinematic_chain = KINEMATIC_CHAINS[skeleton_type]
+ color_names = ["red", "green", "blue", "cyan", "magenta"]
+ s_points = []
+ e_points = []
+ m_colors = []
+ length = motion.shape[0]
+ device = motion.device
+ for chain, color_name in zip(kinematic_chain, color_names):
+ num_line = len(chain) - 1
+ s_points.append(motion[:, chain[:-1]])
+ e_points.append(motion[:, chain[1:]])
+ if const_color is not None:
+ color_name = const_color
+ color_ = get_const_colors(color_name, partial_shape=(length, num_line), alpha=1.0).to(device) # (L, 4, 4)
+ m_colors.append(color_[..., :3] * 255) # (L, 4, 3)
+
+ s_points = torch.cat(s_points, dim=1) # (L, ?, 3)
+ e_points = torch.cat(e_points, dim=1)
+ m_colors = torch.cat(m_colors, dim=1)
+
+ vertices = []
+ for f in range(length):
+ vertices_, faces, vertex_colors = create_skeleton_mesh(s_points[f], e_points[f], radius=0.02, color=m_colors[f])
+ vertices.append(vertices_)
+ vertices = torch.stack(vertices, dim=0)
+ return vertices, faces, vertex_colors
+
+
+def add_motion_as_lines(motion, wis3d, name="joints22", skeleton_type="smpl22", const_color=None, offset=0):
+ """
+ Args:
+ motion (tensor): (L, J, 3)
+ """
+ vertices, faces, vertex_colors = convert_motion_as_line_mesh(
+ motion, skeleton_type=skeleton_type, const_color=const_color
+ )
+ for f in range(len(vertices)):
+ wis3d.set_scene_id(f + offset)
+ wis3d.add_mesh(vertices[f], faces, vertex_colors, name=name) # Add skeleton as cylinders
+ # Old way to add lines, this may cause problems when the number of lines is large
+ # wis3d.add_lines(s_points[f], e_points[f], m_colors[f], name=name)
+
+
+def add_prog_motion_as_lines(motion, wis3d, name="joints22", skeleton_type="smpl22"):
+ """
+ Args:
+ motion (tensor): (P, L, J, 3)
+ """
+ if isinstance(motion, np.ndarray):
+ motion = torch.from_numpy(motion)
+ P, L, J, _ = motion.shape
+ device = motion.device
+
+ kinematic_chain = KINEMATIC_CHAINS[skeleton_type]
+ color_names = ["red", "green", "blue", "cyan", "magenta"]
+ s_points = []
+ e_points = []
+ m_colors = []
+ for chain, color_name in zip(kinematic_chain, color_names):
+ num_line = len(chain) - 1
+ s_points.append(motion[:, :, chain[:-1]])
+ e_points.append(motion[:, :, chain[1:]])
+ color_ = get_gradient_colors(color_name, L, alpha=1.0).to(device) # (L, 4)
+ color_ = color_[None, :, None, :].repeat(P, 1, num_line, 1) # (P, L, num_line, 4)
+ m_colors.append(color_[..., :3] * 255) # (P, L, num_line, 3)
+ s_points = torch.cat(s_points, dim=-2) # (L, ?, 3)
+ e_points = torch.cat(e_points, dim=-2)
+ m_colors = torch.cat(m_colors, dim=-2)
+
+ s_points = s_points.reshape(P, -1, 3)
+ e_points = e_points.reshape(P, -1, 3)
+ m_colors = m_colors.reshape(P, -1, 3)
+
+ for p in range(P):
+ wis3d.set_scene_id(p)
+ wis3d.add_lines(s_points[p], e_points[p], m_colors[p], name=name)
+
+
+def add_joints_motion_as_spheres(joints, wis3d, radius=0.05, name="joints", label_each_joint=False):
+ """Visualize skeleton as spheres to explore the skeleton.
+ Args:
+ joints: (NF, NJ, 3)
+ wis3d
+ radius: radius of the spheres
+ name
+ label_each_joint: if True, each joints will have a label in wis3d (then you can interact with it, but it's slower)
+ """
+ colors = torch.zeros_like(joints).float()
+ n_frames = joints.shape[0]
+ n_joints = joints.shape[1]
+ for i in range(n_joints):
+ colors[:, i, 1] = 255 / n_joints * i
+ colors[:, i, 2] = 255 / n_joints * (n_joints - i)
+ for f in range(n_frames):
+ wis3d.set_scene_id(f)
+ if label_each_joint:
+ for i in range(n_joints):
+ wis3d.add_spheres(
+ joints[f, i].float(),
+ radius=radius,
+ colors=colors[f, i],
+ name=f"{name}-j{i}",
+ )
+ else:
+ wis3d.add_spheres(
+ joints[f].float(),
+ radius=radius,
+ colors=colors[f],
+ name=f"{name}",
+ )
+
+
+def create_skeleton_mesh(p1, p2, radius, color, resolution=4, return_merged=True):
+ """
+ Create mesh between p1 and p2.
+ Args:
+ p1 (torch.Tensor): (N, 3),
+ p2 (torch.Tensor): (N, 3),
+ radius (float): radius,
+ color (torch.Tensor): (N, 3)
+ resolution (int): number of vertices in one circle, denoted as Q
+ Returns:
+ vertices (torch.Tensor): (N * 2Q, 3), if return_merged is False (N, 2Q, 3)
+ faces (torch.Tensor): (M', 3), if return_merged is False (N, M, 3)
+ vertex_colors (torch.Tensor): (N * 2Q, 3), if return_merged is False (N, 2Q, 3)
+ """
+ N = p1.shape[0]
+
+ # Calculate segment direction
+ seg_dir = p2 - p1 # (N, 3)
+ unit_seg_dir = seg_dir / seg_dir.norm(dim=-1, keepdim=True) # (N, 3)
+
+ # Compute an orthogonal vector
+ x_vec = torch.tensor([1, 0, 0], device=p1.device).float().unsqueeze(0).repeat(N, 1) # (N, 3)
+ y_vec = torch.tensor([0, 1, 0], device=p1.device).float().unsqueeze(0).repeat(N, 1)
+ ortho_vec = torch.cross(unit_seg_dir, x_vec, dim=-1) # (N, 3)
+ ortho_vec_ = torch.cross(unit_seg_dir, y_vec, dim=-1) # (N, 3) backup
+ ortho_vec = torch.where(ortho_vec.norm(dim=-1, keepdim=True) > 1e-3, ortho_vec, ortho_vec_)
+
+ # Get circle points on two ends
+ unit_ortho_vec = ortho_vec / ortho_vec.norm(dim=-1, keepdim=True) # (N, 3)
+ theta = torch.linspace(0, 2 * np.pi, resolution, device=p1.device)
+ rotation_matrix = axis_angle_to_matrix(unit_seg_dir[:, None] * theta[None, :, None]) # (N, Q, 3, 3)
+ rotated_points = einsum(rotation_matrix, unit_ortho_vec, "n q i j, n i -> n q j") * radius # (N, Q, 3)
+ bottom_points = rotated_points + p1.unsqueeze(1) # (N, Q, 3)
+ top_points = rotated_points + p2.unsqueeze(1) # (N, Q, 3)
+
+ # Combine bottom and top points
+ vertices = torch.cat([bottom_points, top_points], dim=1) # (N, 2Q, 3)
+
+ # Generate face
+ indices = torch.arange(0, resolution, device=p1.device)
+ bottom_indices = indices
+ top_indices = indices + resolution
+
+ # outside face
+ face_bottom = torch.stack([bottom_indices[:-2], bottom_indices[1:-1], bottom_indices[-1].repeat(resolution - 2)], 1)
+ face_top = torch.stack([top_indices[1:-1], top_indices[:-2], top_indices[-1].repeat(resolution - 2)], 1)
+ faces = torch.cat(
+ [
+ torch.stack([bottom_indices[1:], bottom_indices[:-1], top_indices[:-1]], 1), # out face
+ torch.stack([bottom_indices[1:], top_indices[:-1], top_indices[1:]], 1), # out face
+ face_bottom,
+ face_top,
+ ]
+ )
+ faces = faces.unsqueeze(0).repeat(p1.shape[0], 1, 1) # (N, M, 3)
+
+ # Assign colors
+ vertex_colors = color.unsqueeze(1).repeat(1, resolution * 2, 1)
+
+ if return_merged:
+ # manully adjust face ids
+ N, V = vertices.shape[:2]
+ faces = faces + torch.arange(0, N, device=p1.device).unsqueeze(1).unsqueeze(1) * V
+ faces = faces.reshape(-1, 3)
+ vertices = vertices.reshape(-1, 3)
+ vertex_colors = vertex_colors.reshape(-1, 3)
+
+ return vertices, faces, vertex_colors
+
+
+def get_lines_of_my_frustum(frustum_points):
+ """
+ frustum_points: (B, 8, 3), in (near {lu ru rd ld}, far {lu ru rd ld})
+ """
+ start_points = frustum_points[:, [0, 1, 2, 3, 0, 1, 2, 3, 4, 5, 6, 7]].cpu().numpy()
+ end_points = frustum_points[:, [4, 5, 6, 7, 1, 2, 3, 0, 5, 6, 7, 4]].cpu().numpy()
+ return start_points, end_points
+
+
+def draw_colored_vec(wis3d, vec, name, radius=0.02, colors="r", starts=None, l=1.0):
+ """
+ Args:
+ vec: (3) or (L, 3), should be the same length as colors, like 'rgb'
+ """
+ if len(vec.shape) == 1:
+ vec = vec[None]
+ else:
+ assert len(vec.shape) == 2
+
+ assert len(vec) == len(colors)
+ # split colors, 'rgb' to 'r', 'g', 'b'
+ color_tensor = torch.zeros((len(colors), 3))
+ c2rgb = {
+ "r": torch.tensor([1, 0, 0]).float(),
+ "g": torch.tensor([0, 1, 0]).float(),
+ "b": torch.tensor([0, 0, 1]).float(),
+ }
+ for i, c in enumerate(colors):
+ color_tensor[i] = c2rgb[c]
+
+ if starts is None:
+ starts = torch.zeros_like(vec)
+ ends = starts + vec * l
+ vertices, faces, vertex_colors = create_skeleton_mesh(starts, ends, radius, color_tensor, resolution=10)
+ wis3d.add_mesh(vertices, faces, vertex_colors, name=name)
+
+
+def draw_T_w2c(wis3d, T_w2c, name, radius=0.01, all_in_one=True, l=0.1):
+ """
+ Draw a camera trajectory in world coordinate.
+ Args:
+ T_w2c: (L, 4, 4)
+ """
+ color_tensor = torch.eye(3)
+ if all_in_one:
+ starts = -T_w2c[:, :3, :3].mT @ T_w2c[:, :3, [3]] # (L, 3, 1)
+ starts = starts[:, None, :, 0].expand(-1, 3, -1).reshape(-1, 3) # (L*3, 3)
+ vec = T_w2c[:, :3, :3].reshape(-1, 3) # (L * 3, 3)
+ ends = starts + vec * l
+ color_tensor = color_tensor[None].expand(T_w2c.size(0), -1, -1).reshape(-1, 3)
+
+ vertices, faces, vertex_colors = create_skeleton_mesh(starts, ends, radius, color_tensor, resolution=10)
+ else:
+ raise NotImplementedError
+ wis3d.add_mesh(vertices, faces, vertex_colors, name=name)
+
+
+def create_checkerboard_mesh(y=0.0, grid_size=1.0, bounds=((-3, -3), (3, 3))):
+ """
+ example usage:
+ vertices, faces, vertex_colors = create_checkerboard_mesh()
+ wis3d.add_mesh(vertices=vertices, faces=faces, vertex_colors=vertex_colors, name="one")
+ """
+ color1 = np.array([236, 240, 241], np.uint8) # light
+ color2 = np.array([120, 120, 120], np.uint8) # dark
+
+ # 扩大范围
+ min_x, min_z = bounds[0]
+ max_x, max_z = bounds[1]
+ min_x = grid_size * np.floor(min_x / grid_size)
+ min_z = grid_size * np.floor(min_z / grid_size)
+ max_x = grid_size * np.ceil(max_x / grid_size)
+ max_z = grid_size * np.ceil(max_z / grid_size)
+
+ vertices = []
+ faces = []
+ vertex_colors = []
+ eps = 1e-4 # HACK: disable smooth color & double-side color artifacts of wis3d
+
+ for i, x in enumerate(np.arange(min_x, max_x, grid_size)):
+ for j, z in enumerate(np.arange(min_z, max_z, grid_size)):
+
+ # Right-hand rule for normal direction
+ x += ((i % 2 * 2) - 1) * eps
+ z += ((j % 2 * 2) - 1) * eps
+ v1 = np.array([x, y, z])
+ v2 = np.array([x, y, z + grid_size])
+ v3 = np.array([x + grid_size, y, z + grid_size])
+ v4 = np.array([x + grid_size, y, z])
+ offset = np.array([0, -eps, 0]) # For visualizing the down-side of the mesh
+
+ vertices.extend([v1, v2, v3, v4, v1 + offset, v2 + offset, v3 + offset, v4 + offset])
+ idx = len(vertices) - 8
+ faces.extend(
+ [
+ [idx, idx + 1, idx + 2],
+ [idx + 2, idx + 3, idx],
+ [idx + 4, idx + 7, idx + 6], # double-sided
+ [idx + 6, idx + 5, idx + 4], # double-sided
+ ]
+ )
+ vertex_color = color1 if (i + j) % 2 == 0 else color2
+ vertex_colors.extend([vertex_color] * 8)
+
+ # To numpy.array and the shape should be (n, 3)
+ vertices = np.array(vertices)
+ faces = np.array(faces)
+ vertex_colors = np.array(vertex_colors)
+ assert len(vertices.shape) == 2 and vertices.shape[1] == 3
+ assert len(faces.shape) == 2 and faces.shape[1] == 3
+ assert len(vertex_colors.shape) == 2 and vertex_colors.shape[1] == 3 and vertex_colors.dtype == np.uint8
+
+ return vertices, faces, vertex_colors
+
+
+def add_a_trimesh(mesh, wis3d, name):
+ mesh.apply_transform(wis3d.three_to_world)
+
+ # filename = wis3d.__get_export_file_name("mesh", name)
+ export_dir = Path(wis3d.out_folder) / wis3d.sequence_name / f"{wis3d.scene_id:05d}" / "meshes"
+ export_dir.mkdir(parents=True, exist_ok=True)
+ assert name is not None
+ filename = export_dir / f"{name}.ply"
+ wis3d.counters["mesh"] += 1
+
+ mesh.export(filename)
diff --git a/third_party/GVHMR/inputs/checkpoints/body_models/smpl/SMPL_FEMALE.pkl b/third_party/GVHMR/inputs/checkpoints/body_models/smpl/SMPL_FEMALE.pkl
new file mode 100644
index 0000000000000000000000000000000000000000..c7aee61286b91ee83f1dc8846e7bab306882f30f
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/body_models/smpl/SMPL_FEMALE.pkl
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6d4a1791b6b94880397e1a3a4539b703a228d2150c57de7b288389a8115f4ef0
+size 247530000
diff --git a/third_party/GVHMR/inputs/checkpoints/body_models/smpl/SMPL_MALE.pkl b/third_party/GVHMR/inputs/checkpoints/body_models/smpl/SMPL_MALE.pkl
new file mode 100644
index 0000000000000000000000000000000000000000..247d55241f21c4190521321279b1dc6f94be02a3
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/body_models/smpl/SMPL_MALE.pkl
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ed4d55bb3041fefc6f73b70694d6c8edc1020c0d07340be5cc651cae2c6a6ae3
+size 247101031
diff --git a/third_party/GVHMR/inputs/checkpoints/body_models/smpl/SMPL_NEUTRAL.pkl b/third_party/GVHMR/inputs/checkpoints/body_models/smpl/SMPL_NEUTRAL.pkl
new file mode 100644
index 0000000000000000000000000000000000000000..65ae47d34e5b26720c9ccdd2614044832f0e30f2
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/body_models/smpl/SMPL_NEUTRAL.pkl
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4924f235e63f7c5d5b690acedf736419c2edb846a2d69fc0956169615fa75688
+size 247186228
diff --git a/third_party/GVHMR/inputs/checkpoints/body_models/smpl_3dpw14_J_regressor_sparse.pt b/third_party/GVHMR/inputs/checkpoints/body_models/smpl_3dpw14_J_regressor_sparse.pt
new file mode 100644
index 0000000000000000000000000000000000000000..8d8cda5d595ef01f036ff4eb4c5c455a71d462fa
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/body_models/smpl_3dpw14_J_regressor_sparse.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f3f4cf476e206d36ac806beab9b728b886c10f1534a74318839391f76ad5ab0a
+size 2791
diff --git a/third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt b/third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt
new file mode 100644
index 0000000000000000000000000000000000000000..4428566cb5e090c1617d46bc0bc975ae3eda9bbd
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:70e3213bd30fe8d8ce37b54675282745e406f915a51511a003aeff99b6da04cf
+size 662187
diff --git a/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_FEMALE.npz b/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_FEMALE.npz
new file mode 100644
index 0000000000000000000000000000000000000000..da0a200cd85eb10f73aa36d44f1d9c509a82dfcc
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_FEMALE.npz
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b2a3686c9d6d218ff6822fba411c607a3c8125a70af340f384ce68bebecabe0e
+size 108794146
diff --git a/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_FEMALE.npz:Zone.Identifier b/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_FEMALE.npz:Zone.Identifier
new file mode 100644
index 0000000000000000000000000000000000000000..41ba1e2adb7d21df35e44a6804597148324f1d48
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_FEMALE.npz:Zone.Identifier
@@ -0,0 +1,4 @@
+[ZoneTransfer]
+ZoneId=3
+ReferrerUrl=https://huggingface.co/
+HostUrl=https://cas-bridge.xethub.hf.co/xet-bridge-us/6672e1fcface4c01c0868225/a39ca190ec6044a71a200e63fa2aefc251cb3657e91f019031df8275c4eaa06c?X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Content-Sha256=UNSIGNED-PAYLOAD&X-Amz-Credential=cas%2F20251228%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Date=20251228T181303Z&X-Amz-Expires=3600&X-Amz-Signature=3e993ea9e8a78a59387c4a923e892d55dfd71411c70bb9d8dee7859d8ccde60d&X-Amz-SignedHeaders=host&X-Xet-Cas-Uid=68135a1d2de2415cdd25fa30&response-content-disposition=attachment%3B+filename*%3DUTF-8%27%27SMPLX_FEMALE.npz%3B+filename%3D%22SMPLX_FEMALE.npz%22%3B&x-id=GetObject&Expires=1766949183&Policy=eyJTdGF0ZW1lbnQiOlt7IkNvbmRpdGlvbiI6eyJEYXRlTGVzc1RoYW4iOnsiQVdTOkVwb2NoVGltZSI6MTc2Njk0OTE4M319LCJSZXNvdXJjZSI6Imh0dHBzOi8vY2FzLWJyaWRnZS54ZXRodWIuaGYuY28veGV0LWJyaWRnZS11cy82NjcyZTFmY2ZhY2U0YzAxYzA4NjgyMjUvYTM5Y2ExOTBlYzYwNDRhNzFhMjAwZTYzZmEyYWVmYzI1MWNiMzY1N2U5MWYwMTkwMzFkZjgyNzVjNGVhYTA2YyoifV19&Signature=ERafkOrZkrAhuDTHdZEGAk1Hid6XJm-MNTwdtuwsjOjl7tZtHhT3Wcp1lfPJsZc01F3bq0m1BEAzHQ86hD-xUAP2opSAQ6vG2CimPMIW0ulVU7oCm0leEAa-kNcdQEgRMXOIpU3mqNa5n0zlbzUKvK3kyONKiXxdRMnyn4o0Ff51s90Rrpt%7EMYzNfmgKvrV7hwDhPkIAPpUlTVIoFNGrO4CqJ4wkebwwNZBHN81%7ES848bUWe7Kw7NDMfhwv%7ERKo0ySGbSyKbSte-3%7EN3JSUU-vKuW9Q9V2HaI2wGDw0UiX-FbUmuyTCTQRvGgNKzuO4P25OFTlx%7EhqUv-uVzvXRjEg__&Key-Pair-Id=K2L8F4GPSG1IFC
diff --git a/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_MALE.npz b/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_MALE.npz
new file mode 100644
index 0000000000000000000000000000000000000000..41fdef3ff2784eb06bb479ebf5fb6887aafbc183
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_MALE.npz
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ab318e3f37d2bfaae26abf4e6fab445c2a610e1d63714794d60379cc263bc2a5
+size 108753445
diff --git a/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_MALE.npz:Zone.Identifier b/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_MALE.npz:Zone.Identifier
new file mode 100644
index 0000000000000000000000000000000000000000..ed01b000963333c415f584a302af730f9b663c50
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_MALE.npz:Zone.Identifier
@@ -0,0 +1,4 @@
+[ZoneTransfer]
+ZoneId=3
+ReferrerUrl=https://huggingface.co/
+HostUrl=https://cas-bridge.xethub.hf.co/xet-bridge-us/6672e1fcface4c01c0868225/537306cc40255fdf7cd753931355227e445d0a31af17b353c9bdb4d0c45fcfd5?X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Content-Sha256=UNSIGNED-PAYLOAD&X-Amz-Credential=cas%2F20251228%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Date=20251228T181307Z&X-Amz-Expires=3600&X-Amz-Signature=5862a029fb54f2d3f8e592ecaef90d02b5e85e64f0f2beb1d2d5f3cd7280d1b2&X-Amz-SignedHeaders=host&X-Xet-Cas-Uid=68135a1d2de2415cdd25fa30&response-content-disposition=attachment%3B+filename*%3DUTF-8%27%27SMPLX_MALE.npz%3B+filename%3D%22SMPLX_MALE.npz%22%3B&x-id=GetObject&Expires=1766949187&Policy=eyJTdGF0ZW1lbnQiOlt7IkNvbmRpdGlvbiI6eyJEYXRlTGVzc1RoYW4iOnsiQVdTOkVwb2NoVGltZSI6MTc2Njk0OTE4N319LCJSZXNvdXJjZSI6Imh0dHBzOi8vY2FzLWJyaWRnZS54ZXRodWIuaGYuY28veGV0LWJyaWRnZS11cy82NjcyZTFmY2ZhY2U0YzAxYzA4NjgyMjUvNTM3MzA2Y2M0MDI1NWZkZjdjZDc1MzkzMTM1NTIyN2U0NDVkMGEzMWFmMTdiMzUzYzliZGI0ZDBjNDVmY2ZkNSoifV19&Signature=UsIZCGdmvPYiEWxGzoMRCGQV3NoQczJL37mTr188fEdMmo8jUWyuuvWJyubkGY7CPTMQaw4NxafY7KCSHBAaidBYcwNKVPx8T2gHM00O59KZi1UpYRJVl1DG2QUCeMiVr7G8tNjbSCV6y9BFSrtXKpQRiKewhvj0pCZBFwF3GukR2b5QVAwz7tXhyBH2WpNrlqKD0xoDyc4%7EO06FNabKSQAnznMnn0FflO6-%7E6Z9b9mctQwfxXUaK42twJpRjQUWkCy0aq6gfFjy7VaOIHSWEQVN0Tx0B6-S6BJcFbJkk6ifxh%7EezkqH6DeM2Yz6y%7ExgarZqwethbrcSZ5D0Vk%7ENfQ__&Key-Pair-Id=K2L8F4GPSG1IFC
diff --git a/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_NEUTRAL.npz b/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_NEUTRAL.npz
new file mode 100644
index 0000000000000000000000000000000000000000..6f42b326bd60123bd813c0fa2df7f4660862a920
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/body_models/smplx/SMPLX_NEUTRAL.npz
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:376021446ddc86e99acacd795182bbef903e61d33b76b9d8b359c2b0865bd992
+size 108752058
diff --git a/third_party/GVHMR/inputs/checkpoints/body_models/smplx2smpl_sparse.pt b/third_party/GVHMR/inputs/checkpoints/body_models/smplx2smpl_sparse.pt
new file mode 100644
index 0000000000000000000000000000000000000000..0595089d31dec584e97b169fa936a4a8dfbd36bb
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/body_models/smplx2smpl_sparse.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0fc821a9e79ec3e76d6a9796b96d5bef8cd67055e18497bbe370d8aed9e07e06
+size 208935
diff --git a/third_party/GVHMR/inputs/checkpoints/dpvo/dpvo.pth b/third_party/GVHMR/inputs/checkpoints/dpvo/dpvo.pth
new file mode 100644
index 0000000000000000000000000000000000000000..25b16864668c8625f38021d17cc534258ff6297f
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/dpvo/dpvo.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:30d02dc2b88a321cf99aad8e4ea1152a44d791b5b65bf95ad036922819c0ff12
+size 14167743
diff --git a/third_party/GVHMR/inputs/checkpoints/droidcalib.pth b/third_party/GVHMR/inputs/checkpoints/droidcalib.pth
new file mode 100644
index 0000000000000000000000000000000000000000..354886b1715d6c0acbc8fadb2c2a5bd623d34729
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/droidcalib.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:67c5b78f727fb5ea7a6bf31fafe2f20c8982a56ca0b4f880ab46769a244b0a6f
+size 16049525
diff --git a/third_party/GVHMR/inputs/checkpoints/droidcalib.pth:Zone.Identifier b/third_party/GVHMR/inputs/checkpoints/droidcalib.pth:Zone.Identifier
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/third_party/GVHMR/inputs/checkpoints/dwpose/dw-ll_ucoco_384.onnx b/third_party/GVHMR/inputs/checkpoints/dwpose/dw-ll_ucoco_384.onnx
new file mode 100644
index 0000000000000000000000000000000000000000..df84ce34881c5701a29e09badd8c96f5c17bd214
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/dwpose/dw-ll_ucoco_384.onnx
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:724f4ff2439ed61afb86fb8a1951ec39c6220682803b4a8bd4f598cd913b1843
+size 134399116
diff --git a/third_party/GVHMR/inputs/checkpoints/dwpose/dw-ll_ucoco_384.pth b/third_party/GVHMR/inputs/checkpoints/dwpose/dw-ll_ucoco_384.pth
new file mode 100644
index 0000000000000000000000000000000000000000..c49c36077e9022019db86ae591ae20003689e99d
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/dwpose/dw-ll_ucoco_384.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0d9408b13cd863c4e95a149dd31232f88f2a12aa6cf8964ed74d7d97748c7a07
+size 406878486
diff --git a/third_party/GVHMR/inputs/checkpoints/groundingdino/GroundingDINO_SwinT_OGC.py b/third_party/GVHMR/inputs/checkpoints/groundingdino/GroundingDINO_SwinT_OGC.py
new file mode 100644
index 0000000000000000000000000000000000000000..9158d5f6260ec74bded95377d382387430d7cd70
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/groundingdino/GroundingDINO_SwinT_OGC.py
@@ -0,0 +1,43 @@
+batch_size = 1
+modelname = "groundingdino"
+backbone = "swin_T_224_1k"
+position_embedding = "sine"
+pe_temperatureH = 20
+pe_temperatureW = 20
+return_interm_indices = [1, 2, 3]
+backbone_freeze_keywords = None
+enc_layers = 6
+dec_layers = 6
+pre_norm = False
+dim_feedforward = 2048
+hidden_dim = 256
+dropout = 0.0
+nheads = 8
+num_queries = 900
+query_dim = 4
+num_patterns = 0
+num_feature_levels = 4
+enc_n_points = 4
+dec_n_points = 4
+two_stage_type = "standard"
+two_stage_bbox_embed_share = False
+two_stage_class_embed_share = False
+transformer_activation = "relu"
+dec_pred_bbox_embed_share = True
+dn_box_noise_scale = 1.0
+dn_label_noise_ratio = 0.5
+dn_label_coef = 1.0
+dn_bbox_coef = 1.0
+embed_init_tgt = True
+dn_labelbook_size = 2000
+max_text_len = 256
+text_encoder_type = "bert-base-uncased"
+use_text_enhancer = True
+use_fusion_layer = True
+use_checkpoint = True
+use_transformer_ckpt = True
+use_text_cross_attention = True
+text_dropout = 0.0
+fusion_dropout = 0.0
+fusion_droppath = 0.1
+sub_sentence_present = True
diff --git a/third_party/GVHMR/inputs/checkpoints/groundingdino/groundingdino_swint_ogc.pth b/third_party/GVHMR/inputs/checkpoints/groundingdino/groundingdino_swint_ogc.pth
new file mode 100644
index 0000000000000000000000000000000000000000..5cdf6bcd10d491abf170a78eca4fcebf76aa791a
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/groundingdino/groundingdino_swint_ogc.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3b3ca2563c77c69f651d7bd133e97139c186df06231157a64c507099c52bc799
+size 693997677
diff --git a/third_party/GVHMR/inputs/checkpoints/gvhmr/gvhmr_siga24_release.ckpt b/third_party/GVHMR/inputs/checkpoints/gvhmr/gvhmr_siga24_release.ckpt
new file mode 100644
index 0000000000000000000000000000000000000000..20649d00dc1c5900648594b95de4eef78f7a27ec
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/gvhmr/gvhmr_siga24_release.ckpt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4fae7da2de388d5da3514cb27a2d003f364dacb280e9cf88972b710e589c6b91
+size 163508011
diff --git a/third_party/GVHMR/inputs/checkpoints/hmr2/epoch=10-step=25000.ckpt b/third_party/GVHMR/inputs/checkpoints/hmr2/epoch=10-step=25000.ckpt
new file mode 100644
index 0000000000000000000000000000000000000000..46b527a17ee53d47a16caf15c3d61c878a4790be
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/hmr2/epoch=10-step=25000.ckpt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2dcf79638109781d1ae5f5c44fee5f55bc83291c210653feead9b7f04fa6f20e
+size 2709494041
diff --git a/third_party/GVHMR/inputs/checkpoints/rtmpose-x/rtmpose-x_8xb32-270e_coco-wholebody-384x288.py b/third_party/GVHMR/inputs/checkpoints/rtmpose-x/rtmpose-x_8xb32-270e_coco-wholebody-384x288.py
new file mode 100644
index 0000000000000000000000000000000000000000..46e27a88deaf72bc038e3650e9404138d019a977
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/rtmpose-x/rtmpose-x_8xb32-270e_coco-wholebody-384x288.py
@@ -0,0 +1,233 @@
+_base_ = ['mmpose::_base_/default_runtime.py']
+
+# common setting
+num_keypoints = 133
+input_size = (288, 384)
+
+# runtime
+max_epochs = 270
+stage2_num_epochs = 30
+base_lr = 4e-3
+train_batch_size = 32
+val_batch_size = 32
+
+train_cfg = dict(max_epochs=max_epochs, val_interval=10)
+randomness = dict(seed=21)
+
+# optimizer
+optim_wrapper = dict(
+ type='OptimWrapper',
+ optimizer=dict(type='AdamW', lr=base_lr, weight_decay=0.05),
+ clip_grad=dict(max_norm=35, norm_type=2),
+ paramwise_cfg=dict(
+ norm_decay_mult=0, bias_decay_mult=0, bypass_duplicate=True))
+
+# learning rate
+param_scheduler = [
+ dict(
+ type='LinearLR',
+ start_factor=1.0e-5,
+ by_epoch=False,
+ begin=0,
+ end=1000),
+ dict(
+ type='CosineAnnealingLR',
+ eta_min=base_lr * 0.05,
+ begin=max_epochs // 2,
+ end=max_epochs,
+ T_max=max_epochs // 2,
+ by_epoch=True,
+ convert_to_iter_based=True),
+]
+
+# automatically scaling LR based on the actual training batch size
+auto_scale_lr = dict(base_batch_size=512)
+
+# codec settings
+codec = dict(
+ type='SimCCLabel',
+ input_size=input_size,
+ sigma=(6., 6.93),
+ simcc_split_ratio=2.0,
+ normalize=False,
+ use_dark=False)
+
+# model settings
+model = dict(
+ type='TopdownPoseEstimator',
+ data_preprocessor=dict(
+ type='PoseDataPreprocessor',
+ mean=[123.675, 116.28, 103.53],
+ std=[58.395, 57.12, 57.375],
+ bgr_to_rgb=True),
+ backbone=dict(
+ _scope_='mmdet',
+ type='CSPNeXt',
+ arch='P5',
+ expand_ratio=0.5,
+ deepen_factor=1.33,
+ widen_factor=1.25,
+ out_indices=(4, ),
+ channel_attention=True,
+ norm_cfg=dict(type='SyncBN'),
+ act_cfg=dict(type='SiLU'),
+ init_cfg=dict(
+ type='Pretrained',
+ prefix='backbone.',
+ checkpoint='https://download.openmmlab.com/mmpose/v1/projects/'
+ 'rtmposev1/cspnext-x_udp-body7_210e-384x288-d28b58e6_20230529.pth' # noqa
+ )),
+ head=dict(
+ type='RTMCCHead',
+ in_channels=1280,
+ out_channels=num_keypoints,
+ input_size=codec['input_size'],
+ in_featuremap_size=tuple([s // 32 for s in codec['input_size']]),
+ simcc_split_ratio=codec['simcc_split_ratio'],
+ final_layer_kernel_size=7,
+ gau_cfg=dict(
+ hidden_dims=256,
+ s=128,
+ expansion_factor=2,
+ dropout_rate=0.,
+ drop_path=0.,
+ act_fn='SiLU',
+ use_rel_bias=False,
+ pos_enc=False),
+ loss=dict(
+ type='KLDiscretLoss',
+ use_target_weight=True,
+ beta=10.,
+ label_softmax=True),
+ decoder=codec),
+ test_cfg=dict(flip_test=True, ))
+
+# base dataset settings
+dataset_type = 'CocoWholeBodyDataset'
+data_mode = 'topdown'
+data_root = 'data/coco/'
+
+backend_args = dict(backend='local')
+
+# pipelines
+train_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='RandomFlip', direction='horizontal'),
+ dict(type='RandomHalfBody'),
+ dict(
+ type='RandomBBoxTransform', scale_factor=[0.5, 1.5], rotate_factor=90),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='mmdet.YOLOXHSVRandomAug'),
+ dict(
+ type='Albumentation',
+ transforms=[
+ dict(type='Blur', p=0.1),
+ dict(type='MedianBlur', p=0.1),
+ dict(
+ type='CoarseDropout',
+ max_holes=1,
+ max_height=0.4,
+ max_width=0.4,
+ min_holes=1,
+ min_height=0.2,
+ min_width=0.2,
+ p=1.0),
+ ]),
+ dict(type='GenerateTarget', encoder=codec),
+ dict(type='PackPoseInputs')
+]
+val_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='PackPoseInputs')
+]
+
+train_pipeline_stage2 = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='RandomFlip', direction='horizontal'),
+ dict(type='RandomHalfBody'),
+ dict(
+ type='RandomBBoxTransform',
+ shift_factor=0.,
+ scale_factor=[0.5, 1.5],
+ rotate_factor=90),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='mmdet.YOLOXHSVRandomAug'),
+ dict(
+ type='Albumentation',
+ transforms=[
+ dict(type='Blur', p=0.1),
+ dict(type='MedianBlur', p=0.1),
+ dict(
+ type='CoarseDropout',
+ max_holes=1,
+ max_height=0.4,
+ max_width=0.4,
+ min_holes=1,
+ min_height=0.2,
+ min_width=0.2,
+ p=0.5),
+ ]),
+ dict(type='GenerateTarget', encoder=codec),
+ dict(type='PackPoseInputs')
+]
+
+# data loaders
+train_dataloader = dict(
+ batch_size=train_batch_size,
+ num_workers=10,
+ persistent_workers=True,
+ sampler=dict(type='DefaultSampler', shuffle=True),
+ dataset=dict(
+ type=dataset_type,
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='annotations/coco_wholebody_train_v1.0.json',
+ data_prefix=dict(img='train2017/'),
+ pipeline=train_pipeline,
+ ))
+val_dataloader = dict(
+ batch_size=val_batch_size,
+ num_workers=10,
+ persistent_workers=True,
+ drop_last=False,
+ sampler=dict(type='DefaultSampler', shuffle=False, round_up=False),
+ dataset=dict(
+ type=dataset_type,
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='annotations/coco_wholebody_val_v1.0.json',
+ data_prefix=dict(img='val2017/'),
+ test_mode=True,
+ bbox_file='data/coco/person_detection_results/'
+ 'COCO_val2017_detections_AP_H_56_person.json',
+ pipeline=val_pipeline,
+ ))
+test_dataloader = val_dataloader
+
+# hooks
+default_hooks = dict(
+ checkpoint=dict(
+ save_best='coco-wholebody/AP', rule='greater', max_keep_ckpts=1))
+
+custom_hooks = [
+ dict(
+ type='EMAHook',
+ ema_type='ExpMomentumEMA',
+ momentum=0.0002,
+ update_buffers=True,
+ priority=49),
+ dict(
+ type='mmdet.PipelineSwitchHook',
+ switch_epoch=max_epochs - stage2_num_epochs,
+ switch_pipeline=train_pipeline_stage2)
+]
+
+# evaluators
+val_evaluator = dict(
+ type='CocoWholeBodyMetric',
+ ann_file=data_root + 'annotations/coco_wholebody_val_v1.0.json')
+test_evaluator = val_evaluator
\ No newline at end of file
diff --git a/third_party/GVHMR/inputs/checkpoints/rtmpose-x/rtmpose-x_simcc-coco-wholebody_pt-body7_270e-384x288-401dfc90_20230629.pth b/third_party/GVHMR/inputs/checkpoints/rtmpose-x/rtmpose-x_simcc-coco-wholebody_pt-body7_270e-384x288-401dfc90_20230629.pth
new file mode 100644
index 0000000000000000000000000000000000000000..7343a83f6f5644e721e8795c480634e5c6439ed5
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/rtmpose-x/rtmpose-x_simcc-coco-wholebody_pt-body7_270e-384x288-401dfc90_20230629.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:401dfc904a12d2cbb44ba83c17cd2a45399fead4e29b6fd73137724cadc03419
+size 227232559
diff --git a/third_party/GVHMR/inputs/checkpoints/rtmpose/humanart_l_256x192.pth b/third_party/GVHMR/inputs/checkpoints/rtmpose/humanart_l_256x192.pth
new file mode 100644
index 0000000000000000000000000000000000000000..5b518e3f92709581fc0f1ba770b57bd4c0d20f3c
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/rtmpose/humanart_l_256x192.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:389f2cb01ce3fd1415c041a462be60ad3a47bed8e365a43be8e5555a3c3f9663
+size 110889408
diff --git a/third_party/GVHMR/inputs/checkpoints/rtmw/rtmw-x_384x288.pth b/third_party/GVHMR/inputs/checkpoints/rtmw/rtmw-x_384x288.pth
new file mode 100644
index 0000000000000000000000000000000000000000..fdd3262ed7a50651e35b60b59f4698b53ef60d56
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/rtmw/rtmw-x_384x288.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f840f2044fe46cb3821b7cea86be83e1f6cba406ccd28f5475ac010412dcda95
+size 369720404
diff --git a/third_party/GVHMR/inputs/checkpoints/sam2/sam2_hiera_large.pt b/third_party/GVHMR/inputs/checkpoints/sam2/sam2_hiera_large.pt
new file mode 100644
index 0000000000000000000000000000000000000000..7198ee4779a9e91db4d79bdc80e188cc182482e0
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/sam2/sam2_hiera_large.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7442e4e9b732a508f80e141e7c2913437a3610ee0c77381a66658c3a445df87b
+size 897952466
diff --git a/third_party/GVHMR/inputs/checkpoints/vitpose/td-hm_ViTPose-huge_8xb64-210e_coco-256x192.py b/third_party/GVHMR/inputs/checkpoints/vitpose/td-hm_ViTPose-huge_8xb64-210e_coco-256x192.py
new file mode 100644
index 0000000000000000000000000000000000000000..0e477cc50ab698d8fcdf95c0274c027451b04bab
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/vitpose/td-hm_ViTPose-huge_8xb64-210e_coco-256x192.py
@@ -0,0 +1,275 @@
+auto_scale_lr = dict(base_batch_size=512)
+backend_args = dict(backend='local')
+codec = dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap')
+custom_hooks = [
+ dict(type='SyncBuffersHook'),
+]
+custom_imports = dict(
+ allow_failed_imports=False,
+ imports=[
+ 'mmpose.engine.optim_wrappers.layer_decay_optim_wrapper',
+ ])
+data_mode = 'topdown'
+data_root = 'data/coco/'
+dataset_type = 'CocoDataset'
+default_hooks = dict(
+ badcase=dict(
+ badcase_thr=5,
+ enable=False,
+ metric_type='loss',
+ out_dir='badcase',
+ type='BadCaseAnalysisHook'),
+ checkpoint=dict(
+ interval=10,
+ max_keep_ckpts=1,
+ rule='greater',
+ save_best='coco/AP',
+ type='CheckpointHook'),
+ logger=dict(interval=50, type='LoggerHook'),
+ param_scheduler=dict(type='ParamSchedulerHook'),
+ sampler_seed=dict(type='DistSamplerSeedHook'),
+ timer=dict(type='IterTimerHook'),
+ visualization=dict(enable=False, type='PoseVisualizationHook'))
+default_scope = 'mmpose'
+env_cfg = dict(
+ cudnn_benchmark=False,
+ dist_cfg=dict(backend='nccl'),
+ mp_cfg=dict(mp_start_method='fork', opencv_num_threads=0))
+load_from = None
+log_level = 'INFO'
+log_processor = dict(
+ by_epoch=True, num_digits=6, type='LogProcessor', window_size=50)
+model = dict(
+ backbone=dict(
+ arch='huge',
+ drop_path_rate=0.55,
+ img_size=(
+ 256,
+ 192,
+ ),
+ init_cfg=dict(
+ checkpoint=
+ 'https://download.openmmlab.com/mmpose/v1/pretrained_models/mae_pretrain_vit_huge_20230913.pth',
+ type='Pretrained'),
+ out_type='featmap',
+ patch_cfg=dict(padding=2),
+ patch_size=16,
+ qkv_bias=True,
+ type='mmpretrain.VisionTransformer',
+ with_cls_token=False),
+ data_preprocessor=dict(
+ bgr_to_rgb=True,
+ mean=[
+ 123.675,
+ 116.28,
+ 103.53,
+ ],
+ std=[
+ 58.395,
+ 57.12,
+ 57.375,
+ ],
+ type='PoseDataPreprocessor'),
+ head=dict(
+ decoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ deconv_kernel_sizes=(
+ 4,
+ 4,
+ ),
+ deconv_out_channels=(
+ 256,
+ 256,
+ ),
+ in_channels=1280,
+ loss=dict(type='KeypointMSELoss', use_target_weight=True),
+ out_channels=17,
+ type='HeatmapHead'),
+ test_cfg=dict(flip_mode='heatmap', flip_test=True, shift_heatmap=False),
+ type='TopdownPoseEstimator')
+optim_wrapper = dict(
+ clip_grad=dict(max_norm=1.0, norm_type=2),
+ constructor='LayerDecayOptimWrapperConstructor',
+ optimizer=dict(
+ betas=(
+ 0.9,
+ 0.999,
+ ), lr=0.0005, type='AdamW', weight_decay=0.1),
+ paramwise_cfg=dict(
+ custom_keys=dict(
+ bias=dict(decay_multi=0.0),
+ norm=dict(decay_mult=0.0),
+ pos_embed=dict(decay_mult=0.0),
+ relative_position_bias_table=dict(decay_mult=0.0)),
+ layer_decay_rate=0.85,
+ num_layers=32))
+param_scheduler = [
+ dict(
+ begin=0, by_epoch=False, end=500, start_factor=0.001, type='LinearLR'),
+ dict(
+ begin=0,
+ by_epoch=True,
+ end=210,
+ gamma=0.1,
+ milestones=[
+ 170,
+ 200,
+ ],
+ type='MultiStepLR'),
+]
+resume = False
+test_cfg = dict()
+test_dataloader = dict(
+ batch_size=32,
+ dataset=dict(
+ ann_file='annotations/person_keypoints_val2017.json',
+ bbox_file=
+ 'data/coco/person_detection_results/COCO_val2017_detections_AP_H_56_person.json',
+ data_mode='topdown',
+ data_prefix=dict(img='val2017/'),
+ data_root='data/coco/',
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+test_evaluator = dict(
+ ann_file='data/coco/annotations/person_keypoints_val2017.json',
+ type='CocoMetric')
+train_cfg = dict(by_epoch=True, max_epochs=210, val_interval=10)
+train_dataloader = dict(
+ batch_size=64,
+ dataset=dict(
+ ann_file='annotations/person_keypoints_train2017.json',
+ data_mode='topdown',
+ data_prefix=dict(img='train2017/'),
+ data_root='data/coco/',
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+ ],
+ type='CocoDataset'),
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(shuffle=True, type='DefaultSampler'))
+train_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+]
+val_cfg = dict()
+val_dataloader = dict(
+ batch_size=32,
+ dataset=dict(
+ ann_file='annotations/person_keypoints_val2017.json',
+ bbox_file=
+ 'data/coco/person_detection_results/COCO_val2017_detections_AP_H_56_person.json',
+ data_mode='topdown',
+ data_prefix=dict(img='val2017/'),
+ data_root='data/coco/',
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+val_evaluator = dict(
+ ann_file='data/coco/annotations/person_keypoints_val2017.json',
+ type='CocoMetric')
+val_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+]
+vis_backends = [
+ dict(type='LocalVisBackend'),
+]
+visualizer = dict(
+ name='visualizer',
+ type='PoseLocalVisualizer',
+ vis_backends=[
+ dict(type='LocalVisBackend'),
+ ])
diff --git a/third_party/GVHMR/inputs/checkpoints/vitpose/vitpose-h-multi-coco.pth b/third_party/GVHMR/inputs/checkpoints/vitpose/vitpose-h-multi-coco.pth
new file mode 100644
index 0000000000000000000000000000000000000000..2072ac0591d04aacf4c55af4104d31ee5eb604cd
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/vitpose/vitpose-h-multi-coco.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:50e33f4077ef2a6bcfd7110c58742b24c5859b7798fb0eedd6d2215e0a8980bc
+size 2549075546
diff --git a/third_party/GVHMR/inputs/checkpoints/yolo/yolov8x.pt b/third_party/GVHMR/inputs/checkpoints/yolo/yolov8x.pt
new file mode 100644
index 0000000000000000000000000000000000000000..a0510bf3bb96a465f97b81dab2dd2f437e2cccbe
--- /dev/null
+++ b/third_party/GVHMR/inputs/checkpoints/yolo/yolov8x.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c4d5a3f000d771762f03fc8b57ebd0aae324aeaefdd6e68492a9c4470f2d1e8b
+size 136867539
diff --git a/third_party/GVHMR/process_data.sh b/third_party/GVHMR/process_data.sh
new file mode 100644
index 0000000000000000000000000000000000000000..8c95561c7654e0ed223a2e7c2201deb375efea09
--- /dev/null
+++ b/third_party/GVHMR/process_data.sh
@@ -0,0 +1 @@
+python tools/demo/process_dataset.py --input /mnt/c/Temp/SyntheticDataset --output ../../processed_dataset --genmo --vitpose_infer --vit_batch_size 512 --consistency_check --ground_to_zero --max_duration_s 10 --camera_source unity --debug_kp2d_overlay
diff --git a/third_party/GVHMR/pyproject.toml b/third_party/GVHMR/pyproject.toml
new file mode 100644
index 0000000000000000000000000000000000000000..ff3593963d94d7cddd23a38ef2f0723f8d342d28
--- /dev/null
+++ b/third_party/GVHMR/pyproject.toml
@@ -0,0 +1,16 @@
+[tool.black]
+line-length = 120
+include = '\.pyi?$'
+exclude = '''
+/(
+ \.git
+ | \.hg
+ | \.mypy_cache
+ | \.tox
+ | \.venv
+ | _build
+ | buck-out
+ | build
+ | dist
+)/
+'''
diff --git a/third_party/GVHMR/pyrightconfig.json b/third_party/GVHMR/pyrightconfig.json
new file mode 100644
index 0000000000000000000000000000000000000000..6b1d43620719f77fd6b708f8483219a1de6a19ce
--- /dev/null
+++ b/third_party/GVHMR/pyrightconfig.json
@@ -0,0 +1,7 @@
+{
+ "exclude": [
+ "./inputs",
+ "./outputs"
+ ],
+ "typeCheckingMode": "off",
+}
diff --git a/third_party/GVHMR/requirements.txt b/third_party/GVHMR/requirements.txt
new file mode 100644
index 0000000000000000000000000000000000000000..1b581a6384f40d2e4d59d193ca3d092722e61006
--- /dev/null
+++ b/third_party/GVHMR/requirements.txt
@@ -0,0 +1,48 @@
+# PyTorch
+--extra-index-url https://download.pytorch.org/whl/cu121
+torch==2.3.0+cu121
+torchvision==0.18.0+cu121
+timm==0.9.12 # For HMR2.0a feature extraction
+
+# Lightning + Hydra
+lightning==2.3.0
+hydra-core==1.3
+hydra-zen
+hydra_colorlog
+rich
+
+# Common utilities
+numpy==1.23.5
+jupyter
+matplotlib
+ipdb
+setuptools>=68.0
+black
+tensorboardX
+opencv-python
+ffmpeg-python
+scikit-image
+termcolor
+einops
+imageio==2.34.1
+av==13.0.0 # imageio[pyav], improved performance over imageio[ffmpeg]
+joblib
+
+# Diffusion
+# diffusers[torch]==0.19.3
+# transformers==4.31.0
+
+# 3D-Vision
+pytorch3d @ https://dl.fbaipublicfiles.com/pytorch3d/packaging/wheels/py310_cu121_pyt230/pytorch3d-0.7.6-cp310-cp310-linux_x86_64.whl
+trimesh
+chumpy
+smplx
+# open3d==0.17.0
+wis3d
+pycolmap
+
+# 2D-Pose
+ultralytics==8.2.42 # YOLO
+cython_bbox
+lapx
+cuda-toolkit=12.1
\ No newline at end of file
diff --git a/third_party/GVHMR/rtmpose_l_humanart_256x192.py b/third_party/GVHMR/rtmpose_l_humanart_256x192.py
new file mode 100644
index 0000000000000000000000000000000000000000..7dd10569e5d4c14a04ac2ffa786a614a96469329
--- /dev/null
+++ b/third_party/GVHMR/rtmpose_l_humanart_256x192.py
@@ -0,0 +1,231 @@
+
+# runtime
+max_epochs = 420
+stage2_num_epochs = 30
+base_lr = 4e-3
+
+train_cfg = dict(max_epochs=max_epochs, val_interval=10)
+randomness = dict(seed=21)
+
+# optimizer
+optim_wrapper = dict(
+ type='OptimWrapper',
+ optimizer=dict(type='AdamW', lr=base_lr, weight_decay=0.05),
+ paramwise_cfg=dict(
+ norm_decay_mult=0, bias_decay_mult=0, bypass_duplicate=True))
+
+# learning rate
+param_scheduler = [
+ dict(
+ type='LinearLR',
+ start_factor=1.0e-5,
+ by_epoch=False,
+ begin=0,
+ end=1000),
+ dict(
+ # use cosine lr from 210 to 420 epoch
+ type='CosineAnnealingLR',
+ eta_min=base_lr * 0.05,
+ begin=max_epochs // 2,
+ end=max_epochs,
+ T_max=max_epochs // 2,
+ by_epoch=True,
+ convert_to_iter_based=True),
+]
+
+# automatically scaling LR based on the actual training batch size
+auto_scale_lr = dict(base_batch_size=1024)
+
+# codec settings
+codec = dict(
+ type='SimCCLabel',
+ input_size=(192, 256),
+ sigma=(4.9, 5.66),
+ simcc_split_ratio=2.0,
+ normalize=False,
+ use_dark=False)
+
+# model settings
+model = dict(
+ type='TopdownPoseEstimator',
+ data_preprocessor=dict(
+ type='PoseDataPreprocessor',
+ mean=[123.675, 116.28, 103.53],
+ std=[58.395, 57.12, 57.375],
+ bgr_to_rgb=True),
+ backbone=dict(
+ _scope_='mmdet',
+ type='CSPNeXt',
+ arch='P5',
+ expand_ratio=0.5,
+ deepen_factor=1.,
+ widen_factor=1.,
+ out_indices=(4, ),
+ channel_attention=True,
+ norm_cfg=dict(type='SyncBN'),
+ act_cfg=dict(type='SiLU'),
+ init_cfg=dict(
+ type='Pretrained',
+ prefix='backbone.',
+ checkpoint='https://download.openmmlab.com/mmpose/v1/projects/'
+ 'rtmpose/cspnext-l_udp-aic-coco_210e-256x192-273b7631_20230130.pth' # noqa
+ )),
+ head=dict(
+ type='RTMCCHead',
+ in_channels=1024,
+ out_channels=17,
+ input_size=codec['input_size'],
+ in_featuremap_size=(6, 8),
+ simcc_split_ratio=codec['simcc_split_ratio'],
+ final_layer_kernel_size=7,
+ gau_cfg=dict(
+ hidden_dims=256,
+ s=128,
+ expansion_factor=2,
+ dropout_rate=0.,
+ drop_path=0.,
+ act_fn='SiLU',
+ use_rel_bias=False,
+ pos_enc=False),
+ loss=dict(
+ type='KLDiscretLoss',
+ use_target_weight=True,
+ beta=10.,
+ label_softmax=True),
+ decoder=codec),
+ test_cfg=dict(flip_test=True))
+
+# base dataset settings
+dataset_type = 'HumanArtDataset'
+data_mode = 'topdown'
+data_root = 'data/'
+
+backend_args = dict(backend='local')
+# backend_args = dict(
+# backend='petrel',
+# path_mapping=dict({
+# f'{data_root}': 's3://openmmlab/datasets/detection/coco/',
+# f'{data_root}': 's3://openmmlab/datasets/detection/coco/'
+# }))
+
+# pipelines
+train_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='RandomFlip', direction='horizontal'),
+ dict(type='RandomHalfBody'),
+ dict(
+ type='RandomBBoxTransform', scale_factor=[0.6, 1.4], rotate_factor=80),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='mmdet.YOLOXHSVRandomAug'),
+ dict(
+ type='Albumentation',
+ transforms=[
+ dict(type='Blur', p=0.1),
+ dict(type='MedianBlur', p=0.1),
+ dict(
+ type='CoarseDropout',
+ max_holes=1,
+ max_height=0.4,
+ max_width=0.4,
+ min_holes=1,
+ min_height=0.2,
+ min_width=0.2,
+ p=1.),
+ ]),
+ dict(type='GenerateTarget', encoder=codec),
+ dict(type='PackPoseInputs')
+]
+val_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='PackPoseInputs')
+]
+
+train_pipeline_stage2 = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='RandomFlip', direction='horizontal'),
+ dict(type='RandomHalfBody'),
+ dict(
+ type='RandomBBoxTransform',
+ shift_factor=0.,
+ scale_factor=[0.75, 1.25],
+ rotate_factor=60),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='mmdet.YOLOXHSVRandomAug'),
+ dict(
+ type='Albumentation',
+ transforms=[
+ dict(type='Blur', p=0.1),
+ dict(type='MedianBlur', p=0.1),
+ dict(
+ type='CoarseDropout',
+ max_holes=1,
+ max_height=0.4,
+ max_width=0.4,
+ min_holes=1,
+ min_height=0.2,
+ min_width=0.2,
+ p=0.5),
+ ]),
+ dict(type='GenerateTarget', encoder=codec),
+ dict(type='PackPoseInputs')
+]
+
+# data loaders
+train_dataloader = dict(
+ batch_size=256,
+ num_workers=10,
+ persistent_workers=True,
+ sampler=dict(type='DefaultSampler', shuffle=True),
+ dataset=dict(
+ type=dataset_type,
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='HumanArt/annotations/training_humanart_coco.json',
+ data_prefix=dict(img=''),
+ pipeline=train_pipeline,
+ ))
+val_dataloader = dict(
+ batch_size=64,
+ num_workers=10,
+ persistent_workers=True,
+ drop_last=False,
+ sampler=dict(type='DefaultSampler', shuffle=False, round_up=False),
+ dataset=dict(
+ type=dataset_type,
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='HumanArt/annotations/validation_humanart.json',
+ # bbox_file=f'{data_root}HumanArt/person_detection_results/'
+ # 'HumanArt_validation_detections_AP_H_56_person.json',
+ data_prefix=dict(img=''),
+ test_mode=True,
+ pipeline=val_pipeline,
+ ))
+test_dataloader = val_dataloader
+
+# hooks
+default_hooks = dict(
+ checkpoint=dict(save_best='coco/AP', rule='greater', max_keep_ckpts=1))
+
+custom_hooks = [
+ dict(
+ type='EMAHook',
+ ema_type='ExpMomentumEMA',
+ momentum=0.0002,
+ update_buffers=True,
+ priority=49),
+ dict(
+ type='mmdet.PipelineSwitchHook',
+ switch_epoch=max_epochs - stage2_num_epochs,
+ switch_pipeline=train_pipeline_stage2)
+]
+
+# evaluators
+val_evaluator = dict(
+ type='CocoMetric',
+ ann_file=data_root + 'HumanArt/annotations/validation_humanart.json')
+test_evaluator = val_evaluator
diff --git a/third_party/GVHMR/rtmw_x_384x288_fixed.py b/third_party/GVHMR/rtmw_x_384x288_fixed.py
new file mode 100644
index 0000000000000000000000000000000000000000..cdb24c1876a185ff956e4230d8b3e2845351e5c7
--- /dev/null
+++ b/third_party/GVHMR/rtmw_x_384x288_fixed.py
@@ -0,0 +1,610 @@
+
+# common setting
+num_keypoints = 133
+input_size = (288, 384)
+
+# runtime
+max_epochs = 270
+stage2_num_epochs = 10
+base_lr = 5e-4
+train_batch_size = 320
+val_batch_size = 32
+
+train_cfg = dict(max_epochs=max_epochs, val_interval=10)
+randomness = dict(seed=21)
+
+# optimizer
+optim_wrapper = dict(
+ type='OptimWrapper',
+ optimizer=dict(type='AdamW', lr=base_lr, weight_decay=0.1),
+ clip_grad=dict(max_norm=35, norm_type=2),
+ paramwise_cfg=dict(
+ norm_decay_mult=0, bias_decay_mult=0, bypass_duplicate=True))
+
+# learning rate
+param_scheduler = [
+ dict(
+ type='LinearLR',
+ start_factor=1.0e-5,
+ by_epoch=False,
+ begin=0,
+ end=1000),
+ dict(
+ # use cosine lr from 150 to 300 epoch
+ type='CosineAnnealingLR',
+ eta_min=base_lr * 0.05,
+ begin=max_epochs // 2,
+ end=max_epochs,
+ T_max=max_epochs // 2,
+ by_epoch=True,
+ convert_to_iter_based=True),
+]
+
+# automatically scaling LR based on the actual training batch size
+auto_scale_lr = dict(base_batch_size=2560)
+
+# codec settings
+codec = dict(
+ type='SimCCLabel',
+ input_size=input_size,
+ sigma=(6., 6.93),
+ simcc_split_ratio=2.0,
+ normalize=False,
+ use_dark=False,
+ decode_visibility=True)
+
+# model settings
+model = dict(
+ type='TopdownPoseEstimator',
+ data_preprocessor=dict(
+ type='PoseDataPreprocessor',
+ mean=[123.675, 116.28, 103.53],
+ std=[58.395, 57.12, 57.375],
+ bgr_to_rgb=True),
+ backbone=dict(
+ type='CSPNeXt',
+ arch='P5',
+ expand_ratio=0.5,
+ deepen_factor=1.33,
+ widen_factor=1.25,
+ channel_attention=True,
+ norm_cfg=dict(type='BN'),
+ act_cfg=dict(type='SiLU'),
+ init_cfg=dict(
+ type='Pretrained',
+ prefix='backbone.',
+ checkpoint='https://download.openmmlab.com/mmpose/v1/'
+ 'wholebody_2d_keypoint/rtmpose/ubody/rtmpose-x_simcc-ucoco_pt-aic-coco_270e-384x288-f5b50679_20230822.pth' # noqa
+ )),
+ neck=dict(
+ type='CSPNeXtPAFPN',
+ in_channels=[320, 640, 1280],
+ out_channels=None,
+ out_indices=(
+ 1,
+ 2,
+ ),
+ num_csp_blocks=2,
+ expand_ratio=0.5,
+ norm_cfg=dict(type='SyncBN'),
+ act_cfg=dict(type='SiLU', inplace=True)),
+ head=dict(
+ type='RTMWHead',
+ in_channels=1280,
+ out_channels=num_keypoints,
+ input_size=input_size,
+ in_featuremap_size=tuple([s // 32 for s in input_size]),
+ simcc_split_ratio=codec['simcc_split_ratio'],
+ final_layer_kernel_size=7,
+ gau_cfg=dict(
+ hidden_dims=256,
+ s=128,
+ expansion_factor=2,
+ dropout_rate=0.,
+ drop_path=0.,
+ act_fn='SiLU',
+ use_rel_bias=False,
+ pos_enc=False),
+ loss=dict(
+ type='KLDiscretLoss',
+ use_target_weight=True,
+ beta=1.,
+ label_softmax=True,
+ label_beta=10.,
+ mask=list(range(23, 91)),
+ mask_weight=0.5,
+ ),
+ decoder=codec),
+ test_cfg=dict(flip_test=True))
+
+# base dataset settings
+dataset_type = 'CocoWholeBodyDataset'
+data_mode = 'topdown'
+data_root = 'data/'
+
+backend_args = dict(backend='local')
+
+# pipelines
+train_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='RandomFlip', direction='horizontal'),
+ dict(type='RandomHalfBody'),
+ dict(
+ type='RandomBBoxTransform', scale_factor=[0.5, 1.5], rotate_factor=90),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='PhotometricDistortion'),
+ dict(
+ type='Albumentation',
+ transforms=[
+ dict(type='Blur', p=0.1),
+ dict(type='MedianBlur', p=0.1),
+ dict(
+ type='CoarseDropout',
+ max_holes=1,
+ max_height=0.4,
+ max_width=0.4,
+ min_holes=1,
+ min_height=0.2,
+ min_width=0.2,
+ p=0.5),
+ ]),
+ dict(
+ type='GenerateTarget',
+ encoder=codec,
+ use_dataset_keypoint_weights=True),
+ dict(type='PackPoseInputs')
+]
+val_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(type='PackPoseInputs')
+]
+train_pipeline_stage2 = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(type='RandomFlip', direction='horizontal'),
+ dict(type='RandomHalfBody'),
+ dict(
+ type='RandomBBoxTransform',
+ shift_factor=0.,
+ scale_factor=[0.5, 1.5],
+ rotate_factor=90),
+ dict(type='TopdownAffine', input_size=codec['input_size']),
+ dict(
+ type='Albumentation',
+ transforms=[
+ dict(type='Blur', p=0.1),
+ dict(type='MedianBlur', p=0.1),
+ ]),
+ dict(
+ type='GenerateTarget',
+ encoder=codec,
+ use_dataset_keypoint_weights=True),
+ dict(type='PackPoseInputs')
+]
+
+# mapping
+
+aic_coco133 = [(0, 6), (1, 8), (2, 10), (3, 5), (4, 7), (5, 9), (6, 12),
+ (7, 14), (8, 16), (9, 11), (10, 13), (11, 15)]
+
+crowdpose_coco133 = [(0, 5), (1, 6), (2, 7), (3, 8), (4, 9), (5, 10), (6, 11),
+ (7, 12), (8, 13), (9, 14), (10, 15), (11, 16)]
+
+mpii_coco133 = [
+ (0, 16),
+ (1, 14),
+ (2, 12),
+ (3, 11),
+ (4, 13),
+ (5, 15),
+ (10, 10),
+ (11, 8),
+ (12, 6),
+ (13, 5),
+ (14, 7),
+ (15, 9),
+]
+
+jhmdb_coco133 = [
+ (3, 6),
+ (4, 5),
+ (5, 12),
+ (6, 11),
+ (7, 8),
+ (8, 7),
+ (9, 14),
+ (10, 13),
+ (11, 10),
+ (12, 9),
+ (13, 16),
+ (14, 15),
+]
+
+halpe_coco133 = [(i, i)
+ for i in range(17)] + [(20, 17), (21, 20), (22, 18), (23, 21),
+ (24, 19),
+ (25, 22)] + [(i, i - 3)
+ for i in range(26, 136)]
+
+posetrack_coco133 = [
+ (0, 0),
+ (3, 3),
+ (4, 4),
+ (5, 5),
+ (6, 6),
+ (7, 7),
+ (8, 8),
+ (9, 9),
+ (10, 10),
+ (11, 11),
+ (12, 12),
+ (13, 13),
+ (14, 14),
+ (15, 15),
+ (16, 16),
+]
+
+humanart_coco133 = [(i, i) for i in range(17)] + [(17, 99), (18, 120),
+ (19, 17), (20, 20)]
+
+# train datasets
+dataset_coco = dict(
+ type=dataset_type,
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='coco/annotations/coco_wholebody_train_v1.0.json',
+ data_prefix=dict(img='detection/coco/train2017/'),
+ pipeline=[],
+)
+
+dataset_aic = dict(
+ type='AicDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='aic/annotations/aic_train.json',
+ data_prefix=dict(img='pose/ai_challenge/ai_challenger_keypoint'
+ '_train_20170902/keypoint_train_images_20170902/'),
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=aic_coco133)
+ ],
+)
+
+dataset_crowdpose = dict(
+ type='CrowdPoseDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='crowdpose/annotations/mmpose_crowdpose_trainval.json',
+ data_prefix=dict(img='pose/CrowdPose/images/'),
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=crowdpose_coco133)
+ ],
+)
+
+dataset_mpii = dict(
+ type='MpiiDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='mpii/annotations/mpii_train.json',
+ data_prefix=dict(img='pose/MPI/images/'),
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=mpii_coco133)
+ ],
+)
+
+dataset_jhmdb = dict(
+ type='JhmdbDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='jhmdb/annotations/Sub1_train.json',
+ data_prefix=dict(img='pose/JHMDB/'),
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=jhmdb_coco133)
+ ],
+)
+
+dataset_halpe = dict(
+ type='HalpeDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='halpe/annotations/halpe_train_v1.json',
+ data_prefix=dict(img='pose/Halpe/hico_20160224_det/images/train2015'),
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=halpe_coco133)
+ ],
+)
+
+dataset_posetrack = dict(
+ type='PoseTrack18Dataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='posetrack18/annotations/posetrack18_train.json',
+ data_prefix=dict(img='pose/PoseChallenge2018/'),
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=posetrack_coco133)
+ ],
+)
+
+dataset_humanart = dict(
+ type='HumanArt21Dataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='HumanArt/annotations/training_humanart.json',
+ filter_cfg=dict(scenes=['real_human']),
+ data_prefix=dict(img='pose/'),
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=humanart_coco133)
+ ])
+
+ubody_scenes = [
+ 'Magic_show', 'Entertainment', 'ConductMusic', 'Online_class', 'TalkShow',
+ 'Speech', 'Fitness', 'Interview', 'Olympic', 'TVShow', 'Singing',
+ 'SignLanguage', 'Movie', 'LiveVlog', 'VideoConference'
+]
+
+ubody_datasets = []
+for scene in ubody_scenes:
+ each = dict(
+ type='UBody2dDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file=f'Ubody/annotations/{scene}/train_annotations.json',
+ data_prefix=dict(img='pose/UBody/images/'),
+ pipeline=[],
+ sample_interval=10)
+ ubody_datasets.append(each)
+
+dataset_ubody = dict(
+ type='CombinedDataset',
+ datasets=ubody_datasets,
+ pipeline=[],
+ test_mode=False,
+)
+
+face_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale', padding=1.25),
+ dict(
+ type='RandomBBoxTransform',
+ shift_factor=0.,
+ scale_factor=[1.5, 2.0],
+ rotate_factor=0),
+]
+
+wflw_coco133 = [(i * 2, 23 + i)
+ for i in range(17)] + [(33 + i, 40 + i) for i in range(5)] + [
+ (42 + i, 45 + i) for i in range(5)
+ ] + [(51 + i, 50 + i)
+ for i in range(9)] + [(60, 59), (61, 60), (63, 61),
+ (64, 62), (65, 63), (67, 64),
+ (68, 65), (69, 66), (71, 67),
+ (72, 68), (73, 69),
+ (75, 70)] + [(76 + i, 71 + i)
+ for i in range(20)]
+dataset_wflw = dict(
+ type='WFLWDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='wflw/annotations/face_landmarks_wflw_train.json',
+ data_prefix=dict(img='pose/WFLW/images/'),
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=wflw_coco133), *face_pipeline
+ ],
+)
+
+mapping_300w_coco133 = [(i, 23 + i) for i in range(68)]
+dataset_300w = dict(
+ type='Face300WDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='300w/annotations/face_landmarks_300w_train.json',
+ data_prefix=dict(img='pose/300w/images/'),
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=mapping_300w_coco133), *face_pipeline
+ ],
+)
+
+cofw_coco133 = [(0, 40), (2, 44), (4, 42), (1, 49), (3, 45), (6, 47), (8, 59),
+ (10, 62), (9, 68), (11, 65), (18, 54), (19, 58), (20, 53),
+ (21, 56), (22, 71), (23, 77), (24, 74), (25, 85), (26, 89),
+ (27, 80), (28, 31)]
+dataset_cofw = dict(
+ type='COFWDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='cofw/annotations/cofw_train.json',
+ data_prefix=dict(img='pose/COFW/images/'),
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=cofw_coco133), *face_pipeline
+ ],
+)
+
+lapa_coco133 = [(i * 2, 23 + i) for i in range(17)] + [
+ (33 + i, 40 + i) for i in range(5)
+] + [(42 + i, 45 + i) for i in range(5)] + [
+ (51 + i, 50 + i) for i in range(4)
+] + [(58 + i, 54 + i) for i in range(5)] + [(66, 59), (67, 60), (69, 61),
+ (70, 62), (71, 63), (73, 64),
+ (75, 65), (76, 66), (78, 67),
+ (79, 68), (80, 69),
+ (82, 70)] + [(84 + i, 71 + i)
+ for i in range(20)]
+dataset_lapa = dict(
+ type='LapaDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='LaPa/annotations/lapa_trainval.json',
+ data_prefix=dict(img='pose/LaPa/'),
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=lapa_coco133), *face_pipeline
+ ],
+)
+
+dataset_wb = dict(
+ type='CombinedDataset',
+ datasets=[dataset_coco, dataset_halpe, dataset_ubody],
+ pipeline=[],
+ test_mode=False,
+)
+
+dataset_body = dict(
+ type='CombinedDataset',
+ datasets=[
+ dataset_aic,
+ dataset_crowdpose,
+ dataset_mpii,
+ dataset_jhmdb,
+ dataset_posetrack,
+ dataset_humanart,
+ ],
+ pipeline=[],
+ test_mode=False,
+)
+
+dataset_face = dict(
+ type='CombinedDataset',
+ datasets=[
+ dataset_wflw,
+ dataset_300w,
+ dataset_cofw,
+ dataset_lapa,
+ ],
+ pipeline=[],
+ test_mode=False,
+)
+
+hand_pipeline = [
+ dict(type='LoadImage', backend_args=backend_args),
+ dict(type='GetBBoxCenterScale'),
+ dict(
+ type='RandomBBoxTransform',
+ shift_factor=0.,
+ scale_factor=[1.5, 2.0],
+ rotate_factor=0),
+]
+
+interhand_left = [(21, 95), (22, 94), (23, 93), (24, 92), (25, 99), (26, 98),
+ (27, 97), (28, 96), (29, 103), (30, 102), (31, 101),
+ (32, 100), (33, 107), (34, 106), (35, 105), (36, 104),
+ (37, 111), (38, 110), (39, 109), (40, 108), (41, 91)]
+interhand_right = [(i - 21, j + 21) for i, j in interhand_left]
+interhand_coco133 = interhand_right + interhand_left
+
+dataset_interhand2d = dict(
+ type='InterHand2DDoubleDataset',
+ data_root=data_root,
+ data_mode=data_mode,
+ ann_file='interhand26m/annotations/all/InterHand2.6M_train_data.json',
+ camera_param_file='interhand26m/annotations/all/'
+ 'InterHand2.6M_train_camera.json',
+ joint_file='interhand26m/annotations/all/'
+ 'InterHand2.6M_train_joint_3d.json',
+ data_prefix=dict(img='interhand2.6m/images/train/'),
+ sample_interval=10,
+ pipeline=[
+ dict(
+ type='KeypointConverter',
+ num_keypoints=num_keypoints,
+ mapping=interhand_coco133,
+ ), *hand_pipeline
+ ],
+)
+
+dataset_hand = dict(
+ type='CombinedDataset',
+ datasets=[dataset_interhand2d],
+ pipeline=[],
+ test_mode=False,
+)
+
+train_datasets = [dataset_wb, dataset_body, dataset_face, dataset_hand]
+
+# data loaders
+train_dataloader = dict(
+ batch_size=train_batch_size,
+ num_workers=4,
+ pin_memory=False,
+ persistent_workers=True,
+ sampler=dict(type='DefaultSampler', shuffle=True),
+ dataset=dict(
+ type='CombinedDataset',
+ datasets=train_datasets,
+ pipeline=train_pipeline,
+ test_mode=False,
+ ))
+
+val_dataloader = dict(
+ batch_size=val_batch_size,
+ num_workers=4,
+ persistent_workers=True,
+ drop_last=False,
+ sampler=dict(type='DefaultSampler', shuffle=False, round_up=False),
+ dataset=dict(
+ type='CocoWholeBodyDataset',
+ ann_file='data/coco/annotations/coco_wholebody_val_v1.0.json',
+ data_prefix=dict(img='data/detection/coco/val2017/'),
+ pipeline=val_pipeline,
+ bbox_file='data/coco/person_detection_results/'
+ 'COCO_val2017_detections_AP_H_56_person.json',
+ test_mode=True))
+
+test_dataloader = val_dataloader
+
+# hooks
+default_hooks = dict(
+ checkpoint=dict(
+ save_best='coco-wholebody/AP', rule='greater', max_keep_ckpts=1))
+
+custom_hooks = [
+ dict(
+ type='EMAHook',
+ ema_type='ExpMomentumEMA',
+ momentum=0.0002,
+ update_buffers=True,
+ priority=49),
+ dict(
+ type='mmdet.PipelineSwitchHook',
+ switch_epoch=max_epochs - stage2_num_epochs,
+ switch_pipeline=train_pipeline_stage2)
+]
+
+# evaluators
+val_evaluator = dict(
+ type='CocoWholeBodyMetric',
+ ann_file='data/coco/annotations/coco_wholebody_val_v1.0.json')
+test_evaluator = val_evaluator
diff --git a/third_party/GVHMR/setup.py b/third_party/GVHMR/setup.py
new file mode 100644
index 0000000000000000000000000000000000000000..24a286894764864806968b00ddab5347a9fe6ea9
--- /dev/null
+++ b/third_party/GVHMR/setup.py
@@ -0,0 +1,11 @@
+from setuptools import setup, find_packages
+
+
+setup(
+ name="gvhmr",
+ version="1.0.0",
+ packages=find_packages(),
+ author="Zehong Shen",
+ description=["GVHMR training and inference"],
+ url="https://github.com/zju3dv/GVHMR",
+)
diff --git a/third_party/GVHMR/tools/demo/SyntheticCameraDriver.cs b/third_party/GVHMR/tools/demo/SyntheticCameraDriver.cs
new file mode 100644
index 0000000000000000000000000000000000000000..79527489a772e6ab951c83c9fa63d0638da07b30
--- /dev/null
+++ b/third_party/GVHMR/tools/demo/SyntheticCameraDriver.cs
@@ -0,0 +1,323 @@
+using UnityEngine;
+
+public class SyntheticCameraDriver : MonoBehaviour
+{
+ public enum CameraStyle
+ {
+ Fillian_Fixed = 0, // Static webcam
+ Biboo_ConeCuts = 1, // Tripod that Cuts if mesh touches bounds
+ Anny_LazySnap = 2, // Frontal cuts
+ Lapwing_HandHeld = 3 // Rigid Handheld (Vlog)
+ }
+
+ [Header("Active Style")]
+ public CameraStyle currentStyle = CameraStyle.Fillian_Fixed;
+
+ [Header("Global Settings")]
+ public float randomizeInterval = 15f;
+
+ [Header("References")]
+ public Transform characterRoot;
+ public Camera cam;
+
+ [Header("Fillian Ranges (Fixed Webcam)")]
+ public Vector2 fillianFovRange = new Vector2(55f, 65f);
+ public Vector3 fillianPosMin = new Vector3(-0.2f, 1.2f, 1.8f);
+ public Vector3 fillianPosMax = new Vector3(0.2f, 1.4f, 2.2f);
+ public Vector3 fillianRotMin = new Vector3(8f, 178f, -2f);
+ public Vector3 fillianRotMax = new Vector3(12f, 182f, 2f);
+
+ [Header("Biboo Ranges (Cone Cuts)")]
+ public Vector2 bibooFovRange = new Vector2(40f, 55f);
+ public Vector2 bibooDistRange = new Vector2(0.65f, 1.1f);
+ public Vector2 bibooConeAngleRange = new Vector2(40f, 60f);
+ public Vector2 bibooPitchRange = new Vector2(15f, 35f);
+
+ [Header("Anny Ranges (Lazy Snap)")]
+ public Vector2 annyFovRange = new Vector2(45f, 55f);
+ public Vector2 annyDistRange = new Vector2(0.7f, 1.0f);
+ public Vector2 annyHeightRange = new Vector2(-0.05f, 0.05f);
+ public float annyViewportMargin = 0.15f;
+
+ [Header("Lapwing Ranges (Vlog)")]
+ public Vector2 lapwingFovRange = new Vector2(75f, 90f);
+ public string handBoneName = "left_wrist";
+ public Vector3 lapwingOffsetMin = new Vector3(0.02f, 0.02f, -0.05f);
+ public Vector3 lapwingOffsetMax = new Vector3(0.10f, 0.10f, 0.05f);
+
+ private float _nextRandomizeTime = 0f;
+ private CameraStyle _prevStyle;
+ private float _curFov;
+ private Vector3 _curFillianPos;
+ private Quaternion _curFillianRot;
+ private float _curBibooConeAngle;
+ private float _curBibooPitch;
+ private float _curAnnyDist;
+ private float _curAnnyHeight;
+ private Vector3 _curLapwingOffset;
+ private Vector3 _snapPos;
+ private Quaternion _snapRot;
+ private bool _hasSnapped = false;
+ private float _cameraNearClip = 0.3f;
+
+ private Transform _headBone;
+ private Transform _handBoneForVlog;
+ private SkinnedMeshRenderer _mainBodyRenderer;
+
+ void Awake()
+ {
+ if (cam == null) cam = GetComponent();
+ if (cam == null) cam = Camera.main;
+ if (cam != null) _cameraNearClip = cam.nearClipPlane;
+ _prevStyle = currentStyle;
+ }
+
+ void Start()
+ {
+ FindComponents();
+ UpdateRandomness(true);
+ }
+
+ // Call this after loading a new world scene to bind the driver to that scene's camera,
+ // while keeping the scene's camera settings/post-process components intact.
+ public void BindAndInit(Camera targetCamera, Transform targetCharacterRoot)
+ {
+ cam = targetCamera;
+ characterRoot = targetCharacterRoot;
+
+ if (cam == null) cam = GetComponent();
+ if (cam != null) _cameraNearClip = cam.nearClipPlane;
+
+ FindComponents();
+ UpdateRandomness(true);
+ _prevStyle = currentStyle;
+ }
+
+ public void SetStyleFromFrameData(int styleIndex)
+ {
+ CameraStyle targetStyle = (CameraStyle)styleIndex;
+ if (currentStyle != targetStyle) currentStyle = targetStyle;
+ }
+
+ public void OnFrame(int frameIndex)
+ {
+ if (currentStyle != _prevStyle)
+ {
+ _hasSnapped = false;
+ FindComponents();
+ UpdateRandomness(true);
+ _prevStyle = currentStyle;
+ }
+
+ if (Time.time >= _nextRandomizeTime) UpdateRandomness(false);
+ if (characterRoot == null) return;
+ if (_headBone == null || _mainBodyRenderer == null) FindComponents();
+
+ cam.fieldOfView = _curFov;
+
+ switch (currentStyle)
+ {
+ case CameraStyle.Fillian_Fixed: ApplyFillian(); break;
+ case CameraStyle.Biboo_ConeCuts: ApplyBiboo(); break;
+ case CameraStyle.Anny_LazySnap: ApplyAnny(); break;
+ case CameraStyle.Lapwing_HandHeld: ApplyLapwing(); break;
+ }
+ }
+
+ void UpdateRandomness(bool force)
+ {
+ _nextRandomizeTime = Time.time + randomizeInterval;
+
+ if (currentStyle == CameraStyle.Fillian_Fixed || force)
+ {
+ _curFov = Random.Range(fillianFovRange.x, fillianFovRange.y);
+
+ // --- FIX IS HERE ---
+ // 1. Calculate the LOCAL offset first
+ Vector3 localPos = new Vector3(
+ Random.Range(fillianPosMin.x, fillianPosMax.x),
+ Random.Range(fillianPosMin.y, fillianPosMax.y),
+ Random.Range(fillianPosMin.z, fillianPosMax.z)
+ );
+
+ Vector3 localRotEuler = new Vector3(
+ Random.Range(fillianRotMin.x, fillianRotMax.x),
+ Random.Range(fillianRotMin.y, fillianRotMax.y),
+ Random.Range(fillianRotMin.z, fillianRotMax.z)
+ );
+
+ // 2. Convert to WORLD SPACE immediately using the character's CURRENT position
+ // This effectively "plants the tripod" on the ground relative to where the character spawned.
+ if (characterRoot != null)
+ {
+ _curFillianPos = characterRoot.TransformPoint(localPos);
+ _curFillianRot = characterRoot.rotation * Quaternion.Euler(localRotEuler);
+ }
+ else
+ {
+ // Fallback if no character (shouldn't happen)
+ _curFillianPos = localPos;
+ _curFillianRot = Quaternion.Euler(localRotEuler);
+ }
+ }
+
+ if (currentStyle == CameraStyle.Biboo_ConeCuts || force)
+ {
+ _curFov = Random.Range(bibooFovRange.x, bibooFovRange.y);
+ _curBibooConeAngle = Random.Range(bibooConeAngleRange.x, bibooConeAngleRange.y);
+ _curBibooPitch = Random.Range(bibooPitchRange.x, bibooPitchRange.y);
+ _hasSnapped = false;
+ }
+
+ if (currentStyle == CameraStyle.Anny_LazySnap || force)
+ {
+ _curFov = Random.Range(annyFovRange.x, annyFovRange.y);
+ _curAnnyDist = Random.Range(annyDistRange.x, annyDistRange.y);
+ _curAnnyHeight = Random.Range(annyHeightRange.x, annyHeightRange.y);
+ _hasSnapped = false;
+ }
+
+ if (currentStyle == CameraStyle.Lapwing_HandHeld || force)
+ {
+ _curFov = Random.Range(lapwingFovRange.x, lapwingFovRange.y);
+ _curLapwingOffset = new Vector3(
+ Random.Range(lapwingOffsetMin.x, lapwingOffsetMax.x),
+ Random.Range(lapwingOffsetMin.y, lapwingOffsetMax.y),
+ Random.Range(lapwingOffsetMin.z, lapwingOffsetMax.z)
+ );
+ }
+ }
+
+ // ---------------------------------------------------------
+ // RELATIVE POSITIONING LOGIC
+ // ---------------------------------------------------------
+
+ void ApplyFillian()
+ {
+ // FIX: Apply position/rotation directly in World Space.
+ // Previously: characterRoot.TransformPoint(...) made it follow the character.
+ cam.transform.position = _curFillianPos;
+ cam.transform.rotation = _curFillianRot;
+ }
+
+ void ApplyBiboo()
+ {
+ if (_headBone == null) return;
+ if (!_hasSnapped || IsHeadOffScreen() || IsCameraInsideMeshBounds(cam.transform.position))
+ SnapBiboo();
+
+ cam.transform.position = _snapPos;
+ cam.transform.LookAt(_headBone.position, Vector3.up);
+ }
+
+ void SnapBiboo()
+ {
+ Vector3 headPos = _headBone.position;
+ Vector3 rootForward = characterRoot.forward;
+
+ for (int i = 0; i < 15; i++)
+ {
+ float dist = Random.Range(bibooDistRange.x, bibooDistRange.y);
+ float yaw = Random.Range(-_curBibooConeAngle, _curBibooConeAngle);
+ float pitch = Random.Range(0f, _curBibooPitch);
+
+ Quaternion coneRot = Quaternion.Euler(pitch, yaw, 0);
+ Vector3 dir = characterRoot.rotation * (coneRot * Vector3.forward);
+ Vector3 candidate = headPos + (dir * dist);
+
+ if (!IsCameraInsideMeshBounds(candidate))
+ {
+ _snapPos = candidate;
+ _hasSnapped = true;
+ return;
+ }
+ }
+ _snapPos = headPos + (rootForward * (bibooDistRange.y + 0.2f)) + (Vector3.up * 0.1f);
+ _hasSnapped = true;
+ }
+
+ void ApplyAnny()
+ {
+ if (_headBone == null) return;
+ if (!_hasSnapped || IsHeadOffScreen() || IsCameraInsideMeshBounds(cam.transform.position))
+ SnapAnny();
+
+ cam.transform.position = _snapPos;
+ cam.transform.rotation = _snapRot;
+ }
+
+ void SnapAnny()
+ {
+ Vector3 headPos = _headBone.position;
+ Vector3 rootForward = characterRoot.forward;
+
+ // Calculate relative position using local character coordinates
+ _snapPos = headPos + (rootForward * _curAnnyDist) + (Vector3.up * _curAnnyHeight);
+ Vector3 dirToHead = (headPos - _snapPos).normalized;
+ _snapRot = (dirToHead != Vector3.zero) ? Quaternion.LookRotation(dirToHead, Vector3.up) : Quaternion.identity;
+
+ _hasSnapped = true;
+ }
+
+ void ApplyLapwing()
+ {
+ if (_handBoneForVlog == null || _headBone == null) return;
+
+ // The offset is relative to the hand's rotation
+ Vector3 worldOffset = _handBoneForVlog.rotation * _curLapwingOffset;
+ cam.transform.position = _handBoneForVlog.position + worldOffset;
+ cam.transform.LookAt(_headBone.position, Vector3.up);
+ }
+
+ public void AfterBboxComputed() { }
+
+ bool IsCameraInsideMeshBounds(Vector3 camPos)
+ {
+ if (_mainBodyRenderer == null)
+ return (_headBone.position - camPos).sqrMagnitude < (0.4f * 0.4f);
+
+ Bounds b = _mainBodyRenderer.bounds;
+ float sqrDist = b.SqrDistance(camPos);
+ float safety = _cameraNearClip + 0.15f;
+ return sqrDist < (safety * safety);
+ }
+
+ bool IsHeadOffScreen()
+ {
+ Vector3 vp = cam.WorldToViewportPoint(_headBone.position);
+ if (vp.z < 0f) return true;
+ float m = annyViewportMargin;
+ return (vp.x < m || vp.x > (1f - m) || vp.y < m || vp.y > (1f - m));
+ }
+
+ void FindComponents()
+ {
+ if (characterRoot == null) return;
+ _headBone = FindDeep(characterRoot, "head");
+
+ var allRenderers = characterRoot.GetComponentsInChildren(true);
+ int maxVerts = -1;
+ foreach(var r in allRenderers)
+ {
+ if (r.sharedMesh != null && r.sharedMesh.vertexCount > maxVerts)
+ {
+ maxVerts = r.sharedMesh.vertexCount;
+ _mainBodyRenderer = r;
+ }
+ }
+
+ if (currentStyle == CameraStyle.Lapwing_HandHeld)
+ _handBoneForVlog = FindDeep(characterRoot, handBoneName);
+ }
+
+ Transform FindDeep(Transform parent, string name)
+ {
+ if (parent.name.ToLower() == name.ToLower()) return parent;
+ foreach (Transform child in parent)
+ {
+ Transform result = FindDeep(child, name);
+ if (result != null) return result;
+ }
+ return null;
+ }
+}
diff --git a/third_party/GVHMR/tools/demo/SyntheticRecorder.cs b/third_party/GVHMR/tools/demo/SyntheticRecorder.cs
new file mode 100644
index 0000000000000000000000000000000000000000..7b11e3e52c7bc4bab10e97655f21c7db2a7b6586
--- /dev/null
+++ b/third_party/GVHMR/tools/demo/SyntheticRecorder.cs
@@ -0,0 +1,925 @@
+using UnityEngine;
+using System;
+using System.IO;
+using System.Collections;
+using System.Collections.Generic;
+using Newtonsoft.Json;
+using UnityEngine.SceneManagement;
+
+public class SyntheticRecorder : MonoBehaviour
+{
+ [System.Serializable]
+ public class AvatarConfig
+ {
+ public string avatarName = "Avatar";
+ public GameObject avatarObject;
+
+ [Header("Retargeting Link")]
+ [Tooltip("Drag the HybridPoseCopier component of THIS avatar here.")]
+ public HybridPoseCopier retargeter;
+
+ public List extraMeshes = new List();
+ public float specificPadding = 40f;
+
+ [Header("Keypoint Markers (COCO-17 order)")]
+ [Tooltip("Drag marker transforms (COCO-17 order). They should be parented to the character bones.")]
+ public List customMarkers = new List();
+ }
+
+ // --- JSON Structures ---
+ public class SequenceData { public List frames; }
+ public class FrameData { public int i; public float[] p, t, b; public int s; }
+
+ public class OutputMeta
+ {
+ public int frame_index;
+ public string image_path;
+
+ // bbox is saved as TOP-LEFT origin: [x, y, w, h]
+ public float[] bbox;
+
+ // keypoints saved as TOP-LEFT origin pixels: [x0,y0, x1,y1, ...]
+ public float[] kpts_2d;
+ public int[] kpts_vis;
+
+ // optional convenience: bbox clipped to image bounds (TOP-LEFT), safer for cropping code
+ public float[] bbox_clip;
+
+ public float[] cam_intrinsics; // [fx, fy, cx, cy]
+ public float[] cam_pos_world;
+ public float[] cam_rot_world; // quat xyzw
+
+ public float[] pelvis_pos_world;
+ public float[] pelvis_rot_world;
+
+ // --- NEW FIELDS: Clean In-Camera Rotation & Translation ---
+ public float[] smpl_incam_quat; // [x, y, z, w]
+ public float[] smpl_incam_transl; // [x, y, z] relative to camera
+ public float[] smpl_root_incam_transl; // [x, y, z] characterRoot relative to camera
+ public float smpl_root_world_scale; // characterRoot lossyScale.x (assumed uniform)
+
+ public float[] kpts_3d_world; // [x,y,z,...] in Unity world space
+
+ public float[] smplx_pose;
+ public float[] smplx_betas;
+ }
+
+ [Header("Settings")]
+ public string inputJsonPath = "Assets/StreamingAssets/capture_sequence.json";
+ public string outputFolder = "C:/Temp/SyntheticDataset";
+ public bool startRecordingOnPlay = true;
+ public bool showDebugUI = true;
+
+ [Header("Sequence Naming")]
+ public string sequenceName = "";
+
+ [Header("Rendering Layers")]
+ public LayerMask fillianLayerMask = 1;
+ public LayerMask normalLayerMask = ~0;
+
+ [Header("Occlusion Check")]
+ [Tooltip("Layers that can occlude keypoints. You typically want environment + character layers.")]
+ public LayerMask occlusionCheckLayerMask = ~0;
+
+ [Header("References")]
+ public GameObject characterRoot; // The SMPL Shadow Root
+ public Camera vtuberCamera;
+ public SyntheticCameraDriver cameraDriver;
+
+ [Header("Randomization")]
+ public List avatarList = new List();
+ [Tooltip("List of Unity scene names (as in Build Settings) to use as baked-lit 'maps'. One is loaded randomly on Start.")]
+ public List worldSceneNames = new List();
+ [Tooltip("How to load the selected world scene. Use Additive for a bootstrap scene workflow.")]
+ public LoadSceneMode worldSceneLoadMode = LoadSceneMode.Additive;
+ [Tooltip("After loading, set the selected world scene as the active scene.")]
+ public bool setLoadedWorldSceneActive = true;
+ [Tooltip("If using worldSceneNames, unload any other scenes in the list that are already loaded.")]
+ public bool unloadOtherWorldScenes = true;
+ [Tooltip("Expected world camera GameObject name (guaranteed by user).")]
+ public string worldMainCameraName = "Main Camera";
+ public string spawnPointToken = "SpawnPoint";
+
+ [Header("BBOX Accuracy")]
+ public bool useBakedSkinnedMeshForBbox = true;
+ public int bakedVertexStride = 8;
+
+ [Header("Calibration")]
+ public float movementScale = 1.0f;
+ public Vector3 translationOffset = new Vector3(0, 0.05f, 0);
+
+ // NOTE: This must match the rotation you apply in ApplyFrame.
+ public Vector3 globalCoordinateCorrection = new Vector3(-90, 180, 0);
+
+ [Header("Debug (In-Game)")]
+ public bool drawPelvisAndRootKeypoints = true;
+ public float debugKeypointSizePx = 16f;
+
+ // Private State
+ private float _activePadding = 40f;
+ private SequenceData _data;
+ private Transform[] _bones;
+ private Transform _pelvisBone;
+
+ private List _activeMarkers = new List();
+ private HybridPoseCopier _activeRetargeter;
+
+ // IMPORTANT: this is the root we use for "self hit" checks
+ private Transform _activeAvatarRoot = null;
+
+ private readonly List _activeBboxRenderers = new List();
+
+ private Mesh _bakeMesh;
+ private readonly List _bakedVerts = new List(8192);
+ private Texture2D _greenTex;
+ private Texture2D _redTex;
+ private Texture2D _blueTex;
+ private Texture2D _occTex;
+
+ private Rect _cachedBbox = new Rect(0, 0, 0, 0);
+ private bool _cachedHasBbox = false;
+ private int[] _cachedMarkerVis = null;
+
+ private const int JOINT_COUNT = 22;
+ private static readonly string[] BONE_NAMES = {
+ "pelvis", "left_hip", "right_hip", "spine1", "left_knee", "right_knee", "spine2",
+ "left_ankle", "right_ankle", "spine3", "left_foot", "right_foot", "neck", "left_collar",
+ "right_collar", "head", "left_shoulder", "right_shoulder", "left_elbow", "right_elbow",
+ "left_wrist", "right_wrist"
+ };
+
+ void Start()
+ {
+ Screen.SetResolution(1280, 720, FullScreenMode.Windowed);
+
+ _greenTex = new Texture2D(1, 1);
+ _greenTex.SetPixel(0, 0, Color.green);
+ _greenTex.Apply();
+
+ _redTex = new Texture2D(1, 1);
+ _redTex.SetPixel(0, 0, new Color(1f, 0.1f, 1f, 1f)); // magenta
+ _redTex.Apply();
+
+ _blueTex = new Texture2D(1, 1);
+ _blueTex.SetPixel(0, 0, new Color(1f, 0.92f, 0.15f, 1f)); // yellow
+ _blueTex.Apply();
+
+ _occTex = new Texture2D(1, 1);
+ _occTex.SetPixel(0, 0, Color.red);
+ _occTex.Apply();
+
+ _bakeMesh = new Mesh();
+ _bakeMesh.MarkDynamic();
+
+ RandomizeAvatarAndGatherRenderers();
+
+ // ---------------------------
+ // LEVEL LOADING (ONLY CHANGE)
+ // ---------------------------
+ if (worldSceneNames != null && worldSceneNames.Count > 0)
+ {
+ // Keep a valid camera reference while the world scene loads (prevents OnGUI null refs).
+ if (vtuberCamera == null) vtuberCamera = Camera.main;
+ StartCoroutine(LoadWorldThenContinue());
+ return;
+ }
+
+ ContinueAfterWorldLoad();
+ }
+
+ private IEnumerator LoadWorldThenContinue()
+ {
+ List candidates = new List(worldSceneNames.Count);
+ for (int i = 0; i < worldSceneNames.Count; i++)
+ {
+ string s = worldSceneNames[i];
+ if (!string.IsNullOrWhiteSpace(s)) candidates.Add(s.Trim());
+ }
+ if (candidates.Count == 0)
+ {
+ Debug.LogError("[Recorder] worldSceneNames is set but empty after filtering; cannot load world.");
+ ContinueAfterWorldLoad();
+ yield break;
+ }
+
+ int randomIndex = UnityEngine.Random.Range(0, candidates.Count);
+ string chosen = candidates[randomIndex];
+
+ if (!Application.CanStreamedLevelBeLoaded(chosen))
+ {
+ Debug.LogError($"[Recorder] World scene '{chosen}' cannot be loaded. Add it to Build Settings (or fix the name).");
+ ContinueAfterWorldLoad();
+ yield break;
+ }
+
+ if (worldSceneLoadMode == LoadSceneMode.Single)
+ {
+ DontDestroyOnLoad(gameObject);
+ if (characterRoot != null) DontDestroyOnLoad(characterRoot);
+ }
+
+ AsyncOperation op = SceneManager.LoadSceneAsync(chosen, worldSceneLoadMode);
+ if (op == null)
+ {
+ Debug.LogError($"[Recorder] Failed to start loading world scene '{chosen}'.");
+ ContinueAfterWorldLoad();
+ yield break;
+ }
+ while (!op.isDone) yield return null;
+ yield return null; // allow Awake/Start of loaded scene objects to run
+
+ Scene loaded = SceneManager.GetSceneByName(chosen);
+ if (!loaded.IsValid()) loaded = SceneManager.GetSceneByPath(chosen);
+ if (loaded.IsValid() && loaded.isLoaded)
+ {
+ if (setLoadedWorldSceneActive) SceneManager.SetActiveScene(loaded);
+
+ ApplyRandomSpawnPoint(loaded);
+ BindToWorldMainCameraOrLog(loaded);
+
+ if (unloadOtherWorldScenes && worldSceneLoadMode == LoadSceneMode.Additive)
+ UnloadOtherLoadedWorldScenes(loaded);
+ }
+ else
+ {
+ Debug.LogError($"[Recorder] World scene '{chosen}' finished loading but could not be resolved as a loaded Scene; continuing without world bindings.");
+ }
+
+ ContinueAfterWorldLoad();
+ }
+
+ private void ContinueAfterWorldLoad()
+ {
+ if (vtuberCamera == null) vtuberCamera = Camera.main;
+ if (vtuberCamera == null)
+ {
+ Debug.LogError($"[Recorder] No camera available. Expected a camera named '{worldMainCameraName}' in the loaded world scene (or a tagged MainCamera).");
+ return;
+ }
+
+ FindAndCacheBones();
+
+ if (startRecordingOnPlay)
+ StartCoroutine(RecordSequence());
+ }
+
+ private void UnloadOtherLoadedWorldScenes(Scene keep)
+ {
+ if (worldSceneNames == null || worldSceneNames.Count == 0) return;
+
+ for (int i = 0; i < worldSceneNames.Count; i++)
+ {
+ string other = worldSceneNames[i];
+ if (string.IsNullOrWhiteSpace(other)) continue;
+ other = other.Trim();
+
+ Scene otherScene = SceneManager.GetSceneByName(other);
+ if (!otherScene.IsValid()) otherScene = SceneManager.GetSceneByPath(other);
+ if (otherScene.IsValid() && otherScene.handle == keep.handle) continue;
+ if (otherScene.IsValid() && otherScene.isLoaded)
+ SceneManager.UnloadSceneAsync(otherScene);
+ }
+ }
+
+ private void BindToWorldMainCameraOrLog(Scene worldScene)
+ {
+ if (!worldScene.IsValid() || !worldScene.isLoaded)
+ {
+ Debug.LogError("[Recorder] Loaded scene is not valid/loaded; cannot bind to world camera.");
+ return;
+ }
+ if (string.IsNullOrWhiteSpace(worldMainCameraName))
+ {
+ Debug.LogError("[Recorder] worldMainCameraName is empty; cannot find world camera.");
+ return;
+ }
+
+ Camera found = null;
+ GameObject[] roots = worldScene.GetRootGameObjects();
+ for (int ri = 0; ri < roots.Length; ri++)
+ {
+ GameObject root = roots[ri];
+ if (root == null) continue;
+
+ Transform[] allChildren = root.GetComponentsInChildren(true);
+ foreach (Transform t in allChildren)
+ {
+ if (t == null) continue;
+ if (t.name != worldMainCameraName) continue;
+
+ Camera c = t.GetComponent();
+ if (c == null) continue;
+
+ found = c;
+ break;
+ }
+
+ if (found != null) break;
+ }
+
+ if (found == null)
+ {
+ Debug.LogError($"[Recorder] Could not find Camera named '{worldMainCameraName}' in scene '{worldScene.name}'.");
+ return;
+ }
+
+ if (!found.enabled) found.enabled = true;
+
+ // If the world camera is under a scaled parent, InverseTransformPoint will scale translations.
+ // Keep hierarchy intact unless scale is non-1, then unparent to ensure lossyScale==1.
+ Vector3 ls = found.transform.lossyScale;
+ if (Mathf.Abs(ls.x - 1f) > 1e-4f || Mathf.Abs(ls.y - 1f) > 1e-4f || Mathf.Abs(ls.z - 1f) > 1e-4f)
+ {
+ found.transform.SetParent(null, true);
+ found.transform.localScale = Vector3.one;
+ }
+
+ // Avoid double-rendering if the bootstrap scene has its own camera.
+ if (vtuberCamera != null && vtuberCamera != found)
+ vtuberCamera.enabled = false;
+
+ vtuberCamera = found;
+
+ // Ensure the driver is attached to this camera.
+ SyntheticCameraDriver driver = found.GetComponent();
+ if (driver == null) driver = found.gameObject.AddComponent();
+ if (characterRoot != null)
+ driver.BindAndInit(found, characterRoot.transform);
+ else
+ {
+ driver.cam = found;
+ driver.characterRoot = null;
+ }
+
+ cameraDriver = driver;
+ }
+
+ void OnGUI()
+ {
+ if (!showDebugUI) return;
+ if (vtuberCamera == null) return;
+
+ // bbox debug
+ if (_cachedHasBbox)
+ {
+ Rect r = _cachedBbox;
+ float invY = Screen.height - (r.y + r.height);
+ GUI.DrawTexture(new Rect(r.x, invY, r.width, 3), _greenTex);
+ GUI.DrawTexture(new Rect(r.x, invY + r.height, r.width, 3), _greenTex);
+ GUI.DrawTexture(new Rect(r.x, invY, 3, r.height), _greenTex);
+ GUI.DrawTexture(new Rect(r.x + r.width, invY, 3, r.height), _greenTex);
+ }
+
+ // marker debug
+ if (_activeMarkers != null)
+ {
+ for (int mi = 0; mi < _activeMarkers.Count; mi++)
+ {
+ var m = _activeMarkers[mi];
+ if (m == null) continue;
+ Vector3 sc = vtuberCamera.WorldToScreenPoint(m.position);
+ if (sc.z > 0)
+ {
+ int vis = (_cachedMarkerVis != null && mi >= 0 && mi < _cachedMarkerVis.Length) ? _cachedMarkerVis[mi] : 2;
+ Texture2D tex = (vis == 1) ? _occTex : _greenTex;
+ GUI.DrawTexture(new Rect(sc.x - 2, Screen.height - sc.y - 2, 4, 4), tex);
+ }
+ }
+ }
+
+ // pelvis/root debug (screen-space)
+ if (drawPelvisAndRootKeypoints && vtuberCamera != null)
+ {
+ float size = Mathf.Max(2f, debugKeypointSizePx);
+ float half = size * 0.5f;
+ if (_pelvisBone != null)
+ {
+ Vector3 sc = vtuberCamera.WorldToScreenPoint(_pelvisBone.position);
+ if (sc.z > 0)
+ {
+ float x = sc.x, y = Screen.height - sc.y;
+ GUI.DrawTexture(new Rect(x - half, y - half, size, size), _redTex);
+ GUI.Label(new Rect(x + half + 6, y - 10, 260, 20), "pelvis (SMPL root)");
+ }
+ }
+
+ if (characterRoot != null)
+ {
+ Vector3 sc = vtuberCamera.WorldToScreenPoint(characterRoot.transform.position);
+ if (sc.z > 0)
+ {
+ float x = sc.x, y = Screen.height - sc.y;
+ GUI.DrawTexture(new Rect(x - half, y - half, size, size), _blueTex);
+ GUI.Label(new Rect(x + half + 6, y + 2, 260, 20), "characterRoot");
+ }
+ }
+
+ // Intrinsics debug (helps diagnose "scale"/FOV mismatch)
+ float H = Screen.height;
+ Vector3 camP0_W = vtuberCamera.transform.TransformPoint(new Vector3(0f, 0f, 1f));
+ Vector3 camPx_W = vtuberCamera.transform.TransformPoint(new Vector3(1f, 0f, 1f));
+ Vector3 camPyDown_W = vtuberCamera.transform.TransformPoint(new Vector3(0f, -1f, 1f));
+ Vector3 s0 = vtuberCamera.WorldToScreenPoint(camP0_W);
+ Vector3 sx = vtuberCamera.WorldToScreenPoint(camPx_W);
+ Vector3 sy = vtuberCamera.WorldToScreenPoint(camPyDown_W);
+ float cx = s0.x;
+ float cy = H - s0.y;
+ float fx = sx.x - s0.x;
+ float fy = (H - sy.y) - cy;
+ GUI.Label(new Rect(10, 10, 520, 20), $"K: fx={fx:F1} fy={fy:F1} cx={cx:F1} cy={cy:F1}");
+ }
+ }
+
+ IEnumerator RecordSequence()
+ {
+ if (!File.Exists(inputJsonPath))
+ {
+ Debug.LogError($"[Recorder] Missing JSON: {inputJsonPath}");
+ yield break;
+ }
+
+ string seqName = string.IsNullOrEmpty(sequenceName) ? Path.GetFileNameWithoutExtension(inputJsonPath) : sequenceName;
+ _data = JsonConvert.DeserializeObject(File.ReadAllText(inputJsonPath));
+
+ if (!Directory.Exists(outputFolder)) Directory.CreateDirectory(outputFolder);
+ string seqImageDir = Path.Combine(outputFolder, "images", seqName);
+ if (!Directory.Exists(seqImageDir)) Directory.CreateDirectory(seqImageDir);
+
+ Texture2D screenTex = new Texture2D(Screen.width, Screen.height, TextureFormat.RGB24, false);
+
+ string jsonlPath = Path.Combine(outputFolder, $"sequence_{seqName}.jsonl");
+ using (var sw = new StreamWriter(jsonlPath, false))
+ {
+ for (int i = 0; i < _data.frames.Count; i++)
+ {
+ // 1) Move SMPL shadow
+ ApplyFrame(_data.frames[i]);
+
+ // 2) Move camera
+ if (cameraDriver != null) cameraDriver.OnFrame(i);
+
+ // 3) Force retarget now (remove 1-frame lag)
+ if (_activeRetargeter != null)
+ _activeRetargeter.ManualUpdatePose();
+ else
+ Debug.LogWarning("No Retargeter assigned in AvatarConfig! Poses might lag.");
+
+ // 4) Force transforms updated before we sample markers/bbox
+ Physics.SyncTransforms();
+
+ // 5) Setup rendering
+ bool isFillian = (_data.frames[i].s == 0);
+ vtuberCamera.clearFlags = CameraClearFlags.SolidColor;
+ vtuberCamera.backgroundColor = isFillian ? new Color(0, 0, 0, 0) : Color.black;
+ vtuberCamera.cullingMask = isFillian ? fillianLayerMask : normalLayerMask;
+
+ // 6) Wait for draw
+ yield return new WaitForEndOfFrame();
+
+ // 7) Capture pixels
+ if (screenTex.width != Screen.width || screenTex.height != Screen.height)
+ screenTex.Reinitialize(Screen.width, Screen.height);
+
+ screenTex.ReadPixels(new Rect(0, 0, Screen.width, Screen.height), 0, 0);
+ screenTex.Apply();
+
+ string imgFile = $"img_{i:D5}{(isFillian ? ".png" : ".jpg")}";
+ byte[] bytes = isFillian ? screenTex.EncodeToPNG() : screenTex.EncodeToJPG();
+ File.WriteAllBytes(Path.Combine(seqImageDir, imgFile), bytes);
+
+ // 8) Compute bbox
+ ComputeBoundingBoxCached();
+
+ float H = Screen.height;
+ float W = Screen.width;
+
+ // Robust intrinsics in pixel units (OpenCV-style, y-down):
+ // derive [fx, fy, cx, cy] from Unity's actual projection by sampling screen projections
+ // of known camera-local points. This stays correct even if Unity uses a custom projection.
+ Vector3 camP0_W = vtuberCamera.transform.TransformPoint(new Vector3(0f, 0f, 1f));
+ Vector3 camPx_W = vtuberCamera.transform.TransformPoint(new Vector3(1f, 0f, 1f));
+ Vector3 camPyDown_W = vtuberCamera.transform.TransformPoint(new Vector3(0f, -1f, 1f)); // y-down in CV
+
+ Vector3 s0 = vtuberCamera.WorldToScreenPoint(camP0_W); // bottom-left origin
+ Vector3 sx = vtuberCamera.WorldToScreenPoint(camPx_W); // bottom-left origin
+ Vector3 sy = vtuberCamera.WorldToScreenPoint(camPyDown_W); // bottom-left origin
+
+ float cx = s0.x;
+ float cy = H - s0.y; // convert to top-left origin
+ float fx = sx.x - s0.x;
+ float fy = (H - sy.y) - cy; // y-down, so this should be positive
+
+ if (fx <= 1e-6f || fy <= 1e-6f)
+ Debug.LogWarning($"[Recorder] Suspicious intrinsics fx={fx} fy={fy} (check camera/projection).");
+
+ Rect rFull = _cachedHasBbox ? _cachedBbox : new Rect(0, 0, 0, 0);
+
+ // Convert cached bbox (bottom-left GUI) to TOP-LEFT
+ float bbox_x = rFull.x;
+ float bbox_y = H - (rFull.y + rFull.height);
+ float bbox_w = rFull.width;
+ float bbox_h = rFull.height;
+
+ float clip_x0 = Mathf.Clamp(bbox_x, 0, W);
+ float clip_y0 = Mathf.Clamp(bbox_y, 0, H);
+ float clip_x1 = Mathf.Clamp(bbox_x + bbox_w, 0, W);
+ float clip_y1 = Mathf.Clamp(bbox_y + bbox_h, 0, H);
+ float clip_w = Mathf.Max(0, clip_x1 - clip_x0);
+ float clip_h = Mathf.Max(0, clip_y1 - clip_y0);
+
+ // Keypoints
+ int M = (_activeMarkers != null) ? _activeMarkers.Count : 0;
+ var kpts2D = new List(Mathf.Max(0, M) * 2);
+ var kptsVis = new List(Mathf.Max(0, M));
+ var kpts3D = new List(Mathf.Max(0, M) * 3);
+
+ if (_activeMarkers != null)
+ {
+ for (int mi = 0; mi < _activeMarkers.Count; mi++)
+ {
+ Transform t = _activeMarkers[mi];
+ if (t == null)
+ {
+ kpts2D.Add(0); kpts2D.Add(0);
+ kptsVis.Add(0);
+ kpts3D.Add(0); kpts3D.Add(0); kpts3D.Add(0);
+ continue;
+ }
+
+ Vector3 wPos = t.position;
+ kpts3D.Add(wPos.x); kpts3D.Add(wPos.y); kpts3D.Add(wPos.z);
+
+ Vector3 sPos = vtuberCamera.WorldToScreenPoint(wPos);
+
+ float x_px = sPos.x;
+ float y_px = H - sPos.y;
+
+ kpts2D.Add(x_px);
+ kpts2D.Add(y_px);
+
+ int vis = 0;
+ bool onScreen = (sPos.z > 0 && x_px >= 0 && x_px <= W && y_px >= 0 && y_px <= H);
+
+ if (onScreen)
+ {
+ vis = 2;
+
+ Vector3 camPos = vtuberCamera.transform.position;
+ Vector3 toPoint = (wPos - camPos);
+ float dist = toPoint.magnitude;
+
+ if (dist > 1e-5f)
+ {
+ Vector3 dirN = toPoint / dist;
+
+ // Start slightly in front of the near plane to avoid "starting inside" issues.
+ float startOffset = Mathf.Min(dist, Mathf.Max(0f, vtuberCamera.nearClipPlane) + 0.01f);
+ float endEps = 0.02f;
+ float rayLen = dist - startOffset - endEps;
+
+ if (rayLen > 1e-5f)
+ {
+ Vector3 origin = camPos + dirN * startOffset;
+ RaycastHit[] hits = Physics.RaycastAll(origin, dirN, rayLen, occlusionCheckLayerMask, QueryTriggerInteraction.UseGlobal);
+ if (hits != null && hits.Length > 0)
+ {
+ Array.Sort(hits, (a, b) => a.distance.CompareTo(b.distance));
+
+ Transform selfRoot = (_activeAvatarRoot != null)
+ ? _activeAvatarRoot
+ : ((characterRoot != null) ? characterRoot.transform : null);
+
+ for (int hi = 0; hi < hits.Length; hi++)
+ {
+ Transform ht = hits[hi].transform;
+ if (ht == null) continue;
+
+ bool isSelf = (selfRoot != null && ht.IsChildOf(selfRoot));
+ if (isSelf)
+ {
+ // Ignore collisions on the same bone chain as the marker to avoid false self-occlusion.
+ if (ht == t || ht.IsChildOf(t) || t.IsChildOf(ht))
+ continue;
+ }
+
+ // Anything else hit before the keypoint counts as occlusion (environment or self).
+ vis = 1;
+ break;
+ }
+ }
+ }
+ }
+ }
+
+ kptsVis.Add(vis);
+ }
+ }
+
+ // Cache for debug UI (0=off, 1=occluded, 2=visible)
+ _cachedMarkerVis = kptsVis.ToArray();
+
+ Transform pelvis = (_bones != null && _bones.Length > 0) ? _bones[0] : null;
+
+ // --- NEW MATH: CALCULATE CLEAN IN-CAMERA ROTATION & TRANSLATION ---
+
+ // 1. Get the exact correction used in ApplyFrame
+ Quaternion correction = Quaternion.Euler(globalCoordinateCorrection);
+
+ // 2. Get Pelvis World Rotation & Position
+ Quaternion pelvisWorld = (pelvis != null) ? pelvis.rotation : Quaternion.identity;
+ Vector3 pelvisPos = (pelvis != null) ? pelvis.position : Vector3.zero;
+
+ // 3. Remove the Correction from the Rotation
+ Quaternion rawSmplWorld = Quaternion.Inverse(correction) * pelvisWorld;
+
+ // 4. Calculate ROTATION relative to the Camera
+ Quaternion incamQ = Quaternion.Inverse(vtuberCamera.transform.rotation) * rawSmplWorld;
+
+ // 5. Calculate TRANSLATION relative to the Camera
+ // InverseTransformPoint converts World Position -> Local Position relative to the Camera
+ Vector3 incamPos = vtuberCamera.transform.InverseTransformPoint(pelvisPos);
+ Vector3 rootIncamPos = (characterRoot != null)
+ ? vtuberCamera.transform.InverseTransformPoint(characterRoot.transform.position)
+ : Vector3.zero;
+ float rootScale = (characterRoot != null) ? characterRoot.transform.lossyScale.x : 1f;
+
+ // ----------------------------------------------------
+
+ var meta = new OutputMeta
+ {
+ frame_index = i,
+ image_path = Path.Combine("images", seqName, imgFile).Replace("\\", "/"),
+
+ bbox = new float[] { bbox_x, bbox_y, bbox_w, bbox_h },
+ bbox_clip = new float[] { clip_x0, clip_y0, clip_w, clip_h },
+
+ kpts_2d = kpts2D.ToArray(),
+ kpts_vis = kptsVis.ToArray(),
+
+ cam_intrinsics = new float[] { fx, fy, cx, cy },
+ cam_pos_world = new float[] {
+ vtuberCamera.transform.position.x,
+ vtuberCamera.transform.position.y,
+ vtuberCamera.transform.position.z
+ },
+ cam_rot_world = new float[] {
+ vtuberCamera.transform.rotation.x,
+ vtuberCamera.transform.rotation.y,
+ vtuberCamera.transform.rotation.z,
+ vtuberCamera.transform.rotation.w
+ },
+
+ pelvis_pos_world = new float[] { pelvisPos.x, pelvisPos.y, pelvisPos.z },
+ pelvis_rot_world = new float[] { pelvisWorld.x, pelvisWorld.y, pelvisWorld.z, pelvisWorld.w },
+
+ // EXPORT clean quaternion
+ smpl_incam_quat = new float[] { incamQ.x, incamQ.y, incamQ.z, incamQ.w },
+
+ // EXPORT clean translation
+ smpl_incam_transl = new float[] { incamPos.x, incamPos.y, incamPos.z },
+ // EXPORT root translation (the transform you actually translate in ApplyFrame)
+ smpl_root_incam_transl = new float[] { rootIncamPos.x, rootIncamPos.y, rootIncamPos.z },
+ smpl_root_world_scale = rootScale,
+
+ kpts_3d_world = kpts3D.ToArray(),
+
+ smplx_pose = _data.frames[i].p,
+ smplx_betas = _data.frames[i].b
+ };
+
+ sw.WriteLine(JsonConvert.SerializeObject(meta));
+ }
+ }
+
+ Destroy(screenTex);
+
+#if UNITY_STANDALONE
+ Application.Quit();
+#endif
+ }
+
+ void ApplyFrame(FrameData f)
+ {
+ if (f.p == null || characterRoot == null) return;
+
+ Quaternion correction = Quaternion.Euler(globalCoordinateCorrection);
+ characterRoot.transform.localPosition = (correction * (new Vector3(-f.t[0], f.t[1], f.t[2]) * movementScale)) + translationOffset;
+
+ int floatIdx = 0;
+ for (int i = 0; i < JOINT_COUNT; i++)
+ {
+ if (floatIdx + 2 >= f.p.Length) break;
+ float x = f.p[floatIdx++], y = f.p[floatIdx++], z = f.p[floatIdx++];
+
+ float angle = Mathf.Sqrt(x * x + y * y + z * z);
+ Quaternion q = Quaternion.identity;
+
+ if (angle > 1e-6f)
+ {
+ float c = Mathf.Cos(angle * 0.5f), s = Mathf.Sin(angle * 0.5f);
+ q = new Quaternion(-(x / angle) * s, (y / angle) * s, (z / angle) * s, -c);
+ }
+
+ if (_bones != null && i < _bones.Length && _bones[i] != null)
+ _bones[i].localRotation = (i == 0) ? (correction * q) : q;
+ }
+
+ if (cameraDriver != null)
+ cameraDriver.SetStyleFromFrameData(f.s);
+ }
+
+ private void FindAndCacheBones()
+ {
+ _bones = new Transform[BONE_NAMES.Length];
+ for (int i = 0; i < BONE_NAMES.Length; i++)
+ _bones[i] = FindDeep(characterRoot.transform, BONE_NAMES[i]);
+
+ _pelvisBone = (_bones != null && _bones.Length > 0) ? _bones[0] : null;
+ if (_pelvisBone == null && characterRoot != null)
+ Debug.LogWarning("[Recorder] Could not find bone named 'pelvis' under characterRoot; pelvis exports/debug will be wrong.");
+ }
+
+ private static Transform FindDeep(Transform root, string name)
+ {
+ if (root.name == name) return root;
+ foreach (Transform child in root)
+ {
+ var res = FindDeep(child, name);
+ if (res) return res;
+ }
+ return null;
+ }
+
+ private void RandomizeAvatarAndGatherRenderers()
+ {
+ if (avatarList == null || avatarList.Count == 0) return;
+
+ _activeBboxRenderers.Clear();
+ _activeMarkers.Clear();
+ _activeRetargeter = null;
+ _activeAvatarRoot = null;
+
+ int randomIndex = UnityEngine.Random.Range(0, avatarList.Count);
+ AvatarConfig selected = avatarList[randomIndex];
+
+ for (int i = 0; i < avatarList.Count; i++)
+ {
+ if (avatarList[i].avatarObject != null)
+ avatarList[i].avatarObject.SetActive(i == randomIndex);
+ }
+
+ _activePadding = selected.specificPadding;
+ _activeRetargeter = selected.retargeter;
+
+ if (selected.avatarObject != null)
+ _activeAvatarRoot = selected.avatarObject.transform;
+
+ if (selected.customMarkers != null)
+ _activeMarkers.AddRange(selected.customMarkers);
+
+ if (characterRoot != null)
+ {
+ var smpRenderers = characterRoot.GetComponentsInChildren(true);
+ foreach (var r in smpRenderers) _activeBboxRenderers.Add(r);
+ }
+
+ foreach (GameObject extra in selected.extraMeshes)
+ {
+ if (extra == null) continue;
+ var childRends = extra.GetComponentsInChildren(true);
+ foreach (var cr in childRends)
+ if (!_activeBboxRenderers.Contains(cr)) _activeBboxRenderers.Add(cr);
+ }
+ }
+
+ private void ApplyRandomSpawnPoint(Scene worldScene)
+ {
+ if (!worldScene.IsValid() || !worldScene.isLoaded) return;
+
+ List spawns = new List();
+ GameObject[] roots = worldScene.GetRootGameObjects();
+ for (int ri = 0; ri < roots.Length; ri++)
+ {
+ GameObject root = roots[ri];
+ if (root == null) continue;
+
+ Transform[] allChildren = root.GetComponentsInChildren(true);
+ foreach (Transform child in allChildren)
+ if (child != null && child.name.Contains(spawnPointToken)) spawns.Add(child);
+ }
+
+ if (spawns.Count > 0)
+ {
+ Transform chosen = spawns[UnityEngine.Random.Range(0, spawns.Count)];
+ this.transform.position = chosen.position;
+ this.transform.rotation = chosen.rotation;
+
+ if (characterRoot != null && characterRoot != this.gameObject)
+ {
+ characterRoot.transform.localPosition = Vector3.zero;
+ characterRoot.transform.localRotation = Quaternion.identity;
+ }
+ }
+ }
+
+ private bool ComputeBoundingBoxCached()
+ {
+ _cachedHasBbox = false;
+ _cachedBbox = new Rect(0, 0, 0, 0);
+
+ if (vtuberCamera == null || _activeBboxRenderers.Count == 0) return false;
+
+ Physics.SyncTransforms();
+
+ float minVX = float.MaxValue, maxVX = float.MinValue;
+ float minVY = float.MaxValue, maxVY = float.MinValue;
+ bool foundAny = false;
+
+ int stride = Mathf.Max(1, bakedVertexStride);
+
+ foreach (var rend in _activeBboxRenderers)
+ {
+ if (rend == null) continue;
+
+ if (useBakedSkinnedMeshForBbox && rend is SkinnedMeshRenderer smr)
+ {
+ if (_bakeMesh == null) _bakeMesh = new Mesh();
+ _bakeMesh.Clear();
+
+ // Baked mesh is in the local space of the SMR, but includes
+ // deformations (which includes the scale of the bones).
+ smr.BakeMesh(_bakeMesh);
+
+ if (_bakeMesh.vertexCount == 0) continue;
+
+ _bakedVerts.Clear();
+ _bakeMesh.GetVertices(_bakedVerts);
+
+ Transform smrTransform = smr.transform;
+
+ // --- FIX STARTS HERE ---
+ // We construct a Matrix that uses the object's World Position and Rotation,
+ // but forces Scale to (1,1,1). This prevents double-scaling.
+ Matrix4x4 localToWorldNoScale = Matrix4x4.TRS(
+ smrTransform.position,
+ smrTransform.rotation,
+ Vector3.one
+ );
+ // -----------------------
+
+ for (int vi = 0; vi < _bakedVerts.Count; vi += stride)
+ {
+ // Use MultiplyPoint3x4 with our custom matrix instead of smrTransform.TransformPoint
+ Vector3 worldPos = localToWorldNoScale.MultiplyPoint3x4(_bakedVerts[vi]);
+
+ Vector3 vp = vtuberCamera.WorldToViewportPoint(worldPos);
+
+ if (vp.z <= 0f) continue;
+
+ foundAny = true;
+ if (vp.x < minVX) minVX = vp.x;
+ if (vp.x > maxVX) maxVX = vp.x;
+ if (vp.y < minVY) minVY = vp.y;
+ if (vp.y > maxVY) maxVY = vp.y;
+ }
+ }
+ else
+ {
+ // Standard Renderers use bounds, which are already in World Space
+ // and handle scale correctly automatically.
+ Bounds b = rend.bounds;
+ Vector3 c = b.center, e = b.extents;
+
+ Vector3[] corners = new Vector3[] {
+ c + new Vector3(-e.x, -e.y, -e.z), c + new Vector3(-e.x, -e.y, e.z),
+ c + new Vector3(-e.x, e.y, -e.z), c + new Vector3(-e.x, e.y, e.z),
+ c + new Vector3( e.x, -e.y, -e.z), c + new Vector3( e.x, -e.y, e.z),
+ c + new Vector3( e.x, e.y, -e.z), c + new Vector3( e.x, e.y, e.z)
+ };
+
+ for (int k = 0; k < 8; k++)
+ {
+ Vector3 vp = vtuberCamera.WorldToViewportPoint(corners[k]);
+ if (vp.z <= 0f) continue;
+
+ foundAny = true;
+ if (vp.x < minVX) minVX = vp.x;
+ if (vp.x > maxVX) maxVX = vp.x;
+ if (vp.y < minVY) minVY = vp.y;
+ if (vp.y > maxVY) maxVY = vp.y;
+ }
+ }
+ }
+
+ if (!foundAny || minVX == float.MaxValue) return false;
+
+ float minX = minVX * Screen.width - _activePadding;
+ float maxX = maxVX * Screen.width + _activePadding;
+ float minY = minVY * Screen.height - _activePadding;
+ float maxY = maxVY * Screen.height + _activePadding;
+
+ _cachedBbox = new Rect(minX, minY, maxX - minX, maxY - minY);
+ _cachedHasBbox = true;
+ return true;
+ }
+}
diff --git a/third_party/GVHMR/tools/demo/colab_demo.ipynb b/third_party/GVHMR/tools/demo/colab_demo.ipynb
new file mode 100644
index 0000000000000000000000000000000000000000..a3825e72b4a1381c488fb6ece2b488543e25f07d
--- /dev/null
+++ b/third_party/GVHMR/tools/demo/colab_demo.ipynb
@@ -0,0 +1,536 @@
+{
+ "cells": [
+ {
+ "cell_type": "markdown",
+ "metadata": {
+ "id": "Dv4XCJqiKtun"
+ },
+ "source": [
+ "GVHMR\n",
+ "\n",
+ " World-Grounded Human Motion Recovery via Gravity-View Coordinates\n",
+ "\n",
+ "\n",
+ "\n",
+ "\n",
+ "\n",
+ "Project Page\n",
+ "|\n",
+ "Paper\n",
+ "\n",
+ "\n",
+ "> World-Grounded Human Motion Recovery via Gravity-View Coordinates \n",
+ "> [Zehong Shen](https://zehongs.github.io/)\\*,\n",
+ "[Huaijin Pi](https://phj128.github.io/)\\*,\n",
+ "[Yan Xia](https://isshikihugh.github.io/scholar),\n",
+ "[Zhi Cen](https://scholar.google.com/citations?user=Xyy-uFMAAAAJ),\n",
+ "[Sida Peng](https://pengsida.net/)†,\n",
+ "[Zechen Hu](https://zju3dv.github.io/gvhmr),\n",
+ "[Hujun Bao](http://www.cad.zju.edu.cn/home/bao/),\n",
+ "[Ruizhen Hu](https://csse.szu.edu.cn/staff/ruizhenhu/),\n",
+ "[Xiaowei Zhou](https://xzhou.me/) \n",
+ "> SIGGRAPH Asia 2024\n"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {
+ "id": "zK2VFE-CEk-l"
+ },
+ "source": [
+ "## Installation\n",
+ "\n",
+ "Check [INSTALL.md](https://github.com/IsshikiHugh/GVHMR/blob/main/docs/INSTALL.md) if you want to install GVHMR in your own machine.\n",
+ "\n",
+ "> Tips: you can fold the section and run the whole installation section at once."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 1,
+ "metadata": {
+ "colab": {
+ "base_uri": "https://localhost:8080/"
+ },
+ "id": "PIPJRQZFGplJ",
+ "outputId": "44c58662-6b8e-4bda-fdb3-d796b6744bd6"
+ },
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "Sat Sep 14 17:16:45 2024 \n",
+ "+---------------------------------------------------------------------------------------+\n",
+ "| NVIDIA-SMI 535.104.05 Driver Version: 535.104.05 CUDA Version: 12.2 |\n",
+ "|-----------------------------------------+----------------------+----------------------+\n",
+ "| GPU Name Persistence-M | Bus-Id Disp.A | Volatile Uncorr. ECC |\n",
+ "| Fan Temp Perf Pwr:Usage/Cap | Memory-Usage | GPU-Util Compute M. |\n",
+ "| | | MIG M. |\n",
+ "|=========================================+======================+======================|\n",
+ "| 0 Tesla T4 Off | 00000000:00:04.0 Off | 0 |\n",
+ "| N/A 52C P8 9W / 70W | 0MiB / 15360MiB | 0% Default |\n",
+ "| | | N/A |\n",
+ "+-----------------------------------------+----------------------+----------------------+\n",
+ " \n",
+ "+---------------------------------------------------------------------------------------+\n",
+ "| Processes: |\n",
+ "| GPU GI CI PID Type Process name GPU Memory |\n",
+ "| ID ID Usage |\n",
+ "|=======================================================================================|\n",
+ "| No running processes found |\n",
+ "+---------------------------------------------------------------------------------------+\n"
+ ]
+ }
+ ],
+ "source": [
+ "# Controlling notebook is connected to NVIDIA drivers with CUDA. If this doesn't load check that GPU is selected as hardware accelerator under Edit -> Notebook settings.\n",
+ "# Google may shut down the GPU if usage has surpassed the allocation\n",
+ "\n",
+ "!nvidia-smi"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {
+ "id": "nUM1o8lsGU0n"
+ },
+ "source": [
+ "### Environment Prepration ~ 15 min\n",
+ "\n",
+ "Check [INSTALL.md#Environment](https://github.com/IsshikiHugh/GVHMR/blob/main/docs/INSTALL.md#environment) for further information."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "metadata": {
+ "id": "cKvUncvK-943"
+ },
+ "outputs": [],
+ "source": [
+ "import os\n",
+ "from pathlib import Path\n",
+ "\n",
+ "# Clone the repo.\n",
+ "!git clone https://github.com/zju3dv/GVHMR\n",
+ "proj_root = str(Path('GVHMR').absolute())\n",
+ "\n",
+ "# Install GVHMR. (If Colab asks you to restart, just click cancel and rerun the block.)\n",
+ "%cd {proj_root}\n",
+ "%pip install -r requirements.txt\n",
+ "%pip install -e ."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "metadata": {
+ "id": "vvyrlqQOIz2g"
+ },
+ "outputs": [],
+ "source": [
+ "# # Install DPVO. (Optional)\n",
+ "# %cd third-party/DPVO\n",
+ "\n",
+ "# !wget https://gitlab.com/libeigen/eigen/-/archive/3.4.0/eigen-3.4.0.zip\n",
+ "# !unzip -o eigen-3.4.0.zip -d thirdparty && rm -rf eigen-3.4.0.zip\n",
+ "\n",
+ "# %pip install torch-scatter -f \"https://data.pyg.org/whl/torch-2.3.0+cu121.html\"\n",
+ "# %pip install numba pypose\n",
+ "\n",
+ "# if 'cuda_home' not in locals():\n",
+ "# cuda_home = '/usr/local/cuda-12'\n",
+ "# if not Path(cuda_home).exists():\n",
+ "# raise FileNotFoundError('CUDA_HOME for cuda 12.x not found!')\n",
+ "\n",
+ "# os.environ['CUDA_HOME'] = cuda_home\n",
+ "# os.environ['PATH'] = os.environ['PATH'] + f':{cuda_home}/bin'\n",
+ "\n",
+ "# %pip install -e .\n",
+ "# %cd {proj_root}"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {
+ "id": "vziM8bI6E5ur"
+ },
+ "source": [
+ "### Data Prepration ~ 1 min\n",
+ "\n",
+ "Check [INSTALL.md#Inputs&Outputs](https://github.com/IsshikiHugh/GVHMR/blob/main/docs/INSTALL.md#inputs--outputs) for further information."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 11,
+ "metadata": {
+ "colab": {
+ "base_uri": "https://localhost:8080/"
+ },
+ "id": "3SYDPesR_1m9",
+ "outputId": "8658d240-9471-4779-c67e-61f9d92142dd"
+ },
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "/content/GVHMR\n",
+ "mkdir: cannot create directory ‘inputs’: File exists\n",
+ "mkdir: cannot create directory ‘outputs’: File exists\n",
+ "aria2 is already the newest version (1.36.0-1).\n",
+ "0 upgraded, 0 newly installed, 0 to remove and 49 not upgraded.\n",
+ "\n",
+ "Download Results:\n",
+ "gid |stat|avg speed |path/URI\n",
+ "======+====+===========+=======================================================\n",
+ "858436|\u001b[1;32mOK\u001b[0m | 0B/s|inputs/checkpoints/body_models/smpl/SMPL_NEUTRAL.pkl\n",
+ "\n",
+ "Status Legend:\n",
+ "(OK):download completed.\n",
+ "\n",
+ "Download Results:\n",
+ "gid |stat|avg speed |path/URI\n",
+ "======+====+===========+=======================================================\n",
+ "da93ed|\u001b[1;32mOK\u001b[0m | 0B/s|inputs/checkpoints/body_models/smplx/SMPLX_NEUTRAL.npz\n",
+ "\n",
+ "Status Legend:\n",
+ "(OK):download completed.\n",
+ "\n",
+ "Download Results:\n",
+ "gid |stat|avg speed |path/URI\n",
+ "======+====+===========+=======================================================\n",
+ "b2dfd7|\u001b[1;32mOK\u001b[0m | 0B/s|inputs/checkpoints/dpvo/dpvo.pth\n",
+ "\n",
+ "Status Legend:\n",
+ "(OK):download completed.\n",
+ "\n",
+ "Download Results:\n",
+ "gid |stat|avg speed |path/URI\n",
+ "======+====+===========+=======================================================\n",
+ "5ce060|\u001b[1;32mOK\u001b[0m | 0B/s|inputs/checkpoints/gvhmr/gvhmr_siga24_release.ckpt\n",
+ "\n",
+ "Status Legend:\n",
+ "(OK):download completed.\n",
+ "\n",
+ "Download Results:\n",
+ "gid |stat|avg speed |path/URI\n",
+ "======+====+===========+=======================================================\n",
+ "d41b3e|\u001b[1;32mOK\u001b[0m | 0B/s|inputs/checkpoints/hmr2/epoch=10-step=25000.ckpt\n",
+ "\n",
+ "Status Legend:\n",
+ "(OK):download completed.\n",
+ "\n",
+ "Download Results:\n",
+ "gid |stat|avg speed |path/URI\n",
+ "======+====+===========+=======================================================\n",
+ "3fe18e|\u001b[1;32mOK\u001b[0m | 0B/s|inputs/checkpoints/vitpose/vitpose-h-multi-coco.pth\n",
+ "\n",
+ "Status Legend:\n",
+ "(OK):download completed.\n",
+ "\n",
+ "Download Results:\n",
+ "gid |stat|avg speed |path/URI\n",
+ "======+====+===========+=======================================================\n",
+ "4fe1eb|\u001b[1;32mOK\u001b[0m | 0B/s|inputs/checkpoints/yolo/yolov8x.pt\n",
+ "\n",
+ "Status Legend:\n",
+ "(OK):download completed.\n"
+ ]
+ }
+ ],
+ "source": [
+ "%cd {proj_root}\n",
+ "!mkdir inputs\n",
+ "!mkdir outputs\n",
+ "\n",
+ "!apt install -y -qq aria2\n",
+ "!aria2c --console-log-level=error -c -x 16 -s 16 -k 1M https://huggingface.co/camenduru/SMPLer-X/resolve/main/SMPL_NEUTRAL.pkl -d inputs/checkpoints/body_models/smpl -o SMPL_NEUTRAL.pkl\n",
+ "!aria2c --console-log-level=error -c -x 16 -s 16 -k 1M https://huggingface.co/camenduru/SMPLer-X/resolve/main/SMPLX_NEUTRAL.npz -d inputs/checkpoints/body_models/smplx -o SMPLX_NEUTRAL.npz\n",
+ "!aria2c --console-log-level=error -c -x 16 -s 16 -k 1M https://huggingface.co/camenduru/GVHMR/resolve/main/dpvo/dpvo.pth -d inputs/checkpoints/dpvo -o dpvo.pth\n",
+ "!aria2c --console-log-level=error -c -x 16 -s 16 -k 1M https://huggingface.co/camenduru/GVHMR/resolve/main/gvhmr/gvhmr_siga24_release.ckpt -d inputs/checkpoints/gvhmr -o gvhmr_siga24_release.ckpt\n",
+ "!aria2c --console-log-level=error -c -x 16 -s 16 -k 1M https://huggingface.co/camenduru/GVHMR/resolve/main/hmr2/epoch%3D10-step%3D25000.ckpt -d inputs/checkpoints/hmr2 -o epoch=10-step=25000.ckpt\n",
+ "!aria2c --console-log-level=error -c -x 16 -s 16 -k 1M https://huggingface.co/camenduru/GVHMR/resolve/main/vitpose/vitpose-h-multi-coco.pth -d inputs/checkpoints/vitpose -o vitpose-h-multi-coco.pth\n",
+ "!aria2c --console-log-level=error -c -x 16 -s 16 -k 1M https://huggingface.co/camenduru/GVHMR/resolve/main/yolo/yolov8x.pt -d inputs/checkpoints/yolo -o yolov8x.pt"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {
+ "id": "IxHWylQ-KJdG"
+ },
+ "source": [
+ "## Run Demo\n",
+ "\n",
+ "Use `-s` to skip visual odometry if you know the camera is static, otherwise the camera will be estimated by DPVO.\n",
+ "\n",
+ "We also provide a script demo_folder.py to inference a entire folder.\n",
+ "\n",
+ "\n",
+ "```shell\n",
+ "python tools/demo/demo.py --video=docs/example_video/tennis.mp4 -s\n",
+ "python tools/demo/demo_folder.py -f inputs/demo/folder_in -d outputs/demo/folder_out -s\n",
+ "\n",
+ "```"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 35,
+ "metadata": {
+ "id": "YHaW4uybOQgG"
+ },
+ "outputs": [],
+ "source": [
+ "import io\n",
+ "import base64\n",
+ "from IPython.display import HTML\n",
+ "from hmr4d.utils.video_io_utils import get_video_lwh\n",
+ "\n",
+ "def display_video(fn):\n",
+ " L, W, H = get_video_lwh(fn)\n",
+ " scale = min(W, 1080) / W\n",
+ " W, H = int(W * scale), int(H * scale)\n",
+ " video_encoded = base64.b64encode(io.open(fn, 'rb').read())\n",
+ " return HTML(data=''''''.format(W, H, video_encoded.decode('ascii')))"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {
+ "id": "3EOGJBa_OCWu"
+ },
+ "source": [
+ "### Demo Tennis"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 12,
+ "metadata": {
+ "colab": {
+ "base_uri": "https://localhost:8080/"
+ },
+ "id": "NQIVoNKMKlhb",
+ "outputId": "e125532f-f095-46b2-dd70-11f227b9307d"
+ },
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "[\u001b[36m09/14 17:37:05\u001b[0m][\u001b[32mINFO\u001b[0m] [Input]: /content/GVHMR/docs/example_video/tennis.mp4\u001b[0m\n",
+ "[\u001b[36m09/14 17:37:05\u001b[0m][\u001b[32mINFO\u001b[0m] (L, W, H) = (312, 812, 720)\u001b[0m\n",
+ "[\u001b[36m09/14 17:37:06\u001b[0m][\u001b[32mINFO\u001b[0m] [Output Dir]: outputs/demo/tennis\u001b[0m\n",
+ "[\u001b[36m09/14 17:37:06\u001b[0m][\u001b[32mINFO\u001b[0m] [Copy Video] /content/GVHMR/docs/example_video/tennis.mp4 -> outputs/demo/tennis/0_input_video.mp4\u001b[0m\n",
+ "Copy: 100% 312/312 [00:09<00:00, 33.08it/s]\n",
+ "[\u001b[36m09/14 17:37:17\u001b[0m][\u001b[32mINFO\u001b[0m] [GPU]: Tesla T4\u001b[0m\n",
+ "[\u001b[36m09/14 17:37:17\u001b[0m][\u001b[32mINFO\u001b[0m] [GPU]: _CudaDeviceProperties(name='Tesla T4', major=7, minor=5, total_memory=15102MB, multi_processor_count=40)\u001b[0m\n",
+ "[\u001b[36m09/14 17:37:17\u001b[0m][\u001b[32mINFO\u001b[0m] [Preprocess] Start!\u001b[0m\n",
+ "YoloV8 Tracking: 100% 312/312 [00:27<00:00, 11.14it/s]\n",
+ "ViTPose: 100% 20/20 [00:42<00:00, 2.13s/it]\n",
+ "HMR2 Feature: 100% 20/20 [00:22<00:00, 1.11s/it]\n",
+ "[\u001b[36m09/14 17:39:36\u001b[0m][\u001b[32mINFO\u001b[0m] [Preprocess] End. Time elapsed: 139.28s\u001b[0m\n",
+ "[\u001b[36m09/14 17:39:36\u001b[0m][\u001b[32mINFO\u001b[0m] [HMR4D] Predicting\u001b[0m\n",
+ "[\u001b[36m09/14 17:39:36\u001b[0m][\u001b[32mINFO\u001b[0m] [EnDecoder] Use MM_V1_AMASS_LOCAL_BEDLAM_CAM for statistics!\u001b[0m\n",
+ "[\u001b[36m09/14 17:39:38\u001b[0m][\u001b[32mINFO\u001b[0m] [PL-Trainer] Loading ckpt type: inputs/checkpoints/gvhmr/gvhmr_siga24_release.ckpt\u001b[0m\n",
+ "[\u001b[36m09/14 17:39:40\u001b[0m][\u001b[32mINFO\u001b[0m] [HMR4D] Elapsed: 1.32s for data-length=10.4s\u001b[0m\n",
+ "Rendering Incam: 100% 312/312 [00:24<00:00, 12.71it/s]\n",
+ "Rendering Global: 100% 312/312 [00:16<00:00, 18.68it/s]\n",
+ "[\u001b[36m09/14 17:40:29\u001b[0m][\u001b[32mINFO\u001b[0m] [Merge Videos]\u001b[0m\n"
+ ]
+ }
+ ],
+ "source": [
+ "# Run demo.\n",
+ "video_fn = f'{proj_root}/docs/example_video/tennis.mp4'\n",
+ "!python {proj_root}/tools/demo/demo.py --video={video_fn} -s"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 36,
+ "metadata": {
+ "colab": {
+ "base_uri": "https://localhost:8080/",
+ "height": 499
+ },
+ "id": "wPam0XZfQLL5",
+ "outputId": "02b63472-4539-42a1-b8fd-c9208f3ded72"
+ },
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ ""
+ ],
+ "text/plain": [
+ ""
+ ]
+ },
+ "execution_count": 36,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "source": [
+ "# Visualize the result.\n",
+ "display_video('outputs/demo/tennis/tennis_3_incam_global_horiz.mp4')"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {
+ "id": "CnOwTAfyRo57"
+ },
+ "source": [
+ "### Custom Demo\n",
+ "\n",
+ "In order to run your own video, you need to follow instructions below:\n",
+ "\n",
+ "0. Run the code block below to initialize `demo_google_drive_video()`.\n",
+ "1. Upload your video on Google drive and set accessbility as \"Anyone with link (Viewer)\".\n",
+ "2. Copy the link and extract the ID. The link should follow this pattern: `https://drive.google.com/file/d//view?usp=sharing`.\n",
+ "3. Call `demo_google_drive_video()` with the URL, and pass `True` if the camera is static.\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 31,
+ "metadata": {
+ "colab": {
+ "base_uri": "https://localhost:8080/"
+ },
+ "id": "jqOda6ONRm_C",
+ "outputId": "93672ab6-a490-407c-d6d9-a8174be9bb5a"
+ },
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "/content/GVHMR\n"
+ ]
+ }
+ ],
+ "source": [
+ "%cd {proj_root}\n",
+ "!mkdir -p inputs/demo\n",
+ "\n",
+ "def demo_google_drive_video(url:str, static_camera:bool):\n",
+ " ''' URL should be like https://drive.google.com/file/d/xxxxxxxx/view?usp=drive_link '''\n",
+ "\n",
+ " print(f'[1/3] 📥 Downloading video...')\n",
+ " gdid = url.split('/')[5]\n",
+ " video_name = f'custom_{gdid}'\n",
+ " download_url = f'\\'https://drive.google.com/uc?id={gdid}&export=download&confirm=t\\''\n",
+ " !gdown {download_url} -O inputs/demo/{video_name}.mp4\n",
+ "\n",
+ " print(f'[2/3] 💃 Start running GVHMR...')\n",
+ " flag = '-s' if static_camera else ''\n",
+ " !python {proj_root}/tools/demo/demo.py --video=inputs/demo/{video_name}.mp4 {flag}\n",
+ "\n",
+ " print(f'[3/3] 📺 Displaying result...')\n",
+ " return display_video(f'outputs/demo/{video_name}/{video_name}_3_incam_global_horiz.mp4')"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 39,
+ "metadata": {
+ "colab": {
+ "base_uri": "https://localhost:8080/",
+ "height": 1000
+ },
+ "id": "ElCtnGFQVHU1",
+ "outputId": "d957a268-8648-4f2e-c360-c82ee56a7b25"
+ },
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "[1/3] 📥 Downloading video...\n",
+ "Downloading...\n",
+ "From: https://drive.google.com/uc?id=1KkTsoAuj9yq5JCJ-qg4_MnNBn56wR47F&export=download&confirm=t\n",
+ "To: /content/GVHMR/inputs/demo/custom_1KkTsoAuj9yq5JCJ-qg4_MnNBn56wR47F.mp4\n",
+ "100% 936k/936k [00:00<00:00, 9.14MB/s]\n",
+ "[2/3] 💃 Start running GVHMR...\n",
+ "[\u001b[36m09/14 18:09:12\u001b[0m][\u001b[32mINFO\u001b[0m] [Input]: inputs/demo/custom_1KkTsoAuj9yq5JCJ-qg4_MnNBn56wR47F.mp4\u001b[0m\n",
+ "[\u001b[36m09/14 18:09:12\u001b[0m][\u001b[32mINFO\u001b[0m] (L, W, H) = (367, 682, 666)\u001b[0m\n",
+ "[\u001b[36m09/14 18:09:13\u001b[0m][\u001b[32mINFO\u001b[0m] [Output Dir]: outputs/demo/custom_1KkTsoAuj9yq5JCJ-qg4_MnNBn56wR47F\u001b[0m\n",
+ "[\u001b[36m09/14 18:09:13\u001b[0m][\u001b[32mINFO\u001b[0m] [Copy Video] inputs/demo/custom_1KkTsoAuj9yq5JCJ-qg4_MnNBn56wR47F.mp4 -> outputs/demo/custom_1KkTsoAuj9yq5JCJ-qg4_MnNBn56wR47F/0_input_video.mp4\u001b[0m\n",
+ "Copy: 100% 367/367 [00:09<00:00, 39.06it/s]\n",
+ "[\u001b[36m09/14 18:09:23\u001b[0m][\u001b[32mINFO\u001b[0m] [GPU]: Tesla T4\u001b[0m\n",
+ "[\u001b[36m09/14 18:09:23\u001b[0m][\u001b[32mINFO\u001b[0m] [GPU]: _CudaDeviceProperties(name='Tesla T4', major=7, minor=5, total_memory=15102MB, multi_processor_count=40)\u001b[0m\n",
+ "[\u001b[36m09/14 18:09:23\u001b[0m][\u001b[32mINFO\u001b[0m] [Preprocess] Start!\u001b[0m\n",
+ "YoloV8 Tracking: 100% 367/367 [00:29<00:00, 12.65it/s]\n",
+ "ViTPose: 100% 23/23 [00:52<00:00, 2.27s/it]\n",
+ "HMR2 Feature: 100% 23/23 [00:25<00:00, 1.12s/it]\n",
+ "[\u001b[36m09/14 18:11:57\u001b[0m][\u001b[32mINFO\u001b[0m] [Preprocess] End. Time elapsed: 154.35s\u001b[0m\n",
+ "[\u001b[36m09/14 18:11:57\u001b[0m][\u001b[32mINFO\u001b[0m] [HMR4D] Predicting\u001b[0m\n",
+ "[\u001b[36m09/14 18:11:57\u001b[0m][\u001b[32mINFO\u001b[0m] [EnDecoder] Use MM_V1_AMASS_LOCAL_BEDLAM_CAM for statistics!\u001b[0m\n",
+ "[\u001b[36m09/14 18:11:59\u001b[0m][\u001b[32mINFO\u001b[0m] [PL-Trainer] Loading ckpt type: inputs/checkpoints/gvhmr/gvhmr_siga24_release.ckpt\u001b[0m\n",
+ "[\u001b[36m09/14 18:12:02\u001b[0m][\u001b[32mINFO\u001b[0m] [HMR4D] Elapsed: 1.88s for data-length=12.2s\u001b[0m\n",
+ "Rendering Incam: 100% 367/367 [00:21<00:00, 16.96it/s]\n",
+ "Rendering Global: 100% 367/367 [00:17<00:00, 21.44it/s]\n",
+ "[\u001b[36m09/14 18:12:47\u001b[0m][\u001b[32mINFO\u001b[0m] [Merge Videos]\u001b[0m\n",
+ "[3/3] 📺 Displaying result...\n"
+ ]
+ },
+ {
+ "data": {
+ "text/html": [
+ ""
+ ],
+ "text/plain": [
+ ""
+ ]
+ },
+ "execution_count": 39,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "source": [
+ "demo_google_drive_video(url='https://drive.google.com/file/d/1KkTsoAuj9yq5JCJ-qg4_MnNBn56wR47F/view?usp=drive_link', static_camera=True)"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "metadata": {
+ "id": "irhhFd76pH5c"
+ },
+ "outputs": [],
+ "source": [
+ "# Try it yourself!\n",
+ "demo_google_drive_video(...)"
+ ]
+ }
+ ],
+ "metadata": {
+ "accelerator": "GPU",
+ "colab": {
+ "gpuType": "T4",
+ "provenance": []
+ },
+ "kernelspec": {
+ "display_name": "Python 3",
+ "name": "python3"
+ },
+ "language_info": {
+ "name": "python"
+ }
+ },
+ "nbformat": 4,
+ "nbformat_minor": 0
+}
diff --git a/third_party/GVHMR/tools/demo/demo.py b/third_party/GVHMR/tools/demo/demo.py
new file mode 100644
index 0000000000000000000000000000000000000000..33da18764808b8ce36ebb21aad52277e718cdb21
--- /dev/null
+++ b/third_party/GVHMR/tools/demo/demo.py
@@ -0,0 +1,409 @@
+import cv2
+import torch
+import pytorch_lightning as pl
+import numpy as np
+import argparse
+from hmr4d.utils.pylogger import Log
+import hydra
+from hydra import initialize_config_module, compose
+from pathlib import Path
+from pytorch3d.transforms import quaternion_to_matrix, matrix_to_axis_angle
+
+from hmr4d.configs import register_store_gvhmr
+from hmr4d.utils.video_io_utils import (
+ get_video_lwh,
+ read_video_np,
+ save_video,
+ merge_videos_horizontal,
+ get_writer,
+ get_video_reader,
+)
+from hmr4d.utils.vis.cv2_utils import draw_bbx_xyxy_on_image_batch, draw_coco17_skeleton_batch
+
+from hmr4d.utils.preproc import Tracker, Extractor, VitPoseExtractor, SimpleVO
+
+from hmr4d.utils.geo.hmr_cam import get_bbx_xys_from_xyxy, estimate_K, convert_K_to_K4, create_camera_sensor
+from hmr4d.utils.geo_transform import compute_cam_angvel
+from hmr4d.model.gvhmr.gvhmr_pl_demo import DemoPL
+from hmr4d.utils.net_utils import detach_to_cpu, to_cuda
+from hmr4d.utils.smplx_utils import make_smplx
+from hmr4d.utils.vis.renderer import Renderer, get_global_cameras_static, get_ground_params_from_points
+from tqdm import tqdm
+from hmr4d.utils.geo_transform import apply_T_on_points, compute_T_ayfz2ay
+from einops import einsum, rearrange
+
+
+CRF = 23 # 17 is lossless, every +6 halves the mp4 size
+
+
+def parse_args_to_cfg():
+ from hydra.core.global_hydra import GlobalHydra
+ GlobalHydra.instance().clear() # Clear before HMR4D / repeated runs
+
+ # Put all args to cfg
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--video", type=str, default="inputs/demo/dance_3.mp4")
+ parser.add_argument("--output_root", type=str, default=None, help="by default to outputs/demo")
+ parser.add_argument("-s", "--static_cam", action="store_true", help="If true, skip DPVO")
+ parser.add_argument("--use_dpvo", action="store_true", help="If true, use DPVO. By default not using DPVO.")
+ parser.add_argument(
+ "--f_mm",
+ type=int,
+ default=None,
+ help="Focal length of fullframe camera in mm. Leave it as None to use default values."
+ "For iPhone 15p, the [0.5x, 1x, 2x, 3x] lens have typical values [13, 24, 48, 77]."
+ "If the camera zoom in a lot, you can try 135, 200 or even larger values.",
+ )
+ parser.add_argument("--verbose", action="store_true", help="If true, draw intermediate results")
+ args = parser.parse_args()
+
+ # Input
+ video_path = Path(args.video)
+ assert video_path.exists(), f"Video not found at {video_path}"
+ length, width, height = get_video_lwh(video_path)
+ Log.info(f"[Input]: {video_path}")
+ Log.info(f"(L, W, H) = ({length}, {width}, {height})")
+ # Cfg
+ if GlobalHydra.instance().is_initialized():
+ GlobalHydra.instance().clear()
+ with initialize_config_module(version_base="1.3", config_module=f"hmr4d.configs"):
+ overrides = [
+ f"video_name={video_path.stem}",
+ f"static_cam={args.static_cam}",
+ f"verbose={args.verbose}",
+ f"use_dpvo={args.use_dpvo}",
+ ]
+ if args.f_mm is not None:
+ overrides.append(f"f_mm={args.f_mm}")
+
+ # Allow to change output root
+ if args.output_root is not None:
+ overrides.append(f"output_root={args.output_root}")
+ register_store_gvhmr()
+ cfg = compose(config_name="demo", overrides=overrides)
+
+ GlobalHydra.instance().clear()
+
+ # Output
+ Log.info(f"[Output Dir]: {cfg.output_dir}")
+ Path(cfg.output_dir).mkdir(parents=True, exist_ok=True)
+ Path(cfg.preprocess_dir).mkdir(parents=True, exist_ok=True)
+
+ # Copy raw-input-video to video_path
+ Log.info(f"[Copy Video] {video_path} -> {cfg.video_path}")
+ if not Path(cfg.video_path).exists() or get_video_lwh(video_path)[0] != get_video_lwh(cfg.video_path)[0]:
+ reader = get_video_reader(video_path)
+ writer = get_writer(cfg.video_path, fps=30, crf=CRF)
+ for img in tqdm(reader, total=get_video_lwh(video_path)[0], desc=f"Copy"):
+ writer.write_frame(img)
+ writer.close()
+ reader.close()
+
+ return cfg
+
+
+@torch.no_grad()
+def run_preprocess(cfg):
+ Log.info(f"[Preprocess] Start!")
+ tic = Log.time()
+ video_path = cfg.video_path
+ paths = cfg.paths
+ static_cam = cfg.static_cam
+ verbose = cfg.verbose
+
+ # Get bbx tracking result
+ if not Path(paths.bbx).exists():
+ tracker = Tracker()
+ bbx_xyxy = tracker.get_one_track(video_path).float() # (L, 4)
+ bbx_xys = get_bbx_xys_from_xyxy(bbx_xyxy, base_enlarge=1.2).float() # (L, 3) apply aspect ratio and enlarge
+ torch.save({"bbx_xyxy": bbx_xyxy, "bbx_xys": bbx_xys}, paths.bbx)
+ del tracker
+ else:
+ bbx_xys = torch.load(paths.bbx)["bbx_xys"]
+ Log.info(f"[Preprocess] bbx (xyxy, xys) from {paths.bbx}")
+ if verbose:
+ video = read_video_np(video_path)
+ bbx_xyxy = torch.load(paths.bbx)["bbx_xyxy"]
+ video_overlay = draw_bbx_xyxy_on_image_batch(bbx_xyxy, video)
+ save_video(video_overlay, cfg.paths.bbx_xyxy_video_overlay)
+
+ # Get VitPose
+ if not Path(paths.vitpose).exists():
+ vitpose_extractor = VitPoseExtractor()
+ vitpose = vitpose_extractor.extract(video_path, bbx_xys)
+ torch.save(vitpose, paths.vitpose)
+ del vitpose_extractor
+ else:
+ vitpose = torch.load(paths.vitpose)
+ Log.info(f"[Preprocess] vitpose from {paths.vitpose}")
+ if verbose:
+ video = read_video_np(video_path)
+ video_overlay = draw_coco17_skeleton_batch(video, vitpose, 0.5)
+ save_video(video_overlay, paths.vitpose_video_overlay)
+
+ # Get vit features
+ if not Path(paths.vit_features).exists():
+ extractor = Extractor()
+ vit_features = extractor.extract_video_features(video_path, bbx_xys)
+ torch.save(vit_features, paths.vit_features)
+ del extractor
+ else:
+ Log.info(f"[Preprocess] vit_features from {paths.vit_features}")
+
+ # Get visual odometry results
+ if not static_cam: # use slam to get cam rotation
+ if not Path(paths.slam).exists():
+ if not cfg.use_dpvo:
+ simple_vo = SimpleVO(cfg.video_path, scale=0.5, step=8, method="sift", f_mm=cfg.f_mm)
+ vo_results = simple_vo.compute() # (L, 4, 4), numpy
+ torch.save(vo_results, paths.slam)
+ else: # DPVO
+ from hmr4d.utils.preproc.slam import SLAMModel
+
+ length, width, height = get_video_lwh(cfg.video_path)
+ K_fullimg = estimate_K(width, height)
+ intrinsics = convert_K_to_K4(K_fullimg)
+ slam = SLAMModel(video_path, width, height, intrinsics, buffer=4000, resize=0.5)
+ bar = tqdm(total=length, desc="DPVO")
+ while True:
+ ret = slam.track()
+ if ret:
+ bar.update()
+ else:
+ break
+ slam_results = slam.process() # (L, 7), numpy
+ torch.save(slam_results, paths.slam)
+ else:
+ Log.info(f"[Preprocess] slam results from {paths.slam}")
+
+ Log.info(f"[Preprocess] End. Time elapsed: {Log.time()-tic:.2f}s")
+
+
+def load_data_dict(cfg):
+ paths = cfg.paths
+ length, width, height = get_video_lwh(cfg.video_path)
+ if cfg.static_cam:
+ R_w2c = torch.eye(3).repeat(length, 1, 1)
+ else:
+ traj = torch.load(cfg.paths.slam)
+ if cfg.use_dpvo: # DPVO
+ traj_quat = torch.from_numpy(traj[:, [6, 3, 4, 5]])
+ R_w2c = quaternion_to_matrix(traj_quat).mT
+ else: # SimpleVO
+ R_w2c = torch.from_numpy(traj[:, :3, :3])
+ if cfg.f_mm is not None:
+ K_fullimg = create_camera_sensor(width, height, cfg.f_mm)[2].repeat(length, 1, 1)
+ else:
+ K_fullimg = estimate_K(width, height).repeat(length, 1, 1)
+
+ data = {
+ "length": torch.tensor(length),
+ "bbx_xys": torch.load(paths.bbx)["bbx_xys"],
+ "kp2d": torch.load(paths.vitpose),
+ "K_fullimg": K_fullimg,
+ "cam_angvel": compute_cam_angvel(R_w2c),
+ "f_imgseq": torch.load(paths.vit_features),
+ }
+ return data
+
+
+def render_incam(cfg):
+ incam_video_path = Path(cfg.paths.incam_video)
+ if incam_video_path.exists():
+ Log.info(f"[Render Incam] Video already exists at {incam_video_path}")
+ return
+
+ pred = torch.load(cfg.paths.hmr4d_results)
+ smplx = make_smplx("supermotion").cuda()
+ smplx2smpl = torch.load("hmr4d/utils/body_model/smplx2smpl_sparse.pt").cuda()
+ faces_smpl = make_smplx("smpl").faces
+
+ # smpl
+ smplx_out = smplx(**to_cuda(pred["smpl_params_incam"]))
+ pred_c_verts = torch.stack([torch.matmul(smplx2smpl, v_) for v_ in smplx_out.vertices])
+
+ # -- rendering code -- #
+ video_path = cfg.video_path
+ length, width, height = get_video_lwh(video_path)
+ K = pred["K_fullimg"][0]
+
+ # renderer
+ renderer = Renderer(width, height, device="cuda", faces=faces_smpl, K=K)
+ reader = get_video_reader(video_path) # (F, H, W, 3), uint8, numpy
+ bbx_xys_render = torch.load(cfg.paths.bbx)["bbx_xys"]
+
+ # -- render mesh -- #
+ verts_incam = pred_c_verts
+ writer = get_writer(incam_video_path, fps=30, crf=CRF)
+ for i, img_raw in tqdm(enumerate(reader), total=get_video_lwh(video_path)[0], desc=f"Rendering Incam"):
+ img = renderer.render_mesh(verts_incam[i].cuda(), img_raw, [0.8, 0.8, 0.8])
+
+ # # bbx
+ # bbx_xys_ = bbx_xys_render[i].cpu().numpy()
+ # lu_point = (bbx_xys_[:2] - bbx_xys_[2:] / 2).astype(int)
+ # rd_point = (bbx_xys_[:2] + bbx_xys_[2:] / 2).astype(int)
+ # img = cv2.rectangle(img, lu_point, rd_point, (255, 178, 102), 2)
+
+ writer.write_frame(img)
+ writer.close()
+ reader.close()
+
+
+def render_global(cfg):
+ global_video_path = Path(cfg.paths.global_video)
+ if global_video_path.exists():
+ Log.info(f"[Render Global] Video already exists at {global_video_path}")
+ return
+
+ debug_cam = False
+ pred = torch.load(cfg.paths.hmr4d_results)
+ smplx = make_smplx("supermotion").cuda()
+ smplx2smpl = torch.load("hmr4d/utils/body_model/smplx2smpl_sparse.pt").cuda()
+ faces_smpl = make_smplx("smpl").faces
+ J_regressor = torch.load("hmr4d/utils/body_model/smpl_neutral_J_regressor.pt").cuda()
+
+ # smpl
+ smplx_out = smplx(**to_cuda(pred["smpl_params_global"]))
+ pred_ay_verts = torch.stack([torch.matmul(smplx2smpl, v_) for v_ in smplx_out.vertices])
+
+ def move_to_start_point_face_z(verts):
+ "XZ to origin, Start from the ground, Face-Z"
+ # position
+ verts = verts.clone() # (L, V, 3)
+ offset = einsum(J_regressor, verts[0], "j v, v i -> j i")[0] # (3)
+ offset[1] = verts[:, :, [1]].min()
+ verts = verts - offset
+ # face direction
+ T_ay2ayfz = compute_T_ayfz2ay(einsum(J_regressor, verts[[0]], "j v, l v i -> l j i"), inverse=True)
+ verts = apply_T_on_points(verts, T_ay2ayfz)
+ return verts
+
+ verts_glob = move_to_start_point_face_z(pred_ay_verts)
+ joints_glob = einsum(J_regressor, verts_glob, "j v, l v i -> l j i") # (L, J, 3)
+ global_R, global_T, global_lights = get_global_cameras_static(
+ verts_glob.cpu(),
+ beta=2.0,
+ cam_height_degree=20,
+ target_center_height=1.0,
+ )
+
+ # -- rendering code -- #
+ video_path = cfg.video_path
+ length, width, height = get_video_lwh(video_path)
+ _, _, K = create_camera_sensor(width, height, 24) # render as 24mm lens
+
+ # renderer
+ renderer = Renderer(width, height, device="cuda", faces=faces_smpl, K=K)
+ # renderer = Renderer(width, height, device="cuda", faces=faces_smpl, K=K, bin_size=0)
+
+ # -- render mesh -- #
+ scale, cx, cz = get_ground_params_from_points(joints_glob[:, 0], verts_glob)
+ renderer.set_ground(scale * 1.5, cx, cz)
+ color = torch.ones(3).float().cuda() * 0.8
+
+ render_length = length if not debug_cam else 8
+ writer = get_writer(global_video_path, fps=30, crf=CRF)
+ for i in tqdm(range(render_length), desc=f"Rendering Global"):
+ cameras = renderer.create_camera(global_R[i], global_T[i])
+ img = renderer.render_with_ground(verts_glob[[i]], color[None], cameras, global_lights)
+ writer.write_frame(img)
+ writer.close()
+
+
+def save_output_npz(cfg, pred):
+ """
+ Saves the HMR4D prediction in the specific SMPL-X NPZ format requested:
+ - Root and Body poses in Axis-Angle.
+ - Concatenated and padded with 99 zeros.
+ """
+ output_npz_path = Path(cfg.output_dir) / f"{cfg.video_name}_smplx.npz"
+ Log.info(f"[Export NPZ] Saving SMPL-X output to: {output_npz_path}")
+
+ params = pred["smpl_params_global"]
+ device = params['global_orient'].device
+
+ # --- 1. Root Orient -> Ensure (N, 3) Axis Angle ---
+ root_orient = params['global_orient'] # Could be (N, 3) or (N, 1, 3, 3) or (N, 3, 3)
+
+ if root_orient.ndim == 4: # (N, 1, 3, 3)
+ root_orient = root_orient.squeeze(1)
+
+ if root_orient.ndim == 3 and root_orient.shape[-1] == 3: # (N, 3, 3) -> Convert
+ root_orient = matrix_to_axis_angle(root_orient)
+ elif root_orient.ndim == 2 and root_orient.shape[-1] == 3: # (N, 3) -> Already AA
+ pass
+
+ L = root_orient.shape[0]
+ root_orient = root_orient.reshape(L, 3)
+
+ # --- 2. Body Pose -> Ensure (N, 63) Axis Angle ---
+ body_pose = params['body_pose'] # Could be (N, 63), (N, 21, 3), (N, 21, 3, 3)
+
+ if body_pose.ndim == 4: # (N, J, 3, 3) -> Convert
+ body_pose = matrix_to_axis_angle(body_pose)
+
+ # Flatten to (N, 63)
+ body_pose = body_pose.reshape(L, -1)
+
+ # --- 3. Concatenate [Root, Body] -> (L, 66) ---
+ full_pose = torch.cat([
+ root_orient,
+ body_pose
+ ], dim=-1)
+
+ # --- 4. Pad 99 zeros for SMPL-X (Jaw, Eyes, Hands) -> (L, 165) ---
+ padding = torch.zeros(L, 99, device=device)
+ full_pose = torch.cat([full_pose, padding], dim=-1)
+
+ # --- 5. Construct Dictionary ---
+ # Standard AMASS format uses 'gender' as a string/array, 'betas' as (10/16,)
+ out_dict = {
+ 'mocap_framerate': 30,
+ 'gender': 'neutral',
+ 'betas': params['betas'][0, :10].detach().cpu().numpy(), # (10,)
+ 'trans': params['transl'].detach().cpu().numpy(), # (L, 3)
+ 'poses': full_pose.detach().cpu().numpy(), # (L, 165)
+ }
+
+ with open(output_npz_path, 'wb') as f:
+ np.savez(f, **out_dict)
+ Log.info(f"[Export NPZ] Saved successfully.")
+
+
+
+if __name__ == "__main__":
+ cfg = parse_args_to_cfg()
+ paths = cfg.paths
+ Log.info(f"[GPU]: {torch.cuda.get_device_name()}")
+ Log.info(f'[GPU]: {torch.cuda.get_device_properties("cuda")}')
+
+ # ===== Preprocess and save to disk ===== #
+ run_preprocess(cfg)
+ data = load_data_dict(cfg)
+
+ # ===== HMR4D ===== #
+ if not Path(paths.hmr4d_results).exists():
+ Log.info("[HMR4D] Predicting")
+ model: DemoPL = hydra.utils.instantiate(cfg.model, _recursive_=False)
+ model.load_pretrained_model(cfg.ckpt_path)
+ model = model.eval().cuda()
+ tic = Log.sync_time()
+ pred = model.predict(data, static_cam=cfg.static_cam)
+ pred = detach_to_cpu(pred)
+ data_time = data["length"] / 30
+ Log.info(f"[HMR4D] Elapsed: {Log.sync_time() - tic:.2f}s for data-length={data_time:.1f}s")
+ torch.save(pred, paths.hmr4d_results)
+
+ # ===== Render ===== #
+ render_incam(cfg)
+ render_global(cfg)
+
+ # ===== Export NPZ ===== #
+ # Load predictions if they exist (either just created or loaded from disk)
+ if Path(paths.hmr4d_results).exists():
+ pred_for_export = torch.load(paths.hmr4d_results)
+ save_output_npz(cfg, pred_for_export)
+
+ if not Path(paths.incam_global_horiz_video).exists():
+ Log.info("[Merge Videos]")
+ merge_videos_horizontal([paths.incam_video, paths.global_video], paths.incam_global_horiz_video)
diff --git a/third_party/GVHMR/tools/demo/demo_folder.py b/third_party/GVHMR/tools/demo/demo_folder.py
new file mode 100644
index 0000000000000000000000000000000000000000..9dff08d1985c9300ae3ead69c83054f175279e93
--- /dev/null
+++ b/third_party/GVHMR/tools/demo/demo_folder.py
@@ -0,0 +1,29 @@
+import argparse
+from pathlib import Path
+from tqdm import tqdm
+from hmr4d.utils.pylogger import Log
+import subprocess
+import os
+
+
+if __name__ == "__main__":
+ parser = argparse.ArgumentParser()
+ parser.add_argument("-f", "--folder", type=str)
+ parser.add_argument("-d", "--output_root", type=str, default=None)
+ parser.add_argument("-s", "--static_cam", action="store_true", help="If true, skip DPVO")
+ args = parser.parse_args()
+
+ folder = Path(args.folder)
+ output_root = args.output_root
+
+ # Run demo.py for each .mp4 file
+ mp4_paths = sorted(list(folder.glob("*.mp4")) + list(folder.glob("*.MP4")))
+ Log.info(f"Found {len(mp4_paths)} .mp4 files in {folder}")
+ for mp4_path in tqdm(mp4_paths):
+ command = ["python", "tools/demo/demo.py", "--video", str(mp4_path)]
+ if output_root is not None:
+ command += ["--output_root", output_root]
+ if args.static_cam:
+ command += ["-s"]
+ Log.info(f"Running: {' '.join(command)}")
+ subprocess.run(command, env=dict(os.environ), check=True)
diff --git a/third_party/GVHMR/tools/demo/extract_features.py b/third_party/GVHMR/tools/demo/extract_features.py
new file mode 100644
index 0000000000000000000000000000000000000000..2ab19105fe3e13a9d90865c707db5dc0194bac7c
--- /dev/null
+++ b/third_party/GVHMR/tools/demo/extract_features.py
@@ -0,0 +1,178 @@
+import sys
+import os
+import torch
+import numpy as np
+import cv2
+import argparse
+from pathlib import Path
+from tqdm import tqdm
+import gc
+import concurrent.futures
+
+# Ensure repo root is on sys.path
+REPO_ROOT = Path(__file__).resolve().parents[2]
+if str(REPO_ROOT) not in sys.path:
+ sys.path.insert(0, str(REPO_ROOT))
+
+from hmr4d.utils.pylogger import Log
+
+# Standard ImageNet Normalization
+IMAGENET_MEAN = np.array([0.485, 0.456, 0.406], dtype=np.float32)
+IMAGENET_STD = np.array([0.229, 0.224, 0.225], dtype=np.float32)
+
+def _require_extractor():
+ gvhmr_root = REPO_ROOT / "third_party" / "GVHMR"
+ if gvhmr_root.exists() and str(gvhmr_root) not in sys.path:
+ sys.path.insert(0, str(gvhmr_root))
+ try:
+ from hmr4d.utils.preproc.vitfeat_extractor import Extractor
+ except Exception as e:
+ raise RuntimeError("Could not import Extractor from GVHMR.") from e
+ return Extractor
+
+# --- FAST IMAGE LOADER ---
+def process_single_image(args):
+ path, cx, cy, scale, img_size = args
+ img = cv2.imread(path)
+ if img is None:
+ return np.zeros((3, img_size, img_size), dtype=np.float32)
+
+ H, W = img.shape[:2]
+ max_side = float(max(H, W, 1))
+ try:
+ cx = float(cx)
+ cy = float(cy)
+ scale = float(scale)
+ except Exception as e:
+ raise RuntimeError(f"Bad bbx_xys types for {path}: cx={cx} cy={cy} scale={scale}") from e
+ if not (np.isfinite(cx) and np.isfinite(cy) and np.isfinite(scale)):
+ raise RuntimeError(f"Bad bbx_xys (non-finite) for {path}: cx={cx} cy={cy} scale={scale}")
+ if scale <= 1.0 or scale > max_side * 20.0:
+ raise RuntimeError(f"Bad bbx_xys (scale) for {path}: (H,W)=({H},{W}) cx={cx} cy={cy} scale={scale}")
+
+ half = scale / 2.0
+ x0, y0 = int(cx - half), int(cy - half)
+ x1, y1 = int(cx + half), int(cy + half)
+
+ pad_l, pad_t = max(0, -x0), max(0, -y0)
+ pad_r, pad_b = max(0, x1 - W), max(0, y1 - H)
+
+ # Fail loudly instead of letting OpenCV try to allocate absurdly large padded images.
+ if max(pad_l, pad_t, pad_r, pad_b) > int(max_side * 4.0):
+ raise RuntimeError(
+ f"Insane crop for {path}: (H,W)=({H},{W}) cx={cx:.2f} cy={cy:.2f} scale={scale:.2f} "
+ f"pads(l,t,r,b)=({pad_l},{pad_t},{pad_r},{pad_b})"
+ )
+
+ if pad_l or pad_t or pad_r or pad_b:
+ img = cv2.copyMakeBorder(img, pad_t, pad_b, pad_l, pad_r, cv2.BORDER_CONSTANT, value=(0,0,0))
+ x0 += pad_l; y0 += pad_t; x1 += pad_l; y1 += pad_t
+
+ crop = img[y0:y1, x0:x1]
+ if crop.size == 0:
+ raise RuntimeError(
+ f"Empty crop for {path}: (H,W)=({H},{W}) cx={cx:.2f} cy={cy:.2f} scale={scale:.2f} "
+ f"xyxy=({x0},{y0},{x1},{y1})"
+ )
+ if crop.shape[0] != img_size or crop.shape[1] != img_size:
+ crop = cv2.resize(crop, (img_size, img_size), interpolation=cv2.INTER_LINEAR)
+
+ crop = crop[:, :, ::-1].astype(np.float32) / 255.0
+ crop = (crop - IMAGENET_MEAN) / IMAGENET_STD
+ return crop.transpose(2, 0, 1)
+
+def load_images_parallel(image_paths, bbx_xys, img_size=256, workers=12):
+ if isinstance(bbx_xys, torch.Tensor): bbx_xys = bbx_xys.cpu().numpy()
+ tasks = [(str(p), b[0], b[1], b[2], img_size) for p, b in zip(image_paths, bbx_xys)]
+
+ with concurrent.futures.ThreadPoolExecutor(max_workers=workers) as executor:
+ results = list(executor.map(process_single_image, tasks))
+
+ return torch.from_numpy(np.stack(results))
+
+# --- OPTIMIZED INFERENCE LOOP ---
+def fast_inference(model, tensor, batch_size=64):
+ """
+ Replaces the slow extractor loop.
+ """
+ model.eval()
+ F = tensor.shape[0]
+ features = []
+
+ # Pre-allocate pinned memory for faster transfer
+ tensor = tensor.contiguous()
+
+ with torch.inference_mode():
+ for j in range(0, F, batch_size):
+ # Non-blocking transfer
+ batch = tensor[j : j + batch_size].cuda(non_blocking=True)
+
+ # AMP (Automatic Mixed Precision) -> 2x Speedup
+ with torch.amp.autocast("cuda"):
+ # HMR2 expects dictionary input
+ feat = model({"img": batch})
+
+ features.append(feat.detach().cpu())
+
+ return torch.cat(features, dim=0)
+
+def main():
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--dataset_root", required=True)
+ parser.add_argument("--batch_size", type=int, default=256, help="Increase this if VRAM allows")
+ parser.add_argument("--workers", type=int, default=4)
+ parser.add_argument("--overwrite", action="store_true")
+ args = parser.parse_args()
+
+ dataset_root = Path(args.dataset_root)
+ feat_dir = dataset_root / "genmo_features"
+ images_root = dataset_root
+
+ if not feat_dir.exists():
+ Log.error("Feature dir not found")
+ return
+
+ Log.info("Initializing ViT Model...")
+ ExtractorClass = _require_extractor()
+ extractor_wrapper = ExtractorClass(tqdm_leave=False)
+ # Get the inner torch module (HMR2)
+ model = extractor_wrapper.extractor
+
+ pt_files = sorted(list(feat_dir.glob("*.pt")))
+ Log.info(f"Processing {len(pt_files)} sequences. Batch Size: {args.batch_size}")
+
+ for pt_file in tqdm(pt_files, desc="Dataset Progress"):
+ try:
+ data = torch.load(pt_file, map_location="cpu", weights_only=False)
+
+ if not args.overwrite and "f_imgseq" in data:
+ f = data["f_imgseq"]
+ if isinstance(f, torch.Tensor) and f.ndim == 2 and f.shape[1] > 0:
+ continue
+
+ # Load Images
+ img_rel_paths = data["imgname"]
+ bbx_xys = data["bbx_xys"]
+ abs_img_paths = [images_root / p for p in img_rel_paths]
+
+ if not abs_img_paths[0].exists():
+ continue
+
+ # 1. Load & Process (CPU Parallel)
+ input_tensor = load_images_parallel(abs_img_paths, bbx_xys, workers=args.workers)
+
+ # 2. Fast Inference (GPU FP16)
+ vit_features = fast_inference(model, input_tensor, batch_size=args.batch_size)
+
+ # 3. Save
+ data["f_imgseq"] = vit_features.float() # Save as float32 for compatibility
+ torch.save(data, pt_file)
+
+ except Exception as e:
+ Log.error(f"Error {pt_file.stem}: {e}")
+ continue
+
+if __name__ == "__main__":
+ # Optimize CUDA allocator
+ os.environ["PYTORCH_CUDA_ALLOC_CONF"] = "expandable_segments:True"
+ main()
diff --git a/third_party/GVHMR/tools/demo/process_dataset.py b/third_party/GVHMR/tools/demo/process_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..df1b96d6df0e93c8882a5be3086a015bb6de3ff4
--- /dev/null
+++ b/third_party/GVHMR/tools/demo/process_dataset.py
@@ -0,0 +1,1556 @@
+import sys
+import os
+import json
+import argparse
+import numpy as np
+import zlib
+from glob import glob
+from tqdm import tqdm
+import cv2
+import torch
+from scipy.spatial.transform import Rotation as R
+import time
+import threading
+import queue
+import gc
+from pathlib import Path
+from PIL import Image, ImageDraw, ImageFont
+
+# Suppress libpng warnings and OpenCV noise
+os.environ["OPENCV_LOG_LEVEL"] = "FATAL"
+
+# --- SETUP PATHS ---
+REPO_ROOT = Path(__file__).resolve().parents[2]
+if str(REPO_ROOT) not in sys.path:
+ sys.path.insert(0, str(REPO_ROOT))
+
+from hmr4d.utils.preproc.vitfeat_extractor import Extractor
+from hmr4d.utils.pylogger import Log
+from hmr4d.utils.geo.hmr_cam import get_bbx_xys_from_xyxy
+
+# --- SPEED OPTIMIZATIONS ---
+if "OMP_NUM_THREADS" in os.environ: del os.environ["OMP_NUM_THREADS"]
+cv2.setNumThreads(1)
+torch.set_num_threads(os.cpu_count())
+torch.backends.cudnn.benchmark = True
+os.environ["PYTORCH_CUDA_ALLOC_CONF"] = "expandable_segments:True"
+
+FPS = 30.0
+DEBUG_NUM_FRAMES = 60
+IMAGENET_MEAN = np.array([0.485, 0.456, 0.406], dtype=np.float32)
+IMAGENET_STD = np.array([0.229, 0.224, 0.225], dtype=np.float32)
+
+# COCO-17 skeleton edges (OpenPose-style order used by many toolchains)
+_COCO17_EDGES = [
+ (5, 7),
+ (7, 9),
+ (6, 8),
+ (8, 10),
+ (5, 6),
+ (5, 11),
+ (6, 12),
+ (11, 12),
+ (11, 13),
+ (13, 15),
+ (12, 14),
+ (14, 16),
+ (0, 1),
+ (0, 2),
+ (1, 3),
+ (2, 4),
+ (0, 5),
+ (0, 6),
+]
+
+# --- GLOBAL ASSET CACHE (FAST) ---
+UI_ASSET_CACHE = {}
+def _get_ui_assets(ui_dir):
+ if not ui_dir: return []
+ if ui_dir in UI_ASSET_CACHE: return UI_ASSET_CACHE[ui_dir]
+ imgs = []
+ if os.path.isdir(ui_dir):
+ for name in sorted(os.listdir(ui_dir)):
+ p = os.path.join(ui_dir, name)
+ if name.lower().endswith(('.png', '.jpg', '.jpeg', '.webp')):
+ im = cv2.imread(p, cv2.IMREAD_UNCHANGED)
+ if im is not None: imgs.append(im)
+ UI_ASSET_CACHE[ui_dir] = imgs
+ return imgs
+
+# --- SMPLX->SMPL VERTEX MAP CACHE ---
+_SMPLX2SMPL = None
+def _get_smplx2smpl(device="cuda"):
+ global _SMPLX2SMPL
+ if _SMPLX2SMPL is not None:
+ return _SMPLX2SMPL
+ p = Path("third_party/GVHMR/hmr4d/utils/body_model/smplx2smpl_sparse.pt")
+ if not p.exists():
+ _SMPLX2SMPL = None
+ return None
+ _SMPLX2SMPL = torch.load(p, map_location=device)
+ return _SMPLX2SMPL
+
+# --- THREADED VIT PROCESSOR (FAST) ---
+class VitInferenceThread(threading.Thread):
+ def __init__(self, model, batch_size, device, mean, std):
+ super().__init__()
+ self.model = model
+ self.batch_size = batch_size
+ self.device = device
+ self.mean = mean
+ self.std = std
+ self.input_queue = queue.Queue(maxsize=batch_size * 4)
+ self.results = []
+ self.stop_signal = False
+ self.proc_time = 0.0
+
+ def run(self):
+ batch = []
+ while not self.stop_signal or not self.input_queue.empty():
+ try:
+ raw_crop = self.input_queue.get(timeout=0.1)
+ if raw_crop is None: continue
+ if raw_crop.size == 0:
+ resized = np.zeros((3, 256, 256), dtype=np.uint8)
+ else:
+ resized = cv2.resize(raw_crop, (256, 256), interpolation=cv2.INTER_LINEAR)
+ resized = resized[:, :, ::-1].transpose(2, 0, 1).copy()
+
+ batch.append(torch.from_numpy(resized))
+ if len(batch) >= self.batch_size:
+ self._process_batch(batch)
+ batch = []
+ except queue.Empty: continue
+ if batch: self._process_batch(batch)
+
+ def _process_batch(self, batch):
+ t0 = time.perf_counter()
+ batch_t = torch.stack(batch).to(self.device, non_blocking=True).float()
+ batch_t = (batch_t / 255.0 - self.mean) / self.std
+ with torch.inference_mode(), torch.amp.autocast("cuda"):
+ feats = self.model({"img": batch_t})
+ self.results.append(feats.detach().cpu())
+ self.proc_time += (time.perf_counter() - t0)
+
+# --- FAST OVERLAY LOGIC ---
+def _alpha_blend_fast(roi, src):
+ if src.shape[2] < 4:
+ roi[:] = src[:, :, :3]
+ return
+ alpha = src[:, :, 3].astype(np.uint16)
+ inv_alpha = 255 - alpha
+ for c in range(3):
+ roi[:, :, c] = ((src[:, :, c].astype(np.uint16) * alpha + roi[:, :, c].astype(np.uint16) * inv_alpha) >> 8).astype(np.uint8)
+
+class SimpleUIOverlay:
+ def __init__(self, width, height, seed=0, ui_dir=None, max_images=4, show_prob=0.6, min_hold_frames=20, max_hold_frames=120):
+ self.W, self.H, self.rng = width, height, np.random.default_rng(seed)
+ self.max_images, self.show_prob = max_images, show_prob
+ self.min_hold, self.max_hold = min_hold_frames, max_hold_frames
+ self.assets = _get_ui_assets(ui_dir if ui_dir else os.path.join(os.getcwd(), "UI"))
+ self._ttl, self._active = 0, []
+ def _pick_state(self):
+ self._ttl = self.rng.integers(self.min_hold, self.max_hold + 1)
+ self._active = []
+ if self.assets and self.rng.random() < self.show_prob:
+ k = min(self.max_images, len(self.assets))
+ for idx in self.rng.choice(len(self.assets), size=k, replace=False):
+ im = self.assets[idx]
+ h, w = im.shape[:2]
+ self._active.append((im, self.rng.integers(-w//4, self.W-w), self.rng.integers(-h//4, self.H-h)))
+ def draw(self, img_bgr):
+ if self._ttl <= 0: self._pick_state()
+ self._ttl -= 1
+ for im, x, y in self._active:
+ x0, y0 = max(x, 0), max(y, 0)
+ x1, y1 = min(x + im.shape[1], self.W), min(y + im.shape[0], self.H)
+ if x1 > x0 and y1 > y0:
+ _alpha_blend_fast(img_bgr[y0:y1, x0:x1], im[(y0-y):(y1-y), (x0-x):(x1-x)])
+
+class SimpleChatOverlay:
+ # Random usernames and messages for variety
+ USERNAMES = ["xXSlayerXx", "PogChamp42", "GamerGirl99", "NoobMaster69", "CoolDude", "ProPlayer", "LootGoblin", "HeadShot", "NinjaCat", "PixelKing", "ShadowFox", "DragonBoi", "IcyHot", "ZeroTwo", "MemeL0rd", "FiliArmy", "StreamSniper", "TwitchChatter", "SubGifter", "ModsAsleep"]
+ MESSAGES = [
+ # Short messages
+ "pog", "lol", "gg", "nice!", "omg", "bruh", "sheesh", "F", "W", "L", "based", "cringe", "sus", "sadge", "copium", "kekw", "monkaS", "Prayge", "EZ Clap", "LULW", "xD", "haha", "wow", "damn", "yooo", "fire", "goated", "cap", "no cap", "ratio",
+ # Medium messages
+ "lets goooo", "no way bro", "this is insane", "banger stream", "she's so cracked", "wait what", "LETS GOOO", "im dead lmao", "this is peak", "I cant believe this", "actual content", "real and true",
+ # Long messages
+ "Filian you are not mogging us bro just chill", "yo chat we are so back right now", "this might be the best stream I've ever seen", "bro really said that with a straight face lmao", "can we get some Ws in chat for the homie", "okay but that was actually kinda fire ngl", "the way she did that was lowkey impressive", "chat I need you to understand something real quick", "nah this is actually crazy when you think about it", "imagine watching this and not being entertained smh"
+ ]
+ COLORS = [(255, 0, 0), (0, 255, 0), (0, 128, 255), (255, 165, 0), (255, 0, 255), (0, 255, 255), (255, 255, 0), (128, 0, 255), (255, 128, 0), (0, 255, 128)]
+
+ def __init__(self, width, height, seed=0, num_lines=7, region_w=420, region_h=180, margin=18, font_path=None):
+ from collections import deque
+ self.W, self.H, self.rng = width, height, np.random.default_rng(seed)
+ self.num_lines = num_lines
+ # Initialize with random messages
+ self.messages = deque(maxlen=num_lines)
+ for _ in range(num_lines):
+ self._add_random_message()
+ self.region_w, self.region_h, self.margin = region_w, region_h, margin
+ self.font = ImageFont.truetype(font_path, 18) if font_path and os.path.exists(font_path) else None
+ self._cached, self._dirty = None, True
+
+ def _add_random_message(self):
+ user = self.rng.choice(self.USERNAMES)
+ text = self.rng.choice(self.MESSAGES)
+ color = tuple(self.rng.choice(self.COLORS))
+ self.messages.append({"user": user, "text": text, "color": color})
+
+ def maybe_append(self, idx):
+ if idx % 15 == 0:
+ self._add_random_message()
+ self._dirty = True
+
+ def draw(self, img_bgr):
+ if self._dirty:
+ pil = Image.new("RGBA", (self.region_w, self.region_h), (0,0,0,0))
+ draw = ImageDraw.Draw(pil)
+ for i, m in enumerate(self.messages):
+ # Draw username in its color, then message in white
+ user_text = f"{m['user']}: "
+ msg_text = m['text']
+ # Get username width for positioning
+ if self.font:
+ user_bbox = draw.textbbox((0, 0), user_text, font=self.font)
+ user_width = user_bbox[2] - user_bbox[0]
+ else:
+ user_width = len(user_text) * 8 # Approximate
+ draw.text((5, i*24), user_text, font=self.font, fill=m['color'])
+ draw.text((5 + user_width, i*24), msg_text, font=self.font, fill=(255, 255, 255))
+ self._cached = cv2.cvtColor(np.asarray(pil), cv2.COLOR_RGBA2BGRA)
+ self._dirty = False
+ y_off = self.H - self.region_h - self.margin
+ _alpha_blend_fast(img_bgr[y_off:y_off+self.region_h, self.margin:self.margin+self.region_w], self._cached)
+
+# --- UI PATH RESOLUTION ---
+def _resolve_ui_dir_and_font(ui_dir_arg):
+ """
+ Resolve UI assets directory + optional font path robustly.
+
+ `process_data.sh` is typically executed from `third_party/GVHMR`, so
+ `os.getcwd()/UI` often doesn't exist. Prefer the repo-level `UI/` folder
+ (GENMO/UI) when available.
+ """
+ # Prefer explicit arg
+ if ui_dir_arg:
+ p = Path(str(ui_dir_arg)).expanduser()
+ if p.is_dir():
+ font = p / "Inter_18pt-Bold.ttf"
+ return str(p), (str(font) if font.exists() else None)
+
+ candidates = []
+ try:
+ candidates.append(Path(os.getcwd()) / "UI")
+ except Exception:
+ pass
+ try:
+ candidates.append(Path(REPO_ROOT) / "UI") # third_party/GVHMR/UI (unlikely)
+ except Exception:
+ pass
+ try:
+ # GENMO repo root is `.../GENMO`; this file lives at `.../GENMO/third_party/GVHMR/tools/demo/...`
+ candidates.append(Path(REPO_ROOT).parents[2] / "UI")
+ except Exception:
+ pass
+
+ for c in candidates:
+ if c.is_dir():
+ font = c / "Inter_18pt-Bold.ttf"
+ return str(c), (str(font) if font.exists() else None)
+
+ return None, None
+
+# --- ORIGINAL MATH & HELPERS ---
+def k4_to_K3(k4): return np.array([[k4[0], 0, k4[2]], [0, k4[1], k4[3]], [0, 0, 1]], dtype=np.float32)
+def bbox_xywh_to_bbx_xys(bbox, scale=1.0):
+ return np.array([bbox[0] + 0.5*bbox[2], bbox[1] + 0.5*bbox[3], max(bbox[2], bbox[3])*scale], dtype=np.float32)
+
+def bbx_xywh_from_bbx_xys(bbx_xys, W, H):
+ """Convert [cx, cy, size] to clamped xywh (square crop)."""
+ cx, cy, s = float(bbx_xys[0]), float(bbx_xys[1]), float(bbx_xys[2])
+ x1 = cx - 0.5 * s
+ y1 = cy - 0.5 * s
+ x2 = cx + 0.5 * s
+ y2 = cy + 0.5 * s
+ x1, y1 = max(0.0, x1), max(0.0, y1)
+ x2, y2 = min(float(W), x2), min(float(H), y2)
+ w = max(1.0, x2 - x1)
+ h = max(1.0, y2 - y1)
+ return [int(x1), int(y1), int(w), int(h)]
+def clamp_bbox(bbox, W, H):
+ x, y, w, h = [float(v) for v in bbox]
+ x1, y1 = np.clip(x, 0, W-1.0), np.clip(y, 0, H-1.0)
+ x2, y2 = np.clip(x+w, 0, W), np.clip(y+h, 0, H)
+ return [int(x1), int(y1), int(max(1, x2-x1)), int(max(1, y2-y1))]
+
+def build_T_wc(pos_world, quat_world_xyzw):
+ T = np.eye(4, dtype=np.float64)
+ T[:3, :3] = R.from_quat(np.asarray(quat_world_xyzw, dtype=np.float64)).as_matrix()
+ T[:3, 3] = np.asarray(pos_world, dtype=np.float64)
+ return T
+
+def compute_velocity(mats, fps=30.0):
+ """Compute angular velocity (6D rotation) and translation velocity.
+
+ GVHMR expects cam_angvel as a 6D rotation representation (N, 6) produced by
+ pytorch3d's `matrix_to_rotation_6d` (first two COLUMNS, flattened).
+ First frame should be the identity rotation [1,0,0,0,1,0].
+ """
+ from pytorch3d.transforms import matrix_to_rotation_6d
+ N = len(mats)
+ if N < 2:
+ # Return identity rotation for single frame
+ angvel = np.array([[1, 0, 0, 0, 1, 0]], dtype=np.float32)
+ return angvel, np.zeros((N, 3), dtype=np.float32)
+
+ R_curr = mats[:, :3, :3].astype(np.float32)
+ # R @ R0 = R1 => R = R1 @ R0^T
+ R_diff = np.matmul(R_curr[1:], np.transpose(R_curr[:-1], (0, 2, 1))).astype(np.float32)
+
+ angvel_6d = np.zeros((N, 6), dtype=np.float32)
+ angvel_6d[0, :] = [1, 0, 0, 0, 1, 0]
+ # pytorch3d expects torch tensor
+ angvel_6d[1:, :] = matrix_to_rotation_6d(torch.from_numpy(R_diff)).numpy().astype(np.float32)
+
+ t_curr = mats[:, :3, 3]
+ tvel = np.zeros((N, 3), dtype=np.float32)
+ tvel[1:] = t_curr[1:] - t_curr[:-1]
+ return angvel_6d.astype(np.float32), tvel.astype(np.float32)
+
+# DPVO trajectory (N,7) -> T_w2c (N,4,4), matching `scripts/demo/infer_video.py`.
+def _dpvo7_to_T_w2c(traj_7d: np.ndarray) -> np.ndarray:
+ """
+ DPVO output: [x, y, z, qx, qy, qz, qw] (Camera-to-World).
+ Returns: T_w2c (World-to-Camera) matrices (N,4,4).
+ """
+ traj_7d = np.asarray(traj_7d, dtype=np.float32)
+ t_c2w = traj_7d[:, :3] # (N,3)
+ qx, qy, qz, qw = traj_7d[:, 3], traj_7d[:, 4], traj_7d[:, 5], traj_7d[:, 6]
+ N = traj_7d.shape[0]
+
+ # Quaternion -> rotation (c2w)
+ R_c2w = np.zeros((N, 3, 3), dtype=np.float32)
+ R_c2w[:, 0, 0] = 1 - 2 * (qy * qy + qz * qz)
+ R_c2w[:, 1, 1] = 1 - 2 * (qx * qx + qz * qz)
+ R_c2w[:, 2, 2] = 1 - 2 * (qx * qx + qy * qy)
+
+ R_c2w[:, 0, 1] = 2 * (qx * qy - qz * qw)
+ R_c2w[:, 0, 2] = 2 * (qx * qz + qy * qw)
+ R_c2w[:, 1, 0] = 2 * (qx * qy + qz * qw)
+ R_c2w[:, 1, 2] = 2 * (qy * qz - qx * qw)
+ R_c2w[:, 2, 0] = 2 * (qx * qz - qy * qw)
+ R_c2w[:, 2, 1] = 2 * (qy * qz + qx * qw)
+
+ # C2W -> W2C
+ R_w2c = np.transpose(R_c2w, (0, 2, 1))
+ t_w2c = -np.einsum("nij,nj->ni", R_w2c, t_c2w).astype(np.float32)
+
+ T_w2c = np.tile(np.eye(4, dtype=np.float32)[None], (N, 1, 1))
+ T_w2c[:, :3, :3] = R_w2c
+ T_w2c[:, :3, 3] = t_w2c
+ return T_w2c
+
+# --- VISUALIZATION HELPERS (Restored Original Colors/Text) ---
+def vis_label_and_color(v: int):
+ # Original logic: 2=VIS(Green), 1=OCC(Orange), 0=OFF(Grey)
+ if v == 2: return "VIS", (0, 255, 0)
+ if v == 1: return "OCC", (0, 165, 255)
+ return "OFF", (160, 160, 160)
+
+def draw_vis_text_and_points(img_bgr, kpts2d_xy, vis17):
+ for k in range(17):
+ v = int(vis17[k])
+ label, color = vis_label_and_color(v)
+ x, y = int(round(kpts2d_xy[k, 0])), int(round(kpts2d_xy[k, 1]))
+ if v > 0: cv2.circle(img_bgr, (x, y), 4, color, -1)
+ cv2.putText(img_bgr, f"{k}:{label}", (x + 6, y - 6), cv2.FONT_HERSHEY_SIMPLEX, 0.45, color, 1, cv2.LINE_AA)
+
+def draw_bbox_xywh_and_center(img_bgr, bbox_xywh, color=(255, 255, 0)):
+ x, y, w, h = [float(v) for v in bbox_xywh]
+ cv2.rectangle(img_bgr, (int(x), int(y)), (int(x+w), int(y+h)), color, 2)
+ cv2.circle(img_bgr, (int(x+w/2), int(y+h/2)), 4, (0, 0, 255), -1)
+
+def draw_kp2d_conf_overlay(img_bgr, kp2d_17x3, edges=_COCO17_EDGES):
+ """Draw COCO17 kp2d with confidence (x,y,score)."""
+ kp2d = np.asarray(kp2d_17x3, dtype=np.float32).reshape(17, 3)
+ pts = kp2d[:, :2]
+ conf = kp2d[:, 2]
+
+ # Edges first (faint)
+ for i, j in edges:
+ ci = float(conf[i])
+ cj = float(conf[j])
+ if ci <= 0.05 or cj <= 0.05:
+ continue
+ pi = (int(round(float(pts[i, 0]))), int(round(float(pts[i, 1]))))
+ pj = (int(round(float(pts[j, 0]))), int(round(float(pts[j, 1]))))
+ cv2.line(img_bgr, pi, pj, (200, 200, 200), 2, cv2.LINE_AA)
+
+ # Points (color by confidence)
+ for k in range(17):
+ x, y = int(round(float(pts[k, 0]))), int(round(float(pts[k, 1])))
+ c = float(conf[k])
+ if c <= 0.01:
+ continue
+ # red->green
+ g = int(np.clip(255.0 * c, 0.0, 255.0))
+ r = int(np.clip(255.0 * (1.0 - c), 0.0, 255.0))
+ color = (0, g, r)
+ cv2.circle(img_bgr, (x, y), 4, color, -1, cv2.LINE_AA)
+ cv2.putText(
+ img_bgr,
+ f"{k}:{c:.2f}",
+ (x + 6, y - 6),
+ cv2.FONT_HERSHEY_SIMPLEX,
+ 0.45,
+ color,
+ 1,
+ cv2.LINE_AA,
+ )
+
+def _render_kp2d_overlay_video(
+ video_path,
+ kp2d_seq,
+ bbx_xys_seq,
+ out_path,
+ title=None,
+ max_frames=DEBUG_NUM_FRAMES,
+ ui_dir=None,
+ ui_seed=0,
+ ui_show_prob=0.25,
+ ui_max_images=3,
+ ui_hold_min_s=0.7,
+ ui_hold_max_s=5.0,
+ draw_ui=True,
+):
+ cap = cv2.VideoCapture(video_path)
+ if not cap.isOpened():
+ raise RuntimeError(f"Failed to open video: {video_path}")
+ W = int(cap.get(3))
+ H = int(cap.get(4))
+ writer = cv2.VideoWriter(str(out_path), cv2.VideoWriter_fourcc(*"mp4v"), FPS, (W, H))
+ chat = None
+ ui = None
+ if draw_ui:
+ resolved_ui_dir, resolved_font = _resolve_ui_dir_and_font(ui_dir)
+ chat = SimpleChatOverlay(W, H, seed=int(ui_seed or 0), font_path=resolved_font)
+ min_hold_frames = int(max(1, round(float(ui_hold_min_s) * FPS)))
+ max_hold_frames = int(max(min_hold_frames, round(float(ui_hold_max_s) * FPS)))
+ ui = SimpleUIOverlay(
+ W,
+ H,
+ seed=int(ui_seed or 0),
+ ui_dir=resolved_ui_dir,
+ max_images=int(ui_max_images),
+ show_prob=float(ui_show_prob),
+ min_hold_frames=min_hold_frames,
+ max_hold_frames=max_hold_frames,
+ )
+ try:
+ n = min(int(len(kp2d_seq)), int(len(bbx_xys_seq)), int(max_frames))
+ for i in range(n):
+ ret, img = cap.read()
+ if not ret:
+ break
+ if chat is not None:
+ chat.maybe_append(i)
+ chat.draw(img)
+ if ui is not None:
+ ui.draw(img)
+ bbx = bbx_xywh_from_bbx_xys(bbx_xys_seq[i], W, H)
+ draw_bbox_xywh_and_center(img, bbx, color=(255, 255, 0))
+ draw_kp2d_conf_overlay(img, kp2d_seq[i])
+ if title:
+ cv2.putText(img, str(title), (12, 28), cv2.FONT_HERSHEY_SIMPLEX, 0.9, (255, 255, 255), 2, cv2.LINE_AA)
+ writer.write(img)
+ finally:
+ writer.release()
+ cap.release()
+
+# --- PARSING: Derive from raw Unity transforms (matching process_dataset_fromincam.py) ---
+def parse_smpl_inputs_from_row(row, override_betas10=None, keep_unity_scale=False, transl_source="pelvis", transl_y_offset_m=0.0):
+ """Parse SMPL inputs from Unity export and convert to CV convention.
+
+ Derives incam rotation from pelvis_rot_world + cam_rot_world + Z-180° fix.
+ This is the proven approach from process_dataset_fromincam.py.
+ """
+ C = np.diag([1.0, -1.0, 1.0]).astype(np.float64)
+
+ # Get raw Unity quaternions
+ cam_rot_w_quat = np.array(row["cam_rot_world"], dtype=np.float64)
+ R_cam_w = R.from_quat(cam_rot_w_quat).as_matrix()
+ pel_rot_w_quat = np.array(row["pelvis_rot_world"], dtype=np.float64)
+ R_pel_w = R.from_quat(pel_rot_w_quat).as_matrix()
+
+ # Derive incam rotation: camera^-1 @ pelvis, then convert to CV with Z-180° fix
+ R_rel_unity = R_cam_w.T @ R_pel_w
+ R_cv = C @ R_rel_unity @ C
+ R_final = R_cv @ R.from_euler("z", 180, degrees=True).as_matrix()
+ global_orient_aa = R.from_matrix(R_final).as_rotvec().astype(np.float32)
+
+ # Translation (already Y-flipped in Unity export)
+ smpl_scale = float(row.get("smpl_root_world_scale", 1.0))
+ pelvis_cam_cv = np.asarray(row["smpl_incam_transl"], dtype=np.float64).reshape(3)
+ root_cam_cv = np.asarray(row.get("smpl_root_incam_transl", [0.0, 0.0, 0.0]), dtype=np.float64).reshape(3)
+ pelvis_cam_cv = pelvis_cam_cv + np.array([0.0, float(transl_y_offset_m), 0.0], dtype=np.float64)
+
+ if str(transl_source).strip().lower() == "root":
+ target_cam_cv = root_cam_cv
+ else:
+ if bool(keep_unity_scale):
+ target_cam_cv = pelvis_cam_cv
+ else:
+ if abs(smpl_scale) > 1e-8:
+ target_cam_cv = root_cam_cv + (pelvis_cam_cv - root_cam_cv) / smpl_scale
+ else:
+ target_cam_cv = pelvis_cam_cv
+
+ pose = np.asarray(row["smplx_pose"], dtype=np.float32)
+ body_pose = pose[3:66].astype(np.float32)
+ betas10 = np.zeros(10, dtype=np.float32)
+ if override_betas10 is not None:
+ betas10[:min(10, override_betas10.size)] = override_betas10.flatten()[:10]
+
+ return {
+ "global_orient": global_orient_aa,
+ "body_pose": body_pose,
+ "betas": betas10,
+ "target_cam_cv": target_cam_cv,
+ "cam_rot_w_quat": cam_rot_w_quat,
+ "cam_pos_world": np.asarray(row["cam_pos_world"], dtype=np.float64).reshape(3),
+ "pelvis_pos_world": np.asarray(row["pelvis_pos_world"], dtype=np.float64).reshape(3),
+ "smpl_scale": smpl_scale,
+ "root_cam_cv": root_cam_cv
+ }
+
+def batch_smpl_forward(betas, global_orient, body_pose, device):
+ from hmr4d.utils.smplx_utils import make_smplx
+ model = make_smplx("supermotion").to(device).eval()
+ pelvis_list = []
+ with torch.no_grad():
+ for i in range(0, len(betas), 2048):
+ b_bt = torch.from_numpy(betas[i:i+2048]).to(device).float()
+ b_go = torch.from_numpy(global_orient[i:i+2048]).to(device).float()
+ b_bp = torch.from_numpy(body_pose[i:i+2048]).to(device).float()
+ b_tr = torch.zeros((len(b_bt), 3), dtype=torch.float32, device=device)
+ out = model(betas=b_bt, global_orient=b_go, body_pose=b_bp, transl=b_tr)
+ pelvis_list.append(out.joints[:, 0, :].cpu().numpy())
+ return np.concatenate(pelvis_list, axis=0)
+
+
+def _rot_err_deg(R_a: np.ndarray, R_b: np.ndarray) -> float:
+ R_rel = R_a.T @ R_b
+ return float(np.linalg.norm(R.from_matrix(R_rel).as_rotvec()) * (180.0 / np.pi))
+
+
+def _kabsch_align(src, dst):
+ """Rigid alignment (no scale). src/dst: (J, 3) numpy arrays.
+ Returns R, t such that src @ R + t ≈ dst.
+ """
+ src_mean = src.mean(axis=0)
+ dst_mean = dst.mean(axis=0)
+ X = src - src_mean
+ Y = dst - dst_mean
+ H = X.T @ Y
+ U, _, Vt = np.linalg.svd(H)
+ R_align = Vt.T @ U.T
+ if np.linalg.det(R_align) < 0:
+ Vt[-1, :] *= -1
+ R_align = Vt.T @ U.T
+ t_align = dst_mean - src_mean @ R_align
+ return R_align, t_align
+
+
+def verify_camera_smpl_consistency(smpl_params_c, smpl_params_w, T_w2c_exported, smplx_model, device="cuda"):
+ """Verify camera-SMPL consistency using Kabsch alignment.
+
+ Computes T_c2w from incam->global joint alignment and compares to exported T_w2c.
+
+ Returns:
+ angle_err: Rotation error in degrees
+ t_err: Translation error in meters
+ """
+ with torch.no_grad():
+ # Get joints from incam SMPL
+ params_c = {k: torch.from_numpy(v[None]).to(device).float() for k, v in smpl_params_c.items()}
+ joints_c = smplx_model(**params_c).joints[0, :22].cpu().numpy() # (22, 3)
+
+ # Get joints from global SMPL
+ params_w = {k: torch.from_numpy(v[None]).to(device).float() for k, v in smpl_params_w.items()}
+ joints_w = smplx_model(**params_w).joints[0, :22].cpu().numpy() # (22, 3)
+
+ # Kabsch: find T_c2w that maps incam joints -> global joints
+ R_c2w_kabsch, t_c2w_kabsch = _kabsch_align(joints_c, joints_w)
+
+ # Build T_c2w from Kabsch
+ T_c2w_kabsch = np.eye(4, dtype=np.float64)
+ T_c2w_kabsch[:3, :3] = R_c2w_kabsch
+ T_c2w_kabsch[:3, 3] = t_c2w_kabsch
+
+ # Get T_c2w from exported T_w2c
+ T_c2w_exported = np.linalg.inv(T_w2c_exported)
+
+ # Compute rotation error
+ R_diff = T_c2w_kabsch[:3, :3] @ T_c2w_exported[:3, :3].T
+ trace = np.clip(np.trace(R_diff), -1.0, 3.0)
+ angle_err = np.arccos(np.clip((trace - 1) / 2, -1.0, 1.0)) * 180.0 / np.pi
+
+ # Compute translation error
+ t_err = np.linalg.norm(T_c2w_kabsch[:3, 3] - T_c2w_exported[:3, 3])
+
+ return float(angle_err), float(t_err), T_c2w_kabsch, T_c2w_exported
+
+
+def compute_camera_transforms_from_smpl(
+ smpl_params_c_list,
+ smpl_params_w_list,
+ smplx_model,
+ device="cuda",
+ smooth_alpha=0.2,
+ batch_size=2048,
+ unity_mats_wc=None,
+ cam_jump_rot_deg: float = 5.0,
+ cam_jump_trans_m: float = 0.05,
+):
+ """Compute camera transforms (T_w2c) via Kabsch alignment from incam->global joints.
+
+ OPTIMIZED: Only computes Kabsch for frame 0, then propagates using Unity's relative
+ camera motion. This reduces SMPL forward passes from 2N to just 2.
+
+ Args:
+ smpl_params_c_list: List of dicts with incam SMPL params per frame
+ smpl_params_w_list: List of dicts with global SMPL params per frame
+ smplx_model: SMPLX model instance
+ device: torch device
+ smooth_alpha: EMA smoothing factor (lower = more smoothing)
+ batch_size: Batch size for SMPL forward passes
+ unity_mats_wc: (N, 4, 4) Unity camera-to-world matrices for relative motion
+
+ Returns:
+ mats_w2c: (N, 4, 4) world-to-camera matrices
+ mats_wc: (N, 4, 4) camera-to-world matrices
+ """
+ N = len(smpl_params_c_list)
+
+ # Only need frame 0 for Kabsch - get joints for first frame only
+ with torch.no_grad():
+ params_c = {
+ "global_orient": torch.from_numpy(smpl_params_c_list[0]["global_orient"][None]).to(device).float(),
+ "body_pose": torch.from_numpy(smpl_params_c_list[0]["body_pose"][None]).to(device).float(),
+ "betas": torch.from_numpy(smpl_params_c_list[0]["betas"][None]).to(device).float(),
+ "transl": torch.from_numpy(smpl_params_c_list[0]["transl"][None]).to(device).float()
+ }
+ joints_c_0 = smplx_model(**params_c).joints[0, :22].cpu().numpy()
+
+ params_w = {
+ "global_orient": torch.from_numpy(smpl_params_w_list[0]["global_orient"][None]).to(device).float(),
+ "body_pose": torch.from_numpy(smpl_params_w_list[0]["body_pose"][None]).to(device).float(),
+ "betas": torch.from_numpy(smpl_params_w_list[0]["betas"][None]).to(device).float(),
+ "transl": torch.from_numpy(smpl_params_w_list[0]["transl"][None]).to(device).float()
+ }
+ joints_w_0 = smplx_model(**params_w).joints[0, :22].cpu().numpy()
+
+ # Kabsch alignment for frame 0 only
+ R_c2w_0, t_c2w_0 = _kabsch_align(joints_c_0, joints_w_0)
+ T_c2w_0_kabsch = np.eye(4, dtype=np.float64)
+ T_c2w_0_kabsch[:3, :3] = R_c2w_0
+ T_c2w_0_kabsch[:3, 3] = t_c2w_0
+
+ # If Unity camera matrices provided, propagate with per-shot correction.
+ # Important for datasets with camera cuts: a single correction computed at frame 0 is not valid
+ # after switching to a different Unity camera object.
+ if unity_mats_wc is not None and len(unity_mats_wc) == N:
+ unity_mats_wc = np.asarray(unity_mats_wc, dtype=np.float64)
+
+ # Detect camera cuts by large jumps in Unity camera pose (between consecutive frames).
+ cam_jump_rot_deg = float(cam_jump_rot_deg)
+ cam_jump_trans_m = float(cam_jump_trans_m)
+ cut_starts = [0]
+ try:
+ R_u = unity_mats_wc[:, :3, :3]
+ t_u = unity_mats_wc[:, :3, 3]
+ R_diff = np.matmul(R_u[1:], np.transpose(R_u[:-1], (0, 2, 1)))
+ ang = np.linalg.norm(R.from_matrix(R_diff).as_rotvec(), axis=1) * (180.0 / np.pi)
+ dt = np.linalg.norm(t_u[1:] - t_u[:-1], axis=1)
+ cut_idxs = np.where((ang > cam_jump_rot_deg) | (dt > cam_jump_trans_m))[0]
+ # cut_idxs are indices of transitions (i-1 -> i). Segment starts at i.
+ for ci in cut_idxs.tolist():
+ si = int(ci) + 1
+ if si not in cut_starts and 0 < si < N:
+ cut_starts.append(si)
+ cut_starts = sorted(cut_starts)
+ except Exception:
+ cut_starts = [0]
+
+ # Always include end as a boundary
+ cut_ends = cut_starts[1:] + [N]
+
+ # Compute Kabsch correction per segment start.
+ mats_c2w = np.zeros((N, 4, 4), dtype=np.float64)
+ for seg_start, seg_end in zip(cut_starts, cut_ends):
+ # Kabsch for the first frame of this segment.
+ with torch.no_grad():
+ params_c = {
+ "global_orient": torch.from_numpy(smpl_params_c_list[seg_start]["global_orient"][None]).to(device).float(),
+ "body_pose": torch.from_numpy(smpl_params_c_list[seg_start]["body_pose"][None]).to(device).float(),
+ "betas": torch.from_numpy(smpl_params_c_list[seg_start]["betas"][None]).to(device).float(),
+ "transl": torch.from_numpy(smpl_params_c_list[seg_start]["transl"][None]).to(device).float(),
+ }
+ joints_c = smplx_model(**params_c).joints[0, :22].cpu().numpy()
+ params_w = {
+ "global_orient": torch.from_numpy(smpl_params_w_list[seg_start]["global_orient"][None]).to(device).float(),
+ "body_pose": torch.from_numpy(smpl_params_w_list[seg_start]["body_pose"][None]).to(device).float(),
+ "betas": torch.from_numpy(smpl_params_w_list[seg_start]["betas"][None]).to(device).float(),
+ "transl": torch.from_numpy(smpl_params_w_list[seg_start]["transl"][None]).to(device).float(),
+ }
+ joints_w = smplx_model(**params_w).joints[0, :22].cpu().numpy()
+
+ R_c2w_s, t_c2w_s = _kabsch_align(joints_c, joints_w)
+ T_c2w_s_kabsch = np.eye(4, dtype=np.float64)
+ T_c2w_s_kabsch[:3, :3] = R_c2w_s
+ T_c2w_s_kabsch[:3, 3] = t_c2w_s
+
+ # Correction from Unity camera to Kabsch camera for this segment.
+ T_unity_c2w_s = unity_mats_wc[seg_start]
+ T_unity_w2c_s = np.linalg.inv(T_unity_c2w_s)
+ T_correction = T_c2w_s_kabsch @ T_unity_w2c_s
+
+ # Apply within segment.
+ for i in range(seg_start, seg_end):
+ mats_c2w[i] = T_correction @ unity_mats_wc[i]
+ else:
+ # Fallback: use same transform for all frames (static camera assumption)
+ mats_c2w = np.tile(T_c2w_0_kabsch[None], (N, 1, 1))
+
+ # Apply temporal smoothing (EMA) - optional.
+ # If Unity cuts exist, do not smooth across cut boundaries.
+ if smooth_alpha < 1.0 and N > 1:
+ try:
+ # Recompute cut boundaries for smoothing when unity_mats_wc is available.
+ if unity_mats_wc is not None and len(unity_mats_wc) == N:
+ R_u = unity_mats_wc[:, :3, :3]
+ t_u = unity_mats_wc[:, :3, 3]
+ R_diff = np.matmul(R_u[1:], np.transpose(R_u[:-1], (0, 2, 1)))
+ ang = np.linalg.norm(R.from_matrix(R_diff).as_rotvec(), axis=1) * (180.0 / np.pi)
+ dt = np.linalg.norm(t_u[1:] - t_u[:-1], axis=1)
+ cut_mask = (ang > cam_jump_rot_deg) | (dt > cam_jump_trans_m)
+ cut_starts_smooth = [0] + (np.where(cut_mask)[0] + 1).astype(int).tolist()
+ cut_starts_smooth = sorted(list(set([x for x in cut_starts_smooth if 0 <= x < N])))
+ cut_ends_smooth = cut_starts_smooth[1:] + [N]
+ else:
+ cut_starts_smooth = [0]
+ cut_ends_smooth = [N]
+
+ smoothed = mats_c2w.copy()
+ for seg_start, seg_end in zip(cut_starts_smooth, cut_ends_smooth):
+ if seg_end - seg_start <= 1:
+ continue
+ # EMA within segment
+ for i in range(seg_start + 1, seg_end):
+ R_blend = (1.0 - smooth_alpha) * smoothed[i - 1, :3, :3] + smooth_alpha * mats_c2w[i, :3, :3]
+ U, _, Vt = np.linalg.svd(R_blend)
+ R_smooth = U @ Vt
+ if np.linalg.det(R_smooth) < 0:
+ Vt[-1, :] *= -1
+ R_smooth = U @ Vt
+ smoothed[i, :3, :3] = R_smooth
+ smoothed[i, :3, 3] = (1.0 - smooth_alpha) * smoothed[i - 1, :3, 3] + smooth_alpha * mats_c2w[i, :3, 3]
+ smoothed[i, 3, 3] = 1.0
+ mats_c2w = smoothed
+ except Exception:
+ pass
+
+ # Compute inverses
+ mats_wc = mats_c2w.astype(np.float32)
+ mats_w2c = np.linalg.inv(mats_c2w).astype(np.float32)
+
+ return mats_w2c, mats_wc
+
+
+_SMPLX_MODEL = None
+_SMPLX_DEVICE = None
+def _get_smplx_model(device):
+ global _SMPLX_MODEL, _SMPLX_DEVICE
+ if _SMPLX_MODEL is not None and _SMPLX_DEVICE == device: return _SMPLX_MODEL
+ from hmr4d.utils.smplx_utils import make_smplx
+ _SMPLX_MODEL = make_smplx("supermotion").to(device).eval()
+ _SMPLX_DEVICE = device
+ return _SMPLX_MODEL
+
+class SmplIncamRenderer:
+ def __init__(self, width, height, K4, device="cuda"):
+ from hmr4d.utils.vis.renderer import Renderer
+ self.device = device
+ self.smplx = _get_smplx_model(device)
+ self.faces = self.smplx.faces
+ self.renderer = Renderer(width, height, device=device, faces=self.faces, K=torch.from_numpy(k4_to_K3(K4)).to(device))
+ @torch.no_grad()
+ def render(self, img_rgb, go, bp, bt, tr, fl, pp):
+ K3 = torch.from_numpy(np.array([[fl[0], 0, pp[0]], [0, fl[1], pp[1]], [0, 0, 1]], dtype=np.float32)).to(self.device)
+ self.renderer.set_intrinsic(K3)
+ params = {"global_orient": torch.from_numpy(go[None]).to(self.device).float(), "body_pose": torch.from_numpy(bp[None]).to(self.device).float(), "betas": torch.from_numpy(bt[None]).to(self.device).float(), "transl": torch.from_numpy(tr[None]).to(self.device).float()}
+ verts = self.smplx(**params).vertices[0]
+ return self.renderer.render_mesh(verts, img_rgb, [0.8, 0.8, 0.8])
+ @torch.no_grad()
+ def get_verts(self, go, bp, bt, tr):
+ params = {"global_orient": torch.from_numpy(go[None]).to(self.device).float(), "body_pose": torch.from_numpy(bp[None]).to(self.device).float(), "betas": torch.from_numpy(bt[None]).to(self.device).float(), "transl": torch.from_numpy(tr[None]).to(self.device).float()}
+ return self.smplx(**params).vertices[0]
+
+def _compute_vitpose_selected_indices(num_frames, fps, bucket_seconds, frames_per_bucket, sampling="uniform", seed=123):
+ rng = np.random.default_rng(seed)
+ selected = []
+ bucket_len = max(1, int(round(bucket_seconds * fps)))
+ for b_start in range(0, num_frames, bucket_len):
+ b_end = min(num_frames, b_start + bucket_len)
+ k = min(frames_per_bucket, b_end - b_start)
+ if k <= 0: continue
+ if sampling == "random": idxs = np.sort(rng.choice(np.arange(b_start, b_end), size=k, replace=False)).tolist()
+ elif sampling == "linspace": idxs = np.linspace(b_start, b_end - 1, k, dtype=int).tolist()
+ else: idxs = [b_start + (b_end - b_start) // 2] if k == 1 else [min(b_start + i * ((b_end-b_start)//k), b_end-1) for i in range(k)]
+ selected.extend(idxs)
+ return sorted(list(set(selected)))
+
+def main():
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--input", required=True); parser.add_argument("--output", required=True)
+ parser.add_argument("--debug", action="store_true"); parser.add_argument("--vitpose", action="store_true")
+ parser.add_argument("--genmo", action="store_true"); parser.add_argument("--dpvo", action="store_true")
+ parser.add_argument("--smplx", action="store_true"); parser.add_argument("--debug_no_coco", action="store_true")
+ parser.add_argument("--shape_npz", default=os.path.join(os.path.dirname(__file__), "shape.npz"))
+ parser.add_argument("--vitpose_use_all_frames", action="store_true"); parser.add_argument("--vitpose_bucket_seconds", type=float, default=12.0)
+ parser.add_argument("--vitpose_frames_per_bucket", type=int, default=36); parser.add_argument("--vitpose_sampling", type=str, default="random")
+ parser.add_argument("--vitpose_seed", type=int, default=123); parser.add_argument("--ui_dir", type=str, default=None)
+ parser.add_argument("--ui_show_prob", type=float, default=0.25); parser.add_argument("--ui_max_images", type=int, default=3)
+ parser.add_argument("--ui_hold_min_s", type=float, default=0.7); parser.add_argument("--ui_hold_max_s", type=float, default=5.0)
+ parser.add_argument("--ui_seed", type=int, default=None); parser.add_argument("--keep_unity_scale", action="store_true")
+ parser.add_argument("--transl_source", type=str, default="pelvis"); parser.add_argument("--transl_y_offset_m", type=float, default=-0.020)
+ parser.add_argument("--consistency_check", action="store_true",
+ help="Report Unity incam/world consistency for a few frames.")
+ parser.add_argument("--consistency_check_frames", type=int, default=5,
+ help="Number of frames to check when --consistency_check is set.")
+ parser.add_argument(
+ "--camera_source",
+ type=str,
+ default="unity",
+ choices=["unity", "kabsch"],
+ help="Which world->camera transform to export for GENMO features. "
+ "`unity` uses the recorded Unity camera extrinsics (recommended); "
+ "`kabsch` uses SMPL Kabsch alignment (can diverge from the image camera).",
+ )
+ parser.add_argument("--ground_to_zero", action="store_true",
+ help="Shift world so GT ground is at y=0 (recommended for GENMO/GVHMR postprocess).")
+ parser.add_argument("--ground_frames", type=int, default=5,
+ help="Frames to sample (spread across the clip) to estimate ground height when --ground_to_zero is set.")
+ parser.add_argument("--max_duration_s", type=float, default=None,
+ help="Optionally clip each sequence to the first N seconds (e.g., 10.0).")
+ parser.add_argument(
+ "--vitpose_infer",
+ action="store_true",
+ help="Run ViTPose inference and export its kp2d as GENMO conditioning (overrides Unity-exported kp2d in the .pt).",
+ )
+ parser.add_argument(
+ "--dpvo_infer",
+ action="store_true",
+ help="Run DPVO (SLAM) and export its camera motion as GENMO conditioning (writes T_w2c_dpvo + cam_angvel/cam_tvel from it).",
+ )
+ parser.add_argument(
+ "--dpvo_resize",
+ type=float,
+ default=0.5,
+ help="DPVO internal resize factor (matches infer_video's `slam_resize`).",
+ )
+ parser.add_argument(
+ "--dpvo_buffer",
+ type=int,
+ default=4000,
+ help="DPVO buffer size.",
+ )
+ # Keep Unity->CV conversion minimal; no extra correction/rotation toggles.
+ parser.add_argument("--world_y_offset_m", type=float, default=0.0); parser.add_argument("--vit_batch_size", type=int, default=2048)
+ parser.add_argument("--max_samples", type=int, default=None)
+ parser.add_argument(
+ "--debug_kp2d_overlay",
+ action="store_true",
+ help="Write debug mp4 overlays for kp2d (Unity and, if enabled, ViTPose) on top of the original mp4.",
+ )
+ args = parser.parse_args()
+
+ if not (args.vitpose or args.genmo or args.dpvo or args.smplx):
+ args.vitpose = args.genmo = args.dpvo = args.smplx = True
+
+ device = "cuda"
+ vitpose_extractor_global = None
+ if args.vitpose_infer:
+ try:
+ from hmr4d.utils.preproc.vitpose import VitPoseExtractor as _VitPoseExtractor
+
+ # Initialize once (very expensive to reload the 2.5GB ckpt per sequence).
+ vitpose_extractor_global = _VitPoseExtractor(tqdm_leave=True)
+ except Exception as e:
+ print(f"[WARN] Failed to init ViTPose extractor: {e}")
+
+ # Load J Regressor if available (Used for Global Debug Render)
+ global_J_reg = None
+ j_reg_path = "third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt"
+ if os.path.exists(j_reg_path) and device == "cuda":
+ global_J_reg = torch.load(j_reg_path, map_location=device)
+
+ extractor_wrapper = Extractor(tqdm_leave=False) if args.genmo else None
+ mean_gpu = torch.tensor(IMAGENET_MEAN).view(1, 3, 1, 1).to(device)
+ std_gpu = torch.tensor(IMAGENET_STD).view(1, 3, 1, 1).to(device)
+ override_betas = np.load(args.shape_npz)["betas"]
+ jsonl_files = sorted(glob(os.path.join(args.input, "sequence_*.jsonl")))
+
+ processed_count = 0
+ for jsonl_idx, jsonl_path in enumerate(jsonl_files):
+ if args.max_samples is not None and processed_count >= args.max_samples:
+ print(f"Reached max_samples limit ({args.max_samples}). Stopping.")
+ break
+
+ seq_name = os.path.basename(jsonl_path).replace("sequence_", "").replace(".jsonl", "")
+ print(f"[{jsonl_idx+1}/{len(jsonl_files)}] Processing {seq_name}...")
+ processed_count += 1
+
+ prof = {"BatchPrep": 0.0, "Read": 0.0, "Overlay": 0.0, "ViT": 0.0, "DbgRend": 0.0, "SaveFiles": 0.0}
+ t_start_seq = time.perf_counter()
+
+ with open(jsonl_path, "r") as f: all_lines = f.readlines()
+ lines = all_lines[1:] if len(all_lines) > 1 else []
+ if args.max_duration_s is not None:
+ try:
+ max_frames = int(round(float(args.max_duration_s) * FPS))
+ if max_frames > 0:
+ lines = lines[:max_frames]
+ except Exception:
+ pass
+ num_frames = len(lines)
+ if num_frames == 0: continue
+
+ jsonl_dir = os.path.dirname(jsonl_path)
+ video_path = os.path.join(jsonl_dir, f"video_{seq_name}.mp4")
+ if not os.path.exists(video_path): video_path = os.path.join(jsonl_dir, "video.mp4")
+
+ cap = cv2.VideoCapture(video_path)
+ if not cap.isOpened():
+ print(f" [ERROR] Skipping {seq_name} - video not found.")
+ continue
+
+ W, H = int(cap.get(3)), int(cap.get(4))
+ cap.read() # Burn frame 0
+
+ resolved_ui_dir, resolved_font = _resolve_ui_dir_and_font(args.ui_dir)
+ ui_seed = int(getattr(args, "ui_seed", 0) or 0)
+ min_hold_frames = int(max(1, round(float(getattr(args, "ui_hold_min_s", 0.7)) * FPS)))
+ max_hold_frames = int(max(min_hold_frames, round(float(getattr(args, "ui_hold_max_s", 5.0)) * FPS)))
+ chat = SimpleChatOverlay(W, H, seed=ui_seed, font_path=resolved_font)
+ ui = SimpleUIOverlay(
+ W,
+ H,
+ seed=ui_seed,
+ ui_dir=resolved_ui_dir,
+ max_images=int(getattr(args, "ui_max_images", 3)),
+ show_prob=float(getattr(args, "ui_show_prob", 0.25)),
+ min_hold_frames=min_hold_frames,
+ max_hold_frames=max_hold_frames,
+ )
+
+ # --- PRE-CALCULATION (matching process_dataset_fromincam.py) ---
+ t0_pre = time.perf_counter()
+ parsed = [parse_smpl_inputs_from_row(json.loads(l), override_betas, args.keep_unity_scale, args.transl_source, args.transl_y_offset_m) for l in lines]
+ pel0 = batch_smpl_forward(np.stack([p['betas'] for p in parsed]), np.stack([p['global_orient'] for p in parsed]), np.stack([p['body_pose'] for p in parsed]), device)
+
+ # Coordinate conversion matrices (matching process_dataset_fromincam.py)
+ fix_rot = R.from_euler("z", 180, degrees=True).as_matrix()
+ C = np.diag([1.0, -1.0, 1.0])
+ C4 = np.diag([1.0, -1.0, 1.0, 1.0])
+ fix_mat = np.eye(4); fix_mat[:3, :3] = fix_rot
+
+ # Derive global SMPL from incam × camera (ensures consistency)
+ all_go_w, all_pel_w_cv = [], []
+ for p in parsed:
+ R_cam_w_unity = R.from_quat(p['cam_rot_w_quat']).as_matrix()
+ R_cam_w_cv = fix_rot @ (C @ R_cam_w_unity @ C)
+ R_pelvis_c_cv = R.from_rotvec(p['global_orient']).as_matrix()
+ R_pelvis_w_cv = R_cam_w_cv @ R_pelvis_c_cv
+ all_go_w.append(R.from_matrix(R_pelvis_w_cv).as_rotvec().astype(np.float32))
+
+ pos_cv_raw = (C @ p['pelvis_pos_world'])
+ pelvis_pos_w_cv = fix_rot @ pos_cv_raw
+ all_pel_w_cv.append(pelvis_pos_w_cv)
+
+ pel0_w = batch_smpl_forward(np.stack([p['betas'] for p in parsed]), np.stack(all_go_w), np.stack([p['body_pose'] for p in parsed]), device)
+ prof["BatchPrep"] = time.perf_counter() - t0_pre
+
+ smpl_renderer, vid_incam, vid_global = None, None, None
+ debug_verts = []
+ if args.debug:
+ os.makedirs(os.path.join(args.output, "debug_renders"), exist_ok=True)
+ try:
+ K4_init = np.asarray(json.loads(lines[0])["cam_intrinsics"], dtype=np.float32)
+ smpl_renderer = SmplIncamRenderer(W, H, K4_init, device=device)
+ vid_incam = cv2.VideoWriter(os.path.join(args.output, "debug_renders", f"{seq_name}_incam.mp4"), cv2.VideoWriter_fourcc(*'mp4v'), FPS, (W, H))
+ vid_global = cv2.VideoWriter(os.path.join(args.output, "debug_renders", f"{seq_name}_global.mp4"), cv2.VideoWriter_fourcc(*'mp4v'), FPS, (960, 540))
+ except: pass
+
+ vit_thread = None
+ if args.genmo:
+ vit_thread = VitInferenceThread(extractor_wrapper.extractor.to(device), args.vit_batch_size, device, mean_gpu, std_gpu)
+ vit_thread.start()
+
+ selected_set = set(_compute_vitpose_selected_indices(num_frames, FPS, args.vitpose_bucket_seconds, args.vitpose_frames_per_bucket, args.vitpose_sampling, args.vitpose_seed)) if args.vitpose else set()
+ coco_subset, img_paths = [], []
+ results = {"go_c": [], "tr_c": [], "go_w": [], "tr_w": [], "bp": [], "bt": [], "K": [], "kp2d": [], "kp2d_unity": [], "bbx": [], "mats_wc": [], "mats_w2c": []}
+
+ try:
+ for idx in tqdm(range(num_frames), leave=False):
+ t0 = time.perf_counter(); ret, img = cap.read(); prof["Read"] += (time.perf_counter() - t0)
+ if not ret: break
+
+ t0 = time.perf_counter(); chat.maybe_append(idx); chat.draw(img); ui.draw(img); prof["Overlay"] += (time.perf_counter() - t0)
+
+ row = json.loads(lines[idx]); p = parsed[idx]; bbox = clamp_bbox(row["bbox"], W, H)
+ img_rel = os.path.join("images", seq_name, f"img_{idx:05d}.jpg").replace("\\", "/")
+ img_paths.append(img_rel)
+
+ tr_c = (p["target_cam_cv"] - pel0[idx]).astype(np.float32)
+ if str(args.transl_source) == "root": tr_w = all_pel_w_cv[idx].astype(np.float32)
+ else: tr_w = (all_pel_w_cv[idx] - pel0_w[idx]).astype(np.float32)
+
+ # BEDLAM/3DPW-style bbx_xys (center/size) derived from bbox xyxy with base_enlarge.
+ # This matches the normalization path used by GENMO (normalize_kp2d uses bbx_xys).
+ x, y, w, h = bbox
+ bbx_xyxy = torch.tensor([[x, y, x + w, y + h]], dtype=torch.float32)
+ bbx_xys = (
+ get_bbx_xys_from_xyxy(bbx_xyxy, base_enlarge=1.2)[0]
+ .cpu()
+ .numpy()
+ .astype(np.float32)
+ )
+
+ # Use bbx_xys square crop for ViT feature extraction (keeps f_imgseq consistent).
+ bbox_crop = bbx_xywh_from_bbx_xys(bbx_xys, W, H)
+ if vit_thread:
+ vit_thread.input_queue.put(
+ img[
+ bbox_crop[1] : bbox_crop[1] + bbox_crop[3],
+ bbox_crop[0] : bbox_crop[0] + bbox_crop[2],
+ ]
+ )
+
+ results["go_c"].append(p["global_orient"]); results["bp"].append(p["body_pose"]); results["bt"].append(p["betas"])
+ results["tr_c"].append(tr_c); results["go_w"].append(all_go_w[idx]); results["tr_w"].append(tr_w)
+ results["K"].append(k4_to_K3(row["cam_intrinsics"]))
+ results["bbx"].append(bbx_xys)
+
+ kpts_raw = np.array(row["kpts_2d"]).reshape(-1,2)[:17]
+ vis_raw = np.array(row["kpts_vis"])[:17]
+ if len(vis_raw) >= 5: vis_raw[3] = 1; vis_raw[4] = 1
+ kp2d_unity = np.concatenate([kpts_raw, (vis_raw > 0).astype(np.float32)[:, None]], axis=1)
+ results["kp2d"].append(kp2d_unity)
+ results["kp2d_unity"].append(kp2d_unity)
+
+ # Camera matrix with fix_mat (matching process_dataset_fromincam.py)
+ p_w = p["cam_pos_world"]; q_w = p["cam_rot_w_quat"]
+ cam_T_wc = build_T_wc(p_w, q_w)
+ cam_T_wc_cv = fix_mat @ (C4 @ cam_T_wc @ C4)
+ cam_T_w2c_cv = np.linalg.inv(cam_T_wc_cv)
+ results["mats_wc"].append(cam_T_wc_cv); results["mats_w2c"].append(cam_T_w2c_cv)
+
+ if args.vitpose and idx in selected_set:
+ p_out = os.path.join(args.output, "images", seq_name, f"img_{idx:05d}.jpg")
+ os.makedirs(os.path.dirname(p_out), exist_ok=True)
+ cv2.imwrite(p_out, img)
+ kpts_coco = []
+ for k in range(17): kpts_coco.extend([float(kpts_raw[k, 0]), float(kpts_raw[k, 1]), int(vis_raw[k])])
+ coco_subset.append(({"file_name": img_rel, "width": W, "height": H}, {"bbox": bbox, "keypoints": kpts_coco, "category_id": 1, "iscrowd": 0}))
+
+ if args.debug and idx < DEBUG_NUM_FRAMES and smpl_renderer and vid_incam:
+ dbg = smpl_renderer.render(img[:,:,::-1].copy(), p["global_orient"], p["body_pose"], p["betas"], tr_c, row["cam_intrinsics"][:2], row["cam_intrinsics"][2:])
+ if not args.debug_no_coco:
+ draw_bbox_xywh_and_center(dbg, bbox)
+ draw_vis_text_and_points(dbg, kpts_raw, vis_raw)
+ vid_incam.write(dbg[:,:,::-1])
+ debug_verts.append(smpl_renderer.get_verts(all_go_w[idx], p["body_pose"], p["betas"], tr_w).cpu())
+ except KeyboardInterrupt: print("Stopping...")
+ finally: cap.release()
+ prof["FrameLoop"] = time.perf_counter() - t_start_seq - prof["BatchPrep"]
+
+ # --- KABSCH VERIFICATION: Check camera-SMPL consistency (frame 0 only for speed) ---
+ t0_verify = time.perf_counter()
+ if args.consistency_check and len(results["go_c"]) > 0 and len(results["go_w"]) > 0:
+ smplx_model = _get_smplx_model(device)
+ smpl_c = {
+ "global_orient": results["go_c"][0],
+ "body_pose": results["bp"][0],
+ "betas": results["bt"][0],
+ "transl": results["tr_c"][0]
+ }
+ smpl_w = {
+ "global_orient": results["go_w"][0],
+ "body_pose": results["bp"][0],
+ "betas": results["bt"][0],
+ "transl": results["tr_w"][0]
+ }
+ # Derive T_w2c from body params (same formula used at save time)
+ R_c_v = R.from_rotvec(results["go_c"][0]).as_matrix()
+ R_w_v = R.from_rotvec(results["go_w"][0]).as_matrix()
+ R_w2c_v = R_c_v @ R_w_v.T
+ T_w2c = np.eye(4, dtype=np.float32)
+ T_w2c[:3, :3] = R_w2c_v
+
+ # Use raw Unity T_w2c translation for validation instead of
+ # mathematically deriving it (which misses spawn offsets)
+ T_w2c[:3, 3] = results["mats_w2c"][0][:3, 3]
+ angle_err, t_err, _, _ = verify_camera_smpl_consistency(smpl_c, smpl_w, T_w2c, smplx_model, device)
+ # Note: Translation error >0 is expected due to retargeting mismatch between
+ # Unity's rendered skeleton (pelvis.position) and SMPL's rest-pose (smplx_pose).
+ # This does NOT affect saved features - T_w2c uses Kabsch-aligned physical camera.
+ Log.info(f"[Kabsch] Frame 0: Rot err = {angle_err:.2f}°, Transl err = {t_err:.3f}m (retarget bias, expected)")
+ prof["KabschVerify"] = time.perf_counter() - t0_verify
+
+ old_results = []
+ t0_vitwait = time.perf_counter()
+ if vit_thread:
+ vit_thread.stop_signal = True; vit_thread.join(); prof["ViT"] = vit_thread.proc_time; old_results = vit_thread.results
+ prof["VitWait"] = time.perf_counter() - t0_vitwait
+
+ # Free GPU VRAM from the ViT feature extractor before running other GPU-heavy preprocessors
+ # (ViTPose / DPVO). Keep the wrapper object so the next sequence can move it back to CUDA.
+ if args.genmo:
+ try:
+ del vit_thread
+ except Exception:
+ pass
+ try:
+ if extractor_wrapper is not None and hasattr(extractor_wrapper, "extractor"):
+ extractor_wrapper.extractor.to("cpu")
+ except Exception:
+ pass
+ try:
+ gc.collect()
+ torch.cuda.empty_cache()
+ except Exception:
+ pass
+
+ # Optional: run ViTPose inference and use it as kp2d conditioning for GENMO export.
+ # NOTE: This is separate from `--vitpose` (COCO export). This actually produces (L, 17, 3) kp2d.
+ if args.vitpose_infer:
+ try:
+ print(f"[ViTPoseInfer] Running ViTPose on {num_frames} frames for {seq_name}...")
+ bbx_xys = torch.from_numpy(np.stack(results["bbx"]).astype(np.float32))
+ if vitpose_extractor_global is None:
+ raise RuntimeError("ViTPose extractor is not initialized.")
+ vit_kp2d = vitpose_extractor_global.extract(video_path, bbx_xys) # (L, 17, 3)
+ # Override kp2d used by GENMO export; keep Unity kp2d separately in kp2d_unity.
+ results["kp2d"] = [x.numpy() for x in vit_kp2d.cpu()]
+ except Exception as e:
+ print(f"[WARN] vitpose_infer failed for {seq_name}: {e}")
+ vit_kp2d = None
+ else:
+ vit_kp2d = None
+
+ # Optional: debug render kp2d overlays on the original video.
+ if args.debug_kp2d_overlay:
+ try:
+ os.makedirs(os.path.join(args.output, "debug_renders"), exist_ok=True)
+ bbx_seq = np.stack(results["bbx"]).astype(np.float32)
+ # Prefer ViTPose overlay when available; Unity kp2d is only a fallback.
+ if vit_kp2d is not None:
+ _render_kp2d_overlay_video(
+ video_path=video_path,
+ kp2d_seq=np.stack(results["kp2d"]).astype(np.float32),
+ bbx_xys_seq=bbx_seq,
+ out_path=os.path.join(
+ args.output, "debug_renders", f"{seq_name}_kp2d_vitpose.mp4"
+ ),
+ title=f"{seq_name} kp2d_vitpose",
+ ui_dir=args.ui_dir,
+ ui_seed=getattr(args, "ui_seed", 0) or 0,
+ ui_show_prob=getattr(args, "ui_show_prob", 0.25),
+ ui_max_images=getattr(args, "ui_max_images", 3),
+ ui_hold_min_s=getattr(args, "ui_hold_min_s", 0.7),
+ ui_hold_max_s=getattr(args, "ui_hold_max_s", 5.0),
+ )
+ else:
+ _render_kp2d_overlay_video(
+ video_path=video_path,
+ kp2d_seq=np.stack(results["kp2d_unity"]).astype(np.float32),
+ bbx_xys_seq=bbx_seq,
+ out_path=os.path.join(
+ args.output, "debug_renders", f"{seq_name}_kp2d_unity.mp4"
+ ),
+ title=f"{seq_name} kp2d_unity",
+ ui_dir=args.ui_dir,
+ ui_seed=getattr(args, "ui_seed", 0) or 0,
+ ui_show_prob=getattr(args, "ui_show_prob", 0.25),
+ ui_max_images=getattr(args, "ui_max_images", 3),
+ ui_hold_min_s=getattr(args, "ui_hold_min_s", 0.7),
+ ui_hold_max_s=getattr(args, "ui_hold_max_s", 5.0),
+ )
+ except Exception as e:
+ print(f"[WARN] debug_kp2d_overlay failed for {seq_name}: {e}")
+
+ # Optional: run DPVO (SLAM) and export its camera motion for GENMO conditioning.
+ # This matches `scripts/demo/infer_video.py`'s camera-motion path: T_w2c + cam_angvel + cam_tvel.
+ dpvo_T_w2c = None
+ dpvo_cam_angvel = None
+ dpvo_cam_tvel = None
+ if args.dpvo_infer:
+ try:
+ # `REPO_ROOT` is `third_party/GVHMR`, so import via `hmr4d.*` (same as demo pipeline).
+ from hmr4d.utils.preproc.slam import SLAMModel
+ from hmr4d.utils.geo.hmr_cam import estimate_K
+
+ slam_resize = float(args.dpvo_resize)
+ K_est = estimate_K(W, H)
+ if isinstance(K_est, torch.Tensor):
+ K_est = K_est.cpu().numpy()
+ fx, fy, cx, cy = float(K_est[0, 0]), float(K_est[1, 1]), float(K_est[0, 2]), float(K_est[1, 2])
+ fx *= slam_resize
+ fy *= slam_resize
+ cx *= slam_resize
+ cy *= slam_resize
+ intrinsics = torch.tensor([fx, fy, cx, cy], dtype=torch.float32)
+
+ with torch.inference_mode():
+ slam = SLAMModel(
+ video_path,
+ W,
+ H,
+ intrinsics,
+ buffer=int(args.dpvo_buffer),
+ resize=slam_resize,
+ max_frames=int(num_frames),
+ )
+ print(f"[DPVO] Tracking camera for {seq_name} (resize={slam_resize})...")
+ tracked = 0
+ while True:
+ if not slam.track():
+ break
+ tracked += 1
+ if tracked % 200 == 0:
+ print(f"[DPVO] {seq_name}: tracked {tracked} frames...")
+ print(f"[DPVO] {seq_name}: finishing (tracked {tracked} frames), optimizing trajectory...")
+ traj = slam.process() # usually (N,7)
+ print(f"[DPVO] {seq_name}: done.")
+ try:
+ del slam
+ gc.collect()
+ torch.cuda.empty_cache()
+ except Exception:
+ pass
+
+ if isinstance(traj, torch.Tensor):
+ traj = traj.detach().cpu().numpy()
+ if isinstance(traj, np.ndarray) and traj.ndim == 2 and traj.shape[1] == 7:
+ dpvo_T_w2c = _dpvo7_to_T_w2c(traj)
+ elif isinstance(traj, np.ndarray) and traj.ndim == 3 and traj.shape[1:] == (4, 4):
+ dpvo_T_w2c = traj.astype(np.float32)
+
+ if dpvo_T_w2c is not None:
+ # Pad / trim to match sequence length (num_frames)
+ if dpvo_T_w2c.shape[0] < num_frames:
+ pad = np.repeat(dpvo_T_w2c[-1:,:,:], num_frames - dpvo_T_w2c.shape[0], axis=0)
+ dpvo_T_w2c = np.concatenate([dpvo_T_w2c, pad], axis=0)
+ dpvo_T_w2c = dpvo_T_w2c[:num_frames]
+ dpvo_cam_angvel, dpvo_cam_tvel = compute_velocity(dpvo_T_w2c)
+ except Exception as e:
+ print(f"[WARN] dpvo_infer failed for {seq_name}: {e}")
+
+ # --- RESTORED GLOBAL DEBUG RENDERER WITH CAMERA GIZMO ---
+ if args.debug and vid_global and len(debug_verts) > 0:
+ try:
+ from hmr4d.utils.vis.renderer import Renderer, get_global_cameras_static, get_ground_params_from_points, perspective_projection
+ from hmr4d.utils.geo.hmr_cam import create_camera_sensor
+
+ _, _, K_gl = create_camera_sensor(960, 540, 24)
+ gl_rend = Renderer(960, 540, device=device, faces=smpl_renderer.faces, K=K_gl.to(device), bin_size=0)
+ v_seq = torch.stack(debug_verts); off = v_seq[0].mean(0); off[1] = v_seq[0,:,1].min(); v_seq -= off
+
+ # Cam centers calculation
+ cam_centers = None
+ try:
+ F_len = int(v_seq.shape[0])
+ if len(results["mats_wc"]) >= F_len:
+ cam_wc = np.stack(results["mats_wc"][:F_len], axis=0).astype(np.float32)
+ cam_centers = torch.from_numpy(cam_wc[:, :3, 3]).to(device=device) - off.to(device=device)[None]
+ except: cam_centers = None
+
+ g_R, g_T, g_L = get_global_cameras_static(v_seq, beta=2.0, cam_height_degree=20, target_center_height=1.0, device=device)
+
+ if global_J_reg is not None and v_seq.shape[1] == global_J_reg.shape[-1]:
+ roots = torch.einsum("jv,fvk->fjk", global_J_reg.cpu(), v_seq)[:, 0]
+ else: roots = v_seq.mean(1)
+
+ sc, cx, cz = get_ground_params_from_points(roots, v_seq)
+ gl_rend.set_ground(sc*1.5, cx, cz)
+ col = torch.tensor([[0.0, 1.0, 0.0]], device=device)
+ trail = []
+
+ # Helper functions inside local scope to access gl_rend
+ def _project_xy(points_w):
+ return perspective_projection(points_w.view(1,-1,3), gl_rend.K, gl_rend.R, gl_rend.T.reshape(1,3,1))[0]
+ def _draw_polyline(img, pts_xy, color, closed=False, thickness=1):
+ pts = np.asarray(pts_xy, dtype=np.int32).reshape(-1, 1, 2)
+ if len(pts) >= 2: cv2.polylines(img, [pts], bool(closed), color, int(thickness), cv2.LINE_AA)
+ def _draw_camera_box_axes(img, C_w, right, up, fwd, scale=0.25):
+ C_w = C_w.reshape(3); right = right.reshape(3); up = up.reshape(3); fwd = fwd.reshape(3)
+ L = float(scale)
+ _draw_polyline(img, _project_xy(torch.stack([C_w, C_w + L*right])).detach().cpu().numpy(), (0, 0, 255), thickness=2)
+ _draw_polyline(img, _project_xy(torch.stack([C_w, C_w + L*up])).detach().cpu().numpy(), (0, 255, 0), thickness=2)
+ _draw_polyline(img, _project_xy(torch.stack([C_w, C_w + L*fwd])).detach().cpu().numpy(), (255, 0, 0), thickness=2)
+
+ for i in range(len(v_seq)):
+ cam = gl_rend.create_camera(g_R[i], g_T[i])
+ img_g = gl_rend.render_with_ground(v_seq[i].to(device)[None], col, cam, g_L)
+ img_bgr = img_g[:, :, ::-1].copy()
+
+ if cam_centers is not None and i < cam_centers.shape[0]:
+ try:
+ if i < roots.shape[0]:
+ xy_line = _project_xy(torch.stack([cam_centers[i], roots[i].to(device=device)])).detach().cpu().numpy()
+ _draw_polyline(img_bgr, xy_line, (255, 200, 50), closed=False, thickness=1)
+
+ x2d = _project_xy(cam_centers[i].view(1,3))[0]
+ x, y = int(round(float(x2d[0].item()))), int(round(float(x2d[1].item())))
+ if 0 <= x < img_bgr.shape[1] and 0 <= y < img_bgr.shape[0]:
+ trail.append((x, y))
+ cv2.circle(img_bgr, (x,y), 3, (0,0,255), -1)
+ if len(trail) >= 2: cv2.polylines(img_bgr, [np.array(trail, dtype=np.int32)], False, (0,0,255), 1)
+
+ if len(results["mats_wc"]) > i:
+ R_c2w = torch.from_numpy(np.asarray(results["mats_wc"][i], dtype=np.float32)[:3, :3]).to(device=device)
+ _draw_camera_box_axes(img_bgr, cam_centers[i], R_c2w[:,0], R_c2w[:,1], R_c2w[:,2], scale=0.35)
+ except: pass
+ vid_global.write(img_bgr)
+ except Exception as e: pass
+ if vid_global: vid_global.release()
+ if vid_incam: vid_incam.release()
+
+ t0_kabsch = time.perf_counter()
+
+ # --- CHOOSE CAMERA SOURCE FOR EXPORT ---
+ # Compute this ONCE so it can be used for both GenMo .pt export and SMPLx .npz export.
+ mats_w2c_c = None
+ mats_wc_c = None
+ mats_wc = None
+ camera_source = str(getattr(args, "camera_source", "unity")).strip().lower()
+ if camera_source == "kabsch":
+ # Kabsch-derived camera (can diverge from image camera; kept for debugging).
+ if len(results["go_c"]) > 0:
+ smplx_model = _get_smplx_model(device)
+ smpl_params_c_list = [
+ {"global_orient": results["go_c"][i], "body_pose": results["bp"][i],
+ "betas": results["bt"][i], "transl": results["tr_c"][i]}
+ for i in range(len(results["go_c"]))
+ ]
+ smpl_params_w_list = [
+ {"global_orient": results["go_w"][i], "body_pose": results["bp"][i],
+ "betas": results["bt"][i], "transl": results["tr_w"][i]}
+ for i in range(len(results["go_w"]))
+ ]
+ mats_w2c, mats_wc = compute_camera_transforms_from_smpl(
+ smpl_params_c_list,
+ smpl_params_w_list,
+ smplx_model,
+ device,
+ smooth_alpha=0.2,
+ unity_mats_wc=np.stack(results["mats_wc"]),
+ )
+
+ # Recompute global_orient from Kabsch camera (not Unity camera!)
+ all_go_w_kabsch = []
+ for i in range(len(results["go_c"])):
+ R_c2w_kabsch = mats_wc[i, :3, :3] # Kabsch camera-to-world rotation
+ R_pelvis_c = R.from_rotvec(results["go_c"][i]).as_matrix() # Incam pelvis rotation
+ R_pelvis_w_kabsch = R_c2w_kabsch @ R_pelvis_c # Global pelvis rotation
+ all_go_w_kabsch.append(R.from_matrix(R_pelvis_w_kabsch).as_rotvec().astype(np.float32))
+ results["go_w"] = all_go_w_kabsch
+
+ # Apply world offset (needed for final export)
+ trans_w = np.stack(results["tr_w"]).astype(np.float32)
+ world_off = trans_w[0].copy(); world_off[1] -= float(args.world_y_offset_m)
+ T_wp_w = np.eye(4, dtype=np.float32); T_wp_w[:3, 3] = world_off
+ T_w_wp = np.eye(4, dtype=np.float32); T_w_wp[:3, 3] = -world_off
+ mats_w2c_c = np.matmul(mats_w2c, T_wp_w[None])
+ mats_wc_c = np.matmul(T_w_wp[None], mats_wc)
+ else:
+ # Unity camera: export the recorded Unity camera extrinsics (recommended).
+ if len(results["mats_w2c"]) > 0 and len(results["tr_c"]) > 0:
+ mats_w2c = np.stack(results["mats_w2c"]).astype(np.float32)
+ mats_wc = np.stack(results["mats_wc"]).astype(np.float32)
+ trans_c_all = np.stack(results["tr_c"]).astype(np.float32)
+
+ R_w2c = mats_w2c[:, :3, :3].astype(np.float32)
+ t_w2c = mats_w2c[:, :3, 3].astype(np.float32)
+ R_c2w = np.transpose(R_w2c, (0, 2, 1)).astype(np.float32)
+ t_c2w = -np.einsum("nij,nj->ni", R_c2w, t_w2c).astype(np.float32)
+ trans_w = (np.einsum("nij,nj->ni", R_c2w, trans_c_all) + t_c2w).astype(np.float32)
+
+ world_off = trans_w[0].copy()
+ world_off[1] -= float(args.world_y_offset_m)
+ T_wp_w = np.eye(4, dtype=np.float32); T_wp_w[:3, 3] = world_off
+ T_w_wp = np.eye(4, dtype=np.float32); T_w_wp[:3, 3] = -world_off
+ mats_w2c_c = np.matmul(mats_w2c, T_wp_w[None])
+ mats_wc_c = np.matmul(T_w_wp[None], mats_wc)
+ prof["KabschCompute"] = time.perf_counter() - t0_kabsch
+
+ t0_save = time.perf_counter()
+
+ # Save GENMO features if ViT feature extraction ran successfully (old_results populated).
+ if args.genmo and old_results and len(results["tr_c"]) > 0 and mats_w2c_c is not None:
+ save_p = os.path.join(args.output, "genmo_features", f"{seq_name}.pt"); os.makedirs(os.path.dirname(save_p), exist_ok=True)
+ trans_c = np.stack(results["tr_c"]).astype(np.float32)
+ go_c = np.stack(results["go_c"]).astype(np.float32)
+
+ camera_source = str(getattr(args, "camera_source", "unity")).strip().lower()
+ if camera_source == "kabsch":
+ go_w = np.stack(results["go_w"]).astype(np.float32)
+
+ # BEDLAM-style: derive R_w2c from incam/global rotations.
+ R_w = np.stack([R.from_rotvec(go).as_matrix() for go in go_w]) # (N, 3, 3)
+ R_c = np.stack([R.from_rotvec(go).as_matrix() for go in go_c]) # (N, 3, 3)
+ R_w2c_derived = np.matmul(R_c, np.transpose(R_w, (0, 2, 1))) # R_c @ R_w.T
+
+ N = len(go_w)
+ T_w2c_derived = np.eye(4, dtype=np.float32)[None].repeat(N, axis=0)
+ T_w2c_derived[:, :3, :3] = R_w2c_derived.astype(np.float32)
+ t_w2c_derived = mats_w2c_c[:, :3, 3].astype(np.float32)
+ T_w2c_derived[:, :3, 3] = t_w2c_derived
+
+ R_c2w_derived = np.transpose(R_w2c_derived, (0, 2, 1)).astype(np.float32)
+ t_c2w_derived = -np.einsum("nij,nj->ni", R_c2w_derived, t_w2c_derived).astype(np.float32)
+ trans_w_wp_consistent = (np.einsum("nij,nj->ni", R_c2w_derived, trans_c) + t_c2w_derived).astype(np.float32)
+ else:
+ # Unity camera: export the recorded camera and derive world params from it.
+ T_w2c_derived = mats_w2c_c.astype(np.float32)
+ R_w2c_derived = T_w2c_derived[:, :3, :3].astype(np.float32)
+ t_w2c_derived = T_w2c_derived[:, :3, 3].astype(np.float32)
+ R_c2w_derived = np.transpose(R_w2c_derived, (0, 2, 1)).astype(np.float32)
+ t_c2w_derived = -np.einsum("nij,nj->ni", R_c2w_derived, t_w2c_derived).astype(np.float32)
+
+ trans_w_wp_consistent = (np.einsum("nij,nj->ni", R_c2w_derived, trans_c) + t_c2w_derived).astype(np.float32)
+
+ R_pelvis_c = np.stack([R.from_rotvec(go).as_matrix() for go in go_c]).astype(np.float32)
+ R_pelvis_w = np.matmul(R_c2w_derived, R_pelvis_c).astype(np.float32)
+ go_w = np.stack([R.from_matrix(Rw).as_rotvec().astype(np.float32) for Rw in R_pelvis_w]).astype(np.float32)
+
+ # Optional: shift the entire sequence so that GT ground is at y=0.
+ # This is important because `pp_static_joint` always grounds predictions to y=0;
+ # if GT has a large negative floor offset (common in Unity exports), validation will show
+ # a constant +|floor| meter "floating" even when everything else is correct.
+ if args.ground_to_zero:
+ try:
+ smplx_model = _get_smplx_model(device)
+ smplx2smpl = _get_smplx2smpl(device)
+
+ N_total = int(trans_w_wp_consistent.shape[0])
+ k = int(max(1, min(int(args.ground_frames), N_total)))
+ idxs = np.linspace(0, N_total - 1, k).round().astype(np.int64)
+
+ # Build a small batch on GPU for speed.
+ params = {
+ "global_orient": torch.from_numpy(go_w[idxs]).to(device).float(),
+ "body_pose": torch.from_numpy(np.stack(results["bp"])[idxs]).to(device).float(),
+ "betas": torch.from_numpy(np.stack(results["bt"])[idxs]).to(device).float(),
+ "transl": torch.from_numpy(trans_w_wp_consistent[idxs]).to(device).float(),
+ }
+ with torch.no_grad():
+ verts = smplx_model(**params).vertices # (k, Vx, 3)
+ if smplx2smpl is not None:
+ verts = torch.stack([torch.matmul(smplx2smpl, v) for v in verts]) # (k, 6890, 3)
+ ground_y = float(verts[..., 1].min().item())
+
+ # Apply the shift in the source world frame: p' = p - shift, shift=[0,ground_y,0]
+ shift = np.array([0.0, ground_y, 0.0], dtype=np.float32)
+ trans_w_wp_consistent = trans_w_wp_consistent - shift[None]
+
+ # Update T_w2c accordingly: t' = t + R * shift
+ t_w2c_derived = (t_w2c_derived.astype(np.float32) + np.einsum("nij,j->ni", R_w2c_derived.astype(np.float32), shift)).astype(np.float32)
+ T_w2c_derived[:, :3, 3] = t_w2c_derived
+
+ # Keep metadata consistent.
+ world_off = (world_off.astype(np.float32) - shift).astype(np.float32)
+ except Exception as e:
+ print(f"[WARN] ground_to_zero failed for {seq_name}: {e}")
+
+ # Compute cam_angvel from derived T_w2c (Unity camera). If DPVO was requested and succeeded,
+ # use DPVO cam motion for conditioning and keep Unity motion for reference.
+ cam_av_unity, cam_tv_unity = compute_velocity(T_w2c_derived)
+ cam_av, cam_tv = cam_av_unity, cam_tv_unity
+ if dpvo_cam_angvel is not None and dpvo_cam_tvel is not None and dpvo_T_w2c is not None:
+ cam_av, cam_tv = dpvo_cam_angvel, dpvo_cam_tvel
+
+ torch.save({
+ "smpl_params_c": {"global_orient": torch.from_numpy(go_c), "body_pose": torch.from_numpy(np.stack(results["bp"])), "transl": torch.from_numpy(trans_c), "betas": torch.from_numpy(np.stack(results["bt"]))},
+ "smpl_params_w": {"global_orient": torch.from_numpy(go_w), "body_pose": torch.from_numpy(np.stack(results["bp"])), "transl": torch.from_numpy(trans_w_wp_consistent.astype(np.float32)), "betas": torch.from_numpy(np.stack(results["bt"]))},
+ "T_w2c": torch.from_numpy(T_w2c_derived), "K_fullimg": torch.from_numpy(np.stack(results["K"])),
+ "kp2d": torch.from_numpy(np.stack(results["kp2d"])),
+ "kp2d_unity": torch.from_numpy(np.stack(results["kp2d_unity"])),
+ "bbx_xys": torch.from_numpy(np.stack(results["bbx"])),
+ "cam_angvel": torch.from_numpy(cam_av),
+ "cam_tvel": torch.from_numpy(cam_tv),
+ "cam_angvel_unity": torch.from_numpy(cam_av_unity),
+ "cam_tvel_unity": torch.from_numpy(cam_tv_unity),
+ **(
+ {"T_w2c_dpvo": torch.from_numpy(dpvo_T_w2c)}
+ if dpvo_T_w2c is not None
+ else {}
+ ),
+ "f_imgseq": torch.cat(old_results), "imgname": img_paths, "valid_mask": torch.ones(len(img_paths), dtype=torch.float32),
+ "world_offset": torch.from_numpy(world_off.astype(np.float32))
+ }, save_p)
+
+ if args.smplx and len(results["tr_w"]) > 0:
+ os.makedirs(os.path.join(args.output, "smplx_global"), exist_ok=True)
+ trans_c = np.stack(results["tr_c"]).astype(np.float32)
+ go_w = np.stack(results["go_w"]).astype(np.float32)
+ go_c = np.stack(results["go_c"]).astype(np.float32)
+ # Use the same consistency rule as the GenMO export when possible.
+ if mats_w2c_c is not None:
+ # Build the same derived T_w2c as above.
+ R_w = np.stack([R.from_rotvec(go).as_matrix() for go in go_w]) # (N, 3, 3)
+ R_c = np.stack([R.from_rotvec(go).as_matrix() for go in go_c]) # (N, 3, 3)
+ R_w2c_derived = np.matmul(R_c, np.transpose(R_w, (0, 2, 1))).astype(np.float32)
+ t_w2c_derived = mats_w2c_c[:, :3, 3].astype(np.float32)
+ R_c2w_derived = np.transpose(R_w2c_derived, (0, 2, 1))
+ t_c2w_derived = -np.einsum("nij,nj->ni", R_c2w_derived, t_w2c_derived)
+ trans_w_norm = np.einsum("nij,nj->ni", R_c2w_derived, trans_c) + t_c2w_derived
+ else:
+ trans_w = np.stack(results["tr_w"]).astype(np.float32)
+ world_off = trans_w[0].copy(); world_off[1] -= float(args.world_y_offset_m)
+ trans_w_norm = trans_w - world_off[None]
+
+ # Export with corrected camera transform if available
+ out_dict = {
+ "mocap_framerate": int(FPS),
+ "gender": "neutral",
+ "betas": results["bt"][0],
+ "trans": trans_w_norm,
+ "world_offset": world_off,
+ "poses": np.pad(np.concatenate([np.stack(results["go_w"]), np.stack(results["bp"])], axis=1), ((0,0),(0,99)), mode="constant")
+ }
+ if mats_wc_c is not None:
+ out_dict["camera_transform"] = mats_wc_c # Camera-to-World (with world offset applied)
+ if len(results["K"]) > 0:
+ out_dict["K_fullimg"] = np.stack(results["K"])
+
+ np.savez(os.path.join(args.output, "smplx_global", f"{seq_name}_global.npz"), **out_dict)
+
+ if coco_subset:
+ with open(os.path.join(args.output, "vitpose", "temp_annotations", f"{seq_name}.json"), "w") as f: json.dump(coco_subset, f)
+
+ prof["SaveFiles"] = time.perf_counter() - t0_save
+ total_t = time.perf_counter() - t_start_seq
+ print(f" > Done in {total_t:.2f}s | FPS: {num_frames/max(total_t, 0.001):.1f}")
+ print(f" [Breakdown] BatchPrep: {prof['BatchPrep']:.2f}s | FrameLoop: {prof.get('FrameLoop', 0):.2f}s | KabschVerify: {prof.get('KabschVerify', 0):.2f}s | VitWait: {prof.get('VitWait', 0):.2f}s | KabschCompute: {prof.get('KabschCompute', 0):.2f}s | SaveFiles: {prof['SaveFiles']:.2f}s")
+ print(f" [FrameLoop] Read: {prof['Read']:.2f}s | Overlay: {prof['Overlay']:.2f}s | ViT(GPU): {prof['ViT']:.2f}s")
+
+ processed_count += 1
+ if args.max_samples is not None and processed_count >= args.max_samples:
+ print(f"Reached max_samples limit ({args.max_samples}). Stopping.")
+ break
+
+if __name__ == "__main__": main()
diff --git a/third_party/GVHMR/tools/demo/process_dataset_fromincam.py b/third_party/GVHMR/tools/demo/process_dataset_fromincam.py
new file mode 100644
index 0000000000000000000000000000000000000000..c4015fdf8fee60e2c78be54683fc01072ca0b266
--- /dev/null
+++ b/third_party/GVHMR/tools/demo/process_dataset_fromincam.py
@@ -0,0 +1,566 @@
+import sys
+import os
+import json
+import argparse
+import numpy as np
+import zlib
+from glob import glob
+from tqdm import tqdm
+import cv2
+import torch
+from scipy.spatial.transform import Rotation as R
+import time
+import threading
+import queue
+from pathlib import Path
+from PIL import Image, ImageDraw, ImageFont
+
+# Suppress libpng warnings and OpenCV noise
+os.environ["OPENCV_LOG_LEVEL"] = "FATAL"
+
+# --- SETUP PATHS ---
+REPO_ROOT = Path(__file__).resolve().parents[2]
+if str(REPO_ROOT) not in sys.path:
+ sys.path.insert(0, str(REPO_ROOT))
+
+from hmr4d.utils.preproc.vitfeat_extractor import Extractor
+from hmr4d.utils.pylogger import Log
+
+# --- SPEED OPTIMIZATIONS ---
+if "OMP_NUM_THREADS" in os.environ: del os.environ["OMP_NUM_THREADS"]
+cv2.setNumThreads(1)
+torch.set_num_threads(os.cpu_count())
+torch.backends.cudnn.benchmark = True
+os.environ["PYTORCH_CUDA_ALLOC_CONF"] = "expandable_segments:True"
+
+FPS = 30.0
+DEBUG_NUM_FRAMES = 5
+IMAGENET_MEAN = np.array([0.485, 0.456, 0.406], dtype=np.float32)
+IMAGENET_STD = np.array([0.229, 0.224, 0.225], dtype=np.float32)
+
+# --- GLOBAL ASSET CACHE (FAST) ---
+UI_ASSET_CACHE = {}
+def _get_ui_assets(ui_dir):
+ if not ui_dir: return []
+ if ui_dir in UI_ASSET_CACHE: return UI_ASSET_CACHE[ui_dir]
+ imgs = []
+ if os.path.isdir(ui_dir):
+ for name in sorted(os.listdir(ui_dir)):
+ p = os.path.join(ui_dir, name)
+ if name.lower().endswith(('.png', '.jpg', '.jpeg', '.webp')):
+ im = cv2.imread(p, cv2.IMREAD_UNCHANGED)
+ if im is not None: imgs.append(im)
+ UI_ASSET_CACHE[ui_dir] = imgs
+ return imgs
+
+# --- THREADED VIT PROCESSOR (FAST) ---
+class VitInferenceThread(threading.Thread):
+ def __init__(self, model, batch_size, device, mean, std):
+ super().__init__()
+ self.model = model
+ self.batch_size = batch_size
+ self.device = device
+ self.mean = mean
+ self.std = std
+ self.input_queue = queue.Queue(maxsize=batch_size * 4)
+ self.results = []
+ self.stop_signal = False
+ self.proc_time = 0.0
+
+ def run(self):
+ batch = []
+ while not self.stop_signal or not self.input_queue.empty():
+ try:
+ raw_crop = self.input_queue.get(timeout=0.1)
+ if raw_crop is None: continue
+ if raw_crop.size == 0:
+ resized = np.zeros((3, 256, 256), dtype=np.uint8)
+ else:
+ resized = cv2.resize(raw_crop, (256, 256), interpolation=cv2.INTER_LINEAR)
+ resized = resized[:, :, ::-1].transpose(2, 0, 1).copy()
+
+ batch.append(torch.from_numpy(resized))
+ if len(batch) >= self.batch_size:
+ self._process_batch(batch)
+ batch = []
+ except queue.Empty: continue
+ if batch: self._process_batch(batch)
+
+ def _process_batch(self, batch):
+ t0 = time.perf_counter()
+ batch_t = torch.stack(batch).to(self.device, non_blocking=True).float()
+ batch_t = (batch_t / 255.0 - self.mean) / self.std
+ with torch.inference_mode(), torch.amp.autocast("cuda"):
+ feats = self.model({"img": batch_t})
+ self.results.append(feats.detach().cpu())
+ self.proc_time += (time.perf_counter() - t0)
+
+# --- FAST OVERLAY LOGIC ---
+def _alpha_blend_fast(roi, src):
+ if src.shape[2] < 4:
+ roi[:] = src[:, :, :3]
+ return
+ alpha = src[:, :, 3].astype(np.uint16)
+ inv_alpha = 255 - alpha
+ for c in range(3):
+ roi[:, :, c] = ((src[:, :, c].astype(np.uint16) * alpha + roi[:, :, c].astype(np.uint16) * inv_alpha) >> 8).astype(np.uint8)
+
+class SimpleUIOverlay:
+ def __init__(self, width, height, seed=0, ui_dir=None, max_images=4, show_prob=0.6, min_hold_frames=20, max_hold_frames=120):
+ self.W, self.H, self.rng = width, height, np.random.default_rng(seed)
+ self.max_images, self.show_prob = max_images, show_prob
+ self.min_hold, self.max_hold = min_hold_frames, max_hold_frames
+ self.assets = _get_ui_assets(ui_dir if ui_dir else os.path.join(os.getcwd(), "UI"))
+ self._ttl, self._active = 0, []
+ def _pick_state(self):
+ self._ttl = self.rng.integers(self.min_hold, self.max_hold + 1)
+ self._active = []
+ if self.assets and self.rng.random() < self.show_prob:
+ k = min(self.max_images, len(self.assets))
+ for idx in self.rng.choice(len(self.assets), size=k, replace=False):
+ im = self.assets[idx]
+ h, w = im.shape[:2]
+ self._active.append((im, self.rng.integers(-w//4, self.W-w), self.rng.integers(-h//4, self.H-h)))
+ def draw(self, img_bgr):
+ if self._ttl <= 0: self._pick_state()
+ self._ttl -= 1
+ for im, x, y in self._active:
+ x0, y0 = max(x, 0), max(y, 0)
+ x1, y1 = min(x + im.shape[1], self.W), min(y + im.shape[0], self.H)
+ if x1 > x0 and y1 > y0:
+ _alpha_blend_fast(img_bgr[y0:y1, x0:x1], im[(y0-y):(y1-y), (x0-x):(x1-x)])
+
+class SimpleChatOverlay:
+ def __init__(self, width, height, seed=0, num_lines=7, region_w=420, region_h=180, margin=18, font_path=None):
+ from collections import deque
+ self.W, self.H, self.rng = width, height, np.random.default_rng(seed)
+ self.messages = deque([{"user": "bot", "text": "pog", "color": (255, 0, 0)}] * num_lines, maxlen=num_lines)
+ self.region_w, self.region_h, self.margin = region_w, region_h, margin
+ self.font = ImageFont.truetype(font_path, 18) if font_path and os.path.exists(font_path) else None
+ self._cached, self._dirty = None, True
+ def maybe_append(self, idx):
+ if idx % 15 == 0: self._dirty = True
+ def draw(self, img_bgr):
+ if self._dirty:
+ pil = Image.new("RGBA", (self.region_w, self.region_h), (0,0,0,0))
+ draw = ImageDraw.Draw(pil)
+ for i, m in enumerate(self.messages):
+ draw.text((5, i*24), f"{m['user']}: {m['text']}", font=self.font, fill=(255,255,255))
+ self._cached = cv2.cvtColor(np.asarray(pil), cv2.COLOR_RGBA2BGRA)
+ self._dirty = False
+ y_off = self.H - self.region_h - self.margin
+ _alpha_blend_fast(img_bgr[y_off:y_off+self.region_h, self.margin:self.margin+self.region_w], self._cached)
+
+# --- ORIGINAL MATH & HELPERS ---
+def k4_to_K3(k4): return np.array([[k4[0], 0, k4[2]], [0, k4[1], k4[3]], [0, 0, 1]], dtype=np.float32)
+def bbox_xywh_to_bbx_xys(bbox, scale=1.0):
+ return np.array([bbox[0] + 0.5*bbox[2], bbox[1] + 0.5*bbox[3], max(bbox[2], bbox[3])*scale], dtype=np.float32)
+def clamp_bbox(bbox, W, H):
+ x, y, w, h = [float(v) for v in bbox]
+ x1, y1 = np.clip(x, 0, W-1.0), np.clip(y, 0, H-1.0)
+ x2, y2 = np.clip(x+w, 0, W), np.clip(y+h, 0, H)
+ return [int(x1), int(y1), int(max(1, x2-x1)), int(max(1, y2-y1))]
+
+def build_T_wc(pos_world, quat_world_xyzw):
+ T = np.eye(4, dtype=np.float64)
+ T[:3, :3] = R.from_quat(np.asarray(quat_world_xyzw, dtype=np.float64)).as_matrix()
+ T[:3, 3] = np.asarray(pos_world, dtype=np.float64)
+ return T
+
+def compute_velocity(mats, fps=30.0):
+ N = len(mats)
+ if N < 2: return np.zeros((N, 3), dtype=np.float32), np.zeros((N, 3), dtype=np.float32)
+ R_curr = mats[:, :3, :3]
+ R_diff = np.matmul(R_curr[1:], np.transpose(R_curr[:-1], (0, 2, 1)))
+ angvel = np.zeros((N, 3), dtype=np.float32)
+ angvel[1:] = R.from_matrix(R_diff).as_rotvec()
+ t_curr = mats[:, :3, 3]
+ tvel = np.zeros((N, 3), dtype=np.float32)
+ tvel[1:] = t_curr[1:] - t_curr[:-1]
+ return angvel.astype(np.float32), tvel.astype(np.float32)
+
+# --- VISUALIZATION HELPERS (Restored Original Colors/Text) ---
+def vis_label_and_color(v: int):
+ # Original logic: 2=VIS(Green), 1=OCC(Orange), 0=OFF(Grey)
+ if v == 2: return "VIS", (0, 255, 0)
+ if v == 1: return "OCC", (0, 165, 255)
+ return "OFF", (160, 160, 160)
+
+def draw_vis_text_and_points(img_bgr, kpts2d_xy, vis17):
+ for k in range(17):
+ v = int(vis17[k])
+ label, color = vis_label_and_color(v)
+ x, y = int(round(kpts2d_xy[k, 0])), int(round(kpts2d_xy[k, 1]))
+ if v > 0: cv2.circle(img_bgr, (x, y), 4, color, -1)
+ cv2.putText(img_bgr, f"{k}:{label}", (x + 6, y - 6), cv2.FONT_HERSHEY_SIMPLEX, 0.45, color, 1, cv2.LINE_AA)
+
+def draw_bbox_xywh_and_center(img_bgr, bbox_xywh, color=(255, 255, 0)):
+ x, y, w, h = [float(v) for v in bbox_xywh]
+ cv2.rectangle(img_bgr, (int(x), int(y)), (int(x+w), int(y+h)), color, 2)
+ cv2.circle(img_bgr, (int(x+w/2), int(y+h/2)), 4, (0, 0, 255), -1)
+
+# --- ORIGINAL PARSING LOGIC ---
+def parse_smpl_inputs_from_row(row, override_betas10=None, keep_unity_scale=False, transl_source="pelvis", transl_y_offset_m=0.0):
+ C = np.diag([1.0, -1.0, 1.0]).astype(np.float64)
+ cam_rot_w_quat = np.array(row["cam_rot_world"], dtype=np.float64)
+ R_cam_w = R.from_quat(cam_rot_w_quat).as_matrix()
+ pel_rot_w_quat = np.array(row["pelvis_rot_world"], dtype=np.float64)
+ R_pel_w = R.from_quat(pel_rot_w_quat).as_matrix()
+
+ R_rel_unity = R_cam_w.T @ R_pel_w
+ R_cv = C @ R_rel_unity @ C
+ R_final = R_cv @ R.from_euler("z", 180, degrees=True).as_matrix()
+ global_orient_aa = R.from_matrix(R_final).as_rotvec().astype(np.float32)
+
+ smpl_scale = float(row.get("smpl_root_world_scale", 1.0))
+ pelvis_cam_unity = np.asarray(row["smpl_incam_transl"], dtype=np.float64).reshape(3)
+ root_cam_unity = np.asarray(row.get("smpl_root_incam_transl", [0.0, 0.0, 0.0]), dtype=np.float64).reshape(3)
+ pelvis_cam_unity = pelvis_cam_unity + np.array([0.0, float(transl_y_offset_m), 0.0], dtype=np.float64)
+
+ if str(transl_source).strip().lower() == "root": target_cam_unity = root_cam_unity
+ else:
+ if bool(keep_unity_scale): target_cam_unity = pelvis_cam_unity
+ else:
+ if abs(smpl_scale) > 1e-8: target_cam_unity = root_cam_unity + (pelvis_cam_unity - root_cam_unity) / smpl_scale
+ else: target_cam_unity = pelvis_cam_unity
+ target_cam_cv = (C @ target_cam_unity).astype(np.float64)
+
+ pose = np.asarray(row["smplx_pose"], dtype=np.float32)
+ body_pose = pose[3:66].astype(np.float32)
+ betas10 = np.zeros(10, dtype=np.float32)
+ if override_betas10 is not None: betas10[:min(10, override_betas10.size)] = override_betas10.flatten()[:10]
+
+ return {
+ "global_orient": global_orient_aa, "body_pose": body_pose, "betas": betas10,
+ "target_cam_cv": target_cam_cv, "cam_rot_w_quat": cam_rot_w_quat,
+ "cam_pos_world": np.asarray(row["cam_pos_world"], dtype=np.float64).reshape(3),
+ "pelvis_pos_world": np.asarray(row["pelvis_pos_world"], dtype=np.float64).reshape(3),
+ "smpl_scale": smpl_scale, "root_cam_unity": root_cam_unity
+ }
+
+def batch_smpl_forward(betas, global_orient, body_pose, device):
+ from hmr4d.utils.smplx_utils import make_smplx
+ model = make_smplx("supermotion").to(device).eval()
+ pelvis_list = []
+ with torch.no_grad():
+ for i in range(0, len(betas), 2048):
+ b_bt = torch.from_numpy(betas[i:i+2048]).to(device).float()
+ b_go = torch.from_numpy(global_orient[i:i+2048]).to(device).float()
+ b_bp = torch.from_numpy(body_pose[i:i+2048]).to(device).float()
+ b_tr = torch.zeros((len(b_bt), 3), dtype=torch.float32, device=device)
+ out = model(betas=b_bt, global_orient=b_go, body_pose=b_bp, transl=b_tr)
+ pelvis_list.append(out.joints[:, 0, :].cpu().numpy())
+ return np.concatenate(pelvis_list, axis=0)
+
+_SMPLX_MODEL = None
+_SMPLX_DEVICE = None
+def _get_smplx_model(device):
+ global _SMPLX_MODEL, _SMPLX_DEVICE
+ if _SMPLX_MODEL is not None and _SMPLX_DEVICE == device: return _SMPLX_MODEL
+ from hmr4d.utils.smplx_utils import make_smplx
+ _SMPLX_MODEL = make_smplx("supermotion").to(device).eval()
+ _SMPLX_DEVICE = device
+ return _SMPLX_MODEL
+
+class SmplIncamRenderer:
+ def __init__(self, width, height, K4, device="cuda"):
+ from hmr4d.utils.vis.renderer import Renderer
+ self.device = device
+ self.smplx = _get_smplx_model(device)
+ self.faces = self.smplx.faces
+ self.renderer = Renderer(width, height, device=device, faces=self.faces, K=torch.from_numpy(k4_to_K3(K4)).to(device))
+ @torch.no_grad()
+ def render(self, img_rgb, go, bp, bt, tr, fl, pp):
+ K3 = torch.from_numpy(np.array([[fl[0], 0, pp[0]], [0, fl[1], pp[1]], [0, 0, 1]], dtype=np.float32)).to(self.device)
+ self.renderer.set_intrinsic(K3)
+ params = {"global_orient": torch.from_numpy(go[None]).to(self.device).float(), "body_pose": torch.from_numpy(bp[None]).to(self.device).float(), "betas": torch.from_numpy(bt[None]).to(self.device).float(), "transl": torch.from_numpy(tr[None]).to(self.device).float()}
+ verts = self.smplx(**params).vertices[0]
+ return self.renderer.render_mesh(verts, img_rgb, [0.8, 0.8, 0.8])
+ @torch.no_grad()
+ def get_verts(self, go, bp, bt, tr):
+ params = {"global_orient": torch.from_numpy(go[None]).to(self.device).float(), "body_pose": torch.from_numpy(bp[None]).to(self.device).float(), "betas": torch.from_numpy(bt[None]).to(self.device).float(), "transl": torch.from_numpy(tr[None]).to(self.device).float()}
+ return self.smplx(**params).vertices[0]
+
+def _compute_vitpose_selected_indices(num_frames, fps, bucket_seconds, frames_per_bucket, sampling="uniform", seed=123):
+ rng = np.random.default_rng(seed)
+ selected = []
+ bucket_len = max(1, int(round(bucket_seconds * fps)))
+ for b_start in range(0, num_frames, bucket_len):
+ b_end = min(num_frames, b_start + bucket_len)
+ k = min(frames_per_bucket, b_end - b_start)
+ if k <= 0: continue
+ if sampling == "random": idxs = np.sort(rng.choice(np.arange(b_start, b_end), size=k, replace=False)).tolist()
+ elif sampling == "linspace": idxs = np.linspace(b_start, b_end - 1, k, dtype=int).tolist()
+ else: idxs = [b_start + (b_end - b_start) // 2] if k == 1 else [min(b_start + i * ((b_end-b_start)//k), b_end-1) for i in range(k)]
+ selected.extend(idxs)
+ return sorted(list(set(selected)))
+
+def main():
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--input", required=True); parser.add_argument("--output", required=True)
+ parser.add_argument("--debug", action="store_true"); parser.add_argument("--vitpose", action="store_true")
+ parser.add_argument("--genmo", action="store_true"); parser.add_argument("--dpvo", action="store_true")
+ parser.add_argument("--smplx", action="store_true"); parser.add_argument("--debug_no_coco", action="store_true")
+ parser.add_argument("--shape_npz", default=os.path.join(os.path.dirname(__file__), "shape.npz"))
+ parser.add_argument("--vitpose_use_all_frames", action="store_true"); parser.add_argument("--vitpose_bucket_seconds", type=float, default=12.0)
+ parser.add_argument("--vitpose_frames_per_bucket", type=int, default=36); parser.add_argument("--vitpose_sampling", type=str, default="random")
+ parser.add_argument("--vitpose_seed", type=int, default=123); parser.add_argument("--ui_dir", type=str, default=None)
+ parser.add_argument("--ui_show_prob", type=float, default=0.25); parser.add_argument("--ui_max_images", type=int, default=3)
+ parser.add_argument("--ui_hold_min_s", type=float, default=0.7); parser.add_argument("--ui_hold_max_s", type=float, default=5.0)
+ parser.add_argument("--ui_seed", type=int, default=None); parser.add_argument("--keep_unity_scale", action="store_true")
+ parser.add_argument("--transl_source", type=str, default="pelvis"); parser.add_argument("--transl_y_offset_m", type=float, default=-0.020)
+ parser.add_argument("--world_y_offset_m", type=float, default=1.3415); parser.add_argument("--vit_batch_size", type=int, default=2048)
+ args = parser.parse_args()
+
+ if not (args.vitpose or args.genmo or args.dpvo or args.smplx):
+ args.vitpose = args.genmo = args.dpvo = args.smplx = True
+
+ device = "cuda"
+
+ # Load J Regressor if available (Used for Global Debug Render)
+ global_J_reg = None
+ j_reg_path = "third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt"
+ if os.path.exists(j_reg_path) and device == "cuda":
+ global_J_reg = torch.load(j_reg_path, map_location=device)
+
+ extractor_wrapper = Extractor(tqdm_leave=False) if args.genmo else None
+ mean_gpu = torch.tensor(IMAGENET_MEAN).view(1, 3, 1, 1).to(device)
+ std_gpu = torch.tensor(IMAGENET_STD).view(1, 3, 1, 1).to(device)
+ override_betas = np.load(args.shape_npz)["betas"]
+ jsonl_files = sorted(glob(os.path.join(args.input, "sequence_*.jsonl")))
+
+ for jsonl_idx, jsonl_path in enumerate(jsonl_files):
+ seq_name = os.path.basename(jsonl_path).replace("sequence_", "").replace(".jsonl", "")
+ print(f"[{jsonl_idx+1}/{len(jsonl_files)}] Processing {seq_name}...")
+
+ prof = {"BatchPrep": 0.0, "Read": 0.0, "Overlay": 0.0, "ViT": 0.0, "DbgRend": 0.0, "SaveFiles": 0.0}
+ t_start_seq = time.perf_counter()
+
+ with open(jsonl_path, "r") as f: all_lines = f.readlines()
+ lines = all_lines[1:] if len(all_lines) > 1 else []
+ num_frames = len(lines)
+ if num_frames == 0: continue
+
+ jsonl_dir = os.path.dirname(jsonl_path)
+ video_path = os.path.join(jsonl_dir, f"video_{seq_name}.mp4")
+ if not os.path.exists(video_path): video_path = os.path.join(jsonl_dir, "video.mp4")
+
+ cap = cv2.VideoCapture(video_path)
+ if not cap.isOpened():
+ print(f" [ERROR] Skipping {seq_name} - video not found.")
+ continue
+
+ W, H = int(cap.get(3)), int(cap.get(4))
+ cap.read() # Burn frame 0
+
+ chat = SimpleChatOverlay(W, H, font_path=os.path.join(os.getcwd(), "UI", "Inter_18pt-Bold.ttf"))
+ ui = SimpleUIOverlay(W, H, ui_dir=args.ui_dir, max_images=args.ui_max_images, show_prob=args.ui_show_prob)
+
+ # --- PRE-CALCULATION ---
+ t0_pre = time.perf_counter()
+ parsed = [parse_smpl_inputs_from_row(json.loads(l), override_betas, args.keep_unity_scale, args.transl_source, args.transl_y_offset_m) for l in lines]
+ pel0 = batch_smpl_forward(np.stack([p['betas'] for p in parsed]), np.stack([p['global_orient'] for p in parsed]), np.stack([p['body_pose'] for p in parsed]), device)
+
+ # Exact Matrix Fix Logic
+ fix_rot = R.from_euler("z", 180, degrees=True).as_matrix()
+ C = np.diag([1.0, -1.0, 1.0])
+ C4 = np.diag([1.0, -1.0, 1.0, 1.0])
+ fix_mat = np.eye(4); fix_mat[:3, :3] = fix_rot
+
+ all_go_w, all_pel_w_cv = [], []
+ for p in parsed:
+ R_cam_w_unity = R.from_quat(p['cam_rot_w_quat']).as_matrix()
+ R_cam_w_cv = fix_rot @ (C @ R_cam_w_unity @ C)
+ R_pelvis_c_cv = R.from_rotvec(p['global_orient']).as_matrix()
+ R_pelvis_w_cv = R_cam_w_cv @ R_pelvis_c_cv
+ all_go_w.append(R.from_matrix(R_pelvis_w_cv).as_rotvec().astype(np.float32))
+
+ pos_cv_raw = (C @ p['pelvis_pos_world'])
+ pelvis_pos_w_cv = fix_rot @ pos_cv_raw
+ all_pel_w_cv.append(pelvis_pos_w_cv)
+
+ pel0_w = batch_smpl_forward(np.stack([p['betas'] for p in parsed]), np.stack(all_go_w), np.stack([p['body_pose'] for p in parsed]), device)
+ prof["BatchPrep"] = time.perf_counter() - t0_pre
+
+ smpl_renderer, vid_incam, vid_global = None, None, None
+ debug_verts = []
+ if args.debug:
+ os.makedirs(os.path.join(args.output, "debug_renders"), exist_ok=True)
+ try:
+ K4_init = np.asarray(json.loads(lines[0])["cam_intrinsics"], dtype=np.float32)
+ smpl_renderer = SmplIncamRenderer(W, H, K4_init, device=device)
+ vid_incam = cv2.VideoWriter(os.path.join(args.output, "debug_renders", f"{seq_name}_incam.mp4"), cv2.VideoWriter_fourcc(*'mp4v'), FPS, (W, H))
+ vid_global = cv2.VideoWriter(os.path.join(args.output, "debug_renders", f"{seq_name}_global.mp4"), cv2.VideoWriter_fourcc(*'mp4v'), FPS, (960, 540))
+ except: pass
+
+ vit_thread = None
+ if args.genmo:
+ vit_thread = VitInferenceThread(extractor_wrapper.extractor.to(device), args.vit_batch_size, device, mean_gpu, std_gpu)
+ vit_thread.start()
+
+ selected_set = set(_compute_vitpose_selected_indices(num_frames, FPS, args.vitpose_bucket_seconds, args.vitpose_frames_per_bucket, args.vitpose_sampling, args.vitpose_seed)) if args.vitpose else set()
+ coco_subset, img_paths = [], []
+ results = {"go_c": [], "tr_c": [], "go_w": [], "tr_w": [], "bp": [], "bt": [], "K": [], "kp2d": [], "bbx": [], "mats_wc": [], "mats_w2c": []}
+
+ try:
+ for idx in tqdm(range(num_frames), leave=False):
+ t0 = time.perf_counter(); ret, img = cap.read(); prof["Read"] += (time.perf_counter() - t0)
+ if not ret: break
+
+ t0 = time.perf_counter(); chat.maybe_append(idx); chat.draw(img); ui.draw(img); prof["Overlay"] += (time.perf_counter() - t0)
+
+ row = json.loads(lines[idx]); p = parsed[idx]; bbox = clamp_bbox(row["bbox"], W, H)
+ img_rel = os.path.join("images", seq_name, f"img_{idx:05d}.jpg").replace("\\", "/")
+ img_paths.append(img_rel)
+
+ tr_c = (p["target_cam_cv"] - pel0[idx]).astype(np.float32)
+ if str(args.transl_source) == "root": tr_w = all_pel_w_cv[idx].astype(np.float32)
+ else: tr_w = (all_pel_w_cv[idx] - pel0_w[idx]).astype(np.float32)
+
+ if vit_thread: vit_thread.input_queue.put(img[bbox[1]:bbox[1]+bbox[3], bbox[0]:bbox[0]+bbox[2]])
+
+ results["go_c"].append(p["global_orient"]); results["bp"].append(p["body_pose"]); results["bt"].append(p["betas"])
+ results["tr_c"].append(tr_c); results["go_w"].append(all_go_w[idx]); results["tr_w"].append(tr_w)
+ results["K"].append(k4_to_K3(row["cam_intrinsics"])); results["bbx"].append(bbox_xywh_to_bbx_xys(bbox))
+
+ kpts_raw = np.array(row["kpts_2d"]).reshape(-1,2)[:17]
+ vis_raw = np.array(row["kpts_vis"])[:17]
+ if len(vis_raw) >= 5: vis_raw[3] = 1; vis_raw[4] = 1
+ results["kp2d"].append(np.concatenate([kpts_raw, (vis_raw > 0).astype(np.float32)[:, None]], axis=1))
+
+ # Exact Matrix Logic
+ p_w = p["cam_pos_world"]; q_w = p["cam_rot_w_quat"]
+ cam_T_wc = build_T_wc(p_w, q_w)
+ cam_T_wc_cv = fix_mat @ (C4 @ cam_T_wc @ C4)
+ cam_T_w2c_cv = np.linalg.inv(cam_T_wc_cv)
+ results["mats_wc"].append(cam_T_wc_cv); results["mats_w2c"].append(cam_T_w2c_cv)
+
+ if args.vitpose and idx in selected_set:
+ p_out = os.path.join(args.output, "images", seq_name, f"img_{idx:05d}.jpg")
+ os.makedirs(os.path.dirname(p_out), exist_ok=True)
+ cv2.imwrite(p_out, img)
+ kpts_coco = []
+ for k in range(17): kpts_coco.extend([float(kpts_raw[k, 0]), float(kpts_raw[k, 1]), int(vis_raw[k])])
+ coco_subset.append(({"file_name": img_rel, "width": W, "height": H}, {"bbox": bbox, "keypoints": kpts_coco, "category_id": 1, "iscrowd": 0}))
+
+ if args.debug and idx < DEBUG_NUM_FRAMES and smpl_renderer and vid_incam:
+ dbg = smpl_renderer.render(img[:,:,::-1].copy(), p["global_orient"], p["body_pose"], p["betas"], tr_c, row["cam_intrinsics"][:2], row["cam_intrinsics"][2:])
+ if not args.debug_no_coco:
+ draw_bbox_xywh_and_center(dbg, bbox)
+ draw_vis_text_and_points(dbg, kpts_raw, vis_raw)
+ vid_incam.write(dbg[:,:,::-1])
+ debug_verts.append(smpl_renderer.get_verts(all_go_w[idx], p["body_pose"], p["betas"], tr_w).cpu())
+ except KeyboardInterrupt: print("Stopping...")
+ finally: cap.release()
+
+ old_results = []
+ if vit_thread:
+ t0 = time.perf_counter(); vit_thread.stop_signal = True; vit_thread.join(); prof["ViT"] = vit_thread.proc_time; old_results = vit_thread.results
+
+ # --- RESTORED GLOBAL DEBUG RENDERER WITH CAMERA GIZMO ---
+ if args.debug and vid_global and len(debug_verts) > 0:
+ try:
+ from hmr4d.utils.vis.renderer import Renderer, get_global_cameras_static, get_ground_params_from_points, perspective_projection
+ from hmr4d.utils.geo.hmr_cam import create_camera_sensor
+
+ _, _, K_gl = create_camera_sensor(960, 540, 24)
+ gl_rend = Renderer(960, 540, device=device, faces=smpl_renderer.faces, K=K_gl.to(device), bin_size=0)
+ v_seq = torch.stack(debug_verts); off = v_seq[0].mean(0); off[1] = v_seq[0,:,1].min(); v_seq -= off
+
+ # Cam centers calculation
+ cam_centers = None
+ try:
+ F_len = int(v_seq.shape[0])
+ if len(results["mats_wc"]) >= F_len:
+ cam_wc = np.stack(results["mats_wc"][:F_len], axis=0).astype(np.float32)
+ cam_centers = torch.from_numpy(cam_wc[:, :3, 3]).to(device=device) - off.to(device=device)[None]
+ except: cam_centers = None
+
+ g_R, g_T, g_L = get_global_cameras_static(v_seq, beta=2.0, cam_height_degree=20, target_center_height=1.0, device=device)
+
+ if global_J_reg is not None and v_seq.shape[1] == global_J_reg.shape[-1]:
+ roots = torch.einsum("jv,fvk->fjk", global_J_reg.cpu(), v_seq)[:, 0]
+ else: roots = v_seq.mean(1)
+
+ sc, cx, cz = get_ground_params_from_points(roots, v_seq)
+ gl_rend.set_ground(sc*1.5, cx, cz)
+ col = torch.tensor([[0.0, 1.0, 0.0]], device=device)
+ trail = []
+
+ # Helper functions inside local scope to access gl_rend
+ def _project_xy(points_w):
+ return perspective_projection(points_w.view(1,-1,3), gl_rend.K, gl_rend.R, gl_rend.T.reshape(1,3,1))[0]
+ def _draw_polyline(img, pts_xy, color, closed=False, thickness=1):
+ pts = np.asarray(pts_xy, dtype=np.int32).reshape(-1, 1, 2)
+ if len(pts) >= 2: cv2.polylines(img, [pts], bool(closed), color, int(thickness), cv2.LINE_AA)
+ def _draw_camera_box_axes(img, C_w, right, up, fwd, scale=0.25):
+ C_w = C_w.reshape(3); right = right.reshape(3); up = up.reshape(3); fwd = fwd.reshape(3)
+ L = float(scale)
+ _draw_polyline(img, _project_xy(torch.stack([C_w, C_w + L*right])).detach().cpu().numpy(), (0, 0, 255), thickness=2)
+ _draw_polyline(img, _project_xy(torch.stack([C_w, C_w + L*up])).detach().cpu().numpy(), (0, 255, 0), thickness=2)
+ _draw_polyline(img, _project_xy(torch.stack([C_w, C_w + L*fwd])).detach().cpu().numpy(), (255, 0, 0), thickness=2)
+
+ for i in range(len(v_seq)):
+ cam = gl_rend.create_camera(g_R[i], g_T[i])
+ img_g = gl_rend.render_with_ground(v_seq[i].to(device)[None], col, cam, g_L)
+ img_bgr = img_g[:, :, ::-1].copy()
+
+ if cam_centers is not None and i < cam_centers.shape[0]:
+ try:
+ if i < roots.shape[0]:
+ xy_line = _project_xy(torch.stack([cam_centers[i], roots[i].to(device=device)])).detach().cpu().numpy()
+ _draw_polyline(img_bgr, xy_line, (255, 200, 50), closed=False, thickness=1)
+
+ x2d = _project_xy(cam_centers[i].view(1,3))[0]
+ x, y = int(round(float(x2d[0].item()))), int(round(float(x2d[1].item())))
+ if 0 <= x < img_bgr.shape[1] and 0 <= y < img_bgr.shape[0]:
+ trail.append((x, y))
+ cv2.circle(img_bgr, (x,y), 3, (0,0,255), -1)
+ if len(trail) >= 2: cv2.polylines(img_bgr, [np.array(trail, dtype=np.int32)], False, (0,0,255), 1)
+
+ if len(results["mats_wc"]) > i:
+ R_c2w = torch.from_numpy(np.asarray(results["mats_wc"][i], dtype=np.float32)[:3, :3]).to(device=device)
+ _draw_camera_box_axes(img_bgr, cam_centers[i], R_c2w[:,0], R_c2w[:,1], R_c2w[:,2], scale=0.35)
+ except: pass
+ vid_global.write(img_bgr)
+ except Exception as e: pass
+ if vid_global: vid_global.release()
+ if vid_incam: vid_incam.release()
+
+ t0 = time.perf_counter()
+ if args.genmo and vit_thread and old_results and len(results["tr_w"]) > 0:
+ save_p = os.path.join(args.output, "genmo_features", f"{seq_name}.pt"); os.makedirs(os.path.dirname(save_p), exist_ok=True)
+ trans_w = np.stack(results["tr_w"]).astype(np.float32)
+ world_off = trans_w[0].copy(); world_off[1] -= float(args.world_y_offset_m)
+ mats_wc = np.stack(results["mats_wc"]).astype(np.float32)
+ mats_w2c = np.stack(results["mats_w2c"]).astype(np.float32)
+ T_wp_w = np.eye(4, dtype=np.float32); T_wp_w[:3, 3] = world_off
+ T_w_wp = np.eye(4, dtype=np.float32); T_w_wp[:3, 3] = -world_off
+ mats_w2c_c = np.matmul(mats_w2c, T_wp_w[None])
+ mats_wc_c = np.matmul(T_w_wp[None], mats_wc)
+ cam_av, cam_tv = compute_velocity(mats_wc_c)
+
+ torch.save({
+ "smpl_params_c": {"global_orient": torch.from_numpy(np.stack(results["go_c"])), "body_pose": torch.from_numpy(np.stack(results["bp"])), "transl": torch.from_numpy(np.stack(results["tr_c"])), "betas": torch.from_numpy(np.stack(results["bt"]))},
+ "smpl_params_w": {"global_orient": torch.from_numpy(np.stack(results["go_w"])), "body_pose": torch.from_numpy(np.stack(results["bp"])), "transl": torch.from_numpy(trans_w - world_off[None]), "betas": torch.from_numpy(np.stack(results["bt"]))},
+ "T_w2c": torch.from_numpy(mats_w2c_c), "K_fullimg": torch.from_numpy(np.stack(results["K"])),
+ "kp2d": torch.from_numpy(np.stack(results["kp2d"])), "bbx_xys": torch.from_numpy(np.stack(results["bbx"])),
+ "cam_angvel": torch.from_numpy(cam_av), "cam_tvel": torch.from_numpy(cam_tv),
+ "f_imgseq": torch.cat(old_results), "imgname": img_paths, "valid_mask": torch.ones(len(img_paths), dtype=torch.float32),
+ "world_offset": torch.from_numpy(world_off.astype(np.float32))
+ }, save_p)
+
+ if args.smplx and len(results["tr_w"]) > 0:
+ os.makedirs(os.path.join(args.output, "smplx_global"), exist_ok=True)
+ trans_w = np.stack(results["tr_w"]).astype(np.float32)
+ world_off = trans_w[0].copy(); world_off[1] -= float(args.world_y_offset_m)
+ np.savez(os.path.join(args.output, "smplx_global", f"{seq_name}_global.npz"), mocap_framerate=int(FPS), gender="neutral", betas=results["bt"][0], trans=trans_w - world_off[None], world_offset=world_off, poses=np.pad(np.concatenate([np.stack(results["go_w"]), np.stack(results["bp"])], axis=1), ((0,0),(0,99)), mode="constant"))
+
+ if coco_subset:
+ with open(os.path.join(args.output, "vitpose", "temp_annotations", f"{seq_name}.json"), "w") as f: json.dump(coco_subset, f)
+
+ prof["SaveFiles"] = time.perf_counter() - t0
+ total_t = time.perf_counter() - t_start_seq
+ print(f" > Done in {total_t:.2f}s | FPS: {num_frames/max(total_t, 0.001):.1f} | [Breakdown] BatchPrep: {prof['BatchPrep']:.2f}s | Read: {prof['Read']:.2f}s | Overlay: {prof['Overlay']:.2f}s | ViT: {prof['ViT']:.2f}s | SaveFiles: {prof['SaveFiles']:.2f}s")
+
+if __name__ == "__main__": main()
\ No newline at end of file
diff --git a/third_party/GVHMR/tools/demo/process_dataset_pre_optimization.py b/third_party/GVHMR/tools/demo/process_dataset_pre_optimization.py
new file mode 100644
index 0000000000000000000000000000000000000000..e20cddd551c6a9f82341247bc6433ea4d0b84e80
--- /dev/null
+++ b/third_party/GVHMR/tools/demo/process_dataset_pre_optimization.py
@@ -0,0 +1,1749 @@
+import sys
+
+import os
+
+import json
+
+import argparse
+
+import numpy as np
+
+import zlib
+
+from glob import glob
+
+from tqdm import tqdm
+
+import cv2
+
+import torch
+
+from scipy.spatial.transform import Rotation as R
+
+import time
+
+import shutil
+
+from pathlib import Path
+
+
+
+# --- SETUP PATHS FOR IMPORTS ---
+
+REPO_ROOT = Path(__file__).resolve().parents[2]
+
+if str(REPO_ROOT) not in sys.path:
+
+ sys.path.insert(0, str(REPO_ROOT))
+
+
+
+from hmr4d.utils.preproc.vitfeat_extractor import Extractor
+
+from hmr4d.utils.pylogger import Log
+
+
+
+# Force single thread
+
+os.environ["OMP_NUM_THREADS"] = "1"
+
+os.environ["PYTORCH_CUDA_ALLOC_CONF"] = "expandable_segments:True"
+
+cv2.setNumThreads(0)
+
+torch.set_num_threads(1)
+
+
+
+FPS = 30.0
+
+DEBUG_NUM_FRAMES = 5
+
+IMAGENET_MEAN = np.array([0.485, 0.456, 0.406], dtype=np.float32)
+
+IMAGENET_STD = np.array([0.229, 0.224, 0.225], dtype=np.float32)
+
+
+
+# --- HELPER FUNCTIONS (No Changes) ---
+
+def _process_image_memory(img_bgr, bbox_xywh, img_size=256):
+
+ if img_bgr is None: return np.zeros((3, img_size, img_size), dtype=np.float32)
+
+ x, y, w, h = bbox_xywh
+
+ cx, cy = x + w/2, y + h/2
+
+ scale = max(w, h) * 1.2
+
+ H, W = img_bgr.shape[:2]
+
+ max_side = float(max(H, W, 1))
+
+ if scale <= 1.0 or scale > max_side * 20.0: scale = max_side * 0.5
+
+ half = scale / 2.0
+
+ x0, y0 = int(cx - half), int(cy - half)
+
+ x1, y1 = int(cx + half), int(cy + half)
+
+ pad_l, pad_t = max(0, -x0), max(0, -y0)
+
+ pad_r, pad_b = max(0, x1 - W), max(0, y1 - H)
+
+ if max(pad_l, pad_t, pad_r, pad_b) > int(max_side * 4.0): return np.zeros((3, img_size, img_size), dtype=np.float32)
+
+ if pad_l or pad_t or pad_r or pad_b:
+
+ img_bgr = cv2.copyMakeBorder(img_bgr, pad_t, pad_b, pad_l, pad_r, cv2.BORDER_CONSTANT, value=(0,0,0))
+
+ x0 += pad_l; y0 += pad_t; x1 += pad_l; y1 += pad_t
+
+ crop = img_bgr[y0:y1, x0:x1]
+
+ if crop.size == 0: return np.zeros((3, img_size, img_size), dtype=np.float32)
+
+ if crop.shape[0] != img_size or crop.shape[1] != img_size:
+
+ crop = cv2.resize(crop, (img_size, img_size), interpolation=cv2.INTER_LINEAR)
+
+ crop = crop[:, :, ::-1].astype(np.float32) / 255.0
+
+ crop = (crop - IMAGENET_MEAN) / IMAGENET_STD
+
+ return crop.transpose(2, 0, 1)
+
+
+
+def _alpha_blend_bgra_onto_bgr(dst_bgr, src_bgra, x, y):
+
+ if dst_bgr is None or src_bgra is None: return dst_bgr
+
+ H, W = dst_bgr.shape[:2]
+
+ h, w = src_bgra.shape[:2]
+
+ if w <= 0 or h <= 0: return dst_bgr
+
+ x0, y0 = max(int(x), 0), max(int(y), 0)
+
+ x1, y1 = min(int(x + w), W), min(int(y + h), H)
+
+ if x1 <= x0 or y1 <= y0: return dst_bgr
+
+ roi = dst_bgr[y0:y1, x0:x1]
+
+ src_crop = src_bgra[(y0 - int(y)):(y0 - int(y)) + (y1 - y0), (x0 - int(x)):(x0 - int(x)) + (x1 - x0)]
+
+ if src_crop.shape[2] == 3:
+
+ roi[:] = src_crop
+
+ return dst_bgr
+
+ alpha = src_crop[:, :, 3].astype(np.uint16)
+
+ inv_alpha = 255 - alpha
+
+ b_src, g_src, r_src = src_crop[:, :, 0], src_crop[:, :, 1], src_crop[:, :, 2]
+
+ b_dst, g_dst, r_dst = roi[:, :, 0], roi[:, :, 1], roi[:, :, 2]
+
+ roi[:, :, 0] = ((b_src * alpha + b_dst * inv_alpha) >> 8).astype(np.uint8)
+
+ roi[:, :, 1] = ((g_src * alpha + g_dst * inv_alpha) >> 8).astype(np.uint8)
+
+ roi[:, :, 2] = ((r_src * alpha + r_dst * inv_alpha) >> 8).astype(np.uint8)
+
+ return dst_bgr
+
+
+
+def _find_ui_dir():
+
+ cand = os.path.join(os.getcwd(), "UI")
+
+ if os.path.isdir(cand): return cand
+
+ return None # Simplified for brevity
+
+
+
+def _find_font_path(ui_dir, filename="Inter_18pt-Bold.ttf"):
+
+ if not ui_dir: return None
+
+ p = os.path.join(ui_dir, filename)
+
+ return p if os.path.isfile(p) else None
+
+
+
+def _load_ui_images(ui_dir):
+
+ if not ui_dir or (not os.path.isdir(ui_dir)): return []
+
+ imgs = []
+
+ for name in sorted(os.listdir(ui_dir)):
+
+ p = os.path.join(ui_dir, name)
+
+ if not os.path.isfile(p): continue
+
+ if name.lower().endswith(('.png', '.jpg', '.jpeg', '.webp')):
+
+ im = cv2.imread(p, cv2.IMREAD_UNCHANGED)
+
+ if im is not None:
+
+ if im.ndim == 2: im = cv2.cvtColor(im, cv2.COLOR_GRAY2BGR)
+
+ imgs.append(im)
+
+ return imgs
+
+
+
+class SimpleUIOverlay:
+
+ def __init__(self, width, height, seed=0, ui_dir=None, max_images=4, show_prob=0.6, min_hold_frames=20, max_hold_frames=120):
+
+ self.W, self.H = int(width), int(height)
+
+ self.rng = np.random.default_rng(int(seed))
+
+ self.max_images = max(0, int(max_images))
+
+ self.show_prob = float(show_prob)
+
+ self.min_hold_frames, self.max_hold_frames = max(1, int(min_hold_frames)), max(1, int(max_hold_frames))
+
+ self.ui_dir = ui_dir if ui_dir else _find_ui_dir()
+
+ self.assets = _load_ui_images(self.ui_dir)
+
+ self._ttl, self._active = 0, []
+
+ def _pick_new_state(self):
+
+ self._ttl = int(self.rng.integers(self.min_hold_frames, self.max_hold_frames + 1))
+
+ self._active = []
+
+ if (not self.assets) or (self.max_images <= 0): return
+
+ if float(self.rng.random()) > self.show_prob: return
+
+ k = min(int(self.rng.integers(1, self.max_images + 1)), len(self.assets))
+
+ idxs = self.rng.choice(len(self.assets), size=k, replace=False)
+
+ for idx in idxs:
+
+ im = self.assets[int(idx)]
+
+ h, w = im.shape[:2]
+
+ if w > 0 and h > 0:
+
+ x = int(self.rng.integers(-w // 4, max(1, self.W - (3 * w // 4))))
+
+ y = int(self.rng.integers(-h // 4, max(1, self.H - (3 * h // 4))))
+
+ self._active.append((im, x, y))
+
+ def draw(self, img_bgr):
+
+ if img_bgr is None: return img_bgr
+
+ if self._ttl <= 0: self._pick_new_state()
+
+ self._ttl -= 1
+
+ for im, x, y in self._active: _alpha_blend_bgra_onto_bgr(img_bgr, im, x, y)
+
+ return img_bgr
+
+
+
+class SimpleChatOverlay:
+
+ def __init__(self, width, height, seed=0, num_lines=7, region_w=420, region_h=180, margin=18, every_n_frames=15, corner=None, font_path=None):
+
+ from collections import deque
+
+ self.W, self.H = int(width), int(height)
+
+ self.rng = np.random.default_rng(int(seed))
+
+ self.num_lines, self.margin, self.every_n_frames = int(num_lines), int(margin), max(1, int(every_n_frames))
+
+ self.region_w, self.region_h = int(region_w), int(region_h)
+
+ self.font_path = font_path
+
+ self._pil_fonts = {}
+
+ self.corner = str(corner) if corner else str(self.rng.choice(["tl", "tr", "bl", "br"]))
+
+ self.messages = deque(maxlen=self.num_lines)
+
+ for _ in range(self.num_lines): self.messages.append(self._random_message())
+
+ self._cached_overlay, self._dirty = None, True
+
+ def _random_message(self):
+
+ user = str(self.rng.choice(["nightbot", "viewer", "catjam", "shadow", "speedrunner", "chattycathy", "kappaking"]))
+
+ if self.rng.random() < 0.5: user += str(self.rng.integers(10, 999))
+
+ text = str(self.rng.choice(["pog", "lol", "gg", "nice", "W", "L", "no shot", "crazy", "clip it", "cooking", "unlucky"]))
+
+ color = tuple(int(x) for x in self.rng.choice([(255, 120, 0), (0, 180, 255), (255, 0, 180), (0, 255, 120)]))
+
+ return {"user": user, "text": text, "color": color}
+
+ def _get_pil_font(self, size_px):
+
+ if not self.font_path: return None
+
+ if size_px in self._pil_fonts: return self._pil_fonts[size_px]
+
+ try:
+
+ from PIL import ImageFont
+
+ return ImageFont.truetype(self.font_path, size=max(1, size_px))
+
+ except: return None
+
+ def maybe_append(self, frame_idx):
+
+ if int(frame_idx) % self.every_n_frames == 0:
+
+ self.messages.append(self._random_message())
+
+ self._dirty = True
+
+ def _render_cache(self):
+
+ rw = min(self.region_w, max(40, self.W - 2 * self.margin))
+
+ rh = min(self.region_h, max(40, self.H - 2 * self.margin))
+
+ pil_font = self._get_pil_font(int(round(np.clip(20.0 * (self.H / 720.0), 14.0, 30.0))))
+
+ if pil_font is None:
+
+ self._cached_overlay = None; return
+
+ try:
+
+ from PIL import Image, ImageDraw
+
+ pil = Image.new("RGBA", (rw, rh), (0, 0, 0, 0))
+
+ draw = ImageDraw.Draw(pil)
+
+ line_h = max(14, int(round(float(getattr(pil_font, "size", 18)) * 1.25)))
+
+ lines = list(self.messages)[-min(self.num_lines, max(1, rh // line_h)):]
+
+ local_y = rh - line_h if self.corner in ("bl", "br") else 0
+
+ for msg in lines:
+
+ user = f"{msg['user']}: "
+
+ draw.text((0, local_y), user, font=pil_font, fill=tuple(msg['color'][::-1]))
+
+ tw = draw.textlength(user, font=pil_font)
+
+ draw.text((tw, local_y), msg['text'], font=pil_font, fill=(240, 240, 240))
+
+ local_y += (-line_h if self.corner in ("bl", "br") else line_h)
+
+ self._cached_overlay = cv2.cvtColor(np.asarray(pil), cv2.COLOR_RGBA2BGRA)
+
+ except: self._cached_overlay = None
+
+ def draw(self, img_bgr):
+
+ if img_bgr is None: return img_bgr
+
+ if self._dirty: self._render_cache(); self._dirty = False
+
+ if self._cached_overlay is not None:
+
+ rw = min(self.region_w, max(40, self.W - 2 * self.margin))
+
+ rh = min(self.region_h, max(40, self.H - 2 * self.margin))
+
+ if self.corner == "tl": x, y = self.margin, self.margin
+
+ elif self.corner == "tr": x, y = self.W - self.margin - rw, self.margin
+
+ elif self.corner == "bl": x, y = self.margin, self.H - self.margin - rh
+
+ else: x, y = self.W - self.margin - rw, self.H - self.margin - rh
+
+ _alpha_blend_bgra_onto_bgr(img_bgr, self._cached_overlay, x, y)
+
+ return img_bgr
+
+
+
+def k4_to_K3(k4): return np.array([[k4[0], 0, k4[2]], [0, k4[1], k4[3]], [0, 0, 1]], dtype=np.float32)
+
+def bbox_xywh_to_bbx_xys(bbox_xywh, base_enlarge=1.0):
+
+ x, y, w, h = [float(v) for v in bbox_xywh]
+
+ return np.array([x + 0.5 * w, y + 0.5 * h, max(w, h) * float(base_enlarge)], dtype=np.float32)
+
+def clamp_bbox_xywh_to_image(bbox_xywh, W, H, min_size=1.0):
+
+ x, y, w, h = [float(v) for v in bbox_xywh]
+
+ W, H = float(W), float(H)
+
+ if W <= 0 or H <= 0: return [0.0, 0.0, 0.0, 0.0]
+
+ x2, y2 = x + w, y + h
+
+ x1c = float(np.clip(x, 0.0, max(0.0, W - 1.0)))
+
+ y1c = float(np.clip(y, 0.0, max(0.0, H - 1.0)))
+
+ x2c = float(np.clip(x2, 0.0, W))
+
+ y2c = float(np.clip(y2, 0.0, H))
+
+ if x2c <= x1c: x2c = min(W, x1c + float(min_size))
+
+ if y2c <= y1c: y2c = min(H, y1c + float(min_size))
+
+ wc = max(0.0, x2c - x1c)
+
+ hc = max(0.0, y2c - y1c)
+
+ return [x1c, y1c, wc, hc]
+
+def draw_bbox_xywh_and_center(img_bgr, bbox_xywh, color=(255, 255, 0)):
+
+ x, y, w, h = [float(v) for v in bbox_xywh]
+
+ cv2.rectangle(img_bgr, (int(x), int(y)), (int(x+w), int(y+h)), color, 2)
+
+ cv2.circle(img_bgr, (int(x+w/2), int(y+h/2)), 4, (0, 0, 255), -1)
+
+def vis_label_and_color(v: int):
+
+ if v == 2: return "VIS", (0, 255, 0)
+
+ if v == 1: return "OCC", (0, 165, 255)
+
+ return "OFF", (160, 160, 160)
+
+def draw_vis_text_and_points(img_bgr, kpts2d_xy, vis17):
+
+ for k in range(17):
+
+ v = int(vis17[k])
+
+ label, color = vis_label_and_color(v)
+
+ x, y = int(round(kpts2d_xy[k, 0])), int(round(kpts2d_xy[k, 1]))
+
+ if v > 0: cv2.circle(img_bgr, (x, y), 4, color, -1)
+
+ cv2.putText(img_bgr, f"{k}:{label}", (x + 6, y - 6), cv2.FONT_HERSHEY_SIMPLEX, 0.45, color, 1, cv2.LINE_AA)
+
+def build_T_wc(pos_world, quat_world_xyzw):
+
+ T = np.eye(4, dtype=np.float64)
+
+ T[:3, :3] = R.from_quat(np.asarray(quat_world_xyzw, dtype=np.float64)).as_matrix()
+
+ T[:3, 3] = np.asarray(pos_world, dtype=np.float64)
+
+ return T
+
+def compute_velocity(mats, fps=30.0):
+
+ N = len(mats)
+
+ if N < 2: return np.zeros((N, 3), dtype=np.float32), np.zeros((N, 3), dtype=np.float32)
+
+ R_curr = mats[:, :3, :3]
+
+ R_diff = np.matmul(R_curr[1:], np.transpose(R_curr[:-1], (0, 2, 1)))
+
+ rv = R.from_matrix(R_diff).as_rotvec()
+
+ angvel = np.zeros((N, 3), dtype=np.float32)
+
+ angvel[1:] = rv
+
+ t_curr = mats[:, :3, 3]
+
+ tvel = np.zeros((N, 3), dtype=np.float32)
+
+ tvel[1:] = t_curr[1:] - t_curr[:-1]
+
+ return angvel.astype(np.float32), tvel.astype(np.float32)
+
+
+
+def _compute_vitpose_selected_indices(num_frames, fps, bucket_seconds, frames_per_bucket, sampling="uniform", seed=123):
+
+ if num_frames <= 0: return []
+
+ rng = np.random.default_rng(int(seed))
+
+ selected = []
+
+ bucket_len = max(1, int(round(float(bucket_seconds) * float(fps))))
+
+ b_start = 0
+
+ while b_start < num_frames:
+
+ b_end = min(num_frames, b_start + bucket_len)
+
+ k = min(int(frames_per_bucket), b_end - b_start)
+
+ if k > 0:
+
+ if sampling == "random": idxs = np.sort(rng.choice(np.arange(b_start, b_end), size=k, replace=False)).tolist()
+
+ elif sampling == "linspace": idxs = sorted(list(set(np.linspace(b_start, b_end - 1, k, dtype=int).tolist())))
+
+ else:
+
+ if k == 1: idxs = [b_start + (b_end - b_start) // 2]
+
+ else: step = (b_end - b_start) // k; idxs = [min(b_start + i * step, b_end - 1) for i in range(k)]
+
+ selected.extend(idxs)
+
+ b_start = b_end
+
+ return sorted(list(set(selected)))
+
+
+
+_SMPLX_MODEL = None
+
+_SMPLX_DEVICE = None
+
+def _get_smplx_model(device):
+
+ global _SMPLX_MODEL, _SMPLX_DEVICE
+
+ if _SMPLX_MODEL is not None and _SMPLX_DEVICE == device: return _SMPLX_MODEL
+
+ from hmr4d.utils.smplx_utils import make_smplx
+
+ _SMPLX_MODEL = make_smplx("supermotion").to(device).eval()
+
+ _SMPLX_DEVICE = device
+
+ return _SMPLX_MODEL
+
+
+
+class SmplIncamRenderer:
+
+ def __init__(self, width, height, K4, device="cuda", smplx2smpl_path="hmr4d/utils/body_model/smplx2smpl_sparse.pt"):
+
+ from hmr4d.utils.smplx_utils import make_smplx
+
+ from hmr4d.utils.vis.renderer import Renderer
+
+ self.torch = torch
+
+ self.device = device
+
+ self.smplx = make_smplx("supermotion").to(device).eval()
+
+ self.smplx2smpl = None; self.faces = None
+
+ try:
+
+ self.smplx2smpl = torch.load(smplx2smpl_path).to(device)
+
+ self.faces = make_smplx("smpl").faces
+
+ except: self.faces = self.smplx.faces
+
+ self.K_torch = torch.from_numpy(k4_to_K3(K4)).to(device)
+
+ self.renderer = Renderer(width, height, device=device, faces=self.faces, K=self.K_torch)
+
+ @torch.no_grad()
+
+ def render(self, img_rgb_uint8, global_orient_aa, body_pose_aa, betas_10, transl_xyz, fl, pp):
+
+ K3_torch = torch.from_numpy(np.array([[fl[0], 0, pp[0]], [0, fl[1], pp[1]], [0, 0, 1]], dtype=np.float32)).to(self.device)
+
+ self.renderer.set_intrinsic(K3_torch)
+
+ params = { "global_orient": torch.from_numpy(global_orient_aa[None]).float().to(self.device), "body_pose": torch.from_numpy(body_pose_aa[None]).float().to(self.device), "betas": torch.from_numpy(betas_10[None]).float().to(self.device), "transl": torch.from_numpy(transl_xyz[None]).float().to(self.device), }
+
+ out = self.smplx(**params); verts = out.vertices[0]
+
+ if self.smplx2smpl is not None and verts.dim() == 2: verts = torch.matmul(self.smplx2smpl, verts)
+
+ img_out = self.renderer.render_mesh(verts, img_rgb_uint8, [0.8, 0.8, 0.8])
+
+ return img_out
+
+ @torch.no_grad()
+
+ def get_verts(self, global_orient_aa, body_pose_aa, betas_10, transl_xyz):
+
+ params = { "global_orient": torch.from_numpy(global_orient_aa[None]).float().to(self.device), "body_pose": torch.from_numpy(body_pose_aa[None]).float().to(self.device), "betas": torch.from_numpy(betas_10[None]).float().to(self.device), "transl": torch.from_numpy(transl_xyz[None]).float().to(self.device), }
+
+ out = self.smplx(**params); verts = out.vertices[0]
+
+ if self.smplx2smpl is not None and verts.dim() == 2: verts = torch.matmul(self.smplx2smpl, verts)
+
+ return verts
+
+
+
+def _as_betas10(betas_any) -> np.ndarray:
+
+ betas = np.asarray(betas_any, dtype=np.float32).reshape(-1)
+
+ betas10 = np.zeros(10, dtype=np.float32); n = min(10, betas.size)
+
+ if n > 0: betas10[:n] = betas[:n]
+
+ return betas10
+
+def load_betas10_from_npz(npz_path, key="betas", index=None):
+
+ with np.load(npz_path, allow_pickle=True) as data: arr = data[key]
+
+ if arr.ndim == 0: arr = np.asarray(arr).reshape(1)
+
+ if arr.ndim == 1: betas = arr
+
+ elif arr.ndim == 2: row_idx = 0 if index is None else int(index); betas = arr[row_idx]
+
+ else: raise ValueError(f"Bad betas shape: {arr.shape}")
+
+ return _as_betas10(betas)
+
+def _default_shape_npz_path() -> str: return os.path.join(os.path.dirname(__file__), "shape.npz")
+
+
+
+def parse_smpl_inputs_from_row(row, override_betas10=None, keep_unity_scale=False, transl_source="pelvis", transl_y_offset_m=0.0):
+
+ C = np.diag([1.0, -1.0, 1.0]).astype(np.float64)
+
+ cam_rot_w_quat = np.array(row["cam_rot_world"], dtype=np.float64)
+
+ R_cam_w = R.from_quat(cam_rot_w_quat).as_matrix()
+
+ pel_rot_w_quat = np.array(row["pelvis_rot_world"], dtype=np.float64)
+
+ R_pel_w = R.from_quat(pel_rot_w_quat).as_matrix()
+
+
+
+ # Relative Rotation (Body to Camera)
+
+ R_rel_unity = R_cam_w.T @ R_pel_w
+
+ R_cv = C @ R_rel_unity @ C
+
+ R_final = R_cv @ R.from_euler("z", 180, degrees=True).as_matrix()
+
+ global_orient_aa = R.from_matrix(R_final).as_rotvec().astype(np.float32)
+
+
+
+ smpl_scale = float(row.get("smpl_root_world_scale", 1.0))
+
+ pelvis_cam_unity = np.asarray(row["smpl_incam_transl"], dtype=np.float64).reshape(3)
+
+ root_cam_unity = np.asarray(row.get("smpl_root_incam_transl", [0.0, 0.0, 0.0]), dtype=np.float64).reshape(3)
+
+ pelvis_cam_unity = pelvis_cam_unity + np.array([0.0, float(transl_y_offset_m), 0.0], dtype=np.float64)
+
+
+
+ if str(transl_source).strip().lower() == "root": target_cam_unity = root_cam_unity
+
+ else:
+
+ if bool(keep_unity_scale): target_cam_unity = pelvis_cam_unity
+
+ else:
+
+ if abs(smpl_scale) > 1e-8: target_cam_unity = root_cam_unity + (pelvis_cam_unity - root_cam_unity) / smpl_scale
+
+ else: target_cam_unity = pelvis_cam_unity
+
+ target_cam_cv = (C @ target_cam_unity).astype(np.float64)
+
+
+
+ pose = np.asarray(row["smplx_pose"], dtype=np.float32)
+
+ body_pose = pose[3:66].astype(np.float32)
+
+ betas10 = _as_betas10(override_betas10)
+
+
+
+ return {
+
+ "global_orient": global_orient_aa, "body_pose": body_pose, "betas": betas10,
+
+ "target_cam_cv": target_cam_cv, "cam_rot_w_quat": cam_rot_w_quat,
+
+ "cam_pos_world": np.asarray(row["cam_pos_world"], dtype=np.float64).reshape(3),
+
+ "pelvis_pos_world": np.asarray(row["pelvis_pos_world"], dtype=np.float64).reshape(3),
+
+ "smpl_scale": smpl_scale, "root_cam_unity": root_cam_unity
+
+ }
+
+
+
+def batch_smpl_forward(betas, global_orient, body_pose, device):
+
+ model = _get_smplx_model(device)
+
+ N = len(betas)
+
+ chunk_size = 4096; pelvis_list = []
+
+ with torch.no_grad():
+
+ for i in range(0, N, chunk_size):
+
+ b_betas = torch.from_numpy(betas[i:i+chunk_size]).float().to(device)
+
+ b_go = torch.from_numpy(global_orient[i:i+chunk_size]).float().to(device)
+
+ b_bp = torch.from_numpy(body_pose[i:i+chunk_size]).float().to(device)
+
+ b_tr = torch.zeros((len(b_betas), 3), dtype=torch.float32, device=device)
+
+ out = model(betas=b_betas, global_orient=b_go, body_pose=b_bp, transl=b_tr)
+
+ pelvis_list.append(out.joints[:, 0, :].detach().cpu().numpy())
+
+ return np.concatenate(pelvis_list, axis=0)
+
+
+
+def main():
+
+ parser = argparse.ArgumentParser()
+
+ parser.add_argument("--input", required=True)
+
+ parser.add_argument("--output", required=True)
+
+ parser.add_argument("--debug", action="store_true")
+
+ parser.add_argument("--vitpose", action="store_true")
+
+ parser.add_argument("--genmo", action="store_true")
+
+ parser.add_argument("--dpvo", action="store_true")
+
+ parser.add_argument("--smplx", action="store_true")
+
+ parser.add_argument("--debug_no_coco", action="store_true")
+
+ parser.add_argument("--shape_npz", default=_default_shape_npz_path())
+
+ parser.add_argument("--vitpose_use_all_frames", action="store_true")
+
+ parser.add_argument("--vitpose_bucket_seconds", type=float, default=12.0)
+
+ parser.add_argument("--vitpose_frames_per_bucket", type=int, default=36)
+
+ parser.add_argument("--vitpose_sampling", type=str, default="random")
+
+ parser.add_argument("--vitpose_seed", type=int, default=123)
+
+ parser.add_argument("--ui_dir", type=str, default=None)
+
+ parser.add_argument("--ui_show_prob", type=float, default=0.25)
+
+ parser.add_argument("--ui_max_images", type=int, default=3)
+
+ parser.add_argument("--ui_hold_min_s", type=float, default=0.7)
+
+ parser.add_argument("--ui_hold_max_s", type=float, default=5.0)
+
+ parser.add_argument("--ui_seed", type=int, default=None)
+
+ parser.add_argument("--keep_unity_scale", action="store_true")
+
+ parser.add_argument("--transl_source", type=str, default="pelvis")
+
+ parser.add_argument("--transl_y_offset_m", type=float, default=-0.020)
+
+ parser.add_argument("--world_y_offset_m", type=float, default=1.3415)
+
+ parser.add_argument("--vit_batch_size", type=int, default=2048, help="Batch size for in-memory ViT extraction")
+
+ args = parser.parse_args()
+
+
+
+ if not (args.vitpose or args.genmo or args.dpvo or args.smplx):
+
+ args.vitpose = args.genmo = args.dpvo = args.smplx = True
+
+
+
+ device = "cuda" if torch.cuda.is_available() else "cpu"
+
+ print(f"Running STREAMING processing on {device.upper()}...")
+
+
+
+ vit_model = None
+
+ if args.genmo and Extractor is not None:
+
+ print("Initializing ViT Extractor (HMR2)...")
+
+ extractor_wrapper = Extractor(tqdm_leave=False)
+
+ vit_model = extractor_wrapper.extractor
+
+ vit_model.eval()
+
+ vit_model.to(device)
+
+
+
+ override_betas10 = load_betas10_from_npz(args.shape_npz, key="betas")
+
+ temp_ann_dir = os.path.join(args.output, "vitpose", "temp_annotations")
+
+ os.makedirs(temp_ann_dir, exist_ok=True)
+
+ jsonl_files = sorted(glob(os.path.join(args.input, "sequence_*.jsonl")))
+
+
+
+ global_J_reg = None
+
+ j_reg_path = "third_party/GVHMR/inputs/checkpoints/body_models/smpl_neutral_J_regressor.pt"
+
+ if os.path.exists(j_reg_path) and device == "cuda":
+
+ global_J_reg = torch.load(j_reg_path, map_location=device)
+
+
+
+ for jsonl_idx, jsonl_path in enumerate(jsonl_files):
+
+ seq_name = os.path.splitext(os.path.basename(jsonl_path))[0].replace("sequence_", "")
+
+ print(f"[{jsonl_idx+1}/{len(jsonl_files)}] Processing {seq_name}...")
+
+
+
+ prof = {"smpl_batch": 0.0, "video_read": 0.0, "overlay": 0.0, "vit_process": 0.0,
+
+ "sparse_write": 0.0, "loop_total": 0.0, "save_files": 0.0, "debug_rend": 0.0, "prep": 0.0}
+
+
+
+ t_start_seq = time.perf_counter()
+
+ jsonl_dir = os.path.dirname(os.path.abspath(jsonl_path))
+
+ video_path = os.path.join(jsonl_dir, f"video_{seq_name}.mp4")
+
+ if not os.path.exists(video_path): video_path = os.path.join(jsonl_dir, "video.mp4")
+
+
+
+ out_img_folder = os.path.join(args.output, "images", seq_name)
+
+ os.makedirs(out_img_folder, exist_ok=True)
+
+
+
+ with open(jsonl_path, "r") as f: lines = f.readlines()
+
+ lines = lines[1:] if len(lines) > 0 else []
+
+ num_frames = len(lines)
+
+ if num_frames <= 0: continue
+
+
+
+ genmo_out = os.path.join(args.output, "genmo_features", f"{seq_name}.pt")
+
+ smplx_out = os.path.join(args.output, "smplx_incam", f"{seq_name}_smplx.npz")
+
+ smplx_global_out = os.path.join(args.output, "smplx_global", f"{seq_name}_global.npz")
+
+ dpvo_dir = os.path.join(args.output, "dpvo", seq_name)
+
+ for p in [genmo_out, smplx_out, smplx_global_out, dpvo_dir]:
+
+ if p: os.makedirs(os.path.dirname(p), exist_ok=True)
+
+
+
+ selected_set = set()
+
+ if args.vitpose:
+
+ if args.vitpose_use_all_frames: selected_indices = list(range(num_frames))
+
+ else:
+
+ selected_indices = _compute_vitpose_selected_indices(
+
+ num_frames, FPS, args.vitpose_bucket_seconds,
+
+ args.vitpose_frames_per_bucket, args.vitpose_sampling, args.vitpose_seed
+
+ )
+
+ selected_set = set(selected_indices)
+
+
+
+ cap = cv2.VideoCapture(video_path)
+
+ if not cap.isOpened(): continue
+
+ W = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
+
+ H = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
+
+
+
+ resolved_ui_dir = args.ui_dir if args.ui_dir else _find_ui_dir()
+
+ chat_font_path = _find_font_path(resolved_ui_dir)
+
+ seq_seed = int(zlib.crc32(seq_name.encode("utf-8")) & 0xFFFFFFFF)
+
+ chat_aug = SimpleChatOverlay(W, H, seed=seq_seed, num_lines=7, font_path=chat_font_path)
+
+ ui_aug = SimpleUIOverlay(W, H, seed=((seq_seed ^ 0xA5A5A5A5) & 0xFFFFFFFF), ui_dir=resolved_ui_dir,
+
+ max_images=args.ui_max_images, show_prob=args.ui_show_prob)
+
+
+
+ # --- BATCH SMPL (GPU) ---
+
+ t0_smpl = time.perf_counter()
+
+
+
+ smpl_precalc_data = []
+
+ debug_global_verts_cpu = []
+
+ parsed_rows = []
+
+
+
+ for line in lines:
+
+ row = json.loads(line)
+
+ parsed_rows.append(parse_smpl_inputs_from_row(row, override_betas10, args.keep_unity_scale, args.transl_source, args.transl_y_offset_m))
+
+
+
+ all_betas = np.stack([d['betas'] for d in parsed_rows])
+
+ all_go = np.stack([d['global_orient'] for d in parsed_rows])
+
+ all_bp = np.stack([d['body_pose'] for d in parsed_rows])
+
+
+
+ # We need an initial batch forward to get local pelvis offsets
+
+ all_pelvis0 = batch_smpl_forward(all_betas, all_go, all_bp, device=device)
+
+
+
+ C = np.diag([1.0, -1.0, 1.0]).astype(np.float64)
+
+ C4 = np.diag([1.0, -1.0, 1.0, 1.0]).astype(np.float64)
+
+ all_go_w, all_pelvis_pos_w_cv = [], []
+
+
+
+ # --- FIX: DEFINE THE FIX ROTATION (Z-180) FOR WORLD ---
+
+ fix_rot = R.from_euler("z", 180, degrees=True).as_matrix()
+
+ fix_mat = np.eye(4, dtype=np.float64)
+
+ fix_mat[:3, :3] = fix_rot
+
+
+
+ for i, d in enumerate(parsed_rows):
+
+ # Unity Rotation (Raw)
+
+ R_cam_w_unity = R.from_quat(d['cam_rot_w_quat']).as_matrix()
+
+
+
+ # --- APPLY FIX HERE: Pre-multiply Camera Rot by Fix (Z-180) ---
+
+ # This ensures the SMPL global orientation is calculated relative to the FIXED Camera
+
+ R_cam_w_cv = fix_rot @ (C @ R_cam_w_unity @ C)
+
+
+
+ R_pelvis_c_cv = R.from_rotvec(d['global_orient'].astype(np.float64)).as_matrix()
+
+ R_pelvis_w_cv = R_cam_w_cv @ R_pelvis_c_cv
+
+ all_go_w.append(R.from_matrix(R_pelvis_w_cv).as_rotvec().astype(np.float32))
+
+
+
+ # Position Logic
+
+ pelvis_pos_w_unity = d['pelvis_pos_world']
+
+ root_pos_w_unity = (R_cam_w_unity @ d['root_cam_unity'] + d['cam_pos_world']).reshape(3)
+
+ smpl_scale = d['smpl_scale']
+
+ transl_source_local = str(args.transl_source).strip().lower()
+
+ if transl_source_local == "root": target_pos_w_unity = root_pos_w_unity
+
+ else:
+
+ if bool(args.keep_unity_scale): target_pos_w_unity = pelvis_pos_w_unity
+
+ else:
+
+ if abs(smpl_scale) > 1e-8: target_pos_w_unity = root_pos_w_unity + (pelvis_pos_w_unity - root_pos_w_unity) / smpl_scale
+
+ else: target_pos_w_unity = pelvis_pos_w_unity
+
+
+
+ # --- APPLY FIX HERE: Pre-multiply Position by Fix (Z-180) ---
+
+ pos_cv_raw = (C @ target_pos_w_unity).astype(np.float64)
+
+ pelvis_pos_w_cv = fix_rot @ pos_cv_raw.reshape(3, 1)
+
+ all_pelvis_pos_w_cv.append(pelvis_pos_w_cv.reshape(3))
+
+
+
+ all_go_w = np.stack(all_go_w)
+
+ # Compute World-Space Pelvis offsets (dependent on global orient)
+
+ all_pelvis0_w = batch_smpl_forward(all_betas, all_go_w, all_bp, device=device)
+
+
+
+ for i in range(num_frames):
+
+ d = parsed_rows[i]
+
+ # Incam Transl
+
+ transl_c = (d['target_cam_cv'] - all_pelvis0[i]).astype(np.float32)
+
+
+
+ # World Transl
+
+ if str(args.transl_source) == "root": transl_w = all_pelvis_pos_w_cv[i].astype(np.float32)
+
+ else: transl_w = (all_pelvis_pos_w_cv[i] - all_pelvis0_w[i]).astype(np.float32)
+
+
+
+ smpl_precalc_data.append({
+
+ "go_c": d['global_orient'], "bp": d['body_pose'], "beta": d['betas'], "tr_c": transl_c,
+
+ "go_w": all_go_w[i], "tr_w": transl_w,
+
+ "cam_rot_w_quat": d["cam_rot_w_quat"], "cam_pos_world": d["cam_pos_world"]
+
+ })
+
+
+
+ prof["smpl_batch"] = time.perf_counter() - t0_smpl
+
+
+
+ # --- Debug Verification ---
+
+ if args.debug:
+
+ try:
+
+ row0 = json.loads(lines[0])
+
+ cam_pos0 = np.asarray(row0["cam_pos_world"], dtype=np.float64).reshape(3)
+
+ cam_q0 = np.asarray(row0["cam_rot_world"], dtype=np.float64).reshape(4)
+
+ pelvis_pos0 = np.asarray(row0["pelvis_pos_world"], dtype=np.float64).reshape(3)
+
+ pelvis_cam_meta0 = np.asarray(row0.get("smpl_incam_transl", [0.0, 0.0, 0.0]), dtype=np.float64).reshape(3)
+
+
+
+ R_cam_w0 = R.from_quat(cam_q0).as_matrix()
+
+ pelvis_cam_est0 = (R_cam_w0.T @ (pelvis_pos0 - cam_pos0).reshape(3, 1)).reshape(3)
+
+ diff0 = pelvis_cam_est0 - pelvis_cam_meta0
+
+
+
+ Log.info(f"[Debug] {seq_name} pelvis_cam_unity check: diff={diff0.round(4)}")
+
+ except Exception as e:
+
+ Log.warning(f"[Debug] {seq_name} pelvis_cam_unity check failed: {e}")
+
+
+
+ t0_gap = time.perf_counter()
+
+ smpl_renderer = None
+
+ vid_incam, vid_global = None, None
+
+ debug_end_frame = min(num_frames, DEBUG_NUM_FRAMES)
+
+ if args.debug:
+
+ os.makedirs(os.path.join(args.output, "debug_renders"), exist_ok=True)
+
+ if debug_end_frame > 0:
+
+ try:
+
+ K4_init = np.asarray(json.loads(lines[0])["cam_intrinsics"], dtype=np.float32)
+
+ smpl_renderer = SmplIncamRenderer(W, H, K4_init, device=device)
+
+ fourcc = cv2.VideoWriter_fourcc(*'mp4v')
+
+ vid_incam = cv2.VideoWriter(os.path.join(args.output, "debug_renders", f"{seq_name}_incam.mp4"), fourcc, FPS, (W, H))
+
+ dbg_gw, dbg_gh = 960, 540
+
+ vid_global = cv2.VideoWriter(os.path.join(args.output, "debug_renders", f"{seq_name}_global.mp4"), fourcc, FPS, (dbg_gw, dbg_gh))
+
+ except: pass
+
+
+
+ # --- MAIN LOOP ---
+
+ coco_subset, img_paths, K_fullimg_all = [], [], []
+
+ cam_T_wc_cv_all, cam_T_w2c_cv_all = [], []
+
+ dpvo_poses, dpvo_intrinsics = [], []
+
+ bboxes, bbx_xys_all, kp2d_all = [], [], []
+
+ global_orient_c_all, transl_c_all, body_pose_all, betas_all = [], [], [], []
+
+ global_orient_w_all, transl_w_all = [], []
+
+ vit_img_batch, all_vit_features = [], []
+
+
+
+ ret, _ = cap.read() # skip 0
+
+ prof["prep"] = time.perf_counter() - t0_gap
+
+
+
+ t_start_loop = time.perf_counter()
+
+ for idx in tqdm(range(num_frames), desc="Frames", leave=False):
+
+ t0_read = time.perf_counter()
+
+ ret, img_bgr = cap.read()
+
+ prof["video_read"] += (time.perf_counter() - t0_read)
+
+ if not ret: break
+
+
+
+ img_filename = f"img_{idx:05d}.jpg"
+
+ img_abs_path = os.path.join(out_img_folder, img_filename)
+
+
+
+ t0_ov = time.perf_counter()
+
+ chat_aug.maybe_append(idx)
+
+ chat_aug.draw(img_bgr)
+
+ ui_aug.draw(img_bgr)
+
+ prof["overlay"] += (time.perf_counter() - t0_ov)
+
+
+
+ row = json.loads(lines[idx])
+
+ K4 = np.asarray(row["cam_intrinsics"], dtype=np.float32)
+
+ kpts_raw = np.asarray(row["kpts_2d"], dtype=np.float32).reshape(-1, 2)[:17]
+
+ vis_raw = np.asarray(row["kpts_vis"], dtype=np.int32)[:17]
+
+ if vis_raw.shape[0] >= 5: vis_raw[3] = 1; vis_raw[4] = 1
+
+ bbox = clamp_bbox_xywh_to_image(row["bbox"], W, H)
+
+
+
+ sd = smpl_precalc_data[idx]
+
+ global_orient_c_all.append(sd['go_c'])
+
+ transl_c_all.append(sd['tr_c'])
+
+ global_orient_w_all.append(sd['go_w'])
+
+ transl_w_all.append(sd['tr_w'])
+
+ body_pose_all.append(sd['bp'])
+
+ betas_all.append(sd['beta'])
+
+
+
+ bboxes.append(np.asarray(bbox, dtype=np.float32))
+
+ bbx_xys_all.append(bbox_xywh_to_bbx_xys(bbox))
+
+ kp2d_all.append(np.concatenate([kpts_raw, (vis_raw > 0).astype(np.float32)[:, None]], axis=1))
+
+ K_fullimg_all.append(k4_to_K3(K4))
+
+
+
+ img_rel = os.path.join("images", seq_name, img_filename).replace("\\", "/")
+
+ img_paths.append(img_rel)
+
+
+
+ # Use raw Unity values
+
+ p_w = np.asarray(sd["cam_pos_world"], dtype=np.float32)
+
+ q_w = np.asarray(sd["cam_rot_w_quat"], dtype=np.float32)
+
+
+
+ # 1. Build the Standard Unity-to-CV Matrix (C @ M @ C)
+
+ cam_T_wc = build_T_wc(p_w, q_w)
+
+ cam_T_wc_cv_raw = (C4 @ cam_T_wc @ C4)
+
+
+
+ # 2. APPLY THE FIX (Z-180) to the Camera, matching precalc loop
+
+ cam_T_wc_cv = (fix_mat @ cam_T_wc_cv_raw).astype(np.float32)
+
+
+
+ # 3. Invert for W2C
+
+ cam_T_w2c_cv = np.linalg.inv(cam_T_wc_cv)
+
+
+
+ cam_T_wc_cv_all.append(cam_T_wc_cv)
+
+ cam_T_w2c_cv_all.append(cam_T_w2c_cv)
+
+ dpvo_poses.append(f"{p_w[0]} {p_w[1]} {p_w[2]} {q_w[0]} {q_w[1]} {q_w[2]} {q_w[3]}")
+
+ dpvo_intrinsics.append(K4.astype(np.float32))
+
+
+
+ if args.genmo and vit_model is not None:
+
+ t0_vit = time.perf_counter()
+
+ img_tensor = _process_image_memory(img_bgr, bbox, img_size=256)
+
+ vit_img_batch.append(img_tensor)
+
+ if len(vit_img_batch) >= args.vit_batch_size:
+
+ batch_np = np.stack(vit_img_batch)
+
+ batch_t = torch.from_numpy(batch_np).to(device, non_blocking=True)
+
+ with torch.inference_mode():
+
+ with torch.amp.autocast("cuda"):
+
+ feats = vit_model({"img": batch_t})
+
+ all_vit_features.append(feats.detach().cpu())
+
+ vit_img_batch = []
+
+ prof["vit_process"] += (time.perf_counter() - t0_vit)
+
+
+
+ if args.vitpose and (idx in selected_set):
+
+ t0_wr = time.perf_counter()
+
+ cv2.imwrite(img_abs_path, img_bgr, [int(cv2.IMWRITE_JPEG_QUALITY), 90])
+
+ kpts_coco = []
+
+ for k in range(17): kpts_coco.extend([float(kpts_raw[k, 0]), float(kpts_raw[k, 1]), int(vis_raw[k])])
+
+ coco_subset.append(({"file_name": img_rel, "width": W, "height": H},
+
+ {"category_id": 1, "bbox": bbox, "area": float(bbox[2]*bbox[3]), "iscrowd": 0, "keypoints": kpts_coco, "num_keypoints": int(np.sum(vis_raw > 0))}))
+
+ prof["sparse_write"] += (time.perf_counter() - t0_wr)
+
+
+
+ if args.debug and idx < debug_end_frame and smpl_renderer:
+
+ t0_dbg = time.perf_counter()
+
+ dbg = img_bgr.copy()
+
+ try: draw_bbox_xywh_and_center(dbg, bbox)
+
+ except: pass
+
+ try:
+
+ rgb = smpl_renderer.render(dbg[:, :, ::-1].copy(), sd['go_c'], sd['bp'], sd['beta'], sd['tr_c'], K4[:2], K4[2:])
+
+ dbg = rgb[:, :, ::-1].copy()
+
+ except: pass
+
+ if not args.debug_no_coco:
+
+ draw_vis_text_and_points(dbg, kpts_raw, vis_raw)
+
+ if vid_incam: vid_incam.write(dbg)
+
+
+
+ if vid_global:
+
+ verts_w = smpl_renderer.get_verts(sd['go_w'], sd['bp'], sd['beta'], sd['tr_w']).float()
+
+ debug_global_verts_cpu.append(verts_w.detach().cpu())
+
+ prof["debug_rend"] += (time.perf_counter() - t0_dbg)
+
+
+
+ if args.genmo and len(vit_img_batch) > 0 and vit_model is not None:
+
+ t0_vit = time.perf_counter()
+
+ batch_np = np.stack(vit_img_batch)
+
+ batch_t = torch.from_numpy(batch_np).to(device, non_blocking=True)
+
+ with torch.inference_mode():
+
+ with torch.amp.autocast("cuda"):
+
+ feats = vit_model({"img": batch_t})
+
+ all_vit_features.append(feats.detach().cpu())
+
+ prof["vit_process"] += (time.perf_counter() - t0_vit)
+
+
+
+ prof["loop_total"] = time.perf_counter() - t_start_loop
+
+ cap.release()
+
+ if vid_incam: vid_incam.release()
+
+
+
+ t0_dbg = time.perf_counter()
+
+ if vid_global and len(debug_global_verts_cpu) > 0:
+
+ try:
+
+ from hmr4d.utils.vis.renderer import (
+
+ Renderer,
+
+ get_global_cameras_static,
+
+ get_ground_params_from_points,
+
+ perspective_projection,
+
+ )
+
+ from hmr4d.utils.geo.hmr_cam import create_camera_sensor
+
+
+
+ dbg_gw, dbg_gh = 960, 540
+
+ _, _, K_global = create_camera_sensor(dbg_gw, dbg_gh, 24)
+
+ global_renderer = Renderer(dbg_gw, dbg_gh, device=device, faces=smpl_renderer.faces, K=K_global.to(device), bin_size=0)
+
+ verts_seq = torch.stack(debug_global_verts_cpu, dim=0)
+
+ off = verts_seq[0].mean(0); off[1] = verts_seq[0, :, 1].min()
+
+ verts_seq = verts_seq - off
+
+
+
+ # Convert CV-cam to GPU tensor for visualizer
+
+ cam_centers = None
+
+ try:
+
+ F = int(verts_seq.shape[0])
+
+ if len(cam_T_wc_cv_all) >= F:
+
+ cam_wc = np.stack(cam_T_wc_cv_all[:F], axis=0).astype(np.float32)
+
+ cam_centers = torch.from_numpy(cam_wc[:, :3, 3]).to(device=device)
+
+ cam_centers = cam_centers - off.to(device=device)[None]
+
+ except Exception:
+
+ cam_centers = None
+
+
+
+ g_R, g_T, g_L = get_global_cameras_static(
+
+ verts_seq, beta=2.0, cam_height_degree=20, target_center_height=1.0, device=device
+
+ )
+
+
+
+ if global_J_reg is not None and verts_seq.shape[1] == global_J_reg.shape[-1]:
+
+ joints_seq = torch.einsum("jv,fvk->fjk", global_J_reg.cpu(), verts_seq)
+
+ roots = joints_seq[:, 0]
+
+ else:
+
+ roots = verts_seq.mean(1)
+
+ sc, cx, cz = get_ground_params_from_points(roots, verts_seq)
+
+ global_renderer.set_ground(sc * 1.5, cx, cz)
+
+ col = torch.tensor([[0.0, 1.0, 0.0]], device=device)
+
+ trail = []
+
+
+
+ def _project_xy(points_w: torch.Tensor):
+
+ P = points_w.view(1, -1, 3)
+
+ x2d = perspective_projection(P, global_renderer.K, global_renderer.R, global_renderer.T.reshape(1, 3, 1))[0]
+
+ return x2d
+
+
+
+ def _draw_polyline(img_bgr, pts_xy, color, closed=False, thickness=1):
+
+ pts = np.asarray(pts_xy, dtype=np.int32).reshape(-1, 1, 2)
+
+ if len(pts) < 2: return
+
+ cv2.polylines(img_bgr, [pts], bool(closed), color, int(thickness), cv2.LINE_AA)
+
+
+
+ def _draw_camera_box_axes(img_bgr, C_w, right, up, fwd, scale=0.25):
+
+ C_w = C_w.reshape(3)
+
+ right = right.reshape(3)
+
+ up = up.reshape(3)
+
+ fwd = fwd.reshape(3)
+
+ L = float(scale)
+
+
+
+ # Draw Axis instead of just box (RGB = XYZ)
+
+ # X (Right) - Red
+
+ p_x = C_w + L * right
+
+ xy_x = _project_xy(torch.stack([C_w, p_x])).detach().cpu().numpy()
+
+ _draw_polyline(img_bgr, xy_x, (0, 0, 255), thickness=2)
+
+
+
+ # Y (Up/Down) - Green
+
+ p_y = C_w + L * up
+
+ xy_y = _project_xy(torch.stack([C_w, p_y])).detach().cpu().numpy()
+
+ _draw_polyline(img_bgr, xy_y, (0, 255, 0), thickness=2)
+
+
+
+ # Z (Fwd) - Blue
+
+ p_z = C_w + L * fwd
+
+ xy_z = _project_xy(torch.stack([C_w, p_z])).detach().cpu().numpy()
+
+ _draw_polyline(img_bgr, xy_z, (255, 0, 0), thickness=2)
+
+
+
+ for i in range(len(verts_seq)):
+
+ cam = global_renderer.create_camera(g_R[i], g_T[i])
+
+ img = global_renderer.render_with_ground(verts_seq[i].to(device)[None], col, cam, g_L)
+
+ img_bgr = img[:, :, ::-1].copy()
+
+
+
+ if cam_centers is not None and i < cam_centers.shape[0]:
+
+ try:
+
+ # Blue ray: camera center -> SMPL root
+
+ if i < roots.shape[0]:
+
+ pts_line = torch.stack([cam_centers[i], roots[i].to(device=device)], dim=0)
+
+ xy_line = _project_xy(pts_line).detach().cpu().numpy()
+
+ _draw_polyline(img_bgr, xy_line, (255, 200, 50), closed=False, thickness=1)
+
+
+
+ P = cam_centers[i].view(1, 3)
+
+ x2d = _project_xy(P)[0]
+
+ x, y = int(round(float(x2d[0].item()))), int(round(float(x2d[1].item())))
+
+ if 0 <= x < img_bgr.shape[1] and 0 <= y < img_bgr.shape[0]:
+
+ trail.append((x, y))
+
+ cv2.circle(img_bgr, (x, y), 3, (0, 0, 255), -1)
+
+ if len(trail) >= 2:
+
+ cv2.polylines(img_bgr, [np.array(trail, dtype=np.int32)], False, (0, 0, 255), 1)
+
+
+
+ if len(cam_T_wc_cv_all) > i:
+
+ R_c2w = torch.from_numpy(np.asarray(cam_T_wc_cv_all[i], dtype=np.float32)[:3, :3]).to(device=device)
+
+ C_w = cam_centers[i]
+
+ right = R_c2w[:, 0]
+
+ up = R_c2w[:, 1]
+
+ fwd = R_c2w[:, 2]
+
+ _draw_camera_box_axes(img_bgr, C_w, right, up, fwd, scale=0.35)
+
+ except Exception: pass
+
+
+
+ vid_global.write(img_bgr)
+
+ except: pass
+
+ vid_global.release()
+
+ prof["debug_rend"] += (time.perf_counter() - t0_dbg)
+
+
+
+ t0_save = time.perf_counter()
+
+ if args.genmo:
+
+ trans_w = np.stack(transl_w_all).astype(np.float32)
+
+ world_off = trans_w[0].copy(); world_off[1] -= float(args.world_y_offset_m)
+
+ trans_w_centered = trans_w - world_off[None]
+
+ mats_w2c = np.stack(cam_T_w2c_cv_all).astype(np.float32)
+
+ mats_wc = np.stack(cam_T_wc_cv_all).astype(np.float32)
+
+ T_wp_w = np.eye(4, dtype=np.float32); T_wp_w[:3, 3] = world_off
+
+ T_w_wp = np.eye(4, dtype=np.float32); T_w_wp[:3, 3] = -world_off
+
+ mats_w2c_c = np.matmul(mats_w2c, T_wp_w[None])
+
+ mats_wc_c = np.matmul(T_w_wp[None], mats_wc)
+
+ cam_av, cam_tv = compute_velocity(mats_wc_c, fps=FPS)
+
+
+
+ f_imgseq = torch.cat(all_vit_features, dim=0).float() if all_vit_features else torch.empty(0)
+
+
+
+ g_dict = {
+
+ "smpl_params_c": {"global_orient": torch.from_numpy(np.stack(global_orient_c_all)), "body_pose": torch.from_numpy(np.stack(body_pose_all)), "transl": torch.from_numpy(np.stack(transl_c_all)), "betas": torch.from_numpy(np.stack(betas_all))},
+
+ "smpl_params_w": {"global_orient": torch.from_numpy(np.stack(global_orient_w_all)), "body_pose": torch.from_numpy(np.stack(body_pose_all)), "transl": torch.from_numpy(trans_w_centered), "betas": torch.from_numpy(np.stack(betas_all))},
+
+ "T_w2c": torch.from_numpy(mats_w2c_c), "K_fullimg": torch.from_numpy(np.stack(K_fullimg_all)),
+
+ "kp2d": torch.from_numpy(np.stack(kp2d_all)), "bbx_xys": torch.from_numpy(np.stack(bbx_xys_all)),
+
+ "cam_angvel": torch.from_numpy(cam_av), "cam_tvel": torch.from_numpy(cam_tv),
+
+ "imgname": img_paths, "valid_mask": torch.ones(len(img_paths), dtype=torch.float32),
+
+ "world_offset": torch.from_numpy(world_off.astype(np.float32)),
+
+ "f_imgseq": f_imgseq
+
+ }
+
+ torch.save(g_dict, genmo_out)
+
+
+
+ if args.smplx:
+
+ poses66 = np.concatenate([np.stack(global_orient_w_all), np.stack(body_pose_all)], axis=1)
+
+ poses165 = np.pad(poses66, ((0,0),(0,99)), mode="constant").astype(np.float32)
+
+ trans_w = np.stack(transl_w_all).astype(np.float32)
+
+ world_off = trans_w[0].copy(); world_off[1] -= float(args.world_y_offset_m)
+
+ trans_w = trans_w - world_off[None]
+
+ np.savez(smplx_global_out, mocap_framerate=int(FPS), gender="neutral", betas=betas_all[0], trans=trans_w, poses=poses165, world_offset=world_off)
+
+
+
+ if args.vitpose and coco_subset:
+
+ with open(os.path.join(temp_ann_dir, f"{seq_name}.json"), "w") as f: json.dump(coco_subset, f)
+
+
+
+ prof["save_files"] = time.perf_counter() - t0_save
+
+ total_t = time.perf_counter() - t_start_seq
+
+
+
+ print(f" > Done in {total_t:.2f}s | FPS: {num_frames/total_t:.1f}")
+
+ print(f" [Breakdown] BatchPrep: {prof['smpl_batch']:.2f}s | Init/Gap: {prof['prep']:.2f}s | Read: {prof['video_read']:.2f}s")
+
+ print(f" Overlay: {prof['overlay']:.2f}s | SparseWrite: {prof['sparse_write']:.2f}s | ViT: {prof['vit_process']:.2f}s")
+
+ print(f" DbgRend: {prof['debug_rend']:.2f}s | SaveFiles: {prof['save_files']:.2f}s")
+
+
+
+ print("All sequences processed.")
+
+
+
+if __name__ == "__main__":
+
+ main()
\ No newline at end of file
diff --git a/third_party/GVHMR/tools/demo/shape.npz b/third_party/GVHMR/tools/demo/shape.npz
new file mode 100644
index 0000000000000000000000000000000000000000..f0e6409d6dfe0ef2921461b3303053e79550c1fc
--- /dev/null
+++ b/third_party/GVHMR/tools/demo/shape.npz
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:823c88143c7c8ac6c910b97da27bda34925136d1dc3dc90b9a0f3ab682fa1fdb
+size 81298
diff --git a/third_party/GVHMR/tools/train.py b/third_party/GVHMR/tools/train.py
new file mode 100644
index 0000000000000000000000000000000000000000..7ec8e9b2e5205a88802033a1de43bac6c29140f7
--- /dev/null
+++ b/third_party/GVHMR/tools/train.py
@@ -0,0 +1,87 @@
+import hydra
+import pytorch_lightning as pl
+from omegaconf import DictConfig, OmegaConf
+from pytorch_lightning.callbacks.checkpoint import Checkpoint
+
+from hmr4d.utils.pylogger import Log
+from hmr4d.configs import register_store_gvhmr
+from hmr4d.utils.vis.rich_logger import print_cfg
+from hmr4d.utils.net_utils import load_pretrained_model, get_resume_ckpt_path
+
+
+def get_callbacks(cfg: DictConfig) -> list:
+ """Parse and instantiate all the callbacks in the config."""
+ if not hasattr(cfg, "callbacks") or cfg.callbacks is None:
+ return None
+ # Handle special callbacks
+ enable_checkpointing = cfg.pl_trainer.get("enable_checkpointing", True)
+ # Instantiate all the callbacks
+ callbacks = []
+ for callback in cfg.callbacks.values():
+ if callback is not None:
+ cb = hydra.utils.instantiate(callback, _recursive_=False)
+ # skip when disable checkpointing and the callback is Checkpoint
+ if not enable_checkpointing and isinstance(cb, Checkpoint):
+ continue
+ else:
+ callbacks.append(cb)
+ return callbacks
+
+
+def train(cfg: DictConfig) -> None:
+ """Train/Test"""
+ Log.info(f"[Exp Name]: {cfg.exp_name}")
+ if cfg.task == "fit":
+ Log.info(f"[GPU x Batch] = {cfg.pl_trainer.devices} x {cfg.data.loader_opts.train.batch_size}")
+ pl.seed_everything(cfg.seed)
+
+ # preparation
+ datamodule: pl.LightningDataModule = hydra.utils.instantiate(cfg.data, _recursive_=False)
+ model: pl.LightningModule = hydra.utils.instantiate(cfg.model, _recursive_=False)
+ if cfg.ckpt_path is not None:
+ load_pretrained_model(model, cfg.ckpt_path)
+
+ # PL callbacks and logger
+ callbacks = get_callbacks(cfg)
+ has_ckpt_cb = any([isinstance(cb, Checkpoint) for cb in callbacks])
+ if not has_ckpt_cb and cfg.pl_trainer.get("enable_checkpointing", True):
+ Log.warning("No checkpoint-callback found. Disabling PL auto checkpointing.")
+ cfg.pl_trainer = {**cfg.pl_trainer, "enable_checkpointing": False}
+ logger = hydra.utils.instantiate(cfg.logger, _recursive_=False)
+
+ # PL-Trainer
+ if cfg.task == "test":
+ Log.info("Test mode forces full-precision.")
+ cfg.pl_trainer = {**cfg.pl_trainer, "precision": 32}
+ trainer = pl.Trainer(
+ accelerator="gpu",
+ logger=logger if logger is not None else False,
+ callbacks=callbacks,
+ **cfg.pl_trainer,
+ )
+
+ if cfg.task == "fit":
+ resume_path = None
+ if cfg.resume_mode is not None:
+ resume_path = get_resume_ckpt_path(cfg.resume_mode, ckpt_dir=cfg.callbacks.model_checkpoint.dirpath)
+ Log.info(f"Resume training from {resume_path}")
+ Log.info("Start Fitiing...")
+ trainer.fit(model, datamodule.train_dataloader(), datamodule.val_dataloader(), ckpt_path=resume_path)
+ elif cfg.task == "test":
+ Log.info("Start Testing...")
+ trainer.test(model, datamodule.test_dataloader())
+ else:
+ raise ValueError(f"Unknown task: {cfg.task}")
+
+ Log.info("End of script.")
+
+
+@hydra.main(version_base="1.3", config_path="../hmr4d/configs", config_name="train")
+def main(cfg) -> None:
+ print_cfg(cfg, use_rich=True)
+ train(cfg)
+
+
+if __name__ == "__main__":
+ register_store_gvhmr()
+ main()
diff --git a/third_party/GVHMR/tools/unitest/make_hydra_cfg.py b/third_party/GVHMR/tools/unitest/make_hydra_cfg.py
new file mode 100644
index 0000000000000000000000000000000000000000..2ecfbab9620fc711dc3c84f25b1492f4c3244cca
--- /dev/null
+++ b/third_party/GVHMR/tools/unitest/make_hydra_cfg.py
@@ -0,0 +1,7 @@
+from hmr4d.configs import parse_args_to_cfg, register_store_gvhmr
+from hmr4d.utils.vis.rich_logger import print_cfg
+
+if __name__ == "__main__":
+ register_store_gvhmr()
+ cfg = parse_args_to_cfg()
+ print_cfg(cfg, use_rich=True)
diff --git a/third_party/GVHMR/tools/unitest/run_dataset.py b/third_party/GVHMR/tools/unitest/run_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..0e1f8af58bd59109350a0de81f86a292f9d91254
--- /dev/null
+++ b/third_party/GVHMR/tools/unitest/run_dataset.py
@@ -0,0 +1,41 @@
+import torch
+from torch.utils.data import DataLoader
+from tqdm import tqdm
+
+
+def get_dataset(DATA_TYPE):
+ if DATA_TYPE == "BEDLAM_V2":
+ from hmr4d.dataset.bedlam.bedlam import BedlamDatasetV2
+
+ return BedlamDatasetV2()
+
+ if DATA_TYPE == "3DPW_TRAIN":
+ from hmr4d.dataset.threedpw.threedpw_motion_train import ThreedpwSmplDataset
+
+ return ThreedpwSmplDataset()
+
+if __name__ == "__main__":
+ DATA_TYPE = "3DPW_TRAIN"
+ dataset = get_dataset(DATA_TYPE)
+ print(len(dataset))
+
+ data = dataset[0]
+
+ from hmr4d.datamodule.mocap_trainX_testY import collate_fn
+
+ loader = DataLoader(
+ dataset,
+ shuffle=False,
+ num_workers=0,
+ persistent_workers=False,
+ pin_memory=False,
+ batch_size=1,
+ collate_fn=collate_fn,
+ )
+ i = 0
+ for batch in tqdm(loader):
+ i += 1
+ # if i == 20:
+ # raise AssertionError
+ # time.sleep(0.2)
+ pass
diff --git a/third_party/GVHMR/tools/video/merge_folder.py b/third_party/GVHMR/tools/video/merge_folder.py
new file mode 100644
index 0000000000000000000000000000000000000000..e159bd69be63a48b7f5c276b81f475d56716806d
--- /dev/null
+++ b/third_party/GVHMR/tools/video/merge_folder.py
@@ -0,0 +1,42 @@
+"""This script will glob two folder, check the mp4 files are one-to-one match precisely, then call merge_horizontal.py to merge them one by one"""
+
+import os
+import argparse
+from pathlib import Path
+
+
+def main():
+ parser = argparse.ArgumentParser()
+ parser.add_argument("input_dir1", type=str)
+ parser.add_argument("input_dir2", type=str)
+ parser.add_argument("output_dir", type=str)
+ parser.add_argument("--vertical", action="store_true") # By default use horizontal
+ args = parser.parse_args()
+
+ # Check input
+ input_dir1 = Path(args.input_dir1)
+ input_dir2 = Path(args.input_dir2)
+ assert input_dir1.exists()
+ assert input_dir2.exists()
+ video_paths1 = sorted(input_dir1.glob("*.mp4"))
+ video_paths2 = sorted(input_dir2.glob("*.mp4"))
+ assert len(video_paths1) == len(video_paths2)
+ for path1, path2 in zip(video_paths1, video_paths2):
+ assert path1.stem == path2.stem
+
+ # Merge to output
+ output_dir = Path(args.output_dir)
+ output_dir.mkdir(parents=True, exist_ok=True)
+
+ for path1, path2 in zip(video_paths1, video_paths2):
+ out_path = output_dir / f"{path1.stem}.mp4"
+ in_paths = [str(path1), str(path2)]
+ print(f"Merging {in_paths} to {out_path}")
+ if args.vertical:
+ os.system(f"python tools/video/merge_vertical.py {' '.join(in_paths)} -o {out_path}")
+ else:
+ os.system(f"python tools/video/merge_horizontal.py {' '.join(in_paths)} -o {out_path}")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/third_party/GVHMR/tools/video/merge_horizontal.py b/third_party/GVHMR/tools/video/merge_horizontal.py
new file mode 100644
index 0000000000000000000000000000000000000000..f06da6d70700f0f5f50b70c1709bb761b98c74c2
--- /dev/null
+++ b/third_party/GVHMR/tools/video/merge_horizontal.py
@@ -0,0 +1,15 @@
+import argparse
+from hmr4d.utils.video_io_utils import merge_videos_horizontal
+
+
+def parse_args():
+ """python tools/video/merge_horizontal.py a.mp4 b.mp4 c.mp4 -o out.mp4"""
+ parser = argparse.ArgumentParser()
+ parser.add_argument("input_videos", nargs="+", help="Input video paths")
+ parser.add_argument("-o", "--output", type=str, required=True, help="Output video path")
+ return parser.parse_args()
+
+
+if __name__ == "__main__":
+ args = parse_args()
+ merge_videos_horizontal(args.input_videos, args.output)
diff --git a/third_party/GVHMR/tools/video/merge_vertical.py b/third_party/GVHMR/tools/video/merge_vertical.py
new file mode 100644
index 0000000000000000000000000000000000000000..8617ec17036be374e680524d02231eb145a252c9
--- /dev/null
+++ b/third_party/GVHMR/tools/video/merge_vertical.py
@@ -0,0 +1,15 @@
+import argparse
+from hmr4d.utils.video_io_utils import merge_videos_vertical
+
+
+def parse_args():
+ """python tools/video/merge_vertical.py a.mp4 b.mp4 c.mp4 -o out.mp4"""
+ parser = argparse.ArgumentParser()
+ parser.add_argument("input_videos", nargs="+", help="Input video paths")
+ parser.add_argument("-o", "--output", type=str, required=True, help="Output video path")
+ return parser.parse_args()
+
+
+if __name__ == "__main__":
+ args = parse_args()
+ merge_videos_vertical(args.input_videos, args.output)
diff --git a/third_party/GVHMR/work_dirs/best_coco_AP_epoch_1.pth b/third_party/GVHMR/work_dirs/best_coco_AP_epoch_1.pth
new file mode 100644
index 0000000000000000000000000000000000000000..316e556035219eac541331787b6e73652e3a6276
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/best_coco_AP_epoch_1.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e3f6e3d5e7d0ca801031879df1da632c0f75ed1fb48b94e25c3b53e6ad5a3c25
+size 2549533578
diff --git a/third_party/GVHMR/work_dirs/best_coco_AP_epoch_1.pth:Zone.Identifier b/third_party/GVHMR/work_dirs/best_coco_AP_epoch_1.pth:Zone.Identifier
new file mode 100644
index 0000000000000000000000000000000000000000..bfbf16a817137e2538457752aea761daaf155486
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/best_coco_AP_epoch_1.pth:Zone.Identifier
@@ -0,0 +1,4 @@
+[ZoneTransfer]
+ZoneId=3
+ReferrerUrl=https://81.166.162.13:10297/lab/tree/workspace/work_dirs/vitpose_finetune
+HostUrl=https://81.166.162.13:10297/files/workspace/work_dirs/vitpose_finetune/best_coco_AP_epoch_1.pth?_xsrf=2%7C37ce6f2f%7C3e91b755e3da02dd6a39d924ac5a69bc%7C1766790473
diff --git a/third_party/GVHMR/work_dirs/epoch_20.pth b/third_party/GVHMR/work_dirs/epoch_20.pth
new file mode 100644
index 0000000000000000000000000000000000000000..957f1c7735da5e0e3be0eaef32a327b8afc52941
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/epoch_20.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b7aae1bf438b2cb5a27d4d2f489df1fce5e7e7662055148f2982d999d8451749
+size 2559219914
diff --git a/third_party/GVHMR/work_dirs/epoch_20.pth:Zone.Identifier b/third_party/GVHMR/work_dirs/epoch_20.pth:Zone.Identifier
new file mode 100644
index 0000000000000000000000000000000000000000..5dfa7658517439ad9f0db4894507582ce352d34f
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/epoch_20.pth:Zone.Identifier
@@ -0,0 +1,4 @@
+[ZoneTransfer]
+ZoneId=3
+ReferrerUrl=https://81.166.162.13:10297/lab/tree/workspace/work_dirs/vitpose_finetune
+HostUrl=https://81.166.162.13:10297/files/workspace/work_dirs/vitpose_finetune/epoch_20.pth?_xsrf=2%7C37ce6f2f%7C3e91b755e3da02dd6a39d924ac5a69bc%7C1766790473
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/20251226_181931.log b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/20251226_181931.log
new file mode 100644
index 0000000000000000000000000000000000000000..d857fcc85f52ea96b4ac7e5ff7c5b2727f918218
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/20251226_181931.log
@@ -0,0 +1,1614 @@
+2025/12/26 18:19:35 - mmengine - INFO -
+------------------------------------------------------------
+System environment:
+ sys.platform: linux
+ Python: 3.10.19 (main, Oct 21 2025, 16:43:05) [GCC 11.2.0]
+ CUDA available: True
+ MUSA available: False
+ numpy_random_seed: 110787734
+ GPU 0: NVIDIA GeForce RTX 3060
+ CUDA_HOME: /root/miniconda3/envs/gvhmr
+ NVCC: Cuda compilation tools, release 12.1, V12.1.105
+ GCC: gcc (Ubuntu 11.4.0-1ubuntu1~22.04.2) 11.4.0
+ PyTorch: 2.3.0+cu121
+ PyTorch compiling details: PyTorch built with:
+ - GCC 9.3
+ - C++ Version: 201703
+ - Intel(R) oneAPI Math Kernel Library Version 2022.2-Product Build 20220804 for Intel(R) 64 architecture applications
+ - Intel(R) MKL-DNN v3.3.6 (Git Hash 86e6af5974177e513fd3fee58425e1063e7f1361)
+ - OpenMP 201511 (a.k.a. OpenMP 4.5)
+ - LAPACK is enabled (usually provided by MKL)
+ - NNPACK is enabled
+ - CPU capability usage: AVX2
+ - CUDA Runtime 12.1
+ - NVCC architecture flags: -gencode;arch=compute_50,code=sm_50;-gencode;arch=compute_60,code=sm_60;-gencode;arch=compute_70,code=sm_70;-gencode;arch=compute_75,code=sm_75;-gencode;arch=compute_80,code=sm_80;-gencode;arch=compute_86,code=sm_86;-gencode;arch=compute_90,code=sm_90
+ - CuDNN 8.9.2
+ - Magma 2.6.1
+ - Build settings: BLAS_INFO=mkl, BUILD_TYPE=Release, CUDA_VERSION=12.1, CUDNN_VERSION=8.9.2, CXX_COMPILER=/opt/rh/devtoolset-9/root/usr/bin/c++, CXX_FLAGS= -D_GLIBCXX_USE_CXX11_ABI=0 -fabi-version=11 -fvisibility-inlines-hidden -DUSE_PTHREADPOOL -DNDEBUG -DUSE_KINETO -DLIBKINETO_NOROCTRACER -DUSE_FBGEMM -DUSE_QNNPACK -DUSE_PYTORCH_QNNPACK -DUSE_XNNPACK -DSYMBOLICATE_MOBILE_DEBUG_HANDLE -O2 -fPIC -Wall -Wextra -Werror=return-type -Werror=non-virtual-dtor -Werror=bool-operation -Wnarrowing -Wno-missing-field-initializers -Wno-type-limits -Wno-array-bounds -Wno-unknown-pragmas -Wno-unused-parameter -Wno-unused-function -Wno-unused-result -Wno-strict-overflow -Wno-strict-aliasing -Wno-stringop-overflow -Wsuggest-override -Wno-psabi -Wno-error=pedantic -Wno-error=old-style-cast -Wno-missing-braces -fdiagnostics-color=always -faligned-new -Wno-unused-but-set-variable -Wno-maybe-uninitialized -fno-math-errno -fno-trapping-math -Werror=format -Wno-stringop-overflow, LAPACK_INFO=mkl, PERF_WITH_AVX=1, PERF_WITH_AVX2=1, PERF_WITH_AVX512=1, TORCH_VERSION=2.3.0, USE_CUDA=ON, USE_CUDNN=ON, USE_CUSPARSELT=1, USE_EXCEPTION_PTR=1, USE_GFLAGS=OFF, USE_GLOG=OFF, USE_GLOO=ON, USE_MKL=ON, USE_MKLDNN=ON, USE_MPI=OFF, USE_NCCL=1, USE_NNPACK=ON, USE_OPENMP=ON, USE_ROCM=OFF, USE_ROCM_KERNEL_ASSERT=OFF,
+
+ TorchVision: 0.18.0+cu121
+ OpenCV: 4.12.0
+ MMEngine: 0.10.7
+
+Runtime environment:
+ cudnn_benchmark: False
+ mp_cfg: {'mp_start_method': 'fork', 'opencv_num_threads': 0}
+ dist_cfg: {'backend': 'nccl'}
+ seed: 110787734
+ Distributed launcher: none
+ Distributed training: False
+ GPU number: 1
+------------------------------------------------------------
+
+2025/12/26 18:19:35 - mmengine - INFO - Config:
+auto_scale_lr = None
+backend_args = dict(backend='local')
+codec = dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap')
+custom_hooks = [
+ dict(type='SyncBuffersHook'),
+]
+custom_imports = dict(
+ allow_failed_imports=False,
+ imports=[
+ 'mmpose.engine.optim_wrappers.layer_decay_optim_wrapper',
+ ])
+data_mode = 'topdown'
+data_root = '/root/miko/puni/train/GVHMR/processed_data/'
+dataset_type = 'CocoDataset'
+default_hooks = dict(
+ badcase=dict(
+ badcase_thr=5,
+ enable=False,
+ metric_type='loss',
+ out_dir='badcase',
+ type='BadCaseAnalysisHook'),
+ checkpoint=dict(
+ interval=1,
+ max_keep_ckpts=1,
+ rule='greater',
+ save_best='coco/AP',
+ save_optimizer=False,
+ type='CheckpointHook'),
+ logger=dict(interval=5, type='LoggerHook'),
+ param_scheduler=dict(type='ParamSchedulerHook'),
+ sampler_seed=dict(type='DistSamplerSeedHook'),
+ timer=dict(type='IterTimerHook'),
+ visualization=dict(
+ enable=True, out_dir='vis_results', type='PoseVisualizationHook'))
+default_scope = 'mmpose'
+env_cfg = dict(
+ cudnn_benchmark=False,
+ dist_cfg=dict(backend='nccl'),
+ mp_cfg=dict(mp_start_method='fork', opencv_num_threads=0))
+launcher = 'none'
+load_from = '/root/miko/puni/train/GVHMR/inputs/checkpoints/vitpose/td-hm_ViTPose-huge_8xb64-210e_coco-256x192-e32adcd4_20230314.pth'
+log_level = 'INFO'
+log_processor = dict(
+ by_epoch=True, num_digits=6, type='LogProcessor', window_size=50)
+metainfo = dict(from_file='configs/_base_/datasets/coco.py')
+model = dict(
+ backbone=dict(
+ arch='huge',
+ drop_path_rate=0.55,
+ img_size=(
+ 256,
+ 192,
+ ),
+ init_cfg=None,
+ out_type='featmap',
+ patch_cfg=dict(padding=2),
+ patch_size=16,
+ qkv_bias=True,
+ type='mmpretrain.VisionTransformer',
+ with_cls_token=False),
+ data_preprocessor=dict(
+ bgr_to_rgb=True,
+ mean=[
+ 123.675,
+ 116.28,
+ 103.53,
+ ],
+ std=[
+ 58.395,
+ 57.12,
+ 57.375,
+ ],
+ type='PoseDataPreprocessor'),
+ head=dict(
+ decoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ deconv_kernel_sizes=(
+ 4,
+ 4,
+ ),
+ deconv_out_channels=(
+ 256,
+ 256,
+ ),
+ in_channels=1280,
+ loss=dict(type='KeypointMSELoss', use_target_weight=True),
+ out_channels=17,
+ type='HeatmapHead'),
+ test_cfg=dict(flip_mode='heatmap', flip_test=True, shift_heatmap=False),
+ type='TopdownPoseEstimator')
+optim_wrapper = dict(
+ clip_grad=dict(max_norm=1.0, norm_type=2),
+ constructor='LayerDecayOptimWrapperConstructor',
+ optimizer=dict(
+ betas=(
+ 0.9,
+ 0.999,
+ ), lr=5e-05, type='AdamW', weight_decay=0.1),
+ paramwise_cfg=dict(
+ custom_keys=dict(
+ bias=dict(decay_multi=0.0),
+ norm=dict(decay_mult=0.0),
+ pos_embed=dict(decay_mult=0.0),
+ relative_position_bias_table=dict(decay_mult=0.0)),
+ layer_decay_rate=0.85,
+ num_layers=32))
+param_scheduler = [
+ dict(
+ begin=0,
+ by_epoch=True,
+ end=20,
+ gamma=0.1,
+ milestones=[
+ 14,
+ 18,
+ ],
+ type='MultiStepLR'),
+]
+resume = False
+test_cfg = dict()
+test_dataloader = dict(
+ batch_size=32,
+ dataset=dict(
+ ann_file='annotations/person_keypoints_val2017.json',
+ bbox_file=
+ 'data/coco/person_detection_results/COCO_val2017_detections_AP_H_56_person.json',
+ data_mode='topdown',
+ data_prefix=dict(img='val2017/'),
+ data_root='data/coco/',
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+test_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+train_cfg = dict(by_epoch=True, max_epochs=20, val_interval=1)
+train_dataloader = dict(
+ batch_size=4,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+ ],
+ type='CocoDataset'),
+ num_workers=0,
+ persistent_workers=False,
+ sampler=dict(shuffle=True, type='DefaultSampler'))
+train_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+]
+val_cfg = dict()
+val_dataloader = dict(
+ batch_size=4,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ bbox_file=None,
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=0,
+ persistent_workers=False,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+val_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+val_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+]
+vis_backends = [
+ dict(type='LocalVisBackend'),
+]
+visualizer = None
+work_dir = '../work_dirs/vitpose_finetune'
+
+2025/12/26 18:19:44 - mmengine - INFO - Distributed training is not used, all SyncBatchNorm (SyncBN) layers in the model will be automatically reverted to BatchNormXd layers if they are used.
+2025/12/26 18:19:44 - mmengine - INFO - Hooks will be executed in the following order:
+before_run:
+(VERY_HIGH ) RuntimeInfoHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+before_train:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+before_train_epoch:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(NORMAL ) DistSamplerSeedHook
+ --------------------
+before_train_iter:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+ --------------------
+after_train_iter:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(BELOW_NORMAL) LoggerHook
+(LOW ) ParamSchedulerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+after_train_epoch:
+(NORMAL ) IterTimerHook
+(NORMAL ) SyncBuffersHook
+(LOW ) ParamSchedulerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+before_val:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+before_val_epoch:
+(NORMAL ) IterTimerHook
+(NORMAL ) SyncBuffersHook
+ --------------------
+before_val_iter:
+(NORMAL ) IterTimerHook
+ --------------------
+after_val_iter:
+(NORMAL ) IterTimerHook
+(NORMAL ) PoseVisualizationHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+after_val_epoch:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(BELOW_NORMAL) LoggerHook
+(LOW ) ParamSchedulerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+after_val:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+after_train:
+(VERY_HIGH ) RuntimeInfoHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+before_test:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+before_test_epoch:
+(NORMAL ) IterTimerHook
+ --------------------
+before_test_iter:
+(NORMAL ) IterTimerHook
+ --------------------
+after_test_iter:
+(NORMAL ) IterTimerHook
+(NORMAL ) PoseVisualizationHook
+(NORMAL ) BadCaseAnalysisHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+after_test_epoch:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(NORMAL ) BadCaseAnalysisHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+after_test:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+after_run:
+(BELOW_NORMAL) LoggerHook
+ --------------------
+Name of parameter - Initialization information
+
+backbone.pos_embed - torch.Size([1, 192, 1280]):
+Initialized by user-defined `init_weights` in VisionTransformer
+
+backbone.patch_embed.projection.weight - torch.Size([1280, 3, 16, 16]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.patch_embed.projection.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.0.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.0.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.0.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.1.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.1.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.1.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.1.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.2.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.2.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.2.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.2.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.3.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.3.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.3.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.3.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.4.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.4.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.4.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.4.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.5.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.5.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.5.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.5.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.6.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.6.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.6.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.6.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.7.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.7.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.7.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.7.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.8.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.8.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.8.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.8.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.9.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.9.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.9.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.9.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.10.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.10.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.10.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.10.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.11.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.11.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.11.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.11.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.12.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.12.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.12.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.12.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.13.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.13.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.13.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.13.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.14.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.14.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.14.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.14.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.15.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.15.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.15.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.15.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.16.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.16.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.16.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.16.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.17.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.17.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.17.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.17.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.18.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.18.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.18.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.18.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.19.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.19.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.19.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.19.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.20.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.20.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.20.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.20.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.21.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.21.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.21.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.21.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.22.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.22.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.22.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.22.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.23.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.23.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.23.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.23.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.24.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.24.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.24.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.24.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.25.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.25.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.25.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.25.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.26.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.26.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.26.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.26.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.27.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.27.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.27.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.27.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.28.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.28.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.28.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.28.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.29.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.29.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.29.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.29.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.30.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.30.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.30.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.30.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.31.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.31.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.31.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.31.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+head.deconv_layers.0.weight - torch.Size([1280, 256, 4, 4]):
+NormalInit: mean=0, std=0.001, bias=0
+
+head.deconv_layers.1.weight - torch.Size([256]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+head.deconv_layers.1.bias - torch.Size([256]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+head.deconv_layers.3.weight - torch.Size([256, 256, 4, 4]):
+NormalInit: mean=0, std=0.001, bias=0
+
+head.deconv_layers.4.weight - torch.Size([256]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+head.deconv_layers.4.bias - torch.Size([256]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+head.final_layer.weight - torch.Size([17, 256, 1, 1]):
+NormalInit: mean=0, std=0.001, bias=0
+
+head.final_layer.bias - torch.Size([17]):
+NormalInit: mean=0, std=0.001, bias=0
+2025/12/26 18:19:51 - mmengine - INFO - Load checkpoint from /root/miko/puni/train/GVHMR/inputs/checkpoints/vitpose/td-hm_ViTPose-huge_8xb64-210e_coco-256x192-e32adcd4_20230314.pth
+2025/12/26 18:19:51 - mmengine - WARNING - "FileClient" will be deprecated in future. Please use io functions in https://mmengine.readthedocs.io/en/latest/api/fileio.html#file-io
+2025/12/26 18:19:51 - mmengine - WARNING - "HardDiskBackend" is the alias of "LocalBackend" and the former will be deprecated in future.
+2025/12/26 18:19:51 - mmengine - INFO - Checkpoints will be saved to /root/miko/puni/train/GVHMR/work_dirs/vitpose_finetune.
+2025/12/26 18:19:55 - mmengine - INFO - Epoch(train) [1][ 5/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:13:09 time: 0.734828 data_time: 0.030003 memory: 10107 grad_norm: 0.022649 loss: 0.000853 loss_kpt: 0.000853 acc_pose: 0.916667
+2025/12/26 18:19:58 - mmengine - INFO - Epoch(train) [1][10/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:32 time: 0.702893 data_time: 0.027864 memory: 10107 grad_norm: 0.017201 loss: 0.000738 loss_kpt: 0.000738 acc_pose: 0.750000
+2025/12/26 18:20:02 - mmengine - INFO - Epoch(train) [1][15/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:18 time: 0.693688 data_time: 0.028670 memory: 10107 grad_norm: 0.016388 loss: 0.000700 loss_kpt: 0.000700 acc_pose: 0.884615
+2025/12/26 18:20:05 - mmengine - INFO - Epoch(train) [1][20/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:12 time: 0.691037 data_time: 0.029586 memory: 10107 grad_norm: 0.017376 loss: 0.000673 loss_kpt: 0.000673 acc_pose: 0.916667
+2025/12/26 18:20:09 - mmengine - INFO - Epoch(train) [1][25/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:08 time: 0.690860 data_time: 0.029728 memory: 10107 grad_norm: 0.016256 loss: 0.000613 loss_kpt: 0.000613 acc_pose: 1.000000
+2025/12/26 18:20:15 - mmengine - INFO - Epoch(train) [1][30/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:13:36 time: 0.777490 data_time: 0.029428 memory: 10107 grad_norm: 0.015789 loss: 0.000599 loss_kpt: 0.000599 acc_pose: 0.903846
+2025/12/26 18:20:18 - mmengine - INFO - Epoch(train) [1][35/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:13:21 time: 0.766886 data_time: 0.029464 memory: 10107 grad_norm: 0.015164 loss: 0.000583 loss_kpt: 0.000583 acc_pose: 0.884615
+2025/12/26 18:20:22 - mmengine - INFO - Epoch(train) [1][40/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:13:09 time: 0.759576 data_time: 0.029568 memory: 10107 grad_norm: 0.014961 loss: 0.000589 loss_kpt: 0.000589 acc_pose: 0.910256
+2025/12/26 18:20:25 - mmengine - INFO - Epoch(train) [1][45/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:56 time: 0.750168 data_time: 0.029520 memory: 10107 grad_norm: 0.014750 loss: 0.000571 loss_kpt: 0.000571 acc_pose: 0.955128
+2025/12/26 18:20:29 - mmengine - INFO - Epoch(train) [1][50/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:44 time: 0.742616 data_time: 0.029701 memory: 10107 grad_norm: 0.014235 loss: 0.000546 loss_kpt: 0.000546 acc_pose: 0.961538
+2025/12/26 18:20:31 - mmengine - INFO - Exp name: vitpose_huge_finetune_20251226_181931
+2025/12/26 18:20:31 - mmengine - INFO - Saving checkpoint at 1 epochs
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/vis_data/20251226_181931.json b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/vis_data/20251226_181931.json
new file mode 100644
index 0000000000000000000000000000000000000000..f3e31dcee6fb09bfcf4028fe19ae4adb9d431a3b
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/vis_data/20251226_181931.json
@@ -0,0 +1,10 @@
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03000326156616211, "grad_norm": 0.02264945786446333, "loss": 0.0008531261584721506, "loss_kpt": 0.0008531261584721506, "acc_pose": 0.9166666666666666, "time": 0.7348281383514405, "epoch": 1, "iter": 5, "memory": 10107, "step": 5}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.027863621711730957, "grad_norm": 0.017201234027743338, "loss": 0.0007379717804724351, "loss_kpt": 0.0007379717804724351, "acc_pose": 0.75, "time": 0.7028926372528076, "epoch": 1, "iter": 10, "memory": 10107, "step": 10}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.028670406341552733, "grad_norm": 0.016388442491491635, "loss": 0.0007003831580126037, "loss_kpt": 0.0007003831580126037, "acc_pose": 0.8846153846153845, "time": 0.6936884721120199, "epoch": 1, "iter": 15, "memory": 10107, "step": 15}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.0295857310295105, "grad_norm": 0.01737639047205448, "loss": 0.0006729890854330733, "loss_kpt": 0.0006729890854330733, "acc_pose": 0.9166666666666666, "time": 0.6910372138023376, "epoch": 1, "iter": 20, "memory": 10107, "step": 20}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029728031158447264, "grad_norm": 0.016256342995911836, "loss": 0.0006132185703609139, "loss_kpt": 0.0006132185703609139, "acc_pose": 1.0, "time": 0.6908601188659668, "epoch": 1, "iter": 25, "memory": 10107, "step": 25}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029428418477376303, "grad_norm": 0.015788525979345044, "loss": 0.0005987753878192356, "loss_kpt": 0.0005987753878192356, "acc_pose": 0.9038461538461539, "time": 0.7774903138478597, "epoch": 1, "iter": 30, "memory": 10107, "step": 30}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029464408329554968, "grad_norm": 0.015163798816502094, "loss": 0.0005833704145126311, "loss_kpt": 0.0005833704145126311, "acc_pose": 0.8846153846153846, "time": 0.7668861321040562, "epoch": 1, "iter": 35, "memory": 10107, "step": 35}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029568243026733398, "grad_norm": 0.014960916340351104, "loss": 0.0005887716975848889, "loss_kpt": 0.0005887716975848889, "acc_pose": 0.9102564102564101, "time": 0.7595762133598327, "epoch": 1, "iter": 40, "memory": 10107, "step": 40}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029520204332139758, "grad_norm": 0.014749648856619994, "loss": 0.0005708508054441255, "loss_kpt": 0.0005708508054441255, "acc_pose": 0.9551282051282051, "time": 0.7501680586073134, "epoch": 1, "iter": 45, "memory": 10107, "step": 45}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029701404571533203, "grad_norm": 0.014234937932342292, "loss": 0.000546034719736781, "loss_kpt": 0.000546034719736781, "acc_pose": 0.9615384615384616, "time": 0.7426163625717163, "epoch": 1, "iter": 50, "memory": 10107, "step": 50}
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/vis_data/config.py b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/vis_data/config.py
new file mode 100644
index 0000000000000000000000000000000000000000..ffa57b17f889ecc5dab97c42a84a4251b7ef91f8
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/vis_data/config.py
@@ -0,0 +1,273 @@
+auto_scale_lr = None
+backend_args = dict(backend='local')
+codec = dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap')
+custom_hooks = [
+ dict(type='SyncBuffersHook'),
+]
+custom_imports = dict(
+ allow_failed_imports=False,
+ imports=[
+ 'mmpose.engine.optim_wrappers.layer_decay_optim_wrapper',
+ ])
+data_mode = 'topdown'
+data_root = '/root/miko/puni/train/GVHMR/processed_data/'
+dataset_type = 'CocoDataset'
+default_hooks = dict(
+ badcase=dict(
+ badcase_thr=5,
+ enable=False,
+ metric_type='loss',
+ out_dir='badcase',
+ type='BadCaseAnalysisHook'),
+ checkpoint=dict(
+ interval=1,
+ max_keep_ckpts=1,
+ rule='greater',
+ save_best='coco/AP',
+ save_optimizer=False,
+ type='CheckpointHook'),
+ logger=dict(interval=5, type='LoggerHook'),
+ param_scheduler=dict(type='ParamSchedulerHook'),
+ sampler_seed=dict(type='DistSamplerSeedHook'),
+ timer=dict(type='IterTimerHook'),
+ visualization=dict(
+ enable=True, out_dir='vis_results', type='PoseVisualizationHook'))
+default_scope = 'mmpose'
+env_cfg = dict(
+ cudnn_benchmark=False,
+ dist_cfg=dict(backend='nccl'),
+ mp_cfg=dict(mp_start_method='fork', opencv_num_threads=0))
+launcher = 'none'
+load_from = '/root/miko/puni/train/GVHMR/inputs/checkpoints/vitpose/td-hm_ViTPose-huge_8xb64-210e_coco-256x192-e32adcd4_20230314.pth'
+log_level = 'INFO'
+log_processor = dict(
+ by_epoch=True, num_digits=6, type='LogProcessor', window_size=50)
+metainfo = dict(from_file='configs/_base_/datasets/coco.py')
+model = dict(
+ backbone=dict(
+ arch='huge',
+ drop_path_rate=0.55,
+ img_size=(
+ 256,
+ 192,
+ ),
+ init_cfg=None,
+ out_type='featmap',
+ patch_cfg=dict(padding=2),
+ patch_size=16,
+ qkv_bias=True,
+ type='mmpretrain.VisionTransformer',
+ with_cls_token=False),
+ data_preprocessor=dict(
+ bgr_to_rgb=True,
+ mean=[
+ 123.675,
+ 116.28,
+ 103.53,
+ ],
+ std=[
+ 58.395,
+ 57.12,
+ 57.375,
+ ],
+ type='PoseDataPreprocessor'),
+ head=dict(
+ decoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ deconv_kernel_sizes=(
+ 4,
+ 4,
+ ),
+ deconv_out_channels=(
+ 256,
+ 256,
+ ),
+ in_channels=1280,
+ loss=dict(type='KeypointMSELoss', use_target_weight=True),
+ out_channels=17,
+ type='HeatmapHead'),
+ test_cfg=dict(flip_mode='heatmap', flip_test=True, shift_heatmap=False),
+ type='TopdownPoseEstimator')
+optim_wrapper = dict(
+ clip_grad=dict(max_norm=1.0, norm_type=2),
+ constructor='LayerDecayOptimWrapperConstructor',
+ optimizer=dict(
+ betas=(
+ 0.9,
+ 0.999,
+ ), lr=5e-05, type='AdamW', weight_decay=0.1),
+ paramwise_cfg=dict(
+ custom_keys=dict(
+ bias=dict(decay_multi=0.0),
+ norm=dict(decay_mult=0.0),
+ pos_embed=dict(decay_mult=0.0),
+ relative_position_bias_table=dict(decay_mult=0.0)),
+ layer_decay_rate=0.85,
+ num_layers=32))
+param_scheduler = [
+ dict(
+ begin=0,
+ by_epoch=True,
+ end=20,
+ gamma=0.1,
+ milestones=[
+ 14,
+ 18,
+ ],
+ type='MultiStepLR'),
+]
+resume = False
+test_cfg = dict()
+test_dataloader = dict(
+ batch_size=32,
+ dataset=dict(
+ ann_file='annotations/person_keypoints_val2017.json',
+ bbox_file=
+ 'data/coco/person_detection_results/COCO_val2017_detections_AP_H_56_person.json',
+ data_mode='topdown',
+ data_prefix=dict(img='val2017/'),
+ data_root='data/coco/',
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+test_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+train_cfg = dict(by_epoch=True, max_epochs=20, val_interval=1)
+train_dataloader = dict(
+ batch_size=4,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+ ],
+ type='CocoDataset'),
+ num_workers=0,
+ persistent_workers=False,
+ sampler=dict(shuffle=True, type='DefaultSampler'))
+train_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+]
+val_cfg = dict()
+val_dataloader = dict(
+ batch_size=4,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ bbox_file=None,
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=0,
+ persistent_workers=False,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+val_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+val_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+]
+vis_backends = [
+ dict(type='LocalVisBackend'),
+]
+visualizer = None
+work_dir = '../work_dirs/vitpose_finetune'
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/vis_data/scalars.json b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/vis_data/scalars.json
new file mode 100644
index 0000000000000000000000000000000000000000..f3e31dcee6fb09bfcf4028fe19ae4adb9d431a3b
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_181931/vis_data/scalars.json
@@ -0,0 +1,10 @@
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03000326156616211, "grad_norm": 0.02264945786446333, "loss": 0.0008531261584721506, "loss_kpt": 0.0008531261584721506, "acc_pose": 0.9166666666666666, "time": 0.7348281383514405, "epoch": 1, "iter": 5, "memory": 10107, "step": 5}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.027863621711730957, "grad_norm": 0.017201234027743338, "loss": 0.0007379717804724351, "loss_kpt": 0.0007379717804724351, "acc_pose": 0.75, "time": 0.7028926372528076, "epoch": 1, "iter": 10, "memory": 10107, "step": 10}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.028670406341552733, "grad_norm": 0.016388442491491635, "loss": 0.0007003831580126037, "loss_kpt": 0.0007003831580126037, "acc_pose": 0.8846153846153845, "time": 0.6936884721120199, "epoch": 1, "iter": 15, "memory": 10107, "step": 15}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.0295857310295105, "grad_norm": 0.01737639047205448, "loss": 0.0006729890854330733, "loss_kpt": 0.0006729890854330733, "acc_pose": 0.9166666666666666, "time": 0.6910372138023376, "epoch": 1, "iter": 20, "memory": 10107, "step": 20}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029728031158447264, "grad_norm": 0.016256342995911836, "loss": 0.0006132185703609139, "loss_kpt": 0.0006132185703609139, "acc_pose": 1.0, "time": 0.6908601188659668, "epoch": 1, "iter": 25, "memory": 10107, "step": 25}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029428418477376303, "grad_norm": 0.015788525979345044, "loss": 0.0005987753878192356, "loss_kpt": 0.0005987753878192356, "acc_pose": 0.9038461538461539, "time": 0.7774903138478597, "epoch": 1, "iter": 30, "memory": 10107, "step": 30}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029464408329554968, "grad_norm": 0.015163798816502094, "loss": 0.0005833704145126311, "loss_kpt": 0.0005833704145126311, "acc_pose": 0.8846153846153846, "time": 0.7668861321040562, "epoch": 1, "iter": 35, "memory": 10107, "step": 35}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029568243026733398, "grad_norm": 0.014960916340351104, "loss": 0.0005887716975848889, "loss_kpt": 0.0005887716975848889, "acc_pose": 0.9102564102564101, "time": 0.7595762133598327, "epoch": 1, "iter": 40, "memory": 10107, "step": 40}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029520204332139758, "grad_norm": 0.014749648856619994, "loss": 0.0005708508054441255, "loss_kpt": 0.0005708508054441255, "acc_pose": 0.9551282051282051, "time": 0.7501680586073134, "epoch": 1, "iter": 45, "memory": 10107, "step": 45}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.029701404571533203, "grad_norm": 0.014234937932342292, "loss": 0.000546034719736781, "loss_kpt": 0.000546034719736781, "acc_pose": 0.9615384615384616, "time": 0.7426163625717163, "epoch": 1, "iter": 50, "memory": 10107, "step": 50}
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/20251226_182315.log b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/20251226_182315.log
new file mode 100644
index 0000000000000000000000000000000000000000..9c855d73f4a7116aa7a64dfd578261e06fce3567
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/20251226_182315.log
@@ -0,0 +1,1713 @@
+2025/12/26 18:23:17 - mmengine - INFO -
+------------------------------------------------------------
+System environment:
+ sys.platform: linux
+ Python: 3.10.19 (main, Oct 21 2025, 16:43:05) [GCC 11.2.0]
+ CUDA available: True
+ MUSA available: False
+ numpy_random_seed: 721894494
+ GPU 0: NVIDIA GeForce RTX 3060
+ CUDA_HOME: /root/miniconda3/envs/gvhmr
+ NVCC: Cuda compilation tools, release 12.1, V12.1.105
+ GCC: gcc (Ubuntu 11.4.0-1ubuntu1~22.04.2) 11.4.0
+ PyTorch: 2.3.0+cu121
+ PyTorch compiling details: PyTorch built with:
+ - GCC 9.3
+ - C++ Version: 201703
+ - Intel(R) oneAPI Math Kernel Library Version 2022.2-Product Build 20220804 for Intel(R) 64 architecture applications
+ - Intel(R) MKL-DNN v3.3.6 (Git Hash 86e6af5974177e513fd3fee58425e1063e7f1361)
+ - OpenMP 201511 (a.k.a. OpenMP 4.5)
+ - LAPACK is enabled (usually provided by MKL)
+ - NNPACK is enabled
+ - CPU capability usage: AVX2
+ - CUDA Runtime 12.1
+ - NVCC architecture flags: -gencode;arch=compute_50,code=sm_50;-gencode;arch=compute_60,code=sm_60;-gencode;arch=compute_70,code=sm_70;-gencode;arch=compute_75,code=sm_75;-gencode;arch=compute_80,code=sm_80;-gencode;arch=compute_86,code=sm_86;-gencode;arch=compute_90,code=sm_90
+ - CuDNN 8.9.2
+ - Magma 2.6.1
+ - Build settings: BLAS_INFO=mkl, BUILD_TYPE=Release, CUDA_VERSION=12.1, CUDNN_VERSION=8.9.2, CXX_COMPILER=/opt/rh/devtoolset-9/root/usr/bin/c++, CXX_FLAGS= -D_GLIBCXX_USE_CXX11_ABI=0 -fabi-version=11 -fvisibility-inlines-hidden -DUSE_PTHREADPOOL -DNDEBUG -DUSE_KINETO -DLIBKINETO_NOROCTRACER -DUSE_FBGEMM -DUSE_QNNPACK -DUSE_PYTORCH_QNNPACK -DUSE_XNNPACK -DSYMBOLICATE_MOBILE_DEBUG_HANDLE -O2 -fPIC -Wall -Wextra -Werror=return-type -Werror=non-virtual-dtor -Werror=bool-operation -Wnarrowing -Wno-missing-field-initializers -Wno-type-limits -Wno-array-bounds -Wno-unknown-pragmas -Wno-unused-parameter -Wno-unused-function -Wno-unused-result -Wno-strict-overflow -Wno-strict-aliasing -Wno-stringop-overflow -Wsuggest-override -Wno-psabi -Wno-error=pedantic -Wno-error=old-style-cast -Wno-missing-braces -fdiagnostics-color=always -faligned-new -Wno-unused-but-set-variable -Wno-maybe-uninitialized -fno-math-errno -fno-trapping-math -Werror=format -Wno-stringop-overflow, LAPACK_INFO=mkl, PERF_WITH_AVX=1, PERF_WITH_AVX2=1, PERF_WITH_AVX512=1, TORCH_VERSION=2.3.0, USE_CUDA=ON, USE_CUDNN=ON, USE_CUSPARSELT=1, USE_EXCEPTION_PTR=1, USE_GFLAGS=OFF, USE_GLOG=OFF, USE_GLOO=ON, USE_MKL=ON, USE_MKLDNN=ON, USE_MPI=OFF, USE_NCCL=1, USE_NNPACK=ON, USE_OPENMP=ON, USE_ROCM=OFF, USE_ROCM_KERNEL_ASSERT=OFF,
+
+ TorchVision: 0.18.0+cu121
+ OpenCV: 4.12.0
+ MMEngine: 0.10.7
+
+Runtime environment:
+ cudnn_benchmark: False
+ mp_cfg: {'mp_start_method': 'fork', 'opencv_num_threads': 0}
+ dist_cfg: {'backend': 'nccl'}
+ seed: 721894494
+ Distributed launcher: none
+ Distributed training: False
+ GPU number: 1
+------------------------------------------------------------
+
+2025/12/26 18:23:17 - mmengine - INFO - Config:
+auto_scale_lr = None
+backend_args = dict(backend='local')
+codec = dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap')
+custom_hooks = [
+ dict(type='SyncBuffersHook'),
+]
+custom_imports = dict(
+ allow_failed_imports=False,
+ imports=[
+ 'mmpose.engine.optim_wrappers.layer_decay_optim_wrapper',
+ ])
+data_mode = 'topdown'
+data_root = '/root/miko/puni/train/GVHMR/processed_data/'
+dataset_type = 'CocoDataset'
+default_hooks = dict(
+ badcase=dict(
+ badcase_thr=5,
+ enable=False,
+ metric_type='loss',
+ out_dir='badcase',
+ type='BadCaseAnalysisHook'),
+ checkpoint=dict(
+ interval=1,
+ max_keep_ckpts=1,
+ rule='greater',
+ save_best='coco/AP',
+ save_optimizer=False,
+ type='CheckpointHook'),
+ logger=dict(interval=5, type='LoggerHook'),
+ param_scheduler=dict(type='ParamSchedulerHook'),
+ sampler_seed=dict(type='DistSamplerSeedHook'),
+ timer=dict(type='IterTimerHook'),
+ visualization=dict(
+ enable=True,
+ interval=1,
+ out_dir='vis_results',
+ type='PoseVisualizationHook'))
+default_scope = 'mmpose'
+env_cfg = dict(
+ cudnn_benchmark=False,
+ dist_cfg=dict(backend='nccl'),
+ mp_cfg=dict(mp_start_method='fork', opencv_num_threads=0))
+launcher = 'none'
+load_from = '/root/miko/puni/train/GVHMR/inputs/checkpoints/vitpose/td-hm_ViTPose-huge_8xb64-210e_coco-256x192-e32adcd4_20230314.pth'
+log_level = 'INFO'
+log_processor = dict(
+ by_epoch=True, num_digits=6, type='LogProcessor', window_size=50)
+metainfo = dict(from_file='configs/_base_/datasets/coco.py')
+model = dict(
+ backbone=dict(
+ arch='huge',
+ drop_path_rate=0.55,
+ img_size=(
+ 256,
+ 192,
+ ),
+ init_cfg=None,
+ out_type='featmap',
+ patch_cfg=dict(padding=2),
+ patch_size=16,
+ qkv_bias=True,
+ type='mmpretrain.VisionTransformer',
+ with_cls_token=False),
+ data_preprocessor=dict(
+ bgr_to_rgb=True,
+ mean=[
+ 123.675,
+ 116.28,
+ 103.53,
+ ],
+ std=[
+ 58.395,
+ 57.12,
+ 57.375,
+ ],
+ type='PoseDataPreprocessor'),
+ head=dict(
+ decoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ deconv_kernel_sizes=(
+ 4,
+ 4,
+ ),
+ deconv_out_channels=(
+ 256,
+ 256,
+ ),
+ in_channels=1280,
+ loss=dict(type='KeypointMSELoss', use_target_weight=True),
+ out_channels=17,
+ type='HeatmapHead'),
+ test_cfg=dict(flip_mode='heatmap', flip_test=True, shift_heatmap=False),
+ type='TopdownPoseEstimator')
+optim_wrapper = dict(
+ clip_grad=dict(max_norm=1.0, norm_type=2),
+ constructor='LayerDecayOptimWrapperConstructor',
+ optimizer=dict(
+ betas=(
+ 0.9,
+ 0.999,
+ ), lr=5e-05, type='AdamW', weight_decay=0.1),
+ paramwise_cfg=dict(
+ custom_keys=dict(
+ bias=dict(decay_multi=0.0),
+ norm=dict(decay_mult=0.0),
+ pos_embed=dict(decay_mult=0.0),
+ relative_position_bias_table=dict(decay_mult=0.0)),
+ layer_decay_rate=0.85,
+ num_layers=32))
+param_scheduler = [
+ dict(
+ begin=0,
+ by_epoch=True,
+ end=20,
+ gamma=0.1,
+ milestones=[
+ 14,
+ 18,
+ ],
+ type='MultiStepLR'),
+]
+resume = False
+test_cfg = dict()
+test_dataloader = dict(
+ batch_size=32,
+ dataset=dict(
+ ann_file='annotations/person_keypoints_val2017.json',
+ bbox_file=
+ 'data/coco/person_detection_results/COCO_val2017_detections_AP_H_56_person.json',
+ data_mode='topdown',
+ data_prefix=dict(img='val2017/'),
+ data_root='data/coco/',
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+test_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+train_cfg = dict(by_epoch=True, max_epochs=20, val_interval=1)
+train_dataloader = dict(
+ batch_size=4,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+ ],
+ type='CocoDataset'),
+ num_workers=0,
+ persistent_workers=False,
+ sampler=dict(shuffle=True, type='DefaultSampler'))
+train_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+]
+val_cfg = dict()
+val_dataloader = dict(
+ batch_size=4,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ bbox_file=None,
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=0,
+ persistent_workers=False,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+val_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+val_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+]
+vis_backends = [
+ dict(type='LocalVisBackend'),
+]
+visualizer = dict(
+ name='visualizer',
+ type='PoseLocalVisualizer',
+ vis_backends=[
+ dict(type='LocalVisBackend'),
+ ])
+work_dir = '../work_dirs/vitpose_finetune'
+
+2025/12/26 18:23:22 - mmengine - INFO - Distributed training is not used, all SyncBatchNorm (SyncBN) layers in the model will be automatically reverted to BatchNormXd layers if they are used.
+2025/12/26 18:23:22 - mmengine - INFO - Hooks will be executed in the following order:
+before_run:
+(VERY_HIGH ) RuntimeInfoHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+before_train:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+before_train_epoch:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(NORMAL ) DistSamplerSeedHook
+ --------------------
+before_train_iter:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+ --------------------
+after_train_iter:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(BELOW_NORMAL) LoggerHook
+(LOW ) ParamSchedulerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+after_train_epoch:
+(NORMAL ) IterTimerHook
+(NORMAL ) SyncBuffersHook
+(LOW ) ParamSchedulerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+before_val:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+before_val_epoch:
+(NORMAL ) IterTimerHook
+(NORMAL ) SyncBuffersHook
+ --------------------
+before_val_iter:
+(NORMAL ) IterTimerHook
+ --------------------
+after_val_iter:
+(NORMAL ) IterTimerHook
+(NORMAL ) PoseVisualizationHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+after_val_epoch:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(BELOW_NORMAL) LoggerHook
+(LOW ) ParamSchedulerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+after_val:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+after_train:
+(VERY_HIGH ) RuntimeInfoHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+before_test:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+before_test_epoch:
+(NORMAL ) IterTimerHook
+ --------------------
+before_test_iter:
+(NORMAL ) IterTimerHook
+ --------------------
+after_test_iter:
+(NORMAL ) IterTimerHook
+(NORMAL ) PoseVisualizationHook
+(NORMAL ) BadCaseAnalysisHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+after_test_epoch:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(NORMAL ) BadCaseAnalysisHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+after_test:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+after_run:
+(BELOW_NORMAL) LoggerHook
+ --------------------
+Name of parameter - Initialization information
+
+backbone.pos_embed - torch.Size([1, 192, 1280]):
+Initialized by user-defined `init_weights` in VisionTransformer
+
+backbone.patch_embed.projection.weight - torch.Size([1280, 3, 16, 16]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.patch_embed.projection.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.0.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.0.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.0.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.0.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.1.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.1.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.1.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.1.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.1.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.2.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.2.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.2.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.2.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.2.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.3.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.3.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.3.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.3.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.3.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.4.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.4.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.4.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.4.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.4.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.5.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.5.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.5.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.5.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.5.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.6.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.6.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.6.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.6.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.6.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.7.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.7.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.7.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.7.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.7.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.8.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.8.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.8.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.8.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.8.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.9.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.9.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.9.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.9.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.9.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.10.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.10.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.10.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.10.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.10.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.11.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.11.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.11.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.11.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.11.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.12.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.12.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.12.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.12.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.12.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.13.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.13.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.13.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.13.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.13.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.14.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.14.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.14.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.14.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.14.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.15.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.15.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.15.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.15.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.15.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.16.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.16.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.16.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.16.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.16.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.17.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.17.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.17.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.17.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.17.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.18.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.18.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.18.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.18.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.18.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.19.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.19.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.19.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.19.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.19.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.20.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.20.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.20.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.20.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.20.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.21.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.21.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.21.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.21.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.21.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.22.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.22.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.22.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.22.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.22.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.23.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.23.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.23.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.23.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.23.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.24.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.24.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.24.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.24.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.24.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.25.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.25.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.25.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.25.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.25.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.26.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.26.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.26.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.26.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.26.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.27.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.27.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.27.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.27.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.27.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.28.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.28.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.28.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.28.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.28.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.29.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.29.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.29.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.29.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.29.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.30.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.30.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.30.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.30.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.30.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.31.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.attn.qkv.weight - torch.Size([3840, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.attn.qkv.bias - torch.Size([3840]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.attn.proj.weight - torch.Size([1280, 1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.attn.proj.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.ln2.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.ln2.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.layers.31.ffn.layers.0.0.weight - torch.Size([5120, 1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.31.ffn.layers.0.0.bias - torch.Size([5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.31.ffn.layers.1.weight - torch.Size([1280, 5120]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.layers.31.ffn.layers.1.bias - torch.Size([1280]):
+Initialized by user-defined `init_weights` in TransformerEncoderLayer
+
+backbone.ln1.weight - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+backbone.ln1.bias - torch.Size([1280]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+head.deconv_layers.0.weight - torch.Size([1280, 256, 4, 4]):
+NormalInit: mean=0, std=0.001, bias=0
+
+head.deconv_layers.1.weight - torch.Size([256]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+head.deconv_layers.1.bias - torch.Size([256]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+head.deconv_layers.3.weight - torch.Size([256, 256, 4, 4]):
+NormalInit: mean=0, std=0.001, bias=0
+
+head.deconv_layers.4.weight - torch.Size([256]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+head.deconv_layers.4.bias - torch.Size([256]):
+The value is the same before and after calling `init_weights` of TopdownPoseEstimator
+
+head.final_layer.weight - torch.Size([17, 256, 1, 1]):
+NormalInit: mean=0, std=0.001, bias=0
+
+head.final_layer.bias - torch.Size([17]):
+NormalInit: mean=0, std=0.001, bias=0
+2025/12/26 18:23:25 - mmengine - INFO - Load checkpoint from /root/miko/puni/train/GVHMR/inputs/checkpoints/vitpose/td-hm_ViTPose-huge_8xb64-210e_coco-256x192-e32adcd4_20230314.pth
+2025/12/26 18:23:25 - mmengine - WARNING - "FileClient" will be deprecated in future. Please use io functions in https://mmengine.readthedocs.io/en/latest/api/fileio.html#file-io
+2025/12/26 18:23:25 - mmengine - WARNING - "HardDiskBackend" is the alias of "LocalBackend" and the former will be deprecated in future.
+2025/12/26 18:23:25 - mmengine - INFO - Checkpoints will be saved to /root/miko/puni/train/GVHMR/work_dirs/vitpose_finetune.
+2025/12/26 18:23:29 - mmengine - INFO - Epoch(train) [1][ 5/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:13:41 time: 0.764232 data_time: 0.036754 memory: 10107 grad_norm: 0.019636 loss: 0.001051 loss_kpt: 0.001051 acc_pose: 0.865385
+2025/12/26 18:23:33 - mmengine - INFO - Epoch(train) [1][10/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:13:16 time: 0.744301 data_time: 0.034762 memory: 10107 grad_norm: 0.021607 loss: 0.000937 loss_kpt: 0.000937 acc_pose: 0.730769
+2025/12/26 18:23:36 - mmengine - INFO - Epoch(train) [1][15/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:59 time: 0.731815 data_time: 0.034947 memory: 10107 grad_norm: 0.018354 loss: 0.000807 loss_kpt: 0.000807 acc_pose: 0.923077
+2025/12/26 18:23:40 - mmengine - INFO - Epoch(train) [1][20/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:53 time: 0.729963 data_time: 0.037258 memory: 10107 grad_norm: 0.016740 loss: 0.000742 loss_kpt: 0.000742 acc_pose: 0.833333
+2025/12/26 18:23:45 - mmengine - INFO - Epoch(train) [1][25/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:14:07 time: 0.802908 data_time: 0.036466 memory: 10107 grad_norm: 0.015388 loss: 0.000698 loss_kpt: 0.000698 acc_pose: 0.878205
+2025/12/26 18:23:49 - mmengine - INFO - Epoch(train) [1][30/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:13:47 time: 0.788151 data_time: 0.035721 memory: 10107 grad_norm: 0.015608 loss: 0.000672 loss_kpt: 0.000672 acc_pose: 0.942308
+2025/12/26 18:23:52 - mmengine - INFO - Epoch(train) [1][35/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:13:29 time: 0.774476 data_time: 0.035046 memory: 10107 grad_norm: 0.015550 loss: 0.000639 loss_kpt: 0.000639 acc_pose: 0.807692
+2025/12/26 18:23:56 - mmengine - INFO - Epoch(train) [1][40/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:13:12 time: 0.762040 data_time: 0.034101 memory: 10107 grad_norm: 0.015037 loss: 0.000606 loss_kpt: 0.000606 acc_pose: 0.955128
+2025/12/26 18:23:59 - mmengine - INFO - Epoch(train) [1][45/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:13:00 time: 0.753771 data_time: 0.033316 memory: 10107 grad_norm: 0.014563 loss: 0.000580 loss_kpt: 0.000580 acc_pose: 0.980769
+2025/12/26 18:24:03 - mmengine - INFO - Epoch(train) [1][50/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:50 time: 0.748186 data_time: 0.033002 memory: 10107 grad_norm: 0.014457 loss: 0.000564 loss_kpt: 0.000564 acc_pose: 0.935897
+2025/12/26 18:24:05 - mmengine - INFO - Exp name: vitpose_huge_finetune_20251226_182315
+2025/12/26 18:24:05 - mmengine - INFO - Saving checkpoint at 1 epochs
+2025/12/26 18:24:35 - mmengine - INFO - Epoch(val) [1][ 5/54] eta: 0:00:22 time: 0.460444 data_time: 0.121652 memory: 10107
+2025/12/26 18:24:37 - mmengine - INFO - Epoch(val) [1][10/54] eta: 0:00:21 time: 0.492093 data_time: 0.164461 memory: 7682
+2025/12/26 18:24:39 - mmengine - INFO - Epoch(val) [1][15/54] eta: 0:00:17 time: 0.449483 data_time: 0.125413 memory: 7682
+2025/12/26 18:24:41 - mmengine - INFO - Epoch(val) [1][20/54] eta: 0:00:14 time: 0.427144 data_time: 0.106805 memory: 7682
+2025/12/26 18:24:43 - mmengine - INFO - Epoch(val) [1][25/54] eta: 0:00:12 time: 0.413923 data_time: 0.094380 memory: 7682
+2025/12/26 18:24:44 - mmengine - INFO - Epoch(val) [1][30/54] eta: 0:00:09 time: 0.408919 data_time: 0.087910 memory: 7682
+2025/12/26 18:24:46 - mmengine - INFO - Epoch(val) [1][35/54] eta: 0:00:07 time: 0.404102 data_time: 0.082700 memory: 7682
+2025/12/26 18:24:48 - mmengine - INFO - Epoch(val) [1][40/54] eta: 0:00:05 time: 0.399717 data_time: 0.078238 memory: 7682
+2025/12/26 18:24:50 - mmengine - INFO - Epoch(val) [1][45/54] eta: 0:00:03 time: 0.396237 data_time: 0.075097 memory: 7682
+2025/12/26 18:24:55 - mmengine - INFO - Epoch(val) [1][50/54] eta: 0:00:01 time: 0.448834 data_time: 0.072252 memory: 7682
+2025/12/26 18:24:56 - mmengine - INFO - Evaluating CocoMetric...
+2025/12/26 18:24:56 - mmengine - INFO - Epoch(val) [1][54/54] coco/AP: 0.997902 coco/AP .5: 1.000000 coco/AP .75: 1.000000 coco/AP (M): -1.000000 coco/AP (L): 0.997902 coco/AR: 0.998611 coco/AR .5: 1.000000 coco/AR .75: 1.000000 coco/AR (M): -1.000000 coco/AR (L): 0.998611 data_time: 0.070707 time: 0.443549
+2025/12/26 18:25:51 - mmengine - INFO - The best checkpoint with 0.9979 coco/AP at 1 epoch is saved to best_coco_AP_epoch_1.pth.
+2025/12/26 18:27:02 - mmengine - INFO - Epoch(train) [2][ 5/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:39 time: 0.743397 data_time: 0.032746 memory: 10107 grad_norm: 0.012570 loss: 0.000469 loss_kpt: 0.000469 acc_pose: 0.865385
+2025/12/26 18:27:06 - mmengine - INFO - Epoch(train) [2][10/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:30 time: 0.740149 data_time: 0.032812 memory: 10107 grad_norm: 0.011676 loss: 0.000442 loss_kpt: 0.000442 acc_pose: 0.961538
+2025/12/26 18:27:09 - mmengine - INFO - Epoch(train) [2][15/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:23 time: 0.736868 data_time: 0.031395 memory: 10107 grad_norm: 0.011850 loss: 0.000418 loss_kpt: 0.000418 acc_pose: 0.961538
+2025/12/26 18:27:16 - mmengine - INFO - Epoch(train) [2][20/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:56 time: 0.753985 data_time: 0.031221 memory: 10107 grad_norm: 0.011669 loss: 0.000413 loss_kpt: 0.000413 acc_pose: 0.955128
+2025/12/26 18:27:19 - mmengine - INFO - Epoch(train) [2][25/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:47 time: 0.752214 data_time: 0.031313 memory: 10107 grad_norm: 0.010922 loss: 0.000385 loss_kpt: 0.000385 acc_pose: 0.961538
+2025/12/26 18:27:23 - mmengine - INFO - Epoch(train) [2][30/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:37 time: 0.749049 data_time: 0.030978 memory: 10107 grad_norm: 0.010725 loss: 0.000380 loss_kpt: 0.000380 acc_pose: 0.923077
+2025/12/26 18:27:26 - mmengine - INFO - Epoch(train) [2][35/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:28 time: 0.748924 data_time: 0.031602 memory: 10107 grad_norm: 0.009691 loss: 0.000362 loss_kpt: 0.000362 acc_pose: 1.000000
+2025/12/26 18:27:29 - mmengine - INFO - Epoch(train) [2][40/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:20 time: 0.747481 data_time: 0.031720 memory: 10107 grad_norm: 0.009203 loss: 0.000356 loss_kpt: 0.000356 acc_pose: 0.923077
+2025/12/26 18:27:33 - mmengine - INFO - Epoch(train) [2][45/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:13 time: 0.746588 data_time: 0.032247 memory: 10107 grad_norm: 0.008715 loss: 0.000344 loss_kpt: 0.000344 acc_pose: 1.000000
+2025/12/26 18:27:36 - mmengine - INFO - Epoch(train) [2][50/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:06 time: 0.745307 data_time: 0.031835 memory: 10107 grad_norm: 0.008379 loss: 0.000328 loss_kpt: 0.000328 acc_pose: 0.961538
+2025/12/26 18:27:39 - mmengine - INFO - Exp name: vitpose_huge_finetune_20251226_182315
+2025/12/26 18:27:39 - mmengine - INFO - Saving checkpoint at 2 epochs
+2025/12/26 18:28:41 - mmengine - INFO - Epoch(val) [2][ 5/54] eta: 0:00:20 time: 0.430134 data_time: 0.052152 memory: 10107
+2025/12/26 18:28:43 - mmengine - INFO - Epoch(val) [2][10/54] eta: 0:00:17 time: 0.429318 data_time: 0.051575 memory: 7682
+2025/12/26 18:28:45 - mmengine - INFO - Epoch(val) [2][15/54] eta: 0:00:14 time: 0.429192 data_time: 0.051190 memory: 7682
+2025/12/26 18:29:15 - mmengine - INFO - Epoch(val) [2][20/54] eta: 0:00:12 time: 0.429511 data_time: 0.050658 memory: 7682
+2025/12/26 18:29:17 - mmengine - INFO - Epoch(val) [2][25/54] eta: 0:00:44 time: 1.002689 data_time: 0.618865 memory: 7682
+2025/12/26 18:29:19 - mmengine - INFO - Epoch(val) [2][30/54] eta: 0:00:31 time: 1.001245 data_time: 0.618541 memory: 7682
+2025/12/26 18:29:21 - mmengine - INFO - Epoch(val) [2][35/54] eta: 0:00:22 time: 1.001166 data_time: 0.618902 memory: 7682
+2025/12/26 18:29:23 - mmengine - INFO - Epoch(val) [2][40/54] eta: 0:00:15 time: 1.002421 data_time: 0.620477 memory: 7682
+2025/12/26 18:29:25 - mmengine - INFO - Epoch(val) [2][45/54] eta: 0:00:09 time: 0.947005 data_time: 0.620496 memory: 7682
+2025/12/26 18:29:27 - mmengine - INFO - Epoch(val) [2][50/54] eta: 0:00:03 time: 0.946052 data_time: 0.619977 memory: 7682
+2025/12/26 18:29:28 - mmengine - INFO - Evaluating CocoMetric...
+2025/12/26 18:29:28 - mmengine - INFO - Epoch(val) [2][54/54] coco/AP: 0.999001 coco/AP .5: 1.000000 coco/AP .75: 1.000000 coco/AP (M): -1.000000 coco/AP (L): 0.999001 coco/AR: 0.999537 coco/AR .5: 1.000000 coco/AR .75: 1.000000 coco/AR (M): -1.000000 coco/AR (L): 0.999537 data_time: 0.568328 time: 0.893951
+2025/12/26 18:29:28 - mmengine - INFO - The previous best checkpoint /root/miko/puni/train/GVHMR/work_dirs/vitpose_finetune/best_coco_AP_epoch_1.pth is removed
+2025/12/26 18:30:17 - mmengine - INFO - The best checkpoint with 0.9990 coco/AP at 2 epoch is saved to best_coco_AP_epoch_2.pth.
+2025/12/26 18:31:04 - mmengine - INFO - Epoch(train) [3][ 5/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:56 time: 0.741743 data_time: 0.031337 memory: 10107 grad_norm: 0.008334 loss: 0.000314 loss_kpt: 0.000314 acc_pose: 1.000000
+2025/12/26 18:31:08 - mmengine - INFO - Epoch(train) [3][10/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:50 time: 0.741519 data_time: 0.032124 memory: 10107 grad_norm: 0.007767 loss: 0.000309 loss_kpt: 0.000309 acc_pose: 1.000000
+2025/12/26 18:31:11 - mmengine - INFO - Epoch(train) [3][15/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:43 time: 0.681372 data_time: 0.032214 memory: 10107 grad_norm: 0.008256 loss: 0.000298 loss_kpt: 0.000298 acc_pose: 0.903846
+2025/12/26 18:31:15 - mmengine - INFO - Epoch(train) [3][20/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:40 time: 0.687149 data_time: 0.032968 memory: 10107 grad_norm: 0.008197 loss: 0.000288 loss_kpt: 0.000288 acc_pose: 1.000000
+2025/12/26 18:31:21 - mmengine - INFO - Epoch(train) [3][25/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:58 time: 0.754160 data_time: 0.033881 memory: 10107 grad_norm: 0.008475 loss: 0.000286 loss_kpt: 0.000286 acc_pose: 0.955128
+2025/12/26 18:31:25 - mmengine - INFO - Epoch(train) [3][30/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:53 time: 0.759891 data_time: 0.034764 memory: 10107 grad_norm: 0.008205 loss: 0.000277 loss_kpt: 0.000277 acc_pose: 1.000000
+2025/12/26 18:31:29 - mmengine - INFO - Epoch(train) [3][35/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:48 time: 0.764262 data_time: 0.034711 memory: 10107 grad_norm: 0.008561 loss: 0.000275 loss_kpt: 0.000275 acc_pose: 0.942308
+2025/12/26 18:31:32 - mmengine - INFO - Epoch(train) [3][40/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:42 time: 0.764284 data_time: 0.034351 memory: 10107 grad_norm: 0.008108 loss: 0.000263 loss_kpt: 0.000263 acc_pose: 1.000000
+2025/12/26 18:31:35 - mmengine - INFO - Epoch(train) [3][45/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:36 time: 0.764958 data_time: 0.034349 memory: 10107 grad_norm: 0.008033 loss: 0.000261 loss_kpt: 0.000261 acc_pose: 0.980769
+2025/12/26 18:31:39 - mmengine - INFO - Epoch(train) [3][50/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:31 time: 0.765194 data_time: 0.034553 memory: 10107 grad_norm: 0.008028 loss: 0.000258 loss_kpt: 0.000258 acc_pose: 0.980769
+2025/12/26 18:31:42 - mmengine - INFO - Exp name: vitpose_huge_finetune_20251226_182315
+2025/12/26 18:31:42 - mmengine - INFO - Saving checkpoint at 3 epochs
+2025/12/26 18:33:04 - mmengine - INFO - Epoch(val) [3][ 5/54] eta: 0:00:19 time: 0.944663 data_time: 0.617364 memory: 10107
+2025/12/26 18:33:06 - mmengine - INFO - Epoch(val) [3][10/54] eta: 0:00:17 time: 0.946774 data_time: 0.618295 memory: 7682
+2025/12/26 18:33:08 - mmengine - INFO - Epoch(val) [3][15/54] eta: 0:00:14 time: 0.947891 data_time: 0.618772 memory: 7682
+2025/12/26 18:33:10 - mmengine - INFO - Epoch(val) [3][20/54] eta: 0:00:12 time: 0.373469 data_time: 0.049533 memory: 7682
+2025/12/26 18:33:13 - mmengine - INFO - Epoch(val) [3][25/54] eta: 0:00:12 time: 0.398722 data_time: 0.073851 memory: 7682
+2025/12/26 18:33:15 - mmengine - INFO - Epoch(val) [3][30/54] eta: 0:00:09 time: 0.398670 data_time: 0.073494 memory: 7682
+2025/12/26 18:33:17 - mmengine - INFO - Epoch(val) [3][35/54] eta: 0:00:07 time: 0.395516 data_time: 0.071093 memory: 7682
+2025/12/26 18:33:18 - mmengine - INFO - Epoch(val) [3][40/54] eta: 0:00:05 time: 0.395052 data_time: 0.071582 memory: 7682
+2025/12/26 18:33:20 - mmengine - INFO - Epoch(val) [3][45/54] eta: 0:00:03 time: 0.393863 data_time: 0.071700 memory: 7682
+2025/12/26 18:33:22 - mmengine - INFO - Epoch(val) [3][50/54] eta: 0:00:01 time: 0.392369 data_time: 0.070995 memory: 7682
+2025/12/26 18:33:23 - mmengine - INFO - Evaluating CocoMetric...
+2025/12/26 18:33:23 - mmengine - INFO - Epoch(val) [3][54/54] coco/AP: 0.999001 coco/AP .5: 1.000000 coco/AP .75: 1.000000 coco/AP (M): -1.000000 coco/AP (L): 0.999001 coco/AR: 0.999537 coco/AR .5: 1.000000 coco/AR .75: 1.000000 coco/AR (M): -1.000000 coco/AR (L): 0.999537 data_time: 0.069042 time: 0.389932
+2025/12/26 18:33:27 - mmengine - INFO - Epoch(train) [4][ 5/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:21 time: 0.766323 data_time: 0.037657 memory: 10107 grad_norm: 0.007863 loss: 0.000249 loss_kpt: 0.000249 acc_pose: 1.000000
+2025/12/26 18:33:30 - mmengine - INFO - Epoch(train) [4][10/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:16 time: 0.767784 data_time: 0.037230 memory: 10107 grad_norm: 0.007575 loss: 0.000244 loss_kpt: 0.000244 acc_pose: 0.942308
+2025/12/26 18:33:50 - mmengine - INFO - Epoch(train) [4][15/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:33 time: 1.084146 data_time: 0.353098 memory: 10107 grad_norm: 0.006805 loss: 0.000251 loss_kpt: 0.000251 acc_pose: 0.948718
+2025/12/26 18:33:53 - mmengine - INFO - Epoch(train) [4][20/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:25 time: 1.017580 data_time: 0.352255 memory: 10107 grad_norm: 0.006327 loss: 0.000246 loss_kpt: 0.000246 acc_pose: 0.961538
+2025/12/26 18:33:57 - mmengine - INFO - Epoch(train) [4][25/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:17 time: 1.012922 data_time: 0.351080 memory: 10107 grad_norm: 0.006400 loss: 0.000251 loss_kpt: 0.000251 acc_pose: 1.000000
+2025/12/26 18:34:00 - mmengine - INFO - Epoch(train) [4][30/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:10 time: 1.011975 data_time: 0.351117 memory: 10107 grad_norm: 0.006111 loss: 0.000245 loss_kpt: 0.000245 acc_pose: 0.980769
+2025/12/26 18:34:04 - mmengine - INFO - Epoch(train) [4][35/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:04 time: 1.013251 data_time: 0.351485 memory: 10107 grad_norm: 0.006522 loss: 0.000249 loss_kpt: 0.000249 acc_pose: 0.980769
+2025/12/26 18:34:07 - mmengine - INFO - Epoch(train) [4][40/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:57 time: 1.014701 data_time: 0.351380 memory: 10107 grad_norm: 0.006671 loss: 0.000251 loss_kpt: 0.000251 acc_pose: 1.000000
+2025/12/26 18:34:13 - mmengine - INFO - Epoch(train) [4][45/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:01 time: 1.065679 data_time: 0.351483 memory: 10107 grad_norm: 0.006591 loss: 0.000240 loss_kpt: 0.000240 acc_pose: 1.000000
+2025/12/26 18:34:17 - mmengine - INFO - Epoch(train) [4][50/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:11:54 time: 1.064893 data_time: 0.346761 memory: 10107 grad_norm: 0.006291 loss: 0.000244 loss_kpt: 0.000244 acc_pose: 0.897436
+2025/12/26 18:34:19 - mmengine - INFO - Exp name: vitpose_huge_finetune_20251226_182315
+2025/12/26 18:34:19 - mmengine - INFO - Saving checkpoint at 4 epochs
+2025/12/26 18:35:12 - mmengine - INFO - Epoch(val) [4][ 5/54] eta: 0:00:19 time: 0.391559 data_time: 0.070638 memory: 10107
+2025/12/26 18:35:14 - mmengine - INFO - Epoch(val) [4][10/54] eta: 0:00:16 time: 0.391188 data_time: 0.070370 memory: 7682
+2025/12/26 18:35:16 - mmengine - INFO - Epoch(val) [4][15/54] eta: 0:00:14 time: 0.391405 data_time: 0.070491 memory: 7682
+2025/12/26 18:35:17 - mmengine - INFO - Epoch(val) [4][20/54] eta: 0:00:12 time: 0.367268 data_time: 0.046219 memory: 7682
+2025/12/26 18:35:22 - mmengine - INFO - Epoch(val) [4][25/54] eta: 0:00:13 time: 0.417436 data_time: 0.095715 memory: 7682
+2025/12/26 18:35:24 - mmengine - INFO - Epoch(val) [4][30/54] eta: 0:00:11 time: 0.419983 data_time: 0.096329 memory: 7682
+2025/12/26 18:35:26 - mmengine - INFO - Epoch(val) [4][35/54] eta: 0:00:08 time: 0.421417 data_time: 0.095826 memory: 7682
+2025/12/26 18:35:27 - mmengine - INFO - Epoch(val) [4][40/54] eta: 0:00:06 time: 0.423085 data_time: 0.095672 memory: 7682
+2025/12/26 18:35:29 - mmengine - INFO - Epoch(val) [4][45/54] eta: 0:00:03 time: 0.425266 data_time: 0.096119 memory: 7682
+2025/12/26 18:35:31 - mmengine - INFO - Epoch(val) [4][50/54] eta: 0:00:01 time: 0.426419 data_time: 0.095850 memory: 7682
+2025/12/26 18:35:33 - mmengine - INFO - Evaluating CocoMetric...
+2025/12/26 18:35:33 - mmengine - INFO - Epoch(val) [4][54/54] coco/AP: 1.000000 coco/AP .5: 1.000000 coco/AP .75: 1.000000 coco/AP (M): -1.000000 coco/AP (L): 1.000000 coco/AR: 1.000000 coco/AR .5: 1.000000 coco/AR .75: 1.000000 coco/AR (M): -1.000000 coco/AR (L): 1.000000 data_time: 0.091531 time: 0.421307
+2025/12/26 18:35:33 - mmengine - INFO - The previous best checkpoint /root/miko/puni/train/GVHMR/work_dirs/vitpose_finetune/best_coco_AP_epoch_2.pth is removed
+2025/12/26 18:36:43 - mmengine - INFO - The best checkpoint with 1.0000 coco/AP at 4 epoch is saved to best_coco_AP_epoch_4.pth.
+2025/12/26 18:38:38 - mmengine - INFO - Epoch(train) [5][ 5/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:46 time: 1.393512 data_time: 0.593424 memory: 10107 grad_norm: 0.006485 loss: 0.000250 loss_kpt: 0.000250 acc_pose: 0.948718
+2025/12/26 18:38:42 - mmengine - INFO - Epoch(train) [5][10/54] base_lr: 5.000000e-05 lr: 2.343120e-07 eta: 0:12:41 time: 1.090702 data_time: 0.292002 memory: 10107 grad_norm: 0.006210 loss: 0.000237 loss_kpt: 0.000237 acc_pose: 1.000000
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/20251226_182315.json b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/20251226_182315.json
new file mode 100644
index 0000000000000000000000000000000000000000..623b8cd191ef955ecf19757bcd1956c2e0341e89
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/20251226_182315.json
@@ -0,0 +1,46 @@
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03675370216369629, "grad_norm": 0.01963555794209242, "loss": 0.0010510455002076923, "loss_kpt": 0.0010510455002076923, "acc_pose": 0.8653846153846154, "time": 0.7642317295074463, "epoch": 1, "iter": 5, "memory": 10107, "step": 5}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.034762120246887206, "grad_norm": 0.021606831159442664, "loss": 0.000936518149683252, "loss_kpt": 0.000936518149683252, "acc_pose": 0.7307692307692307, "time": 0.7443006753921508, "epoch": 1, "iter": 10, "memory": 10107, "step": 10}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03494704564412435, "grad_norm": 0.01835411675274372, "loss": 0.0008069732614482442, "loss_kpt": 0.0008069732614482442, "acc_pose": 0.9230769230769231, "time": 0.7318146387736003, "epoch": 1, "iter": 15, "memory": 10107, "step": 15}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.037258028984069824, "grad_norm": 0.016740318736992776, "loss": 0.0007415686268359423, "loss_kpt": 0.0007415686268359423, "acc_pose": 0.8333333333333333, "time": 0.7299626350402832, "epoch": 1, "iter": 20, "memory": 10107, "step": 20}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.036465845108032226, "grad_norm": 0.015388216506689787, "loss": 0.0006977964425459504, "loss_kpt": 0.0006977964425459504, "acc_pose": 0.8782051282051282, "time": 0.8029078578948975, "epoch": 1, "iter": 25, "memory": 10107, "step": 25}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03572105566660563, "grad_norm": 0.015608035571252307, "loss": 0.0006719227142942449, "loss_kpt": 0.0006719227142942449, "acc_pose": 0.9423076923076923, "time": 0.7881508429845174, "epoch": 1, "iter": 30, "memory": 10107, "step": 30}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03504582132611956, "grad_norm": 0.015550296221460615, "loss": 0.0006389786140061915, "loss_kpt": 0.0006389786140061915, "acc_pose": 0.8076923076923077, "time": 0.7744760990142823, "epoch": 1, "iter": 35, "memory": 10107, "step": 35}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03410131335258484, "grad_norm": 0.015036797313950957, "loss": 0.0006063677203201224, "loss_kpt": 0.0006063677203201224, "acc_pose": 0.9551282051282051, "time": 0.7620397627353668, "epoch": 1, "iter": 40, "memory": 10107, "step": 40}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03331554200914171, "grad_norm": 0.014563350400163069, "loss": 0.0005804542594382333, "loss_kpt": 0.0005804542594382333, "acc_pose": 0.9807692307692307, "time": 0.7537709024217394, "epoch": 1, "iter": 45, "memory": 10107, "step": 45}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03300153255462646, "grad_norm": 0.014456901978701354, "loss": 0.0005644101899815724, "loss_kpt": 0.0005644101899815724, "acc_pose": 0.9358974358974359, "time": 0.748186445236206, "epoch": 1, "iter": 50, "memory": 10107, "step": 50}
+{"coco/AP": 0.9979022902290229, "coco/AP .5": 1.0, "coco/AP .75": 1.0, "coco/AP (M)": -1.0, "coco/AP (L)": 0.9979022902290229, "coco/AR": 0.9986111111111111, "coco/AR .5": 1.0, "coco/AR .75": 1.0, "coco/AR (M)": -1.0, "coco/AR (L)": 0.9986111111111111, "data_time": 0.07070658383546052, "time": 0.4435488559581615, "step": 1}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03274566650390625, "grad_norm": 0.012570046382024884, "loss": 0.0004693160206079483, "loss_kpt": 0.0004693160206079483, "acc_pose": 0.8653846153846154, "time": 0.7433967685699463, "epoch": 2, "iter": 59, "memory": 10107, "step": 59}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03281200885772705, "grad_norm": 0.011675749067217111, "loss": 0.0004420897128875367, "loss_kpt": 0.0004420897128875367, "acc_pose": 0.9615384615384616, "time": 0.7401487922668457, "epoch": 2, "iter": 64, "memory": 10107, "step": 64}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03139461517333984, "grad_norm": 0.011850018203258515, "loss": 0.0004179099708562717, "loss_kpt": 0.0004179099708562717, "acc_pose": 0.9615384615384616, "time": 0.7368682384490967, "epoch": 2, "iter": 69, "memory": 10107, "step": 69}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.031221256256103516, "grad_norm": 0.011669193208217622, "loss": 0.0004131932323798537, "loss_kpt": 0.0004131932323798537, "acc_pose": 0.9551282051282051, "time": 0.7539845371246338, "epoch": 2, "iter": 74, "memory": 10107, "step": 74}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03131303310394287, "grad_norm": 0.010922268312424422, "loss": 0.0003852099014329724, "loss_kpt": 0.0003852099014329724, "acc_pose": 0.9615384615384616, "time": 0.7522138786315918, "epoch": 2, "iter": 79, "memory": 10107, "step": 79}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.030978074073791505, "grad_norm": 0.01072505726478994, "loss": 0.00038021193788154053, "loss_kpt": 0.00038021193788154053, "acc_pose": 0.9230769230769231, "time": 0.7490486001968384, "epoch": 2, "iter": 84, "memory": 10107, "step": 84}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.0316017484664917, "grad_norm": 0.009690730567090213, "loss": 0.0003619221624103375, "loss_kpt": 0.0003619221624103375, "acc_pose": 1.0, "time": 0.7489235210418701, "epoch": 2, "iter": 89, "memory": 10107, "step": 89}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03172038078308106, "grad_norm": 0.009203163315542042, "loss": 0.00035649942088639364, "loss_kpt": 0.00035649942088639364, "acc_pose": 0.9230769230769231, "time": 0.7474814081192016, "epoch": 2, "iter": 94, "memory": 10107, "step": 94}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03224703788757324, "grad_norm": 0.008715138505212962, "loss": 0.00034365662053460256, "loss_kpt": 0.00034365662053460256, "acc_pose": 1.0, "time": 0.7465879440307617, "epoch": 2, "iter": 99, "memory": 10107, "step": 99}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03183490753173828, "grad_norm": 0.008379335082136095, "loss": 0.0003281010233331472, "loss_kpt": 0.0003281010233331472, "acc_pose": 0.9615384615384616, "time": 0.7453065204620362, "epoch": 2, "iter": 104, "memory": 10107, "step": 104}
+{"coco/AP": 0.9990007334066741, "coco/AP .5": 1.0, "coco/AP .75": 1.0, "coco/AP (M)": -1.0, "coco/AP (L)": 0.9990007334066741, "coco/AR": 0.999537037037037, "coco/AR .5": 1.0, "coco/AR .75": 1.0, "coco/AR (M)": -1.0, "coco/AR (L)": 0.999537037037037, "data_time": 0.5683283415707675, "time": 0.8939508178017356, "step": 2}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.031337161064147946, "grad_norm": 0.008334153429605067, "loss": 0.0003143988369265571, "loss_kpt": 0.0003143988369265571, "acc_pose": 1.0, "time": 0.7417430591583252, "epoch": 3, "iter": 113, "memory": 10107, "step": 113}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03212445259094238, "grad_norm": 0.007767286342568696, "loss": 0.0003093978948891163, "loss_kpt": 0.0003093978948891163, "acc_pose": 1.0, "time": 0.7415185737609863, "epoch": 3, "iter": 118, "memory": 10107, "step": 118}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.0322141695022583, "grad_norm": 0.00825631142128259, "loss": 0.00029793519875966014, "loss_kpt": 0.00029793519875966014, "acc_pose": 0.9038461538461539, "time": 0.6813721513748169, "epoch": 3, "iter": 123, "memory": 10107, "step": 123}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03296780586242676, "grad_norm": 0.008196818535216152, "loss": 0.0002875287723145448, "loss_kpt": 0.0002875287723145448, "acc_pose": 1.0, "time": 0.6871491479873657, "epoch": 3, "iter": 128, "memory": 10107, "step": 128}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03388114452362061, "grad_norm": 0.008475003554485739, "loss": 0.00028574865777045486, "loss_kpt": 0.00028574865777045486, "acc_pose": 0.9551282051282051, "time": 0.754159688949585, "epoch": 3, "iter": 133, "memory": 10107, "step": 133}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03476443767547607, "grad_norm": 0.008205249458551406, "loss": 0.00027695023658452555, "loss_kpt": 0.00027695023658452555, "acc_pose": 1.0, "time": 0.7598909091949463, "epoch": 3, "iter": 138, "memory": 10107, "step": 138}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.034710936546325684, "grad_norm": 0.008560804035514593, "loss": 0.0002747529439511709, "loss_kpt": 0.0002747529439511709, "acc_pose": 0.9423076923076923, "time": 0.7642617273330689, "epoch": 3, "iter": 143, "memory": 10107, "step": 143}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03435100078582764, "grad_norm": 0.008108133752830326, "loss": 0.0002633608930045739, "loss_kpt": 0.0002633608930045739, "acc_pose": 1.0, "time": 0.7642839145660401, "epoch": 3, "iter": 148, "memory": 10107, "step": 148}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.034349379539489744, "grad_norm": 0.008032905287109315, "loss": 0.00026103587937541305, "loss_kpt": 0.00026103587937541305, "acc_pose": 0.9807692307692307, "time": 0.7649582719802857, "epoch": 3, "iter": 153, "memory": 10107, "step": 153}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03455291748046875, "grad_norm": 0.00802791231777519, "loss": 0.0002578745677601546, "loss_kpt": 0.0002578745677601546, "acc_pose": 0.9807692307692307, "time": 0.765194411277771, "epoch": 3, "iter": 158, "memory": 10107, "step": 158}
+{"coco/AP": 0.9990007334066741, "coco/AP .5": 1.0, "coco/AP .75": 1.0, "coco/AP (M)": -1.0, "coco/AP (L)": 0.9990007334066741, "coco/AR": 0.999537037037037, "coco/AR .5": 1.0, "coco/AR .75": 1.0, "coco/AR (M)": -1.0, "coco/AR (L)": 0.999537037037037, "data_time": 0.06904185468500311, "time": 0.3899315747347745, "step": 3}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.037656850814819336, "grad_norm": 0.007862713877111674, "loss": 0.00024945040786406025, "loss_kpt": 0.00024945040786406025, "acc_pose": 1.0, "time": 0.7663229179382324, "epoch": 4, "iter": 167, "memory": 10107, "step": 167}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03722977638244629, "grad_norm": 0.007574681523256004, "loss": 0.00024375192951993087, "loss_kpt": 0.00024375192951993087, "acc_pose": 0.9423076923076923, "time": 0.7677838706970215, "epoch": 4, "iter": 172, "memory": 10107, "step": 172}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.353097767829895, "grad_norm": 0.006804578523151576, "loss": 0.00025081037034397014, "loss_kpt": 0.00025081037034397014, "acc_pose": 0.9487179487179487, "time": 1.084146456718445, "epoch": 4, "iter": 177, "memory": 10107, "step": 177}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.352255220413208, "grad_norm": 0.006327321524731815, "loss": 0.0002461272144864779, "loss_kpt": 0.0002461272144864779, "acc_pose": 0.9615384615384616, "time": 1.0175804281234742, "epoch": 4, "iter": 182, "memory": 10107, "step": 182}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.35108046531677245, "grad_norm": 0.0064000084483996035, "loss": 0.0002514433472242672, "loss_kpt": 0.0002514433472242672, "acc_pose": 1.0, "time": 1.012921724319458, "epoch": 4, "iter": 187, "memory": 10107, "step": 187}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.35111713886260987, "grad_norm": 0.0061109563102945685, "loss": 0.00024460054191877134, "loss_kpt": 0.00024460054191877134, "acc_pose": 0.9807692307692307, "time": 1.0119748735427856, "epoch": 4, "iter": 192, "memory": 10107, "step": 192}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.3514849328994751, "grad_norm": 0.006521861287765205, "loss": 0.0002486645584576763, "loss_kpt": 0.0002486645584576763, "acc_pose": 0.9807692307692307, "time": 1.0132512187957763, "epoch": 4, "iter": 197, "memory": 10107, "step": 197}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.351379828453064, "grad_norm": 0.0066711186710745095, "loss": 0.00025068359565921127, "loss_kpt": 0.00025068359565921127, "acc_pose": 1.0, "time": 1.0147013092041015, "epoch": 4, "iter": 202, "memory": 10107, "step": 202}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.3514827585220337, "grad_norm": 0.006590824937447905, "loss": 0.00024013744172407314, "loss_kpt": 0.00024013744172407314, "acc_pose": 1.0, "time": 1.0656786012649535, "epoch": 4, "iter": 207, "memory": 10107, "step": 207}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.346760892868042, "grad_norm": 0.006290690060704946, "loss": 0.00024445197224849837, "loss_kpt": 0.00024445197224849837, "acc_pose": 0.8974358974358974, "time": 1.0648930549621582, "epoch": 4, "iter": 212, "memory": 10107, "step": 212}
+{"coco/AP": 1.0, "coco/AP .5": 1.0, "coco/AP .75": 1.0, "coco/AP (M)": -1.0, "coco/AP (L)": 1.0, "coco/AR": 1.0, "coco/AR .5": 1.0, "coco/AR .75": 1.0, "coco/AR (M)": -1.0, "coco/AR (L)": 1.0, "data_time": 0.09153128103776412, "time": 0.42130654941905626, "step": 4}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.5934235525131225, "grad_norm": 0.006484583946876228, "loss": 0.00024959974354715086, "loss_kpt": 0.00024959974354715086, "acc_pose": 0.9487179487179487, "time": 1.3935117435455322, "epoch": 5, "iter": 221, "memory": 10107, "step": 221}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.292002477645874, "grad_norm": 0.00621036748867482, "loss": 0.0002368357115483377, "loss_kpt": 0.0002368357115483377, "acc_pose": 1.0, "time": 1.09070237159729, "epoch": 5, "iter": 226, "memory": 10107, "step": 226}
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/config.py b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/config.py
new file mode 100644
index 0000000000000000000000000000000000000000..9e289fd906cd45d7d1a44d3c7630f78898642d9f
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/config.py
@@ -0,0 +1,281 @@
+auto_scale_lr = None
+backend_args = dict(backend='local')
+codec = dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap')
+custom_hooks = [
+ dict(type='SyncBuffersHook'),
+]
+custom_imports = dict(
+ allow_failed_imports=False,
+ imports=[
+ 'mmpose.engine.optim_wrappers.layer_decay_optim_wrapper',
+ ])
+data_mode = 'topdown'
+data_root = '/root/miko/puni/train/GVHMR/processed_data/'
+dataset_type = 'CocoDataset'
+default_hooks = dict(
+ badcase=dict(
+ badcase_thr=5,
+ enable=False,
+ metric_type='loss',
+ out_dir='badcase',
+ type='BadCaseAnalysisHook'),
+ checkpoint=dict(
+ interval=1,
+ max_keep_ckpts=1,
+ rule='greater',
+ save_best='coco/AP',
+ save_optimizer=False,
+ type='CheckpointHook'),
+ logger=dict(interval=5, type='LoggerHook'),
+ param_scheduler=dict(type='ParamSchedulerHook'),
+ sampler_seed=dict(type='DistSamplerSeedHook'),
+ timer=dict(type='IterTimerHook'),
+ visualization=dict(
+ enable=True,
+ interval=1,
+ out_dir='vis_results',
+ type='PoseVisualizationHook'))
+default_scope = 'mmpose'
+env_cfg = dict(
+ cudnn_benchmark=False,
+ dist_cfg=dict(backend='nccl'),
+ mp_cfg=dict(mp_start_method='fork', opencv_num_threads=0))
+launcher = 'none'
+load_from = '/root/miko/puni/train/GVHMR/inputs/checkpoints/vitpose/td-hm_ViTPose-huge_8xb64-210e_coco-256x192-e32adcd4_20230314.pth'
+log_level = 'INFO'
+log_processor = dict(
+ by_epoch=True, num_digits=6, type='LogProcessor', window_size=50)
+metainfo = dict(from_file='configs/_base_/datasets/coco.py')
+model = dict(
+ backbone=dict(
+ arch='huge',
+ drop_path_rate=0.55,
+ img_size=(
+ 256,
+ 192,
+ ),
+ init_cfg=None,
+ out_type='featmap',
+ patch_cfg=dict(padding=2),
+ patch_size=16,
+ qkv_bias=True,
+ type='mmpretrain.VisionTransformer',
+ with_cls_token=False),
+ data_preprocessor=dict(
+ bgr_to_rgb=True,
+ mean=[
+ 123.675,
+ 116.28,
+ 103.53,
+ ],
+ std=[
+ 58.395,
+ 57.12,
+ 57.375,
+ ],
+ type='PoseDataPreprocessor'),
+ head=dict(
+ decoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ deconv_kernel_sizes=(
+ 4,
+ 4,
+ ),
+ deconv_out_channels=(
+ 256,
+ 256,
+ ),
+ in_channels=1280,
+ loss=dict(type='KeypointMSELoss', use_target_weight=True),
+ out_channels=17,
+ type='HeatmapHead'),
+ test_cfg=dict(flip_mode='heatmap', flip_test=True, shift_heatmap=False),
+ type='TopdownPoseEstimator')
+optim_wrapper = dict(
+ clip_grad=dict(max_norm=1.0, norm_type=2),
+ constructor='LayerDecayOptimWrapperConstructor',
+ optimizer=dict(
+ betas=(
+ 0.9,
+ 0.999,
+ ), lr=5e-05, type='AdamW', weight_decay=0.1),
+ paramwise_cfg=dict(
+ custom_keys=dict(
+ bias=dict(decay_multi=0.0),
+ norm=dict(decay_mult=0.0),
+ pos_embed=dict(decay_mult=0.0),
+ relative_position_bias_table=dict(decay_mult=0.0)),
+ layer_decay_rate=0.85,
+ num_layers=32))
+param_scheduler = [
+ dict(
+ begin=0,
+ by_epoch=True,
+ end=20,
+ gamma=0.1,
+ milestones=[
+ 14,
+ 18,
+ ],
+ type='MultiStepLR'),
+]
+resume = False
+test_cfg = dict()
+test_dataloader = dict(
+ batch_size=32,
+ dataset=dict(
+ ann_file='annotations/person_keypoints_val2017.json',
+ bbox_file=
+ 'data/coco/person_detection_results/COCO_val2017_detections_AP_H_56_person.json',
+ data_mode='topdown',
+ data_prefix=dict(img='val2017/'),
+ data_root='data/coco/',
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+test_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+train_cfg = dict(by_epoch=True, max_epochs=20, val_interval=1)
+train_dataloader = dict(
+ batch_size=4,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+ ],
+ type='CocoDataset'),
+ num_workers=0,
+ persistent_workers=False,
+ sampler=dict(shuffle=True, type='DefaultSampler'))
+train_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+]
+val_cfg = dict()
+val_dataloader = dict(
+ batch_size=4,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ bbox_file=None,
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=0,
+ persistent_workers=False,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+val_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+val_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+]
+vis_backends = [
+ dict(type='LocalVisBackend'),
+]
+visualizer = dict(
+ name='visualizer',
+ type='PoseLocalVisualizer',
+ vis_backends=[
+ dict(type='LocalVisBackend'),
+ ])
+work_dir = '../work_dirs/vitpose_finetune'
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/scalars.json b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/scalars.json
new file mode 100644
index 0000000000000000000000000000000000000000..623b8cd191ef955ecf19757bcd1956c2e0341e89
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/scalars.json
@@ -0,0 +1,46 @@
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03675370216369629, "grad_norm": 0.01963555794209242, "loss": 0.0010510455002076923, "loss_kpt": 0.0010510455002076923, "acc_pose": 0.8653846153846154, "time": 0.7642317295074463, "epoch": 1, "iter": 5, "memory": 10107, "step": 5}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.034762120246887206, "grad_norm": 0.021606831159442664, "loss": 0.000936518149683252, "loss_kpt": 0.000936518149683252, "acc_pose": 0.7307692307692307, "time": 0.7443006753921508, "epoch": 1, "iter": 10, "memory": 10107, "step": 10}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03494704564412435, "grad_norm": 0.01835411675274372, "loss": 0.0008069732614482442, "loss_kpt": 0.0008069732614482442, "acc_pose": 0.9230769230769231, "time": 0.7318146387736003, "epoch": 1, "iter": 15, "memory": 10107, "step": 15}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.037258028984069824, "grad_norm": 0.016740318736992776, "loss": 0.0007415686268359423, "loss_kpt": 0.0007415686268359423, "acc_pose": 0.8333333333333333, "time": 0.7299626350402832, "epoch": 1, "iter": 20, "memory": 10107, "step": 20}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.036465845108032226, "grad_norm": 0.015388216506689787, "loss": 0.0006977964425459504, "loss_kpt": 0.0006977964425459504, "acc_pose": 0.8782051282051282, "time": 0.8029078578948975, "epoch": 1, "iter": 25, "memory": 10107, "step": 25}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03572105566660563, "grad_norm": 0.015608035571252307, "loss": 0.0006719227142942449, "loss_kpt": 0.0006719227142942449, "acc_pose": 0.9423076923076923, "time": 0.7881508429845174, "epoch": 1, "iter": 30, "memory": 10107, "step": 30}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03504582132611956, "grad_norm": 0.015550296221460615, "loss": 0.0006389786140061915, "loss_kpt": 0.0006389786140061915, "acc_pose": 0.8076923076923077, "time": 0.7744760990142823, "epoch": 1, "iter": 35, "memory": 10107, "step": 35}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03410131335258484, "grad_norm": 0.015036797313950957, "loss": 0.0006063677203201224, "loss_kpt": 0.0006063677203201224, "acc_pose": 0.9551282051282051, "time": 0.7620397627353668, "epoch": 1, "iter": 40, "memory": 10107, "step": 40}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03331554200914171, "grad_norm": 0.014563350400163069, "loss": 0.0005804542594382333, "loss_kpt": 0.0005804542594382333, "acc_pose": 0.9807692307692307, "time": 0.7537709024217394, "epoch": 1, "iter": 45, "memory": 10107, "step": 45}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03300153255462646, "grad_norm": 0.014456901978701354, "loss": 0.0005644101899815724, "loss_kpt": 0.0005644101899815724, "acc_pose": 0.9358974358974359, "time": 0.748186445236206, "epoch": 1, "iter": 50, "memory": 10107, "step": 50}
+{"coco/AP": 0.9979022902290229, "coco/AP .5": 1.0, "coco/AP .75": 1.0, "coco/AP (M)": -1.0, "coco/AP (L)": 0.9979022902290229, "coco/AR": 0.9986111111111111, "coco/AR .5": 1.0, "coco/AR .75": 1.0, "coco/AR (M)": -1.0, "coco/AR (L)": 0.9986111111111111, "data_time": 0.07070658383546052, "time": 0.4435488559581615, "step": 1}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03274566650390625, "grad_norm": 0.012570046382024884, "loss": 0.0004693160206079483, "loss_kpt": 0.0004693160206079483, "acc_pose": 0.8653846153846154, "time": 0.7433967685699463, "epoch": 2, "iter": 59, "memory": 10107, "step": 59}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03281200885772705, "grad_norm": 0.011675749067217111, "loss": 0.0004420897128875367, "loss_kpt": 0.0004420897128875367, "acc_pose": 0.9615384615384616, "time": 0.7401487922668457, "epoch": 2, "iter": 64, "memory": 10107, "step": 64}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03139461517333984, "grad_norm": 0.011850018203258515, "loss": 0.0004179099708562717, "loss_kpt": 0.0004179099708562717, "acc_pose": 0.9615384615384616, "time": 0.7368682384490967, "epoch": 2, "iter": 69, "memory": 10107, "step": 69}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.031221256256103516, "grad_norm": 0.011669193208217622, "loss": 0.0004131932323798537, "loss_kpt": 0.0004131932323798537, "acc_pose": 0.9551282051282051, "time": 0.7539845371246338, "epoch": 2, "iter": 74, "memory": 10107, "step": 74}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03131303310394287, "grad_norm": 0.010922268312424422, "loss": 0.0003852099014329724, "loss_kpt": 0.0003852099014329724, "acc_pose": 0.9615384615384616, "time": 0.7522138786315918, "epoch": 2, "iter": 79, "memory": 10107, "step": 79}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.030978074073791505, "grad_norm": 0.01072505726478994, "loss": 0.00038021193788154053, "loss_kpt": 0.00038021193788154053, "acc_pose": 0.9230769230769231, "time": 0.7490486001968384, "epoch": 2, "iter": 84, "memory": 10107, "step": 84}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.0316017484664917, "grad_norm": 0.009690730567090213, "loss": 0.0003619221624103375, "loss_kpt": 0.0003619221624103375, "acc_pose": 1.0, "time": 0.7489235210418701, "epoch": 2, "iter": 89, "memory": 10107, "step": 89}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03172038078308106, "grad_norm": 0.009203163315542042, "loss": 0.00035649942088639364, "loss_kpt": 0.00035649942088639364, "acc_pose": 0.9230769230769231, "time": 0.7474814081192016, "epoch": 2, "iter": 94, "memory": 10107, "step": 94}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03224703788757324, "grad_norm": 0.008715138505212962, "loss": 0.00034365662053460256, "loss_kpt": 0.00034365662053460256, "acc_pose": 1.0, "time": 0.7465879440307617, "epoch": 2, "iter": 99, "memory": 10107, "step": 99}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03183490753173828, "grad_norm": 0.008379335082136095, "loss": 0.0003281010233331472, "loss_kpt": 0.0003281010233331472, "acc_pose": 0.9615384615384616, "time": 0.7453065204620362, "epoch": 2, "iter": 104, "memory": 10107, "step": 104}
+{"coco/AP": 0.9990007334066741, "coco/AP .5": 1.0, "coco/AP .75": 1.0, "coco/AP (M)": -1.0, "coco/AP (L)": 0.9990007334066741, "coco/AR": 0.999537037037037, "coco/AR .5": 1.0, "coco/AR .75": 1.0, "coco/AR (M)": -1.0, "coco/AR (L)": 0.999537037037037, "data_time": 0.5683283415707675, "time": 0.8939508178017356, "step": 2}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.031337161064147946, "grad_norm": 0.008334153429605067, "loss": 0.0003143988369265571, "loss_kpt": 0.0003143988369265571, "acc_pose": 1.0, "time": 0.7417430591583252, "epoch": 3, "iter": 113, "memory": 10107, "step": 113}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03212445259094238, "grad_norm": 0.007767286342568696, "loss": 0.0003093978948891163, "loss_kpt": 0.0003093978948891163, "acc_pose": 1.0, "time": 0.7415185737609863, "epoch": 3, "iter": 118, "memory": 10107, "step": 118}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.0322141695022583, "grad_norm": 0.00825631142128259, "loss": 0.00029793519875966014, "loss_kpt": 0.00029793519875966014, "acc_pose": 0.9038461538461539, "time": 0.6813721513748169, "epoch": 3, "iter": 123, "memory": 10107, "step": 123}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03296780586242676, "grad_norm": 0.008196818535216152, "loss": 0.0002875287723145448, "loss_kpt": 0.0002875287723145448, "acc_pose": 1.0, "time": 0.6871491479873657, "epoch": 3, "iter": 128, "memory": 10107, "step": 128}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03388114452362061, "grad_norm": 0.008475003554485739, "loss": 0.00028574865777045486, "loss_kpt": 0.00028574865777045486, "acc_pose": 0.9551282051282051, "time": 0.754159688949585, "epoch": 3, "iter": 133, "memory": 10107, "step": 133}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03476443767547607, "grad_norm": 0.008205249458551406, "loss": 0.00027695023658452555, "loss_kpt": 0.00027695023658452555, "acc_pose": 1.0, "time": 0.7598909091949463, "epoch": 3, "iter": 138, "memory": 10107, "step": 138}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.034710936546325684, "grad_norm": 0.008560804035514593, "loss": 0.0002747529439511709, "loss_kpt": 0.0002747529439511709, "acc_pose": 0.9423076923076923, "time": 0.7642617273330689, "epoch": 3, "iter": 143, "memory": 10107, "step": 143}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03435100078582764, "grad_norm": 0.008108133752830326, "loss": 0.0002633608930045739, "loss_kpt": 0.0002633608930045739, "acc_pose": 1.0, "time": 0.7642839145660401, "epoch": 3, "iter": 148, "memory": 10107, "step": 148}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.034349379539489744, "grad_norm": 0.008032905287109315, "loss": 0.00026103587937541305, "loss_kpt": 0.00026103587937541305, "acc_pose": 0.9807692307692307, "time": 0.7649582719802857, "epoch": 3, "iter": 153, "memory": 10107, "step": 153}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03455291748046875, "grad_norm": 0.00802791231777519, "loss": 0.0002578745677601546, "loss_kpt": 0.0002578745677601546, "acc_pose": 0.9807692307692307, "time": 0.765194411277771, "epoch": 3, "iter": 158, "memory": 10107, "step": 158}
+{"coco/AP": 0.9990007334066741, "coco/AP .5": 1.0, "coco/AP .75": 1.0, "coco/AP (M)": -1.0, "coco/AP (L)": 0.9990007334066741, "coco/AR": 0.999537037037037, "coco/AR .5": 1.0, "coco/AR .75": 1.0, "coco/AR (M)": -1.0, "coco/AR (L)": 0.999537037037037, "data_time": 0.06904185468500311, "time": 0.3899315747347745, "step": 3}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.037656850814819336, "grad_norm": 0.007862713877111674, "loss": 0.00024945040786406025, "loss_kpt": 0.00024945040786406025, "acc_pose": 1.0, "time": 0.7663229179382324, "epoch": 4, "iter": 167, "memory": 10107, "step": 167}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.03722977638244629, "grad_norm": 0.007574681523256004, "loss": 0.00024375192951993087, "loss_kpt": 0.00024375192951993087, "acc_pose": 0.9423076923076923, "time": 0.7677838706970215, "epoch": 4, "iter": 172, "memory": 10107, "step": 172}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.353097767829895, "grad_norm": 0.006804578523151576, "loss": 0.00025081037034397014, "loss_kpt": 0.00025081037034397014, "acc_pose": 0.9487179487179487, "time": 1.084146456718445, "epoch": 4, "iter": 177, "memory": 10107, "step": 177}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.352255220413208, "grad_norm": 0.006327321524731815, "loss": 0.0002461272144864779, "loss_kpt": 0.0002461272144864779, "acc_pose": 0.9615384615384616, "time": 1.0175804281234742, "epoch": 4, "iter": 182, "memory": 10107, "step": 182}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.35108046531677245, "grad_norm": 0.0064000084483996035, "loss": 0.0002514433472242672, "loss_kpt": 0.0002514433472242672, "acc_pose": 1.0, "time": 1.012921724319458, "epoch": 4, "iter": 187, "memory": 10107, "step": 187}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.35111713886260987, "grad_norm": 0.0061109563102945685, "loss": 0.00024460054191877134, "loss_kpt": 0.00024460054191877134, "acc_pose": 0.9807692307692307, "time": 1.0119748735427856, "epoch": 4, "iter": 192, "memory": 10107, "step": 192}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.3514849328994751, "grad_norm": 0.006521861287765205, "loss": 0.0002486645584576763, "loss_kpt": 0.0002486645584576763, "acc_pose": 0.9807692307692307, "time": 1.0132512187957763, "epoch": 4, "iter": 197, "memory": 10107, "step": 197}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.351379828453064, "grad_norm": 0.0066711186710745095, "loss": 0.00025068359565921127, "loss_kpt": 0.00025068359565921127, "acc_pose": 1.0, "time": 1.0147013092041015, "epoch": 4, "iter": 202, "memory": 10107, "step": 202}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.3514827585220337, "grad_norm": 0.006590824937447905, "loss": 0.00024013744172407314, "loss_kpt": 0.00024013744172407314, "acc_pose": 1.0, "time": 1.0656786012649535, "epoch": 4, "iter": 207, "memory": 10107, "step": 207}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.346760892868042, "grad_norm": 0.006290690060704946, "loss": 0.00024445197224849837, "loss_kpt": 0.00024445197224849837, "acc_pose": 0.8974358974358974, "time": 1.0648930549621582, "epoch": 4, "iter": 212, "memory": 10107, "step": 212}
+{"coco/AP": 1.0, "coco/AP .5": 1.0, "coco/AP .75": 1.0, "coco/AP (M)": -1.0, "coco/AP (L)": 1.0, "coco/AR": 1.0, "coco/AR .5": 1.0, "coco/AR .75": 1.0, "coco/AR (M)": -1.0, "coco/AR (L)": 1.0, "data_time": 0.09153128103776412, "time": 0.42130654941905626, "step": 4}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.5934235525131225, "grad_norm": 0.006484583946876228, "loss": 0.00024959974354715086, "loss_kpt": 0.00024959974354715086, "acc_pose": 0.9487179487179487, "time": 1.3935117435455322, "epoch": 5, "iter": 221, "memory": 10107, "step": 221}
+{"base_lr": 5e-05, "lr": 2.3431201180750451e-07, "data_time": 0.292002477645874, "grad_norm": 0.00621036748867482, "loss": 0.0002368357115483377, "loss_kpt": 0.0002368357115483377, "acc_pose": 1.0, "time": 1.09070237159729, "epoch": 5, "iter": 226, "memory": 10107, "step": 226}
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_100.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_100.png
new file mode 100644
index 0000000000000000000000000000000000000000..1c53dfb6a38c1bb00834c92040bd8ed798b0559d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_100.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:97f377d38bcc9b8551998bf94bf0fc1d2aa2b6dfb29b499d9051d0d28fede33c
+size 707486
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_101.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_101.png
new file mode 100644
index 0000000000000000000000000000000000000000..e0e073e8c1e6df363b9b09f4b913037bd8dc346f
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_101.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0f7158703053c2872d5f304b249d28ed6acef9768da6ec940b32daf4078ba1db
+size 713020
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_102.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_102.png
new file mode 100644
index 0000000000000000000000000000000000000000..e1444ae2b957bd74c8d4a8f862a501ed4d5f34b3
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_102.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:27cf78fdd3f6185cc35a43cceedd770780049afbddef3e16c732bcff6a260e53
+size 709569
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_103.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_103.png
new file mode 100644
index 0000000000000000000000000000000000000000..191824909fcb5aa523b3eee1a40ac3d1ea80e00f
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_103.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a80be0fe65e67d56ac7e06554f6ce38a83c78d179432719d7f2dfddf30827fc3
+size 865824
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_104.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_104.png
new file mode 100644
index 0000000000000000000000000000000000000000..616787486ff1f6df17d8db4e98142fd10f2f496e
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_104.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:19f360fddf945d58d764b525e6151aefd0fc4a899d60022658e363407ed35d17
+size 876518
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_105.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_105.png
new file mode 100644
index 0000000000000000000000000000000000000000..ac93ffee7e697d9c87f65948fbbc486d9787cfd0
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_105.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:517fd7e22df75555670df6f5e53622aace127857377fb2843e5ab13db213ef12
+size 869820
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_106.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_106.png
new file mode 100644
index 0000000000000000000000000000000000000000..77370cf151e824dfd8618e25110500132225235d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_106.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:afb064e43cefb06335e619023753f6ae944790faa9b4e5a0f055fe3f180d15ca
+size 868904
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_107.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_107.png
new file mode 100644
index 0000000000000000000000000000000000000000..caa571c6e766b5a93b2ce15dae6730b9dd8eeb1c
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_107.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:53d15d872bb05fa7bc5021662d2113defdce75240d90635967b47fd2293172b0
+size 871059
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_108.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_108.png
new file mode 100644
index 0000000000000000000000000000000000000000..7277e367cdb11c59c7a44962a2b1ad940d17e7b5
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_108.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1ff639100d979e97a95e779c43eb7d6d2a2ce1eb64434f63e441382b24af1a06
+size 724466
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_109.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_109.png
new file mode 100644
index 0000000000000000000000000000000000000000..dba7cdfe9e9485f377759d79feac870323c14aeb
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_109.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1b15322b210f70cafcc9a92c527ca69a482ccadb1508f6879dc767c4acf98203
+size 716342
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_110.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_110.png
new file mode 100644
index 0000000000000000000000000000000000000000..d3e64996c637b9186417caed25d56de72716e402
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_110.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c4fd2e39a6923516d7fa57e2f68250cbf380e75b5faf0e5e8eb68c1ed944ee39
+size 712886
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_111.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_111.png
new file mode 100644
index 0000000000000000000000000000000000000000..1b4ebb50a53ad63c2b7493e22057744a457cafc6
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_111.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:aa97903df9f84077d9cdc43ea7f58f8711ef7d9446c436f198aae34d2cd3bca7
+size 699806
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_112.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_112.png
new file mode 100644
index 0000000000000000000000000000000000000000..4c716d0c350bb8a1421af164661dd05193172ef4
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_112.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5532555bd39ec431a4a531f6545b213e2a28cce690988df257afad3f8156f322
+size 723709
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_113.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_113.png
new file mode 100644
index 0000000000000000000000000000000000000000..4f217dbfa9314b63c4cc1739841e135065c815d9
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_113.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c998e22d1007b352a93cd88c20b015fc8e145a21a6077f56717d7ee09b8b2a7a
+size 731839
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_114.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_114.png
new file mode 100644
index 0000000000000000000000000000000000000000..f327ab3d973ea291f5dc84918c97867fe4d17b73
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_114.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d9d35477bc44b1982f57b771ff3cf5d35e0a59ffae0b2ae80cad5e5d55d9fc42
+size 693018
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_115.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_115.png
new file mode 100644
index 0000000000000000000000000000000000000000..d40836f1a77e547c46f820dc6e17ffee5167a739
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_115.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4b08379482de3965598e9c27c82bdaa4b8d430a56e6382e5cd3b686adb1ad617
+size 678809
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_116.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_116.png
new file mode 100644
index 0000000000000000000000000000000000000000..274a1aef24ac2e826417c138d8e501ebb67bdd89
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_116.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3d975505277f0e917aec3fdf2d43e832a99ec041f483e997b77b5dab65c577aa
+size 705484
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_117.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_117.png
new file mode 100644
index 0000000000000000000000000000000000000000..b65072921acb5b197c9db43e8b4c923d7ed036d2
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_117.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0c0bc05c4a3a3d5edd555a1cf0b8ca1d9373fb28dd9b0770d01e607e71a6680c
+size 674903
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_118.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_118.png
new file mode 100644
index 0000000000000000000000000000000000000000..a679ef8858ef73a1db3a6236213553f21e28df19
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_118.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:595f927a5bbfcd9b65482450984828ea0c4ed993cf5ecf5463ffd6b805746547
+size 795439
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_119.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_119.png
new file mode 100644
index 0000000000000000000000000000000000000000..857fd81573e0ee2d86aebbf123e3c41d82792c27
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_119.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a25dec887a6c3bd6f84c8afd4402b742aeea1d54ed22213c5634442165185914
+size 693070
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_120.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_120.png
new file mode 100644
index 0000000000000000000000000000000000000000..ff8b8d3e6c5313cd9b0e40ae50501a949cd088b9
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_120.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7b739aca85a14c167c656d8938c3041de71c004b1d1d3c6b18917904c05c70ea
+size 665658
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_121.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_121.png
new file mode 100644
index 0000000000000000000000000000000000000000..bb418a480d1d890e65409b3a3a85f9b0167017e3
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_121.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7952f4d7e8946b5f6ea90e596dcb6ddd99bf4f6cd407b8dddb1afd9e01a81c53
+size 703397
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_122.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_122.png
new file mode 100644
index 0000000000000000000000000000000000000000..79b9d0ec547c4bc92c3701fc3f2544f40ec12307
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_122.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7e883d0938e39067dd758b9b89aa3555bb4384590df7e4e27fd9f8491b2a853d
+size 718513
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_123.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_123.png
new file mode 100644
index 0000000000000000000000000000000000000000..24adfdbb3f4847f45afb54039881c1b441089dcc
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_123.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1991ec37b78adc1976573207abf10f7f85e53948d824e22decdfc5c7b508a92e
+size 706325
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_124.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_124.png
new file mode 100644
index 0000000000000000000000000000000000000000..87d2c79e23d90e0a65f7c219710e2abbfb4c5251
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_124.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cf7a738f1a1a8c49a63d78879224b8e2dbb72095c6bce162780a2ff05477df8a
+size 661409
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_125.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_125.png
new file mode 100644
index 0000000000000000000000000000000000000000..81d83b9ca339b1b452c56a65acf776612b5935f1
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_125.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:43f088a88bb4b96d8041ef8124dc2e343007d3abd20d2cf6e9ed55c7db2c8ed7
+size 702252
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_126.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_126.png
new file mode 100644
index 0000000000000000000000000000000000000000..d4f4c6cf4a018fbccecf63f48403fc71285e1078
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_126.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b66a347dee22d639e7be91c0e81ac2a4f48192a5f326c348ac47676ecce19236
+size 727339
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_127.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_127.png
new file mode 100644
index 0000000000000000000000000000000000000000..259c749607104b0e9f8df323609b29d71f455571
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_127.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c836b74d86dd3e8e90c78c08b0e22776c2b4ea54ffb16d6715c3787c2d6fe2bd
+size 749851
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_128.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_128.png
new file mode 100644
index 0000000000000000000000000000000000000000..b55be211ded55490b76b3b3285b2bbd2d0031935
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_128.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9f97fa655a56a09c125ecf96d715d25042e20022970c24a8021b671b7bf0156c
+size 724668
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_129.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_129.png
new file mode 100644
index 0000000000000000000000000000000000000000..27c65a59a4ed52f24483ff97c5f9f66b15584cf0
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_129.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c96dba0196a25885e21de9c9032976e36ff7a5e9b634bf4f8607c5abba14422e
+size 742598
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_130.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_130.png
new file mode 100644
index 0000000000000000000000000000000000000000..9249e72ad529bb5b57cabccbd088b48bf80e6fa9
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_130.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8e64448c13259f44a0646e567865a613c794a72b8dc0e43646fae1e9675fc2a0
+size 761730
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_131.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_131.png
new file mode 100644
index 0000000000000000000000000000000000000000..fa0de9d7dfc82e0fc7940d9545b2d2d3a7f49fe6
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_131.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:101c15cedf0b551f205c0991da1505a1f4c05d09757b3df84155df3efb89de7a
+size 740625
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_132.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_132.png
new file mode 100644
index 0000000000000000000000000000000000000000..159f22dcd7d7574c2bd13d568bd0da37888836d2
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_132.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f1dd28c0e10fa0d4999139ce704b8aa9a479e5e4614e146fb7754f82a70607d9
+size 739210
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_133.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_133.png
new file mode 100644
index 0000000000000000000000000000000000000000..00e3bcdbc1e733823886b65ee22ee1a7c08ed246
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_133.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:fc751e1574d1e294fec8a5e9eaa626132322b0772cf0cf1bb600697b82d446fe
+size 693297
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_134.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_134.png
new file mode 100644
index 0000000000000000000000000000000000000000..7aebaf116d6fac2a61ae7326ea57316291a2d8b0
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_134.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:dbda2a89d80fcdb83cd5108adf824215573567145ef84b1b36fe901d6357801f
+size 703527
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_135.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_135.png
new file mode 100644
index 0000000000000000000000000000000000000000..21e6b1cb15f28d6867af268077851b367cdd7e3c
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_135.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8ffd92d03e3cc06767ddd9f52724461f5912f7caf71112ccc63087f7d8787483
+size 698981
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_136.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_136.png
new file mode 100644
index 0000000000000000000000000000000000000000..9118883b9a8473c6eef61f05ce78f505a82d75ec
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_136.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:45057143409a11e4a8a5738571301663cb7cf7bba76fd3bb9cc1e68a17e2853d
+size 699877
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_137.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_137.png
new file mode 100644
index 0000000000000000000000000000000000000000..7fb7aa177f2c1d5abcab7604f74d453f839006f6
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_137.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0efb4aa064f9998012b8c1dec6e53a3253a85091295e1dbbd43c48991cacb75e
+size 725629
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_138.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_138.png
new file mode 100644
index 0000000000000000000000000000000000000000..b2929c954a04b3c7618d66af94feb1e87eb27197
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_138.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ff26f41b4dd0b5bc85ccb193822a8c2e4a087b4621e2b04ac781c63fe70675d5
+size 736300
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_139.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_139.png
new file mode 100644
index 0000000000000000000000000000000000000000..06551d3bb1d17bcc94e41dd5f03f0f48f7163f5e
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_139.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a0eaed7992522d1a9c23108cf10d1f974538363e0f5b0aa933886b80ff0696b9
+size 725720
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_140.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_140.png
new file mode 100644
index 0000000000000000000000000000000000000000..9fb1c09e42c21b04b6e913f6a640eb5446355f37
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_140.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d7361177515070fdb55b9d4af8dec89de7ff4935856459581f073fd650c4bc26
+size 722847
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_141.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_141.png
new file mode 100644
index 0000000000000000000000000000000000000000..ccf3666f95b8ad8d398368d6b11effb416013f8c
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_141.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d38e284a35bcdc929083293b18334d699385d9f91939917302161ee34601be19
+size 720606
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_142.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_142.png
new file mode 100644
index 0000000000000000000000000000000000000000..7826350b19f58bacb32a5fefc0c4bffc5f469b7a
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_142.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e009e332abbe74a41b7b31d11f629b865f34c5e2a04c5241e197eb0cf5134654
+size 733579
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_143.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_143.png
new file mode 100644
index 0000000000000000000000000000000000000000..6ce0dd12541aa0d671ff5a0400d921e7e1bbaee6
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_143.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4484cd18ca9c228ac4e6261c14d42ed7701f329b01d7f68a5b1cae37073ad857
+size 705938
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_144.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_144.png
new file mode 100644
index 0000000000000000000000000000000000000000..bde419cc84db1dafcbe26a708e82ce4f39d84cd9
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_144.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b04dcefbac14c0ac4171ea9fb65ad9ddff6597beb2e1c6f4f6d3ed8c2fb56e49
+size 720701
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_145.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_145.png
new file mode 100644
index 0000000000000000000000000000000000000000..6498ac6b61219eeedd5a621c14233941b748b9e5
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_145.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:16547a89e2f85e03c15d6729526fa951148ae61284f711dfb77d519764a055ae
+size 713081
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_146.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_146.png
new file mode 100644
index 0000000000000000000000000000000000000000..3e7e0fd312e9e4e763555df3d6a771a06be47142
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_146.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:bdaae9f88b7a0fae1f354b95983e3871f06c7377c29e9744a91a2330a2885abd
+size 725057
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_147.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_147.png
new file mode 100644
index 0000000000000000000000000000000000000000..9ca8ca8c6cae484b5757f13f891ec7d896590a56
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_147.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:87a73bf8b22fff465904d5c1e0eae8a7951af8982a388fb2f20fce266e56a824
+size 731587
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_148.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_148.png
new file mode 100644
index 0000000000000000000000000000000000000000..b4bc92f3a8110cc5fd48b98ffad964b8c116583d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_148.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:91774aa9fc6d1119988292507c9fb932231cc749c744e73523e0745e197e9850
+size 782395
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_149.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_149.png
new file mode 100644
index 0000000000000000000000000000000000000000..db8ad82c29d090aff19653342dc9ef3473641563
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_149.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b465ade9134c2c2f2a115b1e5a64934781d1e49cf0794fe3d001147ec306a115
+size 805473
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_150.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_150.png
new file mode 100644
index 0000000000000000000000000000000000000000..38a3b35aac38de4702945a67b4b253322c9ee537
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_150.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:178a57008e75321b7cb78bd91e8685a442b5c602e2d1c5cbcf4d575297570860
+size 774065
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_151.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_151.png
new file mode 100644
index 0000000000000000000000000000000000000000..f1dc5feca04cebf163a91f1d648f8532d79f8943
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_151.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:57ed624e5928c4739368cc2e5396ce6846b78b9987846cc8de47d6c11e4d2e3b
+size 800430
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_152.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_152.png
new file mode 100644
index 0000000000000000000000000000000000000000..e92293d8d71e26534de530c5f75430b35ebd6292
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_152.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0dda82cf125db9028a3b6df6325887df4cb247cb84a893bcd637ae585d39fdc7
+size 711079
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_153.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_153.png
new file mode 100644
index 0000000000000000000000000000000000000000..f9096f0f4ece32bf91d4a2f1cdb9209dfe9e82c3
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_153.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3c37efe513646d5f91b2bfb517c9d3b0c00a38dfe9d5ab2a593e3e2fa3291c88
+size 702637
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_154.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_154.png
new file mode 100644
index 0000000000000000000000000000000000000000..c1184b754ae588e28737147a48b973075b27d40f
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_154.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3f64debefed2b0a466b98fb1d1c54807e6b14a999de3f12a4a99935fa0fcf555
+size 708110
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_155.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_155.png
new file mode 100644
index 0000000000000000000000000000000000000000..aa8b17e20a8a62d3756b63403740f0e89b27c809
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_155.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:61a9e4f9c8d0d68de0dd67c8f840c5b2c9bc7b2d88a78bdd81c38ba471de8fb8
+size 713263
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_156.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_156.png
new file mode 100644
index 0000000000000000000000000000000000000000..430eb22d9307214fafc4ec9a15772d66ccc31b62
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_156.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:48eb5e4898801e2b6d71af4b95f42128f18fcff1f3253047477578372629df7d
+size 709046
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_157.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_157.png
new file mode 100644
index 0000000000000000000000000000000000000000..2340d126abb07165f3b14776d4c51a3de4d87f0d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_157.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a1417b77c931a9d667643aa9fb692181143fd5891a320576e5e0ea1e2bafc7cb
+size 865351
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_158.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_158.png
new file mode 100644
index 0000000000000000000000000000000000000000..057bf27027f4054389f97d9e860801954c877062
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_158.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:fad04dc3f822bb11e3159ebb02d40994325df16ee5be325741c5d18c670f1a50
+size 875895
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_159.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_159.png
new file mode 100644
index 0000000000000000000000000000000000000000..4d2bcf39fc75ab50420f2ad2c3321b682dbe0b5b
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_159.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:39470e2c759801cef1022635f624834607aa5c7feddb8cdb287f6a0940e74e80
+size 869564
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_160.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_160.png
new file mode 100644
index 0000000000000000000000000000000000000000..00d59af4112f66fe4917d279a7bfd47bf8e19ce5
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_160.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:169ebd37a701da93e9fae90e04f503d7952e29bad20c555b177890e872f15530
+size 868355
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_161.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_161.png
new file mode 100644
index 0000000000000000000000000000000000000000..d2e787afefded7664d46a0e91105a242d882c2c5
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_161.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ae7f8b0954b5c044d1d0d1fc6a1daba2574d1e9b2b71caa89e6d1359b3550239
+size 870716
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_162.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_162.png
new file mode 100644
index 0000000000000000000000000000000000000000..66172343c793a0b5737b695ef9ae817bcd73afe2
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_162.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:459dbf37c03bdef2c1d5f043160ee3894a4953bdf1190edde3c5a2889824dae3
+size 724408
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_163.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_163.png
new file mode 100644
index 0000000000000000000000000000000000000000..ce417ce59f69c141fa81bf20fd0f044f36341bf2
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_163.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:537610b1ad085a1e3496c17b2996421388a7eee78b7c6d9a0cdc3c8ad270ef46
+size 716554
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_164.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_164.png
new file mode 100644
index 0000000000000000000000000000000000000000..5483d7821451aa7cb3fb7f5af3dde0c4dfbd8da4
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_164.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:de959981bfdf0c397e1f90c22428bf3434578150d315369c7a44d7d2eacda5ed
+size 712962
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_165.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_165.png
new file mode 100644
index 0000000000000000000000000000000000000000..68368c681fe1c6988531f32a2d214a82eaeb1e18
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_165.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:740824394ce128ed4834b04c2e37713ed560033f22701f7646711553398ee868
+size 699835
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_166.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_166.png
new file mode 100644
index 0000000000000000000000000000000000000000..d0c043ab5493b802c47a3fc423b0081a44779524
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_166.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5295ddf35f001f54511a7152fa56a1a1fe08ebec8c2d897046fc0f3fc2c089d1
+size 723892
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_167.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_167.png
new file mode 100644
index 0000000000000000000000000000000000000000..c1be4debfb9b6ffffb04e799f7a62a6435bd1e20
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_167.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b64a7a2c37f3aa8d33c7ab06e3eeb178430514a78ae30eb1be9a8688a1b6905d
+size 731918
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_168.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_168.png
new file mode 100644
index 0000000000000000000000000000000000000000..947d60d893d51d890b1971a02b29bf1ffbdade0d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_168.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:90b88ec37faae342990ada88e84bf7cd899275de88974c94355920a84bfa5b2e
+size 692914
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_169.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_169.png
new file mode 100644
index 0000000000000000000000000000000000000000..96e8b21be16900dd4127390be2440dfb64803299
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_169.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:dc67470e49d549f450ee48332e6e144584e16b3a68419a5646c4c3b43731940a
+size 678909
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_170.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_170.png
new file mode 100644
index 0000000000000000000000000000000000000000..cb1739d714c8fd773b1df4141bf528733baba608
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_170.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:082664f8969ccb14cc00de8897f7e665122221fb216d284dddbb3ccc267e3970
+size 705494
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_171.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_171.png
new file mode 100644
index 0000000000000000000000000000000000000000..ad5655158657eb4b3f7d5555143e5fded2664a1b
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_171.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e3d1c13bf90426ea7295fb55d85b033c104711e0ab1989ea2a7622bd6799a855
+size 674996
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_172.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_172.png
new file mode 100644
index 0000000000000000000000000000000000000000..21e9977f53c9aaa227411e0be3fa61a44a09c3fc
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_172.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:bcd99a80c1660d6942e707ec211c4651591100a6aec473ff74d419ac24c58c1b
+size 795430
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_173.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_173.png
new file mode 100644
index 0000000000000000000000000000000000000000..9b16586a1867a6d72712e4267f7e3991d192863d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_173.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cd1646dd58287d7d825a77bd15eeb038f6207935ea9c732f0ea81d53351740ce
+size 693125
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_174.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_174.png
new file mode 100644
index 0000000000000000000000000000000000000000..666614e6800eb9c92148a33111376e214625e937
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_174.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:742b68803edcc75bbbfcf7e5a84516b2f0bb33ea5b3f85e104f8861431a7effb
+size 666294
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_175.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_175.png
new file mode 100644
index 0000000000000000000000000000000000000000..0b5757903604136007d743ac447862e56fb1d675
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_175.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b68e4f9f3948936c5c94c0103b14bb1d93b96609af84a44ed993eb8e8d0cd80e
+size 703290
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_176.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_176.png
new file mode 100644
index 0000000000000000000000000000000000000000..b11b39595efbafd1b8117693590f7f7506391eb3
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_176.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3003a02e9726ba5a535ada7eb5b57252019c7a3afb837b94f78aa74775791e80
+size 718507
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_177.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_177.png
new file mode 100644
index 0000000000000000000000000000000000000000..bae5a6204298eb34f33236ebcf6fce9a36c81515
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_177.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7d65b7bbc5bb1c206410bfe2fcc3a9805c0aaf2e4b54fd6f4f31819556ce0755
+size 706181
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_178.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_178.png
new file mode 100644
index 0000000000000000000000000000000000000000..861596973ff5022a34aaf201671d2550e6cbabcc
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_178.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:69fcffff37f42e632fac7ccbfbfe0df519658bae5f2c73df040ddfd5e686e015
+size 661282
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_179.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_179.png
new file mode 100644
index 0000000000000000000000000000000000000000..eaae6455615550849dee24c5a50e5858d0bf2a29
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_179.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c1b966c1c7a051ce8d6242f64189400a2ac66fdf12b1fb65958f42db5a566282
+size 702179
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_180.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_180.png
new file mode 100644
index 0000000000000000000000000000000000000000..d64c90f36aac67dcfaf1e4143b6b86c6d5909c57
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_180.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:29be3bd14ca3285872de1cdda6d3d0926af91020aa7c3750502df01714c55b18
+size 727397
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_181.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_181.png
new file mode 100644
index 0000000000000000000000000000000000000000..9fcd6735673a99108600fb4f4001239be26c1aa6
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_181.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8c61733347376904f8be19764763cc53ac9fd9f934865cd8f4a102bc71476962
+size 749861
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_182.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_182.png
new file mode 100644
index 0000000000000000000000000000000000000000..b198fab97c9fa0903604dd1f8ca02096f76449af
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_182.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:20611579822f5b039e78fef8b22a5b978c1be26fb697b3347748f97e6c31d7b9
+size 724675
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_183.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_183.png
new file mode 100644
index 0000000000000000000000000000000000000000..0d65edc52ce5767f80646a5f2a1ecd07a628a569
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_183.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0d75a55e4776fb6642173744d467706eb909e168714a4b26338c27d9dede1c8e
+size 742730
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_184.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_184.png
new file mode 100644
index 0000000000000000000000000000000000000000..2dca82fedfda5e41dd964c33cb32ab3ce1ffa289
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_184.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:aad625bd3f706004d7710efde634ef3a9350c4b57b462ab0d09bd07423e92859
+size 761748
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_185.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_185.png
new file mode 100644
index 0000000000000000000000000000000000000000..24c6a99009f0ce5fbce5031321e77908cfc79d86
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_185.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:84e72e9ca0a6e5bd1fcbd4be1c0f4c6661edf7cae43b86cd43e7a1532e29a5e9
+size 740706
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_186.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_186.png
new file mode 100644
index 0000000000000000000000000000000000000000..3f2358ac5a7015ce01f84aca7742820b6458c4a0
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_186.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d9f389cc6156e6908b4461b3ba2893eca668fb31ef245a1a855116f7f6c7198f
+size 739168
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_187.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_187.png
new file mode 100644
index 0000000000000000000000000000000000000000..27ebb149f914917f02308b272df37d5b2c5e62ea
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_187.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2cfe137cc444fb55b243d45b1d19ac97fba2eb4f913492cb752c57aac1dfcb80
+size 693415
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_188.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_188.png
new file mode 100644
index 0000000000000000000000000000000000000000..519db1bd4e83f19d32b75c97865f91880949b66e
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_188.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:188cb01a6d19cb45daafcf23c1934ce3af1f24cde3c90961abfe69cbcff24ef9
+size 703525
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_189.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_189.png
new file mode 100644
index 0000000000000000000000000000000000000000..e4a5e34cc78740a65b80d187b396fdb81f361fb9
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_189.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b359eb9682b15250f2983c96722c192d6049322f3909ef8046d194b63b2b5d54
+size 699071
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_190.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_190.png
new file mode 100644
index 0000000000000000000000000000000000000000..b1ca2a80ffa8e226af60b836a70e217f06f37305
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_190.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:23508fc52865a28d820e89cd4ac42507b0320419500eca19a6e8af1c8afb8ace
+size 699416
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_191.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_191.png
new file mode 100644
index 0000000000000000000000000000000000000000..46b9d40cf1bdcce4dbd082be3e67668360f677c9
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_191.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:23840a593a9be9faf14d1cdcd917caed3919c6f87f676a5376585e2d8b5f6350
+size 725650
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_192.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_192.png
new file mode 100644
index 0000000000000000000000000000000000000000..54f4eea67aed0a4064178b84b93a6c6714f7ef4b
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_192.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5b7bbe793010b24e0ba3bf77e1ed20ee41d86555c996a32c6f7463982b97e2b8
+size 736372
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_193.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_193.png
new file mode 100644
index 0000000000000000000000000000000000000000..a65d8fc439de6ab2b7199d1bdbcd2a318cedacad
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_193.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:43b0bef8de39081a50086a84f7676fed11c7dcf220486f45918ad5b7092f9f1a
+size 725762
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_194.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_194.png
new file mode 100644
index 0000000000000000000000000000000000000000..a550df0fad74a902e36f2fbd877b477ec156eda6
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_194.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:929e524072f24e8999edf026e31a48fdccfa1fefddf12c88a4aecf01d9d6d58d
+size 722902
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_195.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_195.png
new file mode 100644
index 0000000000000000000000000000000000000000..cde7bdfd44df48d0a2b9f00495f268b9b69e9736
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_195.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:654f20462d14384c438a968382ea5a521903a46db26e38b140279079ea051a34
+size 720717
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_196.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_196.png
new file mode 100644
index 0000000000000000000000000000000000000000..48851522ee79e8eb56ee4100a568be0980809e3b
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_196.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cdd132497ea29d6698575347cc79554a3b5553f13dd1aa7d596f04c44140d0bc
+size 733523
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_197.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_197.png
new file mode 100644
index 0000000000000000000000000000000000000000..66b5e314966a883da6119c6b181f149dc13ed978
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_197.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:515a7d4592899bc3c4a90099884e26c5a4fafa28437c9d774b159d1af65dabdf
+size 706009
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_198.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_198.png
new file mode 100644
index 0000000000000000000000000000000000000000..6e41f7d1293566f92083485132c4a70b4a98eed6
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_198.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3211b3325c70ada65667c23f196a031478f9716e59e09ab671cd3f33e6bf33a1
+size 720752
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_199.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_199.png
new file mode 100644
index 0000000000000000000000000000000000000000..e493959f04c7fb34108be82dafcf263321a1a175
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_199.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:fce464ba525ca04b119921a6ce8ca884f583189fed170894800d125070734282
+size 713134
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_200.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_200.png
new file mode 100644
index 0000000000000000000000000000000000000000..430fb0dce83e10b21358e7475f2ad5132a3864bb
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_200.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2a8467f6da804e7f0ffb9f464e81f09fe7217ff4d159e338aa5e149caa717a82
+size 725068
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_201.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_201.png
new file mode 100644
index 0000000000000000000000000000000000000000..b85aac4ebad8e62195bc72078601417bfbea6bd6
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_201.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:11dfdfb62872d53fdcaaaa5f778042ea1dd72f53f9c90959892ad8f11ddc4068
+size 731628
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_202.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_202.png
new file mode 100644
index 0000000000000000000000000000000000000000..b1f60eaaedd94f020bf80c6088f3279b6338cce0
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_202.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cf06b46d08cf8b715d4aedc5debc68f0bf119f69d974ee656e1bd2a9640d6201
+size 782623
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_203.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_203.png
new file mode 100644
index 0000000000000000000000000000000000000000..2e8b0fdb35d07783fe6b2a5f927d877cbfc7d054
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_203.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ea2d8f5e53f439c8c2c4d4d2b9daebbea2ecd6aef120e27651c77a94d53e6251
+size 805462
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_204.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_204.png
new file mode 100644
index 0000000000000000000000000000000000000000..f2c9d0e4f8c82c970feaea83b729843c2945836a
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_204.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0f5cb7334b7c636c05f02b864eeefa906889589df5289b355137dbf51e55cdb9
+size 774084
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_205.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_205.png
new file mode 100644
index 0000000000000000000000000000000000000000..c92f1b473bc80c55201da5e7a201168b253f4dee
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_205.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d5a9f9051d22321bb4ce94c507ef5c50079c25db6b481cb2c3a1eb116299ca9b
+size 800550
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_206.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_206.png
new file mode 100644
index 0000000000000000000000000000000000000000..27155f1718623a91f04e3cc189303100824424f8
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_206.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d82a4eeaa765ec163f6dbdcd3c6e72388e24df0c46e85cdc94df3b704290133f
+size 710966
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_207.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_207.png
new file mode 100644
index 0000000000000000000000000000000000000000..605e0ee74093986774a09c582c3e6d057a4f390c
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_207.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f187b799eaa669846c79f2f099cf76e08585a92930feaaa54be0b92366a04b60
+size 702549
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_208.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_208.png
new file mode 100644
index 0000000000000000000000000000000000000000..7db7e5adf05fb63c1e92867b359191213bdd935c
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_208.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:aae8b00ae118f67260ffaa1b347f92fa9b982174a7cbaa348bb1b11e14332ccd
+size 707446
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_209.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_209.png
new file mode 100644
index 0000000000000000000000000000000000000000..a82c26a466f70fb1eae679d48e3f7f19fe8cbe2a
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_209.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:df901f5c268a29f8f1a1edc7be1f2558b0a30046c747534aee77d3d0d0d38420
+size 713138
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_210.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_210.png
new file mode 100644
index 0000000000000000000000000000000000000000..02098df9d895d320f0abb3c14303441198a17003
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_210.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9b4ad61ddffcf6faadecbd3aaf121d65f27204134f2f44a7a64cc1d16029db0d
+size 709247
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_211.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_211.png
new file mode 100644
index 0000000000000000000000000000000000000000..5ff8c93d8663bc149fcd3ca7aeb9095582a6098f
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_211.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:11c3b57af5b5bd3e4ac2f4e3d1bc3e7be6cf42de6e88c152ab4bc35f6db88a50
+size 865321
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_212.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_212.png
new file mode 100644
index 0000000000000000000000000000000000000000..f254ea662157affc25855281859c36100a182aac
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_212.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:501c9164c7beb330be1aa77bf00d3c222dec7fa61a1c1351a218e93e79bf56fb
+size 875528
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_213.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_213.png
new file mode 100644
index 0000000000000000000000000000000000000000..a776b07e252ffd2fed3d958820954d21f1e32fb1
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_213.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:285b8cd91923d802e4351a32df1683db701aa58fa05ea830a9f9bf7adc59f2c9
+size 869374
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_214.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_214.png
new file mode 100644
index 0000000000000000000000000000000000000000..e6a562fb99066e73c5a3f9d41381d69de66f0bf8
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_214.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2f38f8e024bafd2b2583dc491e99844a9eba3b2c20192294bbbc2c070311977d
+size 868238
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_215.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_215.png
new file mode 100644
index 0000000000000000000000000000000000000000..d5cf7869b39f1f9c8494c5b0d1876e44265568b3
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_215.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3a076c9da7e76c5a13aec1be6c5164f58d79d86e6adc96d0779566c806915f61
+size 870537
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_216.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_216.png
new file mode 100644
index 0000000000000000000000000000000000000000..4ec770856ad4a46f5bfdb8d23f7b5e461d49a419
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_216.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:35a58bacc9ee4f0c4a86a5b2a5dbcbc1b53409313458855fe06a6df17cfef446
+size 724408
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_217.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_217.png
new file mode 100644
index 0000000000000000000000000000000000000000..cdd03aea394546aa86a5f55d73824768a538bd92
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_217.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9f3a7accc48d621c1343a061430e59162f2317c8365f77e05fb3fc5a764ee0f7
+size 716553
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_218.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_218.png
new file mode 100644
index 0000000000000000000000000000000000000000..653c3c5da912820b2cd22b2883a63bd49f96d7ee
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_218.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a8df180cdb9e1414a8f1b654ddc330a6e0aec394eb2f5bee9cc15456b5f73b33
+size 712965
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_219.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_219.png
new file mode 100644
index 0000000000000000000000000000000000000000..c7289b2c677d582663cb02d975a22d5d038fb9cb
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_219.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:63d1a353c9ac0d7c3e96a6342839d2f44c44f2999f125c1da92c18a7185dc6d8
+size 699808
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_220.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_220.png
new file mode 100644
index 0000000000000000000000000000000000000000..1143892451639df39ee685db4ce340aa79e41954
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_220.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:30d203d8985e362c6e64ea3227b36530a9d07297e41838c6cb6fc8867714996f
+size 723856
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_221.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_221.png
new file mode 100644
index 0000000000000000000000000000000000000000..c38050ec63e4ae987350dbbe0c44a9828269bc92
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_221.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ec7e103e4f62e2d81d4fb35947c7b3f856966e4df11dc9187baae4598044324c
+size 731916
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_222.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_222.png
new file mode 100644
index 0000000000000000000000000000000000000000..00143b5b0d856ea8492a27e1b80911dd935c2f28
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_222.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:067de2cca64ee38973f9bb8623c5933142f3456791201c8b5f01ef9225fae049
+size 693002
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_223.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_223.png
new file mode 100644
index 0000000000000000000000000000000000000000..b6657699940c2302c010d1d2e603f956b69c6b5c
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_223.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:79216f89b0d339832ab907c08315230e785fb2300d5b14294c9329fe52230979
+size 678857
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_224.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_224.png
new file mode 100644
index 0000000000000000000000000000000000000000..e80d63883a2d9c78dce81bdce2f31cbc9bf4e880
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_224.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1339ec97401420135125fcb495ccb4c1e77dee89729b6d6399d8238cb01a23bb
+size 705512
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_225.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_225.png
new file mode 100644
index 0000000000000000000000000000000000000000..8f1473c3ab6246e8dc10eefd41b38e36f7349b05
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_225.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:39b06069a4dc2f17ffa3747ec67701430febc9e77abb1e5c47e7f26d028c0a9e
+size 674967
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_226.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_226.png
new file mode 100644
index 0000000000000000000000000000000000000000..cee845bbc310edb2d4baafc2877c821cafe626d4
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_226.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ea33225d5e88f450c7b5e98d8f14bc478b76605d5b3a9e46b2e2f09b8024e35f
+size 795321
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_227.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_227.png
new file mode 100644
index 0000000000000000000000000000000000000000..62b09a872ecae7be12396d2d685c4fc3e425dbe1
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_227.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d10e5025bc8e95e6a6ca6902ff9052f9415753d1d7f62378353f8e28954dbaf2
+size 693082
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_228.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_228.png
new file mode 100644
index 0000000000000000000000000000000000000000..78e615c9e1d69ec13393646d2a25615a2ec34fe7
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_228.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5f87382f255b8dc669122fa6399a5dffc459fa8bf52c6aac00f76a1110fdb6bc
+size 666194
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_229.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_229.png
new file mode 100644
index 0000000000000000000000000000000000000000..6ed9ee0b5b84a6ca8e5d2077651cef122804bb4b
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_229.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6f094afa43db3741c3f2688d334675220f124ceaf7786c5fd216fd567d73c65b
+size 703470
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_230.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_230.png
new file mode 100644
index 0000000000000000000000000000000000000000..75ab380ff95d9b88ed369f87adad9e7b8a9df1d2
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_230.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a4ae5dba031ff46e34ee0afe6bba885d16411e4d5f2efb6d83226c6e971b4d66
+size 718487
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_231.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_231.png
new file mode 100644
index 0000000000000000000000000000000000000000..952acad2ee79dcbd31a122db6ddf419b6f035803
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_231.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f06263971ef90263b82db7f0879f2dc5bfceb606538e727c84faf969a2fff52b
+size 706286
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_232.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_232.png
new file mode 100644
index 0000000000000000000000000000000000000000..53345004d58a1c5ce699f1f879e2fc83f1d90679
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_232.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2283798098cf8ab6eae80f7ba956034eb9c6d3ef7022b8069c49b57c688254bb
+size 661265
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_233.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_233.png
new file mode 100644
index 0000000000000000000000000000000000000000..1b058f7a2066e439015b3259f2ff02dac9de1403
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_233.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:62ac1d9f68f708ec524d4b00bde16bb2ee7ddb30cf02433648eea73ff87d2624
+size 702098
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_234.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_234.png
new file mode 100644
index 0000000000000000000000000000000000000000..66220409490a27dd782db3d87528dbe30da8af50
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_234.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4c26451cbd0482c8bc7b83c467e957650941c7e7250b132c3209f073739d6a5f
+size 727376
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_235.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_235.png
new file mode 100644
index 0000000000000000000000000000000000000000..7ab6de64ff3cca8fd7ee15e4f4c3a2113096fbf8
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_235.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8d22ca85a24a054b2a4abac2aeeb74aafef9da713e5a6af506cfd56fdb6f5e41
+size 749851
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_236.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_236.png
new file mode 100644
index 0000000000000000000000000000000000000000..05c5fc12203cdc16c7715a4652fd0ef6f1a2c8ad
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_236.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a4a0ec47369d99128f66c6bf5a24c14a45551cc2bdb209638a615b42fb925a7d
+size 724599
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_237.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_237.png
new file mode 100644
index 0000000000000000000000000000000000000000..24c6d71198882e552000782e40d242a40e93761f
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_237.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:dba3c6aac927b83c6fcb3f77ffb26b6765fd6a818bea7bfcb81bb02a3ec955dc
+size 742623
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_238.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_238.png
new file mode 100644
index 0000000000000000000000000000000000000000..710a0d60c1916cd3671d91ce2040586a6abe0a3f
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_238.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6616f26b9db71cc08de1c627084ba3d9bb461cacc103fdc5adc85b7d3d1b2231
+size 761793
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_239.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_239.png
new file mode 100644
index 0000000000000000000000000000000000000000..b4fced4ed2b63762ae14520172ca4b06065a41e8
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_239.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e863640a44471ee14d8470116fbdbac091636dfa3362bc9ad2c7cfbe05081027
+size 740694
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_240.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_240.png
new file mode 100644
index 0000000000000000000000000000000000000000..b0ab9ec4fb29e3ea887b2024f2fe8c7c1d3e599e
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_240.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d275c151036f649905a46043d5bc17178ee55df1b3c93d46e6afadda93153394
+size 739109
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_241.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_241.png
new file mode 100644
index 0000000000000000000000000000000000000000..f39417b55918344b9e973bda424e550fd68989a2
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_241.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3e977f8d92cb28d2968bab2268f3fbf06bc3287053ef50e425267a652e6f19a9
+size 693243
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_242.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_242.png
new file mode 100644
index 0000000000000000000000000000000000000000..eae174db76936f385754c9eab8db8a2bfefd973d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_242.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e2a67cfd8762d26c3c3d3aea83ba112a9c8fd647f939c345f59a8868c6349782
+size 703398
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_243.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_243.png
new file mode 100644
index 0000000000000000000000000000000000000000..38744f3347cd701705e256787d186bff2cd0737d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_243.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0cbad7eebd8dd566d92b292106ac12efeb16dbc3a26774fd8ef29a6ab5a5a341
+size 699132
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_244.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_244.png
new file mode 100644
index 0000000000000000000000000000000000000000..6238eaba36f4d7544a56f24851206435e996221d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_244.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:88f4239a7fdad23c542fe8fe4acc49c7b214da249f07f825120e60002198e142
+size 699458
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_245.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_245.png
new file mode 100644
index 0000000000000000000000000000000000000000..185e58bad7c975f67244e6a15356afe6b3a142df
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_245.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:bbbb23ddd9ad5a3addfffee78818b319896683869598073caf7d627ee23a0f2e
+size 725103
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_246.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_246.png
new file mode 100644
index 0000000000000000000000000000000000000000..aed515f66978add01cf3b1fee2b43825b1118bb8
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_246.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:84875ea43579f80af999821f7915a7d33b0f89fd4134c0dd31f730ea3590fc28
+size 736289
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_247.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_247.png
new file mode 100644
index 0000000000000000000000000000000000000000..251c07dee9e674df39eaba942191876cce54dd70
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_247.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c515a795feb595737c87786313d7d8006fd2740b710db77ec04dec5cafa86168
+size 725706
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_248.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_248.png
new file mode 100644
index 0000000000000000000000000000000000000000..3f6aea03036c906e0a5da904de64d759b43185e5
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_248.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7a0e3199ebfd4fd0f1aa98461561d5b54f543175dff4cd4046e6991c484099fa
+size 722586
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_249.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_249.png
new file mode 100644
index 0000000000000000000000000000000000000000..97a283e3f16065f51af9c3cc236d810a3f0185ea
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_249.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:44751e7ea426939bbc32532d9567d3743a62a0c115c31af990835cde5f9033b6
+size 720686
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_250.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_250.png
new file mode 100644
index 0000000000000000000000000000000000000000..9992c882098f071d0c367e93a4ea9d954d123027
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_250.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0c8d27bafeaed8f8dec6d971959e119d9936a5d17b90e242415cff8bbf236f29
+size 733450
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_251.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_251.png
new file mode 100644
index 0000000000000000000000000000000000000000..d5e86f151e5ea9eb06949e4757f5f437a0930eb4
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_251.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4d879dc71e2cdc408bf3826f43f199620ac88b9ba8782951e8c792b858526e05
+size 707232
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_252.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_252.png
new file mode 100644
index 0000000000000000000000000000000000000000..6624d9db5149bc5c0e97b795dd6c9fce6d973b19
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_252.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:aa3a28a6b3a063d9d79fbeb9edd5d21135ccc4290476dc1300e74f764f577e01
+size 722013
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_253.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_253.png
new file mode 100644
index 0000000000000000000000000000000000000000..906bedeea947a6034b8c716c3ef0c993677e4deb
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_253.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0caf23481122c841eff655c130840e90148f7890d61f7a93e9316d1c57d6ada5
+size 713106
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_254.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_254.png
new file mode 100644
index 0000000000000000000000000000000000000000..ce2380153f4c7abec2260d5120074621f428dfde
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_254.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:612c8330f4fa1e212440e3389a748d3a31c8e8a133d22b6f8eebca47938d731a
+size 725982
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_255.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_255.png
new file mode 100644
index 0000000000000000000000000000000000000000..98ca3fc9d1393ddbf8ba410399f00d4fa3e41b8d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_255.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cb61fcc0223c04ae077b22d0157766479822bba3a548882285995b3230aca2d6
+size 731738
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_256.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_256.png
new file mode 100644
index 0000000000000000000000000000000000000000..0a522013726b278cc068615b0ba3a49687c0c174
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_256.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9c47f51b6345524050dd24294aa49ed0acb7994589c97dee0cab1858819b53ca
+size 782699
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_257.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_257.png
new file mode 100644
index 0000000000000000000000000000000000000000..b97c3833a7daf12beee2e66b3f51d512086069b8
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_257.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:28f86730201adc887602c545e6abe7597cbe046da2200c83cec6eec37fa6e4be
+size 805439
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_258.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_258.png
new file mode 100644
index 0000000000000000000000000000000000000000..cbc1570632afbe7e687fce1c10912741127a88fe
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_258.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b4975c8c09fc60ce6835a5cafbc90c89e2b8ef3cb1816395b035dbb7749e96b9
+size 776317
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_259.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_259.png
new file mode 100644
index 0000000000000000000000000000000000000000..261f9c949c5e49d2e192f3a9c2b2944812fc49f0
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_259.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:fd656566d513fc20b00d84c610858345cf582fbd8456800b27cbb3f94615c28a
+size 800606
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_260.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_260.png
new file mode 100644
index 0000000000000000000000000000000000000000..87588882590780d8dbd1011bf2a82d40144d6ecb
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_260.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5368ed297e59bc268420d8004ecabd4b73c4290deb7a6e0840c2928af1e37100
+size 710889
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_261.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_261.png
new file mode 100644
index 0000000000000000000000000000000000000000..bb5f9103412c7a832b9a75cbaca346d790342bc9
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_261.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ae09934227ca49de47d4a28bcdf6963381c37d5c86dd7183190d0716f0f80d9d
+size 702635
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_262.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_262.png
new file mode 100644
index 0000000000000000000000000000000000000000..2cdb1c6d3664e103a4f42cacba2fe33fc9fb416e
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_262.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:fc57835e721d4d1c7f864dc6a7c74e31c217a872a2d853427b8ec0473a0c000c
+size 707459
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_263.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_263.png
new file mode 100644
index 0000000000000000000000000000000000000000..a0a14836f5b59ae124c9d2fd00258d14e08e07ee
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_263.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7a01a0b1cf3d3bee5a3fea928fd95dc8fb4ae4f714fbe08f0575edb53c52e67b
+size 713018
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_264.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_264.png
new file mode 100644
index 0000000000000000000000000000000000000000..8aebb3db8093c86f96058d93f15ef1bc5d323ccf
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_264.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:937fc46c6e8143892529ac5192e7f26bd5eb8b6baf26570dbb44ba64bdab28da
+size 709258
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_265.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_265.png
new file mode 100644
index 0000000000000000000000000000000000000000..0d011418d8098000ab95ed68641d51b763390da4
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_265.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:22fe54b8fc639dd1fc074b6a39c07c0fd44978ec7bec5e4d70df2a0a5906dedb
+size 865153
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_266.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_266.png
new file mode 100644
index 0000000000000000000000000000000000000000..ace79302c76d8e36271209e3580b5294fda50333
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_266.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:436e0acf728b0cf898ddd7e6d14166205db92ab3a67dd1b0c4d4644c0f06f1ce
+size 875412
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_267.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_267.png
new file mode 100644
index 0000000000000000000000000000000000000000..91fe4a019fd1cad3bb5b111d74b3c37048360014
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_267.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:458efaa37c1ebf5fb1fd2cdac4d81cbb438dc5a80f848fe437a7a53f15bc60b3
+size 869459
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_268.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_268.png
new file mode 100644
index 0000000000000000000000000000000000000000..caf9537bdae6b561387c4b9ce534f2101d8d0d05
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_268.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:13df7cfc77fb9a94d8e07378911065ee8e3244322f83f08875a49e14abeb785e
+size 867626
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_269.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_269.png
new file mode 100644
index 0000000000000000000000000000000000000000..b00d339c12e7a0669de178c38a6086105e90f982
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_269.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:eac75cf9435583874a747428177df218682777192a0b35bfc067832cd9b9e595
+size 870096
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_54.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_54.png
new file mode 100644
index 0000000000000000000000000000000000000000..517b51ab919d62be7e01e5eda23dbacc5f1dfbad
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_54.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:36dfcb63eea300167a74362cfee875bbbf23a43bba101d509e5662ba03285dd3
+size 724567
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_55.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_55.png
new file mode 100644
index 0000000000000000000000000000000000000000..143d9de85f787bc87029d74c31251ecb85f1ec0b
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_55.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:90bab771a20bffa7e600c63ce91e861946181f452e527f0c2c9da96d0368a3dc
+size 716269
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_56.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_56.png
new file mode 100644
index 0000000000000000000000000000000000000000..8f593f9f236701d94e03aa7530c08060d42b4acf
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_56.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2af5b5b57df66280e4121dc24834b150864326d3d1f19dfc7abebacbca293e67
+size 712862
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_57.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_57.png
new file mode 100644
index 0000000000000000000000000000000000000000..834142679ae2293b37e82c688a21ea65b0c368fc
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_57.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7a647fd41e6fe760b79f2346a5ffcad53c44ed8ea8c973bd16ac2bbc729f2b1f
+size 699503
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_58.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_58.png
new file mode 100644
index 0000000000000000000000000000000000000000..4e847b03168f57209ebb3ac6f0e65fa5423fd2a2
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_58.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:de5b39ae660b0e209ebf8f358991cbd5a946130525ba09bddbffe537856e6378
+size 723813
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_59.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_59.png
new file mode 100644
index 0000000000000000000000000000000000000000..85ba42dd82942c14ae6aebe955f16b562a1c2003
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_59.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f924f8c9ecd96526ccf3517c2e33c02e86fcefd7f95756b5c0fcd2102f33e017
+size 731702
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_60.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_60.png
new file mode 100644
index 0000000000000000000000000000000000000000..3612886dc5a2f9386dbbc1198c1b844f859cfd5e
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_60.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:37835b4eabb57a8e0266f1b5d749073fecabde81c19efe60479bfb73d679594f
+size 693000
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_61.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_61.png
new file mode 100644
index 0000000000000000000000000000000000000000..7799742bd64bd8534cb9750fdeb3e3601ba04569
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_61.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b1f6955093858564228a82629d09d1f4d4056163643015f7d15cc90a45d34640
+size 678747
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_62.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_62.png
new file mode 100644
index 0000000000000000000000000000000000000000..15c69cae3de10e7448571c081d2c6e0e2c2da332
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_62.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:07bdf5e94b62f2b7f2a272640aea7362a7f8429947fe64ee491f582a30a6bc58
+size 705389
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_63.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_63.png
new file mode 100644
index 0000000000000000000000000000000000000000..a8197c8807880a53c240ffddc97305bac61d25f6
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_63.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b398b05ee116beeb5c4e030c4655e78ba598a425b8ac28c4ecbbec5908a80fa6
+size 675049
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_64.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_64.png
new file mode 100644
index 0000000000000000000000000000000000000000..ac4023a2e9f9dfd11107d539a023820e8158c045
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_64.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6f1d2715cc963dd6590150054ecbb415bd42798ed1a7ed21cd2c5c5995f8e51b
+size 795531
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_65.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_65.png
new file mode 100644
index 0000000000000000000000000000000000000000..c087247ba6d0a2ed320da11d40fe3a489edcc6de
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_65.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cb053439c03fe2fccc13e478d72f21d41c164d88bcc41e80aa0c4c494cb4d795
+size 693064
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_66.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_66.png
new file mode 100644
index 0000000000000000000000000000000000000000..030560fbbf009da1d41011ee1f7503d3a567c19c
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_66.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f50fddb94e9277f3f86236c6e13f63f4a2966426e7536dbfc3478dfcec7d12d4
+size 666235
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_67.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_67.png
new file mode 100644
index 0000000000000000000000000000000000000000..7731cb501c76c3a3faef98e2ea5458146cde81d4
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_67.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8ae4abf32a625aa1303fe722e38917b87b9fa4a6be489e94e945a601fcec7eed
+size 703447
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_68.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_68.png
new file mode 100644
index 0000000000000000000000000000000000000000..1162838b3fe75580fffa4f2446848a9d6115f81f
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_68.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8439f78b201475c331f633475a4c5be2f51fdaf7d5d9a946a225e6f2d74aac9d
+size 718429
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_69.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_69.png
new file mode 100644
index 0000000000000000000000000000000000000000..8cc387e2dde4b0320e3f69c65e694c40747a4e73
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_69.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:650104f463b5f1e8c94771d0ebff0ac29e2e0747f17b4c224f17137eec637458
+size 706581
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_70.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_70.png
new file mode 100644
index 0000000000000000000000000000000000000000..40d2a58c47f6e335290442a8f6ed173035ce69f4
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_70.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f22d0a8a8c7ece82d8019dcb9947d2c57c5216f377d95767188da5b7cbc1be75
+size 661317
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_71.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_71.png
new file mode 100644
index 0000000000000000000000000000000000000000..77b0d60a1f66d623fda10e9fed386de17c4ddc85
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_71.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3f996a3cb126396cb2be54681a332c67371bcee2cd5be4afc32cd23febcb40ed
+size 702126
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_72.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_72.png
new file mode 100644
index 0000000000000000000000000000000000000000..437394d41fe21f8452cf600e30997c064e9ad4ec
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_72.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e109133f2ab3b656e5a3cc80f819fdb18b126887cc06591019d7c4b0642a594f
+size 727424
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_73.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_73.png
new file mode 100644
index 0000000000000000000000000000000000000000..460c86a1d4a2ee104ac1602e6dd6d33f964e9268
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_73.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c3cb3de5023805ac34b2161914055d54df5d7df8948ce3310fc6fa13c793e811
+size 749998
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_74.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_74.png
new file mode 100644
index 0000000000000000000000000000000000000000..56069c53fb177da20bb42a1318b283a290b4231a
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_74.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:474e6ea35d6d33bde26cda9d756e492081de69dd9841940cf64c78f699b5fe5b
+size 724597
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_75.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_75.png
new file mode 100644
index 0000000000000000000000000000000000000000..2a076ba1ab9c6e5b34ece585e92284601793ce2b
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_75.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:83c4d041d9b21d7617b5498e6625d133618a109b2628fa2a6edf5249ab1deb50
+size 742629
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_76.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_76.png
new file mode 100644
index 0000000000000000000000000000000000000000..527d9e941ad62e3b457b060bf9c2d6454ccf390b
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_76.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d3bcc826180c8a481932deac9d6b2f14e6d25d6e040c3402048e6ed2ac8a2275
+size 761854
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_77.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_77.png
new file mode 100644
index 0000000000000000000000000000000000000000..aa4b97836c12adfe4bcaf92b7b8e75f89f22e1a3
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_77.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8d99b42a6fd2ec9899bc58d3edcc404b450cfa1550c5f07010d6a6fed1fc73e0
+size 740852
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_78.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_78.png
new file mode 100644
index 0000000000000000000000000000000000000000..26984a475ead4bec3e3921a86236baea3206ce0e
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_78.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e408c592409f15a9344de3f7b193114f221590e9632f1c6d741a29c1212f8948
+size 739191
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_79.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_79.png
new file mode 100644
index 0000000000000000000000000000000000000000..9a329c4b900992442ab2b873ccd4cc02b0e460c7
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_79.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1d0f4f669a4afe28861c31d5c878db5034f501abcb5febfe883a0021e92e3f94
+size 693257
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_80.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_80.png
new file mode 100644
index 0000000000000000000000000000000000000000..3b09b57b8c106a72839f98ff5337d845f2769fe8
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_80.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:db6d997265b0e30cd45fc8fbfe60c68a646f9f34ddf40bc1a90d16bea5949902
+size 703531
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_81.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_81.png
new file mode 100644
index 0000000000000000000000000000000000000000..761062b820f63a4ef5ef90ed9d3e92ac05f33771
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_81.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6dfac82d826ab7d4be38cdc8245776d9e5547821f30a8465390ff172491be860
+size 699420
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_82.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_82.png
new file mode 100644
index 0000000000000000000000000000000000000000..254d6ef0831f057077ec5f7795e20dfe36fde02c
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_82.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:fc91eee725c51d190fd39cdb8cea59e2fe6d4a3d05565e5a81e532eb4cb7ce35
+size 700188
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_83.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_83.png
new file mode 100644
index 0000000000000000000000000000000000000000..098a0b6cc453dfe12e750b04d958290eaa7b1919
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_83.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3a186704e74786f319fd686a06acb57a7654050461b2bc24ea9923f0fbc5bf00
+size 726000
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_84.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_84.png
new file mode 100644
index 0000000000000000000000000000000000000000..e0a12c9719ad6b4b963783cf4423b47f55214a48
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_84.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c39b0ac28c6b1e754bee54845448d799271d4b91d4bd12fd4cce403d9875f385
+size 736131
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_85.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_85.png
new file mode 100644
index 0000000000000000000000000000000000000000..566979342e6d33448db453d44fdec535d2cfa73d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_85.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ba870ef3a524ad0cd895014c15d280fbb81c2880f8480a062a6b184ce6ff0a3d
+size 725592
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_86.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_86.png
new file mode 100644
index 0000000000000000000000000000000000000000..bd81257c41bd60583cee52e2d225eaf8a57709ea
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_86.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ac9e18b2a29950a2be77b051b7bcfb6951340ee572c5136eeb9aed9cc7395565
+size 722367
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_87.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_87.png
new file mode 100644
index 0000000000000000000000000000000000000000..75566f56f1de7d822fa29a422114be43f94e9f1d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_87.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:65c33b32b071134864bb5e00deda5eb8997af133db2306f15143906d3a6f7e4f
+size 720610
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_88.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_88.png
new file mode 100644
index 0000000000000000000000000000000000000000..9d5ae2bfc48c01aa58367a6afdd188cf6753d00b
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_88.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6754b6627f16447e6b74009995a3514e83755adac20259b4d3c3280d23b7d46e
+size 733392
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_89.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_89.png
new file mode 100644
index 0000000000000000000000000000000000000000..551aa7ffc36274894825b7ce832b5e1c00ceca76
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_89.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6fc3a3a7d631203fe3cd134645b364d5713b3fe0be6ee32905b996249a1982a2
+size 707268
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_90.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_90.png
new file mode 100644
index 0000000000000000000000000000000000000000..7bbdac9438c60a6e8db24c7cbfd16c3d56c0b3e5
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_90.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4aa346f87c059630a1ae14b6cb79feea51052a2d41da5c0020d0ea58b854b9d2
+size 721864
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_91.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_91.png
new file mode 100644
index 0000000000000000000000000000000000000000..23365c3afc56b0510bb42fe477edc7522327953d
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_91.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:646be4374beeb23a83e024f6cd27bea039aca61b277679120392102267ea49e7
+size 713117
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_92.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_92.png
new file mode 100644
index 0000000000000000000000000000000000000000..df41602c10c2addc4c269ce70ea4d3a73759c6a0
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_92.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2b657d5c97490e6b7630ad3c30bca18c7180a1839d84e2f3259c710cd924b55e
+size 726576
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_93.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_93.png
new file mode 100644
index 0000000000000000000000000000000000000000..44e7b4f7372ba4a5a8cb8275b3a7a136191a07e8
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_93.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a06469744831fd7f9f1c04cba5d7ce837872ba240d56006258bfc26af4cb9976
+size 731453
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_94.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_94.png
new file mode 100644
index 0000000000000000000000000000000000000000..43ddc9a4ac1bb4f1062c0698aaf73b7a22db6fcf
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_94.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cbc46b2ee69a46888ef2804746a019b774cfca2c1427d320df306558e982a5f6
+size 783215
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_95.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_95.png
new file mode 100644
index 0000000000000000000000000000000000000000..f2049ffdd2c37203421b4cd501faf5e86b640118
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_95.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8ac139e3915a6d1624cb7fa645ffbfd5898e4b84d94905d876a18059f1b912fb
+size 806595
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_96.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_96.png
new file mode 100644
index 0000000000000000000000000000000000000000..d91404d1a8e1d3c07f0c49884646f0e4470188da
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_96.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b7d7c51a20dfa4c9665f177983d1557fd448455add9bad957a4ce68dc8bb9640
+size 776191
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_97.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_97.png
new file mode 100644
index 0000000000000000000000000000000000000000..2f5c923783280ff1c5f236b5c6e0b2accdc3d841
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_97.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:969d1b0e2de13c4c9d5713dd7b0dc2ad86887d17b2b94042bb3448f05b8bdd30
+size 800600
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_98.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_98.png
new file mode 100644
index 0000000000000000000000000000000000000000..4147906c89b075440b170c63d28cbdd9e2401807
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_98.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:fc1652fc29730ca484e8b3936a974e5e0f5bdc6339ab1cc3736fc74340043045
+size 709735
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_99.png b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_99.png
new file mode 100644
index 0000000000000000000000000000000000000000..153403c7c9fd15e56d1abae8b5cc85105ad77e4c
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_182315/vis_data/vis_image/val_img_99.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:bbb67a74498e52c0d2023753563c24e4acf87736bc1a702a6fa7e4a277e0dbf3
+size 702034
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_194608/20251226_194608.log b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_194608/20251226_194608.log
new file mode 100644
index 0000000000000000000000000000000000000000..13de6f3076f8b3072779b09901c59b2732018c31
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_194608/20251226_194608.log
@@ -0,0 +1,418 @@
+2025/12/26 19:46:11 - mmengine - INFO -
+------------------------------------------------------------
+System environment:
+ sys.platform: linux
+ Python: 3.10.19 (main, Oct 21 2025, 16:43:05) [GCC 11.2.0]
+ CUDA available: True
+ MUSA available: False
+ numpy_random_seed: 1082439054
+ GPU 0: NVIDIA GeForce RTX 3060
+ CUDA_HOME: /root/miniconda3/envs/gvhmr
+ NVCC: Cuda compilation tools, release 12.1, V12.1.105
+ GCC: gcc (Ubuntu 11.4.0-1ubuntu1~22.04.2) 11.4.0
+ PyTorch: 2.3.0+cu121
+ PyTorch compiling details: PyTorch built with:
+ - GCC 9.3
+ - C++ Version: 201703
+ - Intel(R) oneAPI Math Kernel Library Version 2022.2-Product Build 20220804 for Intel(R) 64 architecture applications
+ - Intel(R) MKL-DNN v3.3.6 (Git Hash 86e6af5974177e513fd3fee58425e1063e7f1361)
+ - OpenMP 201511 (a.k.a. OpenMP 4.5)
+ - LAPACK is enabled (usually provided by MKL)
+ - NNPACK is enabled
+ - CPU capability usage: AVX2
+ - CUDA Runtime 12.1
+ - NVCC architecture flags: -gencode;arch=compute_50,code=sm_50;-gencode;arch=compute_60,code=sm_60;-gencode;arch=compute_70,code=sm_70;-gencode;arch=compute_75,code=sm_75;-gencode;arch=compute_80,code=sm_80;-gencode;arch=compute_86,code=sm_86;-gencode;arch=compute_90,code=sm_90
+ - CuDNN 8.9.2
+ - Magma 2.6.1
+ - Build settings: BLAS_INFO=mkl, BUILD_TYPE=Release, CUDA_VERSION=12.1, CUDNN_VERSION=8.9.2, CXX_COMPILER=/opt/rh/devtoolset-9/root/usr/bin/c++, CXX_FLAGS= -D_GLIBCXX_USE_CXX11_ABI=0 -fabi-version=11 -fvisibility-inlines-hidden -DUSE_PTHREADPOOL -DNDEBUG -DUSE_KINETO -DLIBKINETO_NOROCTRACER -DUSE_FBGEMM -DUSE_QNNPACK -DUSE_PYTORCH_QNNPACK -DUSE_XNNPACK -DSYMBOLICATE_MOBILE_DEBUG_HANDLE -O2 -fPIC -Wall -Wextra -Werror=return-type -Werror=non-virtual-dtor -Werror=bool-operation -Wnarrowing -Wno-missing-field-initializers -Wno-type-limits -Wno-array-bounds -Wno-unknown-pragmas -Wno-unused-parameter -Wno-unused-function -Wno-unused-result -Wno-strict-overflow -Wno-strict-aliasing -Wno-stringop-overflow -Wsuggest-override -Wno-psabi -Wno-error=pedantic -Wno-error=old-style-cast -Wno-missing-braces -fdiagnostics-color=always -faligned-new -Wno-unused-but-set-variable -Wno-maybe-uninitialized -fno-math-errno -fno-trapping-math -Werror=format -Wno-stringop-overflow, LAPACK_INFO=mkl, PERF_WITH_AVX=1, PERF_WITH_AVX2=1, PERF_WITH_AVX512=1, TORCH_VERSION=2.3.0, USE_CUDA=ON, USE_CUDNN=ON, USE_CUSPARSELT=1, USE_EXCEPTION_PTR=1, USE_GFLAGS=OFF, USE_GLOG=OFF, USE_GLOO=ON, USE_MKL=ON, USE_MKLDNN=ON, USE_MPI=OFF, USE_NCCL=1, USE_NNPACK=ON, USE_OPENMP=ON, USE_ROCM=OFF, USE_ROCM_KERNEL_ASSERT=OFF,
+
+ TorchVision: 0.18.0+cu121
+ OpenCV: 4.12.0
+ MMEngine: 0.10.7
+
+Runtime environment:
+ cudnn_benchmark: False
+ mp_cfg: {'mp_start_method': 'fork', 'opencv_num_threads': 0}
+ dist_cfg: {'backend': 'nccl'}
+ seed: 1082439054
+ Distributed launcher: none
+ Distributed training: False
+ GPU number: 1
+------------------------------------------------------------
+
+2025/12/26 19:46:11 - mmengine - INFO - Config:
+auto_scale_lr = None
+backend_args = dict(backend='local')
+batch_size_per_gpu = 8
+codec = dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap')
+custom_hooks = [
+ dict(type='SyncBuffersHook'),
+]
+custom_imports = dict(
+ allow_failed_imports=False,
+ imports=[
+ 'mmpose.engine.optim_wrappers.layer_decay_optim_wrapper',
+ ])
+data_mode = 'topdown'
+data_root = '/root/miko/puni/train/GVHMR/processed_data/'
+dataset_type = 'CocoDataset'
+default_hooks = dict(
+ badcase=dict(
+ badcase_thr=5,
+ enable=False,
+ metric_type='loss',
+ out_dir='badcase',
+ type='BadCaseAnalysisHook'),
+ checkpoint=dict(
+ interval=1,
+ max_keep_ckpts=2,
+ rule='greater',
+ save_best='coco/AP',
+ save_optimizer=False,
+ type='CheckpointHook'),
+ logger=dict(interval=50, type='LoggerHook'),
+ param_scheduler=dict(type='ParamSchedulerHook'),
+ sampler_seed=dict(type='DistSamplerSeedHook'),
+ timer=dict(type='IterTimerHook'),
+ visualization=dict(
+ enable=True,
+ interval=100,
+ out_dir='vis_results',
+ type='PoseVisualizationHook'))
+default_scope = 'mmpose'
+env_cfg = dict(
+ cudnn_benchmark=False,
+ dist_cfg=dict(backend='nccl'),
+ mp_cfg=dict(mp_start_method='fork', opencv_num_threads=0))
+launcher = 'none'
+load_from = '/root/miko/puni/train/GVHMR/inputs/checkpoints/vitpose/td-hm_ViTPose-huge_8xb64-210e_coco-256x192-e32adcd4_20230314.pth'
+log_level = 'INFO'
+log_processor = dict(
+ by_epoch=True, num_digits=6, type='LogProcessor', window_size=50)
+metainfo = dict(from_file='configs/_base_/datasets/coco.py')
+model = dict(
+ backbone=dict(
+ arch='huge',
+ drop_path_rate=0.55,
+ img_size=(
+ 256,
+ 192,
+ ),
+ init_cfg=None,
+ out_type='featmap',
+ patch_cfg=dict(padding=2),
+ patch_size=16,
+ qkv_bias=True,
+ type='mmpretrain.VisionTransformer',
+ with_cls_token=False),
+ data_preprocessor=dict(
+ bgr_to_rgb=True,
+ mean=[
+ 123.675,
+ 116.28,
+ 103.53,
+ ],
+ std=[
+ 58.395,
+ 57.12,
+ 57.375,
+ ],
+ type='PoseDataPreprocessor'),
+ head=dict(
+ decoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ deconv_kernel_sizes=(
+ 4,
+ 4,
+ ),
+ deconv_out_channels=(
+ 256,
+ 256,
+ ),
+ in_channels=1280,
+ loss=dict(type='KeypointMSELoss', use_target_weight=True),
+ out_channels=17,
+ type='HeatmapHead'),
+ test_cfg=dict(flip_mode='heatmap', flip_test=True, shift_heatmap=False),
+ type='TopdownPoseEstimator')
+num_workers = 4
+optim_wrapper = dict(
+ clip_grad=dict(max_norm=1.0, norm_type=2),
+ constructor='LayerDecayOptimWrapperConstructor',
+ dtype='float16',
+ optimizer=dict(
+ betas=(
+ 0.9,
+ 0.999,
+ ), lr=5e-05, type='AdamW', weight_decay=0.1),
+ paramwise_cfg=dict(
+ custom_keys=dict(
+ bias=dict(decay_multi=0.0),
+ norm=dict(decay_mult=0.0),
+ pos_embed=dict(decay_mult=0.0),
+ relative_position_bias_table=dict(decay_mult=0.0)),
+ layer_decay_rate=0.85,
+ num_layers=32),
+ type='AmpOptimWrapper')
+param_scheduler = [
+ dict(
+ begin=0,
+ by_epoch=True,
+ end=20,
+ gamma=0.1,
+ milestones=[
+ 14,
+ 18,
+ ],
+ type='MultiStepLR'),
+]
+resume = False
+test_cfg = dict()
+test_dataloader = dict(
+ batch_size=32,
+ dataset=dict(
+ ann_file='annotations/person_keypoints_val2017.json',
+ bbox_file=
+ 'data/coco/person_detection_results/COCO_val2017_detections_AP_H_56_person.json',
+ data_mode='topdown',
+ data_prefix=dict(img='val2017/'),
+ data_root='data/coco/',
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+test_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+train_cfg = dict(by_epoch=True, max_epochs=20, val_interval=1)
+train_dataloader = dict(
+ batch_size=8,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+ ],
+ type='CocoDataset'),
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(shuffle=True, type='DefaultSampler'))
+train_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+]
+val_cfg = dict()
+val_dataloader = dict(
+ batch_size=8,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ bbox_file=None,
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+val_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+val_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+]
+vis_backends = [
+ dict(type='LocalVisBackend'),
+]
+visualizer = dict(
+ name='visualizer',
+ type='PoseLocalVisualizer',
+ vis_backends=[
+ dict(type='LocalVisBackend'),
+ ])
+work_dir = '../work_dirs/vitpose_finetune'
+
+2025/12/26 19:46:18 - mmengine - INFO - Distributed training is not used, all SyncBatchNorm (SyncBN) layers in the model will be automatically reverted to BatchNormXd layers if they are used.
+2025/12/26 19:46:18 - mmengine - INFO - Hooks will be executed in the following order:
+before_run:
+(VERY_HIGH ) RuntimeInfoHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+before_train:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+before_train_epoch:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(NORMAL ) DistSamplerSeedHook
+ --------------------
+before_train_iter:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+ --------------------
+after_train_iter:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(BELOW_NORMAL) LoggerHook
+(LOW ) ParamSchedulerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+after_train_epoch:
+(NORMAL ) IterTimerHook
+(NORMAL ) SyncBuffersHook
+(LOW ) ParamSchedulerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+before_val:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+before_val_epoch:
+(NORMAL ) IterTimerHook
+(NORMAL ) SyncBuffersHook
+ --------------------
+before_val_iter:
+(NORMAL ) IterTimerHook
+ --------------------
+after_val_iter:
+(NORMAL ) IterTimerHook
+(NORMAL ) PoseVisualizationHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+after_val_epoch:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(BELOW_NORMAL) LoggerHook
+(LOW ) ParamSchedulerHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+after_val:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+after_train:
+(VERY_HIGH ) RuntimeInfoHook
+(VERY_LOW ) CheckpointHook
+ --------------------
+before_test:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+before_test_epoch:
+(NORMAL ) IterTimerHook
+ --------------------
+before_test_iter:
+(NORMAL ) IterTimerHook
+ --------------------
+after_test_iter:
+(NORMAL ) IterTimerHook
+(NORMAL ) PoseVisualizationHook
+(NORMAL ) BadCaseAnalysisHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+after_test_epoch:
+(VERY_HIGH ) RuntimeInfoHook
+(NORMAL ) IterTimerHook
+(NORMAL ) BadCaseAnalysisHook
+(BELOW_NORMAL) LoggerHook
+ --------------------
+after_test:
+(VERY_HIGH ) RuntimeInfoHook
+ --------------------
+after_run:
+(BELOW_NORMAL) LoggerHook
+ --------------------
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_194608/vis_data/config.py b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_194608/vis_data/config.py
new file mode 100644
index 0000000000000000000000000000000000000000..c97d4b7b066ab202a8b4c5a39bee5f125b4b53c7
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/20251226_194608/vis_data/config.py
@@ -0,0 +1,285 @@
+auto_scale_lr = None
+backend_args = dict(backend='local')
+batch_size_per_gpu = 8
+codec = dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap')
+custom_hooks = [
+ dict(type='SyncBuffersHook'),
+]
+custom_imports = dict(
+ allow_failed_imports=False,
+ imports=[
+ 'mmpose.engine.optim_wrappers.layer_decay_optim_wrapper',
+ ])
+data_mode = 'topdown'
+data_root = '/root/miko/puni/train/GVHMR/processed_data/'
+dataset_type = 'CocoDataset'
+default_hooks = dict(
+ badcase=dict(
+ badcase_thr=5,
+ enable=False,
+ metric_type='loss',
+ out_dir='badcase',
+ type='BadCaseAnalysisHook'),
+ checkpoint=dict(
+ interval=1,
+ max_keep_ckpts=2,
+ rule='greater',
+ save_best='coco/AP',
+ save_optimizer=False,
+ type='CheckpointHook'),
+ logger=dict(interval=50, type='LoggerHook'),
+ param_scheduler=dict(type='ParamSchedulerHook'),
+ sampler_seed=dict(type='DistSamplerSeedHook'),
+ timer=dict(type='IterTimerHook'),
+ visualization=dict(
+ enable=True,
+ interval=100,
+ out_dir='vis_results',
+ type='PoseVisualizationHook'))
+default_scope = 'mmpose'
+env_cfg = dict(
+ cudnn_benchmark=False,
+ dist_cfg=dict(backend='nccl'),
+ mp_cfg=dict(mp_start_method='fork', opencv_num_threads=0))
+launcher = 'none'
+load_from = '/root/miko/puni/train/GVHMR/inputs/checkpoints/vitpose/td-hm_ViTPose-huge_8xb64-210e_coco-256x192-e32adcd4_20230314.pth'
+log_level = 'INFO'
+log_processor = dict(
+ by_epoch=True, num_digits=6, type='LogProcessor', window_size=50)
+metainfo = dict(from_file='configs/_base_/datasets/coco.py')
+model = dict(
+ backbone=dict(
+ arch='huge',
+ drop_path_rate=0.55,
+ img_size=(
+ 256,
+ 192,
+ ),
+ init_cfg=None,
+ out_type='featmap',
+ patch_cfg=dict(padding=2),
+ patch_size=16,
+ qkv_bias=True,
+ type='mmpretrain.VisionTransformer',
+ with_cls_token=False),
+ data_preprocessor=dict(
+ bgr_to_rgb=True,
+ mean=[
+ 123.675,
+ 116.28,
+ 103.53,
+ ],
+ std=[
+ 58.395,
+ 57.12,
+ 57.375,
+ ],
+ type='PoseDataPreprocessor'),
+ head=dict(
+ decoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ deconv_kernel_sizes=(
+ 4,
+ 4,
+ ),
+ deconv_out_channels=(
+ 256,
+ 256,
+ ),
+ in_channels=1280,
+ loss=dict(type='KeypointMSELoss', use_target_weight=True),
+ out_channels=17,
+ type='HeatmapHead'),
+ test_cfg=dict(flip_mode='heatmap', flip_test=True, shift_heatmap=False),
+ type='TopdownPoseEstimator')
+num_workers = 4
+optim_wrapper = dict(
+ clip_grad=dict(max_norm=1.0, norm_type=2),
+ constructor='LayerDecayOptimWrapperConstructor',
+ dtype='float16',
+ optimizer=dict(
+ betas=(
+ 0.9,
+ 0.999,
+ ), lr=5e-05, type='AdamW', weight_decay=0.1),
+ paramwise_cfg=dict(
+ custom_keys=dict(
+ bias=dict(decay_multi=0.0),
+ norm=dict(decay_mult=0.0),
+ pos_embed=dict(decay_mult=0.0),
+ relative_position_bias_table=dict(decay_mult=0.0)),
+ layer_decay_rate=0.85,
+ num_layers=32),
+ type='AmpOptimWrapper')
+param_scheduler = [
+ dict(
+ begin=0,
+ by_epoch=True,
+ end=20,
+ gamma=0.1,
+ milestones=[
+ 14,
+ 18,
+ ],
+ type='MultiStepLR'),
+]
+resume = False
+test_cfg = dict()
+test_dataloader = dict(
+ batch_size=32,
+ dataset=dict(
+ ann_file='annotations/person_keypoints_val2017.json',
+ bbox_file=
+ 'data/coco/person_detection_results/COCO_val2017_detections_AP_H_56_person.json',
+ data_mode='topdown',
+ data_prefix=dict(img='val2017/'),
+ data_root='data/coco/',
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+test_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+train_cfg = dict(by_epoch=True, max_epochs=20, val_interval=1)
+train_dataloader = dict(
+ batch_size=8,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+ ],
+ type='CocoDataset'),
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(shuffle=True, type='DefaultSampler'))
+train_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+]
+val_cfg = dict()
+val_dataloader = dict(
+ batch_size=8,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ bbox_file=None,
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/root/miko/puni/train/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+val_evaluator = dict(
+ ann_file=
+ '/root/miko/puni/train/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+val_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+]
+vis_backends = [
+ dict(type='LocalVisBackend'),
+]
+visualizer = dict(
+ name='visualizer',
+ type='PoseLocalVisualizer',
+ vis_backends=[
+ dict(type='LocalVisBackend'),
+ ])
+work_dir = '../work_dirs/vitpose_finetune'
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/best_coco_AP_epoch_4.pth b/third_party/GVHMR/work_dirs/vitpose_finetune/best_coco_AP_epoch_4.pth
new file mode 100644
index 0000000000000000000000000000000000000000..a6c239a72c0a9bc0018c66f27a405c6b44938bcd
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/best_coco_AP_epoch_4.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b70e78ed66a196b7add605dcbf462c3bdf7b403cd03477a13c35d9e31377ce64
+size 2549057548
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/epoch_4.pth b/third_party/GVHMR/work_dirs/vitpose_finetune/epoch_4.pth
new file mode 100644
index 0000000000000000000000000000000000000000..9ff56ea9d85b4124d90f2f5fac839a42151035ea
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/epoch_4.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7f8486d0cf827e2be4afc46a456b382f079382e9e8930a62d532b3e93abf49bf
+size 2549059148
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/last_checkpoint b/third_party/GVHMR/work_dirs/vitpose_finetune/last_checkpoint
new file mode 100644
index 0000000000000000000000000000000000000000..51ada83f688881b55bec13cfe703b88009cf41b7
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/last_checkpoint
@@ -0,0 +1 @@
+/root/miko/puni/train/GVHMR/work_dirs/vitpose_finetune/epoch_4.pth
\ No newline at end of file
diff --git a/third_party/GVHMR/work_dirs/vitpose_finetune/vitpose_huge_finetune.py b/third_party/GVHMR/work_dirs/vitpose_finetune/vitpose_huge_finetune.py
new file mode 100644
index 0000000000000000000000000000000000000000..5ec128fd2e544d90c2334c020a589061273ab941
--- /dev/null
+++ b/third_party/GVHMR/work_dirs/vitpose_finetune/vitpose_huge_finetune.py
@@ -0,0 +1,285 @@
+auto_scale_lr = None
+backend_args = dict(backend='local')
+batch_size_per_gpu = 8
+codec = dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap')
+custom_hooks = [
+ dict(type='SyncBuffersHook'),
+]
+custom_imports = dict(
+ allow_failed_imports=False,
+ imports=[
+ 'mmpose.engine.optim_wrappers.layer_decay_optim_wrapper',
+ ])
+data_mode = 'topdown'
+data_root = '/workspace/GVHMR/processed_data/'
+dataset_type = 'CocoDataset'
+default_hooks = dict(
+ badcase=dict(
+ badcase_thr=5,
+ enable=False,
+ metric_type='loss',
+ out_dir='badcase',
+ type='BadCaseAnalysisHook'),
+ checkpoint=dict(
+ interval=1,
+ max_keep_ckpts=2,
+ rule='greater',
+ save_best='coco/AP',
+ save_optimizer=False,
+ type='CheckpointHook'),
+ logger=dict(interval=50, type='LoggerHook'),
+ param_scheduler=dict(type='ParamSchedulerHook'),
+ sampler_seed=dict(type='DistSamplerSeedHook'),
+ timer=dict(type='IterTimerHook'),
+ visualization=dict(
+ enable=True,
+ interval=100,
+ out_dir='vis_results',
+ type='PoseVisualizationHook'))
+default_scope = 'mmpose'
+env_cfg = dict(
+ cudnn_benchmark=False,
+ dist_cfg=dict(backend='nccl'),
+ mp_cfg=dict(mp_start_method='fork', opencv_num_threads=0))
+launcher = 'none'
+load_from = '/workspace/td-hm_ViTPose-huge_8xb64-210e_coco-256x192-e32adcd4_20230314.pth'
+log_level = 'INFO'
+log_processor = dict(
+ by_epoch=True, num_digits=6, type='LogProcessor', window_size=50)
+metainfo = dict(from_file='configs/_base_/datasets/coco.py')
+model = dict(
+ backbone=dict(
+ arch='huge',
+ drop_path_rate=0.55,
+ img_size=(
+ 256,
+ 192,
+ ),
+ init_cfg=None,
+ out_type='featmap',
+ patch_cfg=dict(padding=2),
+ patch_size=16,
+ qkv_bias=True,
+ type='mmpretrain.VisionTransformer',
+ with_cls_token=False),
+ data_preprocessor=dict(
+ bgr_to_rgb=True,
+ mean=[
+ 123.675,
+ 116.28,
+ 103.53,
+ ],
+ std=[
+ 58.395,
+ 57.12,
+ 57.375,
+ ],
+ type='PoseDataPreprocessor'),
+ head=dict(
+ decoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ deconv_kernel_sizes=(
+ 4,
+ 4,
+ ),
+ deconv_out_channels=(
+ 256,
+ 256,
+ ),
+ in_channels=1280,
+ loss=dict(type='KeypointMSELoss', use_target_weight=True),
+ out_channels=17,
+ type='HeatmapHead'),
+ test_cfg=dict(flip_mode='heatmap', flip_test=True, shift_heatmap=False),
+ type='TopdownPoseEstimator')
+num_workers = 4
+optim_wrapper = dict(
+ clip_grad=dict(max_norm=1.0, norm_type=2),
+ constructor='LayerDecayOptimWrapperConstructor',
+ dtype='float16',
+ optimizer=dict(
+ betas=(
+ 0.9,
+ 0.999,
+ ), lr=5e-05, type='AdamW', weight_decay=0.1),
+ paramwise_cfg=dict(
+ custom_keys=dict(
+ bias=dict(decay_multi=0.0),
+ norm=dict(decay_mult=0.0),
+ pos_embed=dict(decay_mult=0.0),
+ relative_position_bias_table=dict(decay_mult=0.0)),
+ layer_decay_rate=0.85,
+ num_layers=32),
+ type='AmpOptimWrapper')
+param_scheduler = [
+ dict(
+ begin=0,
+ by_epoch=True,
+ end=20,
+ gamma=0.1,
+ milestones=[
+ 14,
+ 18,
+ ],
+ type='MultiStepLR'),
+]
+resume = False
+test_cfg = dict()
+test_dataloader = dict(
+ batch_size=32,
+ dataset=dict(
+ ann_file='annotations/person_keypoints_val2017.json',
+ bbox_file=
+ 'data/coco/person_detection_results/COCO_val2017_detections_AP_H_56_person.json',
+ data_mode='topdown',
+ data_prefix=dict(img='val2017/'),
+ data_root='data/coco/',
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+test_evaluator = dict(
+ ann_file=
+ '/workspace/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+train_cfg = dict(by_epoch=True, max_epochs=20, val_interval=1)
+train_dataloader = dict(
+ batch_size=8,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/workspace/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+ ],
+ type='CocoDataset'),
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(shuffle=True, type='DefaultSampler'))
+train_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(direction='horizontal', type='RandomFlip'),
+ dict(type='RandomHalfBody'),
+ dict(type='RandomBBoxTransform'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(
+ encoder=dict(
+ heatmap_size=(
+ 48,
+ 64,
+ ),
+ input_size=(
+ 192,
+ 256,
+ ),
+ sigma=2,
+ type='UDPHeatmap'),
+ type='GenerateTarget'),
+ dict(type='PackPoseInputs'),
+]
+val_cfg = dict()
+val_dataloader = dict(
+ batch_size=8,
+ dataset=dict(
+ ann_file='vitpose/annotations/person_keypoints_train.json',
+ bbox_file=None,
+ data_mode='topdown',
+ data_prefix=dict(img=''),
+ data_root='/workspace/GVHMR/processed_data/',
+ metainfo=dict(from_file='configs/_base_/datasets/coco.py'),
+ pipeline=[
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+ ],
+ test_mode=True,
+ type='CocoDataset'),
+ drop_last=False,
+ num_workers=4,
+ persistent_workers=True,
+ sampler=dict(round_up=False, shuffle=False, type='DefaultSampler'))
+val_evaluator = dict(
+ ann_file=
+ '/workspace/GVHMR/processed_data/vitpose/annotations/person_keypoints_train.json',
+ type='CocoMetric')
+val_pipeline = [
+ dict(type='LoadImage'),
+ dict(type='GetBBoxCenterScale'),
+ dict(input_size=(
+ 192,
+ 256,
+ ), type='TopdownAffine', use_udp=True),
+ dict(type='PackPoseInputs'),
+]
+vis_backends = [
+ dict(type='LocalVisBackend'),
+]
+visualizer = dict(
+ name='visualizer',
+ type='PoseLocalVisualizer',
+ vis_backends=[
+ dict(type='LocalVisBackend'),
+ ])
+work_dir = '../work_dirs/vitpose_finetune'
diff --git a/third_party/hamer/.dockerignore b/third_party/hamer/.dockerignore
new file mode 100644
index 0000000000000000000000000000000000000000..72c966ee7e809a7e6f4573030acce21cdba4bfbb
--- /dev/null
+++ b/third_party/hamer/.dockerignore
@@ -0,0 +1,13 @@
+# Virtual environment ignores:
+venv/
+.venv/
+.hamer/
+
+# Ignoring the .tar.gz files because of the fetching scripts:
+*.tar.gz
+_DATA/
+
+# Ignoring .zip files becuase of the mano_v1_2.zip download with
+# the MANO_RIGHT.pkl file inside:
+*.zip
+mano_v1_2/
\ No newline at end of file
diff --git a/third_party/hamer/.gitignore b/third_party/hamer/.gitignore
new file mode 100644
index 0000000000000000000000000000000000000000..f7975d8606aed6f4212877ce6d5fdfe68bbe56df
--- /dev/null
+++ b/third_party/hamer/.gitignore
@@ -0,0 +1,154 @@
+# Specific
+/logs*/
+/results/
+/sandbox/
+*.lock
+*.pt
+*.npy
+/example_data/downloaded*
+*.tar
+*.tar.gz
+/discord_sandbox/
+/demo_out/
+token_channel.csv
+/_DATA/
+/hamer_training_data/
+/outputs/
+
+mano_v1_2/
+
+
+# Byte-compiled / optimized / DLL files
+__pycache__/
+*.py[cod]
+*$py.class
+
+# C extensions
+*.so
+
+# Distribution / packaging
+.Python
+build/
+develop-eggs/
+dist/
+downloads/
+eggs/
+.eggs/
+lib/
+lib64/
+parts/
+sdist/
+var/
+wheels/
+pip-wheel-metadata/
+share/python-wheels/
+*.egg-info/
+.installed.cfg
+*.egg
+MANIFEST
+
+# PyInstaller
+# Usually these files are written by a python script from a template
+# before PyInstaller builds the exe, so as to inject date/other infos into it.
+*.manifest
+*.spec
+
+# Installer logs
+pip-log.txt
+pip-delete-this-directory.txt
+
+# Unit test / coverage reports
+htmlcov/
+.tox/
+.nox/
+.coverage
+.coverage.*
+.cache
+nosetests.xml
+coverage.xml
+*.cover
+*.py,cover
+.hypothesis/
+.pytest_cache/
+
+# Translations
+*.mo
+*.pot
+
+# Django stuff:
+*.log
+local_settings.py
+db.sqlite3
+db.sqlite3-journal
+
+# Flask stuff:
+instance/
+.webassets-cache
+
+# Scrapy stuff:
+.scrapy
+
+# Sphinx documentation
+docs/_build/
+
+# PyBuilder
+target/
+
+# Jupyter Notebook
+.ipynb_checkpoints
+
+# IPython
+profile_default/
+ipython_config.py
+
+# pyenv
+.python-version
+
+# pipenv
+# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
+# However, in case of collaboration, if having platform-specific dependencies or dependencies
+# having no cross-platform support, pipenv may install dependencies that don't work, or not
+# install all needed dependencies.
+#Pipfile.lock
+
+# PEP 582; used by e.g. github.com/David-OConnor/pyflow
+__pypackages__/
+
+# Celery stuff
+celerybeat-schedule
+celerybeat.pid
+
+# SageMath parsed files
+*.sage.py
+
+# Environments
+.env
+.venv
+env/
+venv/
+ENV/
+env.bak/
+venv.bak/
+.hamer/
+
+# Spyder project settings
+.spyderproject
+.spyproject
+
+# Rope project settings
+.ropeproject
+
+# mkdocs documentation
+/site
+
+# mypy
+.mypy_cache/
+.dmypy.json
+dmypy.json
+
+# Pyre type checker
+.pyre/
+/checkpoints/
+/data/
+
+*.zip
diff --git a/third_party/hamer/.gitmodules b/third_party/hamer/.gitmodules
new file mode 100644
index 0000000000000000000000000000000000000000..03f59f27837f42c6a3778f484ecec3c6e54a97b7
--- /dev/null
+++ b/third_party/hamer/.gitmodules
@@ -0,0 +1,4 @@
+[submodule "third-party/ViTPose"]
+ path = third-party/ViTPose
+ url = https://github.com/ViTAE-Transformer/ViTPose.git
+ branch = main
diff --git a/third_party/hamer/LICENSE.md b/third_party/hamer/LICENSE.md
new file mode 100644
index 0000000000000000000000000000000000000000..209a6f5b97b2fc76a146c550a1a8d1d91af5eba7
--- /dev/null
+++ b/third_party/hamer/LICENSE.md
@@ -0,0 +1,21 @@
+MIT License
+
+Copyright (c) 2023 UC Regents, Georgios Pavlakos
+
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.
diff --git a/third_party/hamer/README.md b/third_party/hamer/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..697943c58055d7bf97c98bd406f521b927b9400d
--- /dev/null
+++ b/third_party/hamer/README.md
@@ -0,0 +1,129 @@
+# HaMeR: Hand Mesh Recovery
+Code repository for the paper:
+**Reconstructing Hands in 3D with Transformers**
+
+[Georgios Pavlakos](https://geopavlakos.github.io/), [Dandan Shan](https://ddshan.github.io/), [Ilija Radosavovic](https://people.eecs.berkeley.edu/~ilija/), [Angjoo Kanazawa](https://people.eecs.berkeley.edu/~kanazawa/), [David Fouhey](https://cs.nyu.edu/~fouhey/), [Jitendra Malik](http://people.eecs.berkeley.edu/~malik/)
+
+[](https://arxiv.org/pdf/2312.05251.pdf) [](https://geopavlakos.github.io/hamer/) [](https://colab.research.google.com/drive/1rQbQzegFWGVOm1n1d-S6koOWDo7F2ucu?usp=sharing) [](https://huggingface.co/spaces/geopavlakos/HaMeR)
+
+
+
+## News
+
+- [2024/06] HaMeR received the 2nd place award in the Ego-Pose Hands task of the Ego-Exo4D Challenge! Please check the [validation report](https://www.cs.utexas.edu/~pavlakos/hamer/resources/egoexo4d_challenge.pdf).
+- [2024/05] We have released the evaluation pipeline!
+- [2024/05] We have released the HInt dataset annotations! Please check [here](https://github.com/ddshan/hint).
+- [2023/12] Original release!
+
+## Installation
+First you need to clone the repo:
+```
+git clone --recursive https://github.com/geopavlakos/hamer.git
+cd hamer
+```
+
+We recommend creating a virtual environment for HaMeR. You can use venv:
+```bash
+python3.10 -m venv .hamer
+source .hamer/bin/activate
+```
+
+or alternatively conda:
+```bash
+conda create --name hamer python=3.10
+conda activate hamer
+```
+
+Then, you can install the rest of the dependencies. This is for CUDA 11.7, but you can adapt accordingly:
+```bash
+pip install torch torchvision --index-url https://download.pytorch.org/whl/cu117
+pip install -e .[all]
+pip install -v -e third-party/ViTPose
+```
+
+You also need to download the trained models:
+```bash
+bash fetch_demo_data.sh
+```
+
+Besides these files, you also need to download the MANO model. Please visit the [MANO website](https://mano.is.tue.mpg.de) and register to get access to the downloads section. We only require the right hand model. You need to put `MANO_RIGHT.pkl` under the `_DATA/data/mano` folder.
+
+### Docker Compose
+
+If you wish to use HaMeR with Docker, you can use the following command:
+
+```
+docker compose -f ./docker/docker-compose.yml up -d
+```
+
+After the image is built successfully, enter the container and run the steps as above:
+
+```
+docker compose -f ./docker/docker-compose.yml exec hamer-dev /bin/bash
+```
+
+Continue with the installation steps:
+
+```bash
+bash fetch_demo_data.sh
+```
+
+## Demo
+```bash
+python demo.py \
+ --img_folder example_data --out_folder demo_out \
+ --batch_size=48 --side_view --save_mesh --full_frame
+```
+
+## HInt Dataset
+We have released the annotations for the HInt dataset. Please follow the instructions [here](https://github.com/ddshan/hint)
+
+## Training
+First, download the training data to `./hamer_training_data/` by running:
+```
+bash fetch_training_data.sh
+```
+
+Then you can start training using the following command:
+```
+python train.py exp_name=hamer data=mix_all experiment=hamer_vit_transformer trainer=gpu launcher=local
+```
+Checkpoints and logs will be saved to `./logs/`.
+
+## Evaluation
+Download the [evaluation metadata](https://www.dropbox.com/scl/fi/7ip2vnnu355e2kqbyn1bc/hamer_evaluation_data.tar.gz?rlkey=nb4x10uc8mj2qlfq934t5mdlh) to `./hamer_evaluation_data/`. Additionally, download the FreiHAND, HO-3D, and HInt dataset images and update the corresponding paths in `hamer/configs/datasets_eval.yaml`.
+
+Run evaluation on multiple datasets as follows, results are stored in `results/eval_regression.csv`.
+```bash
+python eval.py --dataset 'FREIHAND-VAL,HO3D-VAL,NEWDAYS-TEST-ALL,NEWDAYS-TEST-VIS,NEWDAYS-TEST-OCC,EPICK-TEST-ALL,EPICK-TEST-VIS,EPICK-TEST-OCC,EGO4D-TEST-ALL,EGO4D-TEST-VIS,EGO4D-TEST-OCC'
+```
+
+Results for HInt are stored in `results/eval_regression.csv`. For [FreiHAND](https://github.com/lmb-freiburg/freihand) and [HO-3D](https://codalab.lisn.upsaclay.fr/competitions/4318) you get as output a `.json` file that can be used for evaluation using their corresponding evaluation processes.
+
+## Acknowledgements
+Parts of the code are taken or adapted from the following repos:
+- [4DHumans](https://github.com/shubham-goel/4D-Humans)
+- [SLAHMR](https://github.com/vye16/slahmr)
+- [ProHMR](https://github.com/nkolot/ProHMR)
+- [SPIN](https://github.com/nkolot/SPIN)
+- [SMPLify-X](https://github.com/vchoutas/smplify-x)
+- [HMR](https://github.com/akanazawa/hmr)
+- [ViTPose](https://github.com/ViTAE-Transformer/ViTPose)
+- [Detectron2](https://github.com/facebookresearch/detectron2)
+
+Additionally, we thank [StabilityAI](https://stability.ai/) for a generous compute grant that enabled this work.
+
+## Open-Source Contributions
+- [Wentao Hu](https://vincenthu19.github.io/) integrated the hand parameters predicted by HaMeR into SMPL-X - [Mano2Smpl-X](https://github.com/VincentHu19/Mano2Smpl-X)
+
+## Citing
+If you find this code useful for your research, please consider citing the following paper:
+
+```bibtex
+@inproceedings{pavlakos2024reconstructing,
+ title={Reconstructing Hands in 3{D} with Transformers},
+ author={Pavlakos, Georgios and Shan, Dandan and Radosavovic, Ilija and Kanazawa, Angjoo and Fouhey, David and Malik, Jitendra},
+ booktitle={CVPR},
+ year={2024}
+}
+```
diff --git a/third_party/hamer/_DATA/data/mano/MANO_RIGHT.pkl b/third_party/hamer/_DATA/data/mano/MANO_RIGHT.pkl
new file mode 100644
index 0000000000000000000000000000000000000000..8e7ac7faf64ad51096ec1da626ea13757ed7f665
--- /dev/null
+++ b/third_party/hamer/_DATA/data/mano/MANO_RIGHT.pkl
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:45d60aa3b27ef9107a7afd4e00808f307fd91111e1cfa35afd5c4a62de264767
+size 3821356
diff --git a/third_party/hamer/_DATA/data/mano/MANO_RIGHT.pkl:Zone.Identifier b/third_party/hamer/_DATA/data/mano/MANO_RIGHT.pkl:Zone.Identifier
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/third_party/hamer/_DATA/data/mano_mean_params.npz b/third_party/hamer/_DATA/data/mano_mean_params.npz
new file mode 100644
index 0000000000000000000000000000000000000000..dc294b01fb78a9cd6636c87a69b59cf82d28d15b
--- /dev/null
+++ b/third_party/hamer/_DATA/data/mano_mean_params.npz
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:efc0ec58e4a5cef78f3abfb4e8f91623b8950be9eff8b8e0dbb0d036ebc63988
+size 1178
diff --git a/third_party/hamer/_DATA/hamer_ckpts/checkpoints/hamer.ckpt b/third_party/hamer/_DATA/hamer_ckpts/checkpoints/hamer.ckpt
new file mode 100644
index 0000000000000000000000000000000000000000..c5d0dae12e9a553336d196e22dea6b4ed74df351
--- /dev/null
+++ b/third_party/hamer/_DATA/hamer_ckpts/checkpoints/hamer.ckpt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e5cc06f294d88a92dee24e603480aab04de532b49f0e08200804ee7d90e16f53
+size 2689536166
diff --git a/third_party/hamer/_DATA/hamer_ckpts/dataset_config.yaml b/third_party/hamer/_DATA/hamer_ckpts/dataset_config.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..77b67251770062f769fdddfb0c8ffa4cc7720a80
--- /dev/null
+++ b/third_party/hamer/_DATA/hamer_ckpts/dataset_config.yaml
@@ -0,0 +1,42 @@
+COCOW-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/cocow-train/{000000..000036}.tar
+ epoch_size: 78666
+DEX-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/dex-train/{000000..000406}.tar
+ epoch_size: 406888
+FREIHAND-MOCAP:
+ DATASET_FILE: hamer_training_data/freihand_mocap.npz
+FREIHAND-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/freihand-train/{000000..000130}.tar
+ epoch_size: 130240
+H2O3D-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/h2o3d-train/{000000..000060}.tar
+ epoch_size: 121996
+HALPE-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/halpe-train/{000000..000022}.tar
+ epoch_size: 34289
+HO3D-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/ho3d-train/{000000..000083}.tar
+ epoch_size: 83325
+INTERHAND26M-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/interhand26m-train/{000000..001056}.tar
+ epoch_size: 1424632
+MPIINZSL-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/mpiinzsl-train/{000000..000015}.tar
+ epoch_size: 15184
+MTC-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/mtc-train/{000000..000306}.tar
+ epoch_size: 363947
+RHD-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/rhd-train/{000000..000041}.tar
+ epoch_size: 61705
diff --git a/third_party/hamer/_DATA/hamer_ckpts/model_config.yaml b/third_party/hamer/_DATA/hamer_ckpts/model_config.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..035c0685c809135e1b05c670486f2f70a8f3e5f2
--- /dev/null
+++ b/third_party/hamer/_DATA/hamer_ckpts/model_config.yaml
@@ -0,0 +1,111 @@
+task_name: train
+tags:
+- dev
+train: true
+test: false
+ckpt_path: null
+seed: null
+DATASETS:
+ TRAIN:
+ FREIHAND-TRAIN:
+ WEIGHT: 0.25
+ INTERHAND26M-TRAIN:
+ WEIGHT: 0.25
+ MTC-TRAIN:
+ WEIGHT: 0.1
+ RHD-TRAIN:
+ WEIGHT: 0.05
+ COCOW-TRAIN:
+ WEIGHT: 0.1
+ HALPE-TRAIN:
+ WEIGHT: 0.05
+ MPIINZSL-TRAIN:
+ WEIGHT: 0.05
+ HO3D-TRAIN:
+ WEIGHT: 0.05
+ H2O3D-TRAIN:
+ WEIGHT: 0.05
+ DEX-TRAIN:
+ WEIGHT: 0.05
+ VAL:
+ FREIHAND-TRAIN:
+ WEIGHT: 1.0
+ MOCAP: FREIHAND-MOCAP
+ BETAS_REG: true
+ CONFIG:
+ SCALE_FACTOR: 0.3
+ ROT_FACTOR: 30
+ TRANS_FACTOR: 0.02
+ COLOR_SCALE: 0.2
+ ROT_AUG_RATE: 0.6
+ TRANS_AUG_RATE: 0.5
+ DO_FLIP: false
+ FLIP_AUG_RATE: 0.0
+ EXTREME_CROP_AUG_RATE: 0.0
+ EXTREME_CROP_AUG_LEVEL: 1
+extras:
+ ignore_warnings: false
+ enforce_tags: true
+ print_config: true
+exp_name: hamer
+MANO:
+ DATA_DIR: _DATA/data/
+ MODEL_PATH: data/mano
+ GENDER: neutral
+ NUM_HAND_JOINTS: 15
+ MEAN_PARAMS: data/mano_mean_params.npz
+ CREATE_BODY_POSE: false
+EXTRA:
+ FOCAL_LENGTH: 5000
+ NUM_LOG_IMAGES: 4
+ NUM_LOG_SAMPLES_PER_IMAGE: 8
+ PELVIS_IND: 0
+GENERAL:
+ TOTAL_STEPS: 1000000
+ LOG_STEPS: 1000
+ VAL_STEPS: 1000
+ CHECKPOINT_STEPS: 10000
+ CHECKPOINT_SAVE_TOP_K: 1
+ NUM_WORKERS: 8
+ PREFETCH_FACTOR: 2
+TRAIN:
+ LR: 1.0e-05
+ WEIGHT_DECAY: 0.0001
+ BATCH_SIZE: 32
+ LOSS_REDUCTION: mean
+ NUM_TRAIN_SAMPLES: 2
+ NUM_TEST_SAMPLES: 64
+ POSE_2D_NOISE_RATIO: 0.01
+ SMPL_PARAM_NOISE_RATIO: 0.005
+MODEL:
+ IMAGE_SIZE: 256
+ IMAGE_MEAN:
+ - 0.485
+ - 0.456
+ - 0.406
+ IMAGE_STD:
+ - 0.229
+ - 0.224
+ - 0.225
+ BACKBONE:
+ TYPE: vit
+ PRETRAINED_WEIGHTS: hamer_training_data/vitpose_backbone.pth
+ MANO_HEAD:
+ TYPE: transformer_decoder
+ IN_CHANNELS: 2048
+ TRANSFORMER_DECODER:
+ depth: 6
+ heads: 8
+ mlp_dim: 1024
+ dim_head: 64
+ dropout: 0.0
+ emb_dropout: 0.0
+ norm: layer
+ context_dim: 1280
+LOSS_WEIGHTS:
+ KEYPOINTS_3D: 0.05
+ KEYPOINTS_2D: 0.01
+ GLOBAL_ORIENT: 0.001
+ HAND_POSE: 0.001
+ BETAS: 0.0005
+ ADVERSARIAL: 0.0005
diff --git a/third_party/hamer/demo.py b/third_party/hamer/demo.py
new file mode 100644
index 0000000000000000000000000000000000000000..d8a414149fa0f044230e97e2842e7be389e3dbce
--- /dev/null
+++ b/third_party/hamer/demo.py
@@ -0,0 +1,208 @@
+from pathlib import Path
+import torch
+import argparse
+import os
+import cv2
+import numpy as np
+
+from hamer.configs import CACHE_DIR_HAMER
+from hamer.models import HAMER, download_models, load_hamer, DEFAULT_CHECKPOINT
+from hamer.utils import recursive_to
+from hamer.datasets.vitdet_dataset import ViTDetDataset, DEFAULT_MEAN, DEFAULT_STD
+from hamer.utils.renderer import Renderer, cam_crop_to_full
+
+LIGHT_BLUE=(0.65098039, 0.74117647, 0.85882353)
+
+from vitpose_model import ViTPoseModel
+
+import json
+from typing import Dict, Optional
+
+def main():
+ parser = argparse.ArgumentParser(description='HaMeR demo code')
+ parser.add_argument('--checkpoint', type=str, default=DEFAULT_CHECKPOINT, help='Path to pretrained model checkpoint')
+ parser.add_argument('--img_folder', type=str, default='images', help='Folder with input images')
+ parser.add_argument('--out_folder', type=str, default='out_demo', help='Output folder to save rendered results')
+ parser.add_argument('--side_view', dest='side_view', action='store_true', default=False, help='If set, render side view also')
+ parser.add_argument('--full_frame', dest='full_frame', action='store_true', default=True, help='If set, render all people together also')
+ parser.add_argument('--save_mesh', dest='save_mesh', action='store_true', default=False, help='If set, save meshes to disk also')
+ parser.add_argument('--batch_size', type=int, default=1, help='Batch size for inference/fitting')
+ parser.add_argument('--rescale_factor', type=float, default=2.0, help='Factor for padding the bbox')
+ parser.add_argument('--body_detector', type=str, default='vitdet', choices=['vitdet', 'regnety'], help='Using regnety improves runtime and reduces memory')
+ parser.add_argument('--file_type', nargs='+', default=['*.jpg', '*.png'], help='List of file extensions to consider')
+
+ args = parser.parse_args()
+
+ # Download and load checkpoints
+ download_models(CACHE_DIR_HAMER)
+ model, model_cfg = load_hamer(args.checkpoint)
+
+ # Setup HaMeR model
+ device = torch.device('cuda') if torch.cuda.is_available() else torch.device('cpu')
+ model = model.to(device)
+ model.eval()
+
+ # Load detector
+ from hamer.utils.utils_detectron2 import DefaultPredictor_Lazy
+ if args.body_detector == 'vitdet':
+ from detectron2.config import LazyConfig
+ import hamer
+ cfg_path = Path(hamer.__file__).parent/'configs'/'cascade_mask_rcnn_vitdet_h_75ep.py'
+ detectron2_cfg = LazyConfig.load(str(cfg_path))
+ detectron2_cfg.train.init_checkpoint = "https://dl.fbaipublicfiles.com/detectron2/ViTDet/COCO/cascade_mask_rcnn_vitdet_h/f328730692/model_final_f05665.pkl"
+ for i in range(3):
+ detectron2_cfg.model.roi_heads.box_predictors[i].test_score_thresh = 0.25
+ detector = DefaultPredictor_Lazy(detectron2_cfg)
+ elif args.body_detector == 'regnety':
+ from detectron2 import model_zoo
+ from detectron2.config import get_cfg
+ detectron2_cfg = model_zoo.get_config('new_baselines/mask_rcnn_regnety_4gf_dds_FPN_400ep_LSJ.py', trained=True)
+ detectron2_cfg.model.roi_heads.box_predictor.test_score_thresh = 0.5
+ detectron2_cfg.model.roi_heads.box_predictor.test_nms_thresh = 0.4
+ detector = DefaultPredictor_Lazy(detectron2_cfg)
+
+ # keypoint detector
+ cpm = ViTPoseModel(device)
+
+ # Setup the renderer
+ renderer = Renderer(model_cfg, faces=model.mano.faces)
+
+ # Make output directory if it does not exist
+ os.makedirs(args.out_folder, exist_ok=True)
+
+ # Get all demo images ends with .jpg or .png
+ img_paths = [img for end in args.file_type for img in Path(args.img_folder).glob(end)]
+
+ # Iterate over all images in folder
+ for img_path in img_paths:
+ img_cv2 = cv2.imread(str(img_path))
+
+ # Detect humans in image
+ det_out = detector(img_cv2)
+ img = img_cv2.copy()[:, :, ::-1]
+
+ det_instances = det_out['instances']
+ valid_idx = (det_instances.pred_classes==0) & (det_instances.scores > 0.5)
+ pred_bboxes=det_instances.pred_boxes.tensor[valid_idx].cpu().numpy()
+ pred_scores=det_instances.scores[valid_idx].cpu().numpy()
+
+ # Detect human keypoints for each person
+ vitposes_out = cpm.predict_pose(
+ img,
+ [np.concatenate([pred_bboxes, pred_scores[:, None]], axis=1)],
+ )
+
+ bboxes = []
+ is_right = []
+
+ # Use hands based on hand keypoint detections
+ for vitposes in vitposes_out:
+ left_hand_keyp = vitposes['keypoints'][-42:-21]
+ right_hand_keyp = vitposes['keypoints'][-21:]
+
+ # Rejecting not confident detections
+ keyp = left_hand_keyp
+ valid = keyp[:,2] > 0.5
+ if sum(valid) > 3:
+ bbox = [keyp[valid,0].min(), keyp[valid,1].min(), keyp[valid,0].max(), keyp[valid,1].max()]
+ bboxes.append(bbox)
+ is_right.append(0)
+ keyp = right_hand_keyp
+ valid = keyp[:,2] > 0.5
+ if sum(valid) > 3:
+ bbox = [keyp[valid,0].min(), keyp[valid,1].min(), keyp[valid,0].max(), keyp[valid,1].max()]
+ bboxes.append(bbox)
+ is_right.append(1)
+
+ if len(bboxes) == 0:
+ continue
+
+ boxes = np.stack(bboxes)
+ right = np.stack(is_right)
+
+ # Run reconstruction on all detected hands
+ dataset = ViTDetDataset(model_cfg, img_cv2, boxes, right, rescale_factor=args.rescale_factor)
+ dataloader = torch.utils.data.DataLoader(dataset, batch_size=8, shuffle=False, num_workers=0)
+
+ all_verts = []
+ all_cam_t = []
+ all_right = []
+
+ for batch in dataloader:
+ batch = recursive_to(batch, device)
+ with torch.no_grad():
+ out = model(batch)
+
+ multiplier = (2*batch['right']-1)
+ pred_cam = out['pred_cam']
+ pred_cam[:,1] = multiplier*pred_cam[:,1]
+ box_center = batch["box_center"].float()
+ box_size = batch["box_size"].float()
+ img_size = batch["img_size"].float()
+ multiplier = (2*batch['right']-1)
+ scaled_focal_length = model_cfg.EXTRA.FOCAL_LENGTH / model_cfg.MODEL.IMAGE_SIZE * img_size.max()
+ pred_cam_t_full = cam_crop_to_full(pred_cam, box_center, box_size, img_size, scaled_focal_length).detach().cpu().numpy()
+
+ # Render the result
+ batch_size = batch['img'].shape[0]
+ for n in range(batch_size):
+ # Get filename from path img_path
+ img_fn, _ = os.path.splitext(os.path.basename(img_path))
+ person_id = int(batch['personid'][n])
+ white_img = (torch.ones_like(batch['img'][n]).cpu() - DEFAULT_MEAN[:,None,None]/255) / (DEFAULT_STD[:,None,None]/255)
+ input_patch = batch['img'][n].cpu() * (DEFAULT_STD[:,None,None]/255) + (DEFAULT_MEAN[:,None,None]/255)
+ input_patch = input_patch.permute(1,2,0).numpy()
+
+ regression_img = renderer(out['pred_vertices'][n].detach().cpu().numpy(),
+ out['pred_cam_t'][n].detach().cpu().numpy(),
+ batch['img'][n],
+ mesh_base_color=LIGHT_BLUE,
+ scene_bg_color=(1, 1, 1),
+ )
+
+ if args.side_view:
+ side_img = renderer(out['pred_vertices'][n].detach().cpu().numpy(),
+ out['pred_cam_t'][n].detach().cpu().numpy(),
+ white_img,
+ mesh_base_color=LIGHT_BLUE,
+ scene_bg_color=(1, 1, 1),
+ side_view=True)
+ final_img = np.concatenate([input_patch, regression_img, side_img], axis=1)
+ else:
+ final_img = np.concatenate([input_patch, regression_img], axis=1)
+
+ cv2.imwrite(os.path.join(args.out_folder, f'{img_fn}_{person_id}.png'), 255*final_img[:, :, ::-1])
+
+ # Add all verts and cams to list
+ verts = out['pred_vertices'][n].detach().cpu().numpy()
+ is_right = batch['right'][n].cpu().numpy()
+ verts[:,0] = (2*is_right-1)*verts[:,0]
+ cam_t = pred_cam_t_full[n]
+ all_verts.append(verts)
+ all_cam_t.append(cam_t)
+ all_right.append(is_right)
+
+ # Save all meshes to disk
+ if args.save_mesh:
+ camera_translation = cam_t.copy()
+ tmesh = renderer.vertices_to_trimesh(verts, camera_translation, LIGHT_BLUE, is_right=is_right)
+ tmesh.export(os.path.join(args.out_folder, f'{img_fn}_{person_id}.obj'))
+
+ # Render front view
+ if args.full_frame and len(all_verts) > 0:
+ misc_args = dict(
+ mesh_base_color=LIGHT_BLUE,
+ scene_bg_color=(1, 1, 1),
+ focal_length=scaled_focal_length,
+ )
+ cam_view = renderer.render_rgba_multiple(all_verts, cam_t=all_cam_t, render_res=img_size[n], is_right=all_right, **misc_args)
+
+ # Overlay image
+ input_img = img_cv2.astype(np.float32)[:,:,::-1]/255.0
+ input_img = np.concatenate([input_img, np.ones_like(input_img[:,:,:1])], axis=2) # Add alpha channel
+ input_img_overlay = input_img[:,:,:3] * (1-cam_view[:,:,3:]) + cam_view[:,:,:3] * cam_view[:,:,3:]
+
+ cv2.imwrite(os.path.join(args.out_folder, f'{img_fn}_all.jpg'), 255*input_img_overlay[:, :, ::-1])
+
+if __name__ == '__main__':
+ main()
diff --git a/third_party/hamer/docker/docker-compose.yml b/third_party/hamer/docker/docker-compose.yml
new file mode 100644
index 0000000000000000000000000000000000000000..216336f4a270e95f05614deb388f55a1f7128992
--- /dev/null
+++ b/third_party/hamer/docker/docker-compose.yml
@@ -0,0 +1,16 @@
+services:
+ hamer-dev:
+ build:
+ context: ../
+ dockerfile: ./docker/hamer-dev.Dockerfile
+ volumes:
+ - ../:/app
+ tty: true
+ deploy:
+ resources:
+ reservations:
+ devices:
+ - driver: nvidia
+ count: 1 # alternatively, use `count: all` for all GPUs
+ capabilities: [gpu]
+
diff --git a/third_party/hamer/docker/hamer-dev.Dockerfile b/third_party/hamer/docker/hamer-dev.Dockerfile
new file mode 100644
index 0000000000000000000000000000000000000000..fa693e18e3872bd44e1f0707d9f5f3e47656d31d
--- /dev/null
+++ b/third_party/hamer/docker/hamer-dev.Dockerfile
@@ -0,0 +1,53 @@
+ARG BASE=nvidia/cuda:12.6.2-devel-ubuntu22.04
+FROM ${BASE} AS hamer
+
+# Install OS dependencies:
+RUN apt-get update && apt-get upgrade -y
+RUN apt-get install -y --no-install-recommends --fix-missing \
+ gcc g++ \
+ make \
+ python3 python3-dev python3-pip python3-venv python3-wheel \
+ espeak-ng libsndfile1-dev \
+ git \
+ wget \
+ ffmpeg \
+ libsm6 libxext6 \
+ libglfw3-dev libgles2-mesa-dev \
+ && rm -rf /var/lib/apt/lists/*
+
+# Install hamer:
+WORKDIR /app
+
+# Create virtual environment:
+RUN python3 -m venv /opt/venv
+
+# Add virtual environment to PATH
+ENV PATH="/opt/venv/bin:$PATH"
+
+# Activate virtual environment and install dependencies:
+# REVIEW: We need to install/upgrade wheel and setuptools first because otherwise installation fails:
+RUN --mount=type=cache,target=/root/.cache/pip \
+ pip install --upgrade wheel setuptools
+
+# Install torch and torchvision:
+RUN --mount=type=cache,target=/root/.cache/pip \
+ pip install torch==2.2.0 torchvision==0.17.0 --index-url https://download.pytorch.org/whl/cu118
+
+# REVIEW: Numpy is installed separately because otherwise installation fails:
+RUN --mount=type=cache,target=/root/.cache/pip \
+ pip install numpy
+
+# Install gdown (used for fetching scripts):
+RUN --mount=type=cache,target=/root/.cache/pip \
+ pip install gdown
+
+# Install third-party dependencies ViTPose:
+COPY third-party/ third-party/
+RUN --mount=type=cache,target=/root/.cache/pip \
+ pip install -v -e third-party/ViTPose
+
+# Install project dependencies:
+COPY . .
+# Install hamer:
+RUN --mount=type=cache,target=/root/.cache/pip \
+ pip install -e .[all]
diff --git a/third_party/hamer/eval.py b/third_party/hamer/eval.py
new file mode 100644
index 0000000000000000000000000000000000000000..f119d27661fc7361ff6546f469211925f8ab2763
--- /dev/null
+++ b/third_party/hamer/eval.py
@@ -0,0 +1,166 @@
+import argparse
+import os
+import json
+from pathlib import Path
+import traceback
+from typing import List, Optional
+
+import pandas as pd
+import torch
+from filelock import FileLock
+from hamer.configs import dataset_eval_config
+from hamer.datasets import create_dataset
+from hamer.utils import Evaluator, recursive_to
+from tqdm import tqdm
+
+from hamer.configs import CACHE_DIR_HAMER
+from hamer.models import HAMER, download_models, load_hamer, DEFAULT_CHECKPOINT
+
+def main():
+ parser = argparse.ArgumentParser(description='Evaluate trained models')
+ parser.add_argument('--checkpoint', type=str, default=DEFAULT_CHECKPOINT, help='Path to pretrained model checkpoint')
+ parser.add_argument('--results_folder', type=str, default='results', help='Path to results folder.')
+ parser.add_argument('--dataset', type=str, default='FREIHAND-VAL,HO3D-VAL,NEWDAYS-TEST-ALL,NEWDAYS-TEST-VIS,NEWDAYS-TEST-OCC,EPICK-TEST-ALL,EPICK-TEST-VIS,EPICK-TEST-OCC,EGO4D-TEST-ALL,EGO4D-TEST-VIS,EGO4D-TEST-OCC', help='Dataset to evaluate')
+ parser.add_argument('--batch_size', type=int, default=16, help='Batch size for inference')
+ parser.add_argument('--num_samples', type=int, default=1, help='Number of test samples to draw')
+ parser.add_argument('--num_workers', type=int, default=8, help='Number of workers used for data loading')
+ parser.add_argument('--log_freq', type=int, default=10, help='How often to log results')
+ parser.add_argument('--shuffle', dest='shuffle', action='store_true', default=False, help='Shuffle the dataset during evaluation')
+ parser.add_argument('--exp_name', type=str, default=None, help='Experiment name')
+
+ args = parser.parse_args()
+
+ # Download and load checkpoints
+ download_models(CACHE_DIR_HAMER)
+ model, model_cfg = load_hamer(args.checkpoint)
+
+ # Setup HMR2.0 model
+ device = torch.device('cuda') if torch.cuda.is_available() else torch.device('cpu')
+ model = model.to(device)
+ model.eval()
+
+ # Load config and run eval, one dataset at a time
+ print('Evaluating on datasets: {}'.format(args.dataset), flush=True)
+ for dataset in args.dataset.split(','):
+ dataset_cfg = dataset_eval_config()[dataset]
+ args.dataset = dataset
+ run_eval(model, model_cfg, dataset_cfg, device, args)
+
+def run_eval(model, model_cfg, dataset_cfg, device, args):
+
+ # List of metrics to log
+ if args.dataset in ['FREIHAND-VAL', 'HO3D-VAL']:
+ metrics = None
+ preds = ['vertices', 'keypoints_3d']
+ pck_thresholds = None
+ rescale_factor = -1
+ elif args.dataset in ['NEWDAYS-TEST-ALL', 'NEWDAYS-TEST-VIS', 'NEWDAYS-TEST-OCC',
+ 'EPICK-TEST-ALL', 'EPICK-TEST-VIS', 'EPICK-TEST-OCC',
+ 'EGO4D-TEST-ALL', 'EGO4D-TEST-VIS', 'EGO4D-TEST-OCC']:
+ metrics = ['mode_kpl2']
+ preds = None
+ pck_thresholds = [0.05, 0.1, 0.15]
+ rescale_factor = 2
+
+ # Create dataset and data loader
+ dataset = create_dataset(model_cfg, dataset_cfg, train=False, rescale_factor=rescale_factor)
+ dataloader = torch.utils.data.DataLoader(dataset, args.batch_size, shuffle=args.shuffle, num_workers=args.num_workers)
+
+ # Setup evaluator object
+ evaluator = Evaluator(
+ dataset_length=dataset.__len__(),
+ dataset=args.dataset,
+ keypoint_list=dataset_cfg.KEYPOINT_LIST,
+ pelvis_ind=model_cfg.EXTRA.PELVIS_IND,
+ metrics=metrics,
+ preds=preds,
+ pck_thresholds=pck_thresholds,
+ )
+
+ # Go over the images in the dataset.
+ try:
+ for i, batch in enumerate(tqdm(dataloader)):
+ batch = recursive_to(batch, device)
+ with torch.no_grad():
+ out = model(batch)
+ evaluator(out, batch)
+ if i % args.log_freq == args.log_freq - 1:
+ evaluator.log()
+ evaluator.log()
+ error = None
+ except (Exception, KeyboardInterrupt) as e:
+ traceback.print_exc()
+ error = repr(e)
+ i = 0
+
+ # Append results to file
+ if metrics is not None:
+ metrics_dict = evaluator.get_metrics_dict()
+ results_csv = os.path.join(args.results_folder, 'eval_regression.csv')
+ save_eval_result(results_csv, metrics_dict, args.checkpoint, args.dataset, error=error, iters_done=i, exp_name=args.exp_name)
+ if preds is not None:
+ results_json = os.path.join(args.results_folder, '%s.json' % args.dataset.lower())
+ preds_dict = evaluator.get_preds_dict()
+ save_preds_result(results_json, preds_dict)
+
+def save_eval_result(
+ csv_path: str,
+ metric_dict: float,
+ checkpoint_path: str,
+ dataset_name: str,
+ # start_time: pd.Timestamp,
+ error: Optional[str] = None,
+ iters_done=None,
+ exp_name=None,
+) -> None:
+ """Save evaluation results for a single scene file to a common CSV file."""
+
+ timestamp = pd.Timestamp.now()
+ exists: bool = os.path.exists(csv_path)
+ exp_name = exp_name or Path(checkpoint_path).parent.parent.name
+
+ # save each metric as different row to the csv path
+ metric_names = list(metric_dict.keys())
+ metric_values = list(metric_dict.values())
+ N = len(metric_names)
+ df = pd.DataFrame(
+ dict(
+ timestamp=[timestamp] * N,
+ checkpoint_path=[checkpoint_path] * N,
+ exp_name=[exp_name] * N,
+ dataset=[dataset_name] * N,
+ metric_name=metric_names,
+ metric_value=metric_values,
+ error=[error] * N,
+ iters_done=[iters_done] * N,
+ ),
+ index=list(range(N)),
+ )
+
+ # Lock the file to prevent multiple processes from writing to it at the same time.
+ lock = FileLock(f"{csv_path}.lock", timeout=10)
+ with lock:
+ df.to_csv(csv_path, mode="a", header=not exists, index=False)
+
+def save_preds_result(
+ pred_out_path: str,
+ preds_dict: float,
+) -> None:
+ """ Save predictions into a json file. """
+ xyz_pred_list = preds_dict['keypoints_3d']
+ verts_pred_list = preds_dict['vertices']
+ # make sure its only lists
+ xyz_pred_list = [x.tolist() for x in xyz_pred_list]
+ verts_pred_list = [x.tolist() for x in verts_pred_list]
+
+ # save to a json
+ with open(pred_out_path, 'w') as fo:
+ json.dump(
+ [
+ xyz_pred_list,
+ verts_pred_list
+ ], fo)
+ print('Dumped %d joints and %d verts predictions to %s' % (len(xyz_pred_list), len(verts_pred_list), pred_out_path))
+
+if __name__ == '__main__':
+ main()
diff --git a/third_party/hamer/example_data/test1.jpg b/third_party/hamer/example_data/test1.jpg
new file mode 100644
index 0000000000000000000000000000000000000000..d6bef02a40dbe7475e9d49cffb580bc8fe7cd1dd
--- /dev/null
+++ b/third_party/hamer/example_data/test1.jpg
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8ee090a3f1521367e7bdc320a6ed2cbae1f08dff842cb319d816aabafaab9263
+size 103706
diff --git a/third_party/hamer/example_data/test2.jpg b/third_party/hamer/example_data/test2.jpg
new file mode 100644
index 0000000000000000000000000000000000000000..515cac11d1630c0b432e6b46d246bc30e8558510
Binary files /dev/null and b/third_party/hamer/example_data/test2.jpg differ
diff --git a/third_party/hamer/example_data/test3.jpg b/third_party/hamer/example_data/test3.jpg
new file mode 100644
index 0000000000000000000000000000000000000000..5494616ede570d70b7b8a3f199e2205700f3b7ce
Binary files /dev/null and b/third_party/hamer/example_data/test3.jpg differ
diff --git a/third_party/hamer/example_data/test4.jpg b/third_party/hamer/example_data/test4.jpg
new file mode 100644
index 0000000000000000000000000000000000000000..ce19b8cb39f762ea8863ed86ba986c66923157da
--- /dev/null
+++ b/third_party/hamer/example_data/test4.jpg
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d6664110f21dac2ae853e494f0c2333145f4135dd882c367ec3c59566b6dd8f3
+size 499380
diff --git a/third_party/hamer/example_data/test5.jpg b/third_party/hamer/example_data/test5.jpg
new file mode 100644
index 0000000000000000000000000000000000000000..c049a26b92bc79eabe2c0f41f268291a5bfa156e
--- /dev/null
+++ b/third_party/hamer/example_data/test5.jpg
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a2d00fec511a0b9e40dc80335d8ff25c63e0f071013c4033e5dfd9def7c4d19b
+size 169516
diff --git a/third_party/hamer/fetch_demo_data.sh b/third_party/hamer/fetch_demo_data.sh
new file mode 100644
index 0000000000000000000000000000000000000000..16db0fe48a1cc117697c10ac4b572100132b9f76
--- /dev/null
+++ b/third_party/hamer/fetch_demo_data.sh
@@ -0,0 +1,3 @@
+wget https://www.cs.utexas.edu/~pavlakos/hamer/data/hamer_demo_data.tar.gz
+
+tar --warning=no-unknown-keyword --exclude=".*" -xvf hamer_demo_data.tar.gz
diff --git a/third_party/hamer/fetch_training_data.sh b/third_party/hamer/fetch_training_data.sh
new file mode 100644
index 0000000000000000000000000000000000000000..6c33df0892b3047504f8c98d8cd467cc1a11a174
--- /dev/null
+++ b/third_party/hamer/fetch_training_data.sh
@@ -0,0 +1,19 @@
+# Google drive links to download the training data
+gdown https://drive.google.com/uc?id=1BuKEc9qoBVgF8ApTTgAKRFVvDagWHPt7 # hamer_training_data_part1.tar.gz
+gdown https://drive.google.com/uc?id=1lNqBsifaxMP3NHIV_KKJVCT1zDvUn2T_ # hamer_training_data_part2.tar.gz
+gdown https://drive.google.com/uc?id=16xfV_ALY_M3VZeXKpjlB3MhnCNiS5XSq # hamer_training_data_part3.tar.gz
+gdown https://drive.google.com/uc?id=1SqzFHH2-UI6PlGTMd0Ds2FJKALBBOaEO # hamer_training_data_part4a.tar.gz
+gdown https://drive.google.com/uc?id=1xQxEjpaa3WqJt60UMtX_wpnOZ3bpj-u9 # hamer_training_data_part4b.tar.gz
+gdown https://drive.google.com/uc?id=1ozejeAue1M-p4boIfpnpX0E8RFIeHUZH # hamer_training_data_part4c.tar.gz
+
+# Alternatively, consider using the dropbox links:
+#wget -O hamer_training_data_part1.tar.gz https://www.dropbox.com/scl/fi/bqq0jheev3626q1wiijs1/hamer_training_data_part1.tar.gz?rlkey=8fv4ktvk7r3txofd90q0trgxr
+#wget -O hamer_training_data_part2.tar.gz https://www.dropbox.com/scl/fi/l9l5udalchu0mh4qxnw2t/hamer_training_data_part2.tar.gz?rlkey=i0n2lzix4q6jxmhm4sr5rtmkt
+#wget -O hamer_training_data_part3.tar.gz https://www.dropbox.com/scl/fi/6lamcbwt79ri0oj4knwm3/hamer_training_data_part3.tar.gz?rlkey=j5y7ea7xrlu440ud12otaj2ne
+#wget -O hamer_training_data_part4a.tar.gz https://www.dropbox.com/scl/fi/vp6cw7he8t0eigjf6001l/hamer_training_data_part4a.tar.gz?rlkey=wylmufft4a5nq3yxep2olifrk
+#wget -O hamer_training_data_part4b.tar.gz https://www.dropbox.com/scl/fi/vyjasngr67ru14fb8s108/hamer_training_data_part4b.tar.gz?rlkey=qgotg1v9lkgo5eu78gh8b007t
+#wget -O hamer_training_data_part4c.tar.gz https://www.dropbox.com/scl/fi/nfvz5zpcmhz8hkwzc6ji4/hamer_training_data_part4c.tar.gz?rlkey=ygh0wvse04twhh1ri3xiw2sag
+
+for f in hamer_training_data_part*.tar.gz; do
+ tar --warning=no-unknown-keyword --exclude=".*" -xvf $f
+done
diff --git a/third_party/hamer/hamer.egg-info/PKG-INFO b/third_party/hamer/hamer.egg-info/PKG-INFO
new file mode 100644
index 0000000000000000000000000000000000000000..b29a54c8c81e8e921d613e50461808f1ef92a4ec
--- /dev/null
+++ b/third_party/hamer/hamer.egg-info/PKG-INFO
@@ -0,0 +1,33 @@
+Metadata-Version: 2.4
+Name: hamer
+Version: 0.0.0
+Summary: HaMeR as a package
+License-File: LICENSE.md
+Requires-Dist: gdown
+Requires-Dist: numpy
+Requires-Dist: opencv-python
+Requires-Dist: pyrender
+Requires-Dist: pytorch-lightning
+Requires-Dist: scikit-image
+Requires-Dist: smplx==0.1.28
+Requires-Dist: torch
+Requires-Dist: torchvision
+Requires-Dist: yacs
+Requires-Dist: detectron2@ git+https://github.com/facebookresearch/detectron2
+Requires-Dist: chumpy@ git+https://github.com/mattloper/chumpy
+Requires-Dist: mmcv==1.3.9
+Requires-Dist: timm
+Requires-Dist: einops
+Requires-Dist: xtcocotools
+Requires-Dist: pandas
+Provides-Extra: all
+Requires-Dist: hydra-core; extra == "all"
+Requires-Dist: hydra-submitit-launcher; extra == "all"
+Requires-Dist: hydra-colorlog; extra == "all"
+Requires-Dist: pyrootutils; extra == "all"
+Requires-Dist: rich; extra == "all"
+Requires-Dist: webdataset; extra == "all"
+Dynamic: license-file
+Dynamic: provides-extra
+Dynamic: requires-dist
+Dynamic: summary
diff --git a/third_party/hamer/hamer.egg-info/SOURCES.txt b/third_party/hamer/hamer.egg-info/SOURCES.txt
new file mode 100644
index 0000000000000000000000000000000000000000..81eda66cbbdf343de71425d966047ce2eda28fc1
--- /dev/null
+++ b/third_party/hamer/hamer.egg-info/SOURCES.txt
@@ -0,0 +1,42 @@
+LICENSE.md
+README.md
+setup.py
+hamer/__init__.py
+hamer.egg-info/PKG-INFO
+hamer.egg-info/SOURCES.txt
+hamer.egg-info/dependency_links.txt
+hamer.egg-info/requires.txt
+hamer.egg-info/top_level.txt
+hamer/configs/__init__.py
+hamer/configs/cascade_mask_rcnn_vitdet_h_75ep.py
+hamer/datasets/__init__.py
+hamer/datasets/dataset.py
+hamer/datasets/image_dataset.py
+hamer/datasets/json_dataset.py
+hamer/datasets/mocap_dataset.py
+hamer/datasets/utils.py
+hamer/datasets/vitdet_dataset.py
+hamer/models/__init__.py
+hamer/models/discriminator.py
+hamer/models/hamer.py
+hamer/models/losses.py
+hamer/models/mano_wrapper.py
+hamer/models/backbones/__init__.py
+hamer/models/backbones/vit.py
+hamer/models/components/__init__.py
+hamer/models/components/pose_transformer.py
+hamer/models/components/t_cond_mlp.py
+hamer/models/heads/__init__.py
+hamer/models/heads/mano_head.py
+hamer/utils/__init__.py
+hamer/utils/download.py
+hamer/utils/geometry.py
+hamer/utils/mesh_renderer.py
+hamer/utils/misc.py
+hamer/utils/pose_utils.py
+hamer/utils/pylogger.py
+hamer/utils/render_openpose.py
+hamer/utils/renderer.py
+hamer/utils/rich_utils.py
+hamer/utils/skeleton_renderer.py
+hamer/utils/utils_detectron2.py
\ No newline at end of file
diff --git a/third_party/hamer/hamer.egg-info/dependency_links.txt b/third_party/hamer/hamer.egg-info/dependency_links.txt
new file mode 100644
index 0000000000000000000000000000000000000000..8b137891791fe96927ad78e64b0aad7bded08bdc
--- /dev/null
+++ b/third_party/hamer/hamer.egg-info/dependency_links.txt
@@ -0,0 +1 @@
+
diff --git a/third_party/hamer/hamer.egg-info/requires.txt b/third_party/hamer/hamer.egg-info/requires.txt
new file mode 100644
index 0000000000000000000000000000000000000000..2b0ba3cd29952b808a032be6f6544ec512b0c79b
--- /dev/null
+++ b/third_party/hamer/hamer.egg-info/requires.txt
@@ -0,0 +1,25 @@
+gdown
+numpy
+opencv-python
+pyrender
+pytorch-lightning
+scikit-image
+smplx==0.1.28
+torch
+torchvision
+yacs
+detectron2@ git+https://github.com/facebookresearch/detectron2
+chumpy@ git+https://github.com/mattloper/chumpy
+mmcv==1.3.9
+timm
+einops
+xtcocotools
+pandas
+
+[all]
+hydra-core
+hydra-submitit-launcher
+hydra-colorlog
+pyrootutils
+rich
+webdataset
diff --git a/third_party/hamer/hamer.egg-info/top_level.txt b/third_party/hamer/hamer.egg-info/top_level.txt
new file mode 100644
index 0000000000000000000000000000000000000000..fcfc77d1b852b6363af93ee8daad26275ca50de7
--- /dev/null
+++ b/third_party/hamer/hamer.egg-info/top_level.txt
@@ -0,0 +1 @@
+hamer
diff --git a/third_party/hamer/hamer/__init__.py b/third_party/hamer/hamer/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/third_party/hamer/hamer/configs/__init__.py b/third_party/hamer/hamer/configs/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..47d17a4153c4b71f0410616274689c6c1806e6e7
--- /dev/null
+++ b/third_party/hamer/hamer/configs/__init__.py
@@ -0,0 +1,114 @@
+import os
+from typing import Dict
+from yacs.config import CfgNode as CN
+
+CACHE_DIR_HAMER = "./_DATA"
+
+def to_lower(x: Dict) -> Dict:
+ """
+ Convert all dictionary keys to lowercase
+ Args:
+ x (dict): Input dictionary
+ Returns:
+ dict: Output dictionary with all keys converted to lowercase
+ """
+ return {k.lower(): v for k, v in x.items()}
+
+_C = CN(new_allowed=True)
+
+_C.GENERAL = CN(new_allowed=True)
+_C.GENERAL.RESUME = True
+_C.GENERAL.TIME_TO_RUN = 3300
+_C.GENERAL.VAL_STEPS = 100
+_C.GENERAL.LOG_STEPS = 100
+_C.GENERAL.CHECKPOINT_STEPS = 20000
+_C.GENERAL.CHECKPOINT_DIR = "checkpoints"
+_C.GENERAL.SUMMARY_DIR = "tensorboard"
+_C.GENERAL.NUM_GPUS = 1
+_C.GENERAL.NUM_WORKERS = 4
+_C.GENERAL.MIXED_PRECISION = True
+_C.GENERAL.ALLOW_CUDA = True
+_C.GENERAL.PIN_MEMORY = False
+_C.GENERAL.DISTRIBUTED = False
+_C.GENERAL.LOCAL_RANK = 0
+_C.GENERAL.USE_SYNCBN = False
+_C.GENERAL.WORLD_SIZE = 1
+
+_C.TRAIN = CN(new_allowed=True)
+_C.TRAIN.NUM_EPOCHS = 100
+_C.TRAIN.BATCH_SIZE = 32
+_C.TRAIN.SHUFFLE = True
+_C.TRAIN.WARMUP = False
+_C.TRAIN.NORMALIZE_PER_IMAGE = False
+_C.TRAIN.CLIP_GRAD = False
+_C.TRAIN.CLIP_GRAD_VALUE = 1.0
+_C.LOSS_WEIGHTS = CN(new_allowed=True)
+
+_C.DATASETS = CN(new_allowed=True)
+
+_C.MODEL = CN(new_allowed=True)
+_C.MODEL.IMAGE_SIZE = 224
+
+_C.EXTRA = CN(new_allowed=True)
+_C.EXTRA.FOCAL_LENGTH = 5000
+
+_C.DATASETS.CONFIG = CN(new_allowed=True)
+_C.DATASETS.CONFIG.SCALE_FACTOR = 0.3
+_C.DATASETS.CONFIG.ROT_FACTOR = 30
+_C.DATASETS.CONFIG.TRANS_FACTOR = 0.02
+_C.DATASETS.CONFIG.COLOR_SCALE = 0.2
+_C.DATASETS.CONFIG.ROT_AUG_RATE = 0.6
+_C.DATASETS.CONFIG.TRANS_AUG_RATE = 0.5
+_C.DATASETS.CONFIG.DO_FLIP = False
+_C.DATASETS.CONFIG.FLIP_AUG_RATE = 0.5
+_C.DATASETS.CONFIG.EXTREME_CROP_AUG_RATE = 0.10
+
+def default_config() -> CN:
+ """
+ Get a yacs CfgNode object with the default config values.
+ """
+ # Return a clone so that the defaults will not be altered
+ # This is for the "local variable" use pattern
+ return _C.clone()
+
+def dataset_config(name='datasets_tar.yaml') -> CN:
+ """
+ Get dataset config file
+ Returns:
+ CfgNode: Dataset config as a yacs CfgNode object.
+ """
+ cfg = CN(new_allowed=True)
+ config_file = os.path.join(os.path.dirname(os.path.realpath(__file__)), name)
+ cfg.merge_from_file(config_file)
+ cfg.freeze()
+ return cfg
+
+def dataset_eval_config() -> CN:
+ return dataset_config('datasets_eval.yaml')
+
+def get_config(config_file: str, merge: bool = True, update_cachedir: bool = False) -> CN:
+ """
+ Read a config file and optionally merge it with the default config file.
+ Args:
+ config_file (str): Path to config file.
+ merge (bool): Whether to merge with the default config or not.
+ Returns:
+ CfgNode: Config as a yacs CfgNode object.
+ """
+ if merge:
+ cfg = default_config()
+ else:
+ cfg = CN(new_allowed=True)
+ cfg.merge_from_file(config_file)
+
+ if update_cachedir:
+ def update_path(path: str) -> str:
+ if os.path.isabs(path):
+ return path
+ return os.path.join(CACHE_DIR_HAMER, path)
+
+ cfg.MANO.MODEL_PATH = update_path(cfg.MANO.MODEL_PATH)
+ cfg.MANO.MEAN_PARAMS = update_path(cfg.MANO.MEAN_PARAMS)
+
+ cfg.freeze()
+ return cfg
diff --git a/third_party/hamer/hamer/configs/cascade_mask_rcnn_vitdet_h_75ep.py b/third_party/hamer/hamer/configs/cascade_mask_rcnn_vitdet_h_75ep.py
new file mode 100644
index 0000000000000000000000000000000000000000..0c6ae0eaf48c2c2d3b70529a0d2d915432e43db6
--- /dev/null
+++ b/third_party/hamer/hamer/configs/cascade_mask_rcnn_vitdet_h_75ep.py
@@ -0,0 +1,129 @@
+## coco_loader_lsj.py
+
+import detectron2.data.transforms as T
+from detectron2 import model_zoo
+from detectron2.config import LazyCall as L
+
+# Data using LSJ
+image_size = 1024
+dataloader = model_zoo.get_config("common/data/coco.py").dataloader
+dataloader.train.mapper.augmentations = [
+ L(T.RandomFlip)(horizontal=True), # flip first
+ L(T.ResizeScale)(
+ min_scale=0.1, max_scale=2.0, target_height=image_size, target_width=image_size
+ ),
+ L(T.FixedSizeCrop)(crop_size=(image_size, image_size), pad=False),
+]
+dataloader.train.mapper.image_format = "RGB"
+dataloader.train.total_batch_size = 64
+# recompute boxes due to cropping
+dataloader.train.mapper.recompute_boxes = True
+
+dataloader.test.mapper.augmentations = [
+ L(T.ResizeShortestEdge)(short_edge_length=image_size, max_size=image_size),
+]
+
+from functools import partial
+from fvcore.common.param_scheduler import MultiStepParamScheduler
+
+from detectron2 import model_zoo
+from detectron2.config import LazyCall as L
+from detectron2.solver import WarmupParamScheduler
+from detectron2.modeling.backbone.vit import get_vit_lr_decay_rate
+
+# mask_rcnn_vitdet_b_100ep.py
+
+model = model_zoo.get_config("common/models/mask_rcnn_vitdet.py").model
+
+# Initialization and trainer settings
+train = model_zoo.get_config("common/train.py").train
+train.amp.enabled = True
+train.ddp.fp16_compression = True
+train.init_checkpoint = "detectron2://ImageNetPretrained/MAE/mae_pretrain_vit_base.pth"
+
+
+# Schedule
+# 100 ep = 184375 iters * 64 images/iter / 118000 images/ep
+train.max_iter = 184375
+
+lr_multiplier = L(WarmupParamScheduler)(
+ scheduler=L(MultiStepParamScheduler)(
+ values=[1.0, 0.1, 0.01],
+ milestones=[163889, 177546],
+ num_updates=train.max_iter,
+ ),
+ warmup_length=250 / train.max_iter,
+ warmup_factor=0.001,
+)
+
+# Optimizer
+optimizer = model_zoo.get_config("common/optim.py").AdamW
+optimizer.params.lr_factor_func = partial(get_vit_lr_decay_rate, num_layers=12, lr_decay_rate=0.7)
+optimizer.params.overrides = {"pos_embed": {"weight_decay": 0.0}}
+
+# cascade_mask_rcnn_vitdet_b_100ep.py
+
+from detectron2.config import LazyCall as L
+from detectron2.layers import ShapeSpec
+from detectron2.modeling.box_regression import Box2BoxTransform
+from detectron2.modeling.matcher import Matcher
+from detectron2.modeling.roi_heads import (
+ FastRCNNOutputLayers,
+ FastRCNNConvFCHead,
+ CascadeROIHeads,
+)
+
+# arguments that don't exist for Cascade R-CNN
+[model.roi_heads.pop(k) for k in ["box_head", "box_predictor", "proposal_matcher"]]
+
+model.roi_heads.update(
+ _target_=CascadeROIHeads,
+ box_heads=[
+ L(FastRCNNConvFCHead)(
+ input_shape=ShapeSpec(channels=256, height=7, width=7),
+ conv_dims=[256, 256, 256, 256],
+ fc_dims=[1024],
+ conv_norm="LN",
+ )
+ for _ in range(3)
+ ],
+ box_predictors=[
+ L(FastRCNNOutputLayers)(
+ input_shape=ShapeSpec(channels=1024),
+ test_score_thresh=0.05,
+ box2box_transform=L(Box2BoxTransform)(weights=(w1, w1, w2, w2)),
+ cls_agnostic_bbox_reg=True,
+ num_classes="${...num_classes}",
+ )
+ for (w1, w2) in [(10, 5), (20, 10), (30, 15)]
+ ],
+ proposal_matchers=[
+ L(Matcher)(thresholds=[th], labels=[0, 1], allow_low_quality_matches=False)
+ for th in [0.5, 0.6, 0.7]
+ ],
+)
+
+# cascade_mask_rcnn_vitdet_h_75ep.py
+
+from functools import partial
+
+train.init_checkpoint = "detectron2://ImageNetPretrained/MAE/mae_pretrain_vit_huge_p14to16.pth"
+
+model.backbone.net.embed_dim = 1280
+model.backbone.net.depth = 32
+model.backbone.net.num_heads = 16
+model.backbone.net.drop_path_rate = 0.5
+# 7, 15, 23, 31 for global attention
+model.backbone.net.window_block_indexes = (
+ list(range(0, 7)) + list(range(8, 15)) + list(range(16, 23)) + list(range(24, 31))
+)
+
+optimizer.params.lr_factor_func = partial(get_vit_lr_decay_rate, lr_decay_rate=0.9, num_layers=32)
+optimizer.params.overrides = {}
+optimizer.params.weight_decay_norm = None
+
+train.max_iter = train.max_iter * 3 // 4 # 100ep -> 75ep
+lr_multiplier.scheduler.milestones = [
+ milestone * 3 // 4 for milestone in lr_multiplier.scheduler.milestones
+]
+lr_multiplier.scheduler.num_updates = train.max_iter
diff --git a/third_party/hamer/hamer/configs/datasets_eval.yaml b/third_party/hamer/hamer/configs/datasets_eval.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..492dfedffe05933fa5707d250da77aae9bb10126
--- /dev/null
+++ b/third_party/hamer/hamer/configs/datasets_eval.yaml
@@ -0,0 +1,65 @@
+FREIHAND-VAL:
+ TYPE: ImageDataset
+ DATASET_FILE: hamer_evaluation_data/freihand_val.npz
+ IMG_DIR: /home/user/datasets/FreiHAND_pub_v2_eval/evaluation/rgb/
+ KEYPOINT_LIST: [0] # Dummy
+
+HO3D-VAL:
+ TYPE: ImageDataset
+ DATASET_FILE: hamer_evaluation_data/ho3d_val.npz
+ IMG_DIR: /home/user/datasets/HO3D_v2/evaluation/
+ KEYPOINT_LIST: [0] # Dummy
+
+EGO4D-TEST-ALL:
+ TYPE: ImageDataset
+ DATASET_FILE: hamer_evaluation_data/TEST_ego4d_img_all.npz
+ IMG_DIR: /home/user/datasets/HInt_annotation_partial/TEST_ego4d_img/
+ KEYPOINT_LIST: [0] # Dummy
+
+EGO4D-TEST-VIS:
+ TYPE: ImageDataset
+ DATASET_FILE: hamer_evaluation_data/TEST_ego4d_img_vis.npz
+ IMG_DIR: /home/user/datasets/HInt_annotation_partial/TEST_ego4d_img/
+ KEYPOINT_LIST: [0] # Dummy
+
+EGO4D-TEST-OCC:
+ TYPE: ImageDataset
+ DATASET_FILE: hamer_evaluation_data/TEST_ego4d_img_occ.npz
+ IMG_DIR: /home/user/datasets/HInt_annotation_partial/TEST_ego4d_img/
+ KEYPOINT_LIST: [0] # Dummy
+
+EPICK-TEST-ALL:
+ TYPE: ImageDataset
+ DATASET_FILE: hamer_evaluation_data/TEST_epick_img_all.npz
+ IMG_DIR: /home/user/datasets/HInt_annotation_partial/TEST_epick_img/
+ KEYPOINT_LIST: [0] # Dummy
+
+EPICK-TEST-VIS:
+ TYPE: ImageDataset
+ DATASET_FILE: hamer_evaluation_data/TEST_epick_img_vis.npz
+ IMG_DIR: /home/user/datasets/HInt_annotation_partial/TEST_epick_img/
+ KEYPOINT_LIST: [0] # Dummy
+
+EPICK-TEST-OCC:
+ TYPE: ImageDataset
+ DATASET_FILE: hamer_evaluation_data/TEST_epick_img_occ.npz
+ IMG_DIR: /home/user/datasets/HInt_annotation_partial/TEST_epick_img/
+ KEYPOINT_LIST: [0] # Dummy
+
+NEWDAYS-TEST-ALL:
+ TYPE: ImageDataset
+ DATASET_FILE: hamer_evaluation_data/TEST_newdays_img_all.npz
+ IMG_DIR: /home/user/datasets/HInt_annotation_partial/TEST_newdays_img/
+ KEYPOINT_LIST: [0] # Dummy
+
+NEWDAYS-TEST-VIS:
+ TYPE: ImageDataset
+ DATASET_FILE: hamer_evaluation_data/TEST_newdays_img_vis.npz
+ IMG_DIR: /home/user/datasets/HInt_annotation_partial/TEST_newdays_img/
+ KEYPOINT_LIST: [0] # Dummy
+
+NEWDAYS-TEST-OCC:
+ TYPE: ImageDataset
+ DATASET_FILE: hamer_evaluation_data/TEST_newdays_img_occ.npz
+ IMG_DIR: /home/user/datasets/HInt_annotation_partial/TEST_newdays_img/
+ KEYPOINT_LIST: [0] # Dummy
diff --git a/third_party/hamer/hamer/configs/datasets_tar.yaml b/third_party/hamer/hamer/configs/datasets_tar.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..2ebad8a6404e5fe59db55f9e042af8301053eb66
--- /dev/null
+++ b/third_party/hamer/hamer/configs/datasets_tar.yaml
@@ -0,0 +1,42 @@
+FREIHAND-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/freihand-train/{000000..000130}.tar
+ epoch_size: 130_240
+INTERHAND26M-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/interhand26m-train/{000000..001056}.tar
+ epoch_size: 1_424_632
+HALPE-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/halpe-train/{000000..000022}.tar
+ epoch_size: 34_289
+COCOW-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/cocow-train/{000000..000036}.tar
+ epoch_size: 78_666
+MTC-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/mtc-train/{000000..000306}.tar
+ epoch_size: 363_947
+RHD-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/rhd-train/{000000..000041}.tar
+ epoch_size: 61_705
+MPIINZSL-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/mpiinzsl-train/{000000..000015}.tar
+ epoch_size: 15_184
+HO3D-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/ho3d-train/{000000..000083}.tar
+ epoch_size: 83_325
+H2O3D-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/h2o3d-train/{000000..000060}.tar
+ epoch_size: 121_996
+DEX-TRAIN:
+ TYPE: ImageDataset
+ URLS: hamer_training_data/dataset_tars/dex-train/{000000..000406}.tar
+ epoch_size: 406_888
+FREIHAND-MOCAP:
+ DATASET_FILE: hamer_training_data/freihand_mocap.npz
diff --git a/third_party/hamer/hamer/configs_hydra/data/mix_all.yaml b/third_party/hamer/hamer/configs_hydra/data/mix_all.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..26e0d7102553772cbb9a4893e55863f56e3bc41d
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/data/mix_all.yaml
@@ -0,0 +1,31 @@
+# @package _global_
+defaults:
+ - /data_filtering: low1
+
+DATASETS:
+ TRAIN:
+ FREIHAND-TRAIN:
+ WEIGHT: 0.25
+ INTERHAND26M-TRAIN:
+ WEIGHT: 0.25
+ MTC-TRAIN:
+ WEIGHT: 0.1
+ RHD-TRAIN:
+ WEIGHT: 0.05
+ COCOW-TRAIN:
+ WEIGHT: 0.1
+ HALPE-TRAIN:
+ WEIGHT: 0.05
+ MPIINZSL-TRAIN:
+ WEIGHT: 0.05
+ HO3D-TRAIN:
+ WEIGHT: 0.05
+ H2O3D-TRAIN:
+ WEIGHT: 0.05
+ DEX-TRAIN:
+ WEIGHT: 0.05
+ VAL:
+ FREIHAND-TRAIN:
+ WEIGHT: 1.0
+
+ MOCAP: FREIHAND-MOCAP
diff --git a/third_party/hamer/hamer/configs_hydra/data_filtering/low1.yaml b/third_party/hamer/hamer/configs_hydra/data_filtering/low1.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..bea3b9df8c10100f1de32600546f254aa70a5081
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/data_filtering/low1.yaml
@@ -0,0 +1,13 @@
+# @package _global_
+
+DATASETS:
+ # Data filtering during training
+ SUPPRESS_KP_CONF_THRESH: 0.3
+ FILTER_NUM_KP: 4
+ FILTER_NUM_KP_THRESH: 0.0
+ FILTER_REPROJ_THRESH: 31000
+
+ SUPPRESS_BETAS_THRESH: 3.0
+ SUPPRESS_BAD_POSES: False
+ POSES_BETAS_SIMULTANEOUS: True
+ FILTER_NO_POSES: False # If True, filters images that don't have poses
diff --git a/third_party/hamer/hamer/configs_hydra/experiment/default.yaml b/third_party/hamer/hamer/configs_hydra/experiment/default.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..323acd5b609dfba819334f58040a5fa12baf8734
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/experiment/default.yaml
@@ -0,0 +1,29 @@
+# @package _global_
+
+MANO:
+ DATA_DIR: _DATA/data/
+ MODEL_PATH: ${MANO.DATA_DIR}/mano
+ GENDER: neutral
+ NUM_HAND_JOINTS: 15
+ MEAN_PARAMS: ${MANO.DATA_DIR}/mano_mean_params.npz
+ CREATE_BODY_POSE: FALSE
+
+EXTRA:
+ FOCAL_LENGTH: 5000
+ NUM_LOG_IMAGES: 4
+ NUM_LOG_SAMPLES_PER_IMAGE: 8
+ PELVIS_IND: 0
+
+DATASETS:
+ BETAS_REG: True
+ CONFIG:
+ SCALE_FACTOR: 0.3
+ ROT_FACTOR: 30
+ TRANS_FACTOR: 0.02
+ COLOR_SCALE: 0.2
+ ROT_AUG_RATE: 0.6
+ TRANS_AUG_RATE: 0.5
+ DO_FLIP: False
+ FLIP_AUG_RATE: 0.0
+ EXTREME_CROP_AUG_RATE: 0.0
+ EXTREME_CROP_AUG_LEVEL: 1
diff --git a/third_party/hamer/hamer/configs_hydra/experiment/hamer_vit_transformer.yaml b/third_party/hamer/hamer/configs_hydra/experiment/hamer_vit_transformer.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..1a3a5924a4f2d9e61ddf82615de7433bb2f66518
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/experiment/hamer_vit_transformer.yaml
@@ -0,0 +1,51 @@
+# @package _global_
+
+defaults:
+ - default.yaml
+
+GENERAL:
+ TOTAL_STEPS: 1_000_000
+ LOG_STEPS: 1000
+ VAL_STEPS: 1000
+ CHECKPOINT_STEPS: 1000
+ CHECKPOINT_SAVE_TOP_K: 1
+ NUM_WORKERS: 4
+ PREFETCH_FACTOR: 2
+
+TRAIN:
+ LR: 1e-5
+ WEIGHT_DECAY: 1e-4
+ BATCH_SIZE: 8
+ LOSS_REDUCTION: mean
+ NUM_TRAIN_SAMPLES: 2
+ NUM_TEST_SAMPLES: 64
+ POSE_2D_NOISE_RATIO: 0.01
+ SMPL_PARAM_NOISE_RATIO: 0.005
+
+MODEL:
+ IMAGE_SIZE: 256
+ IMAGE_MEAN: [0.485, 0.456, 0.406]
+ IMAGE_STD: [0.229, 0.224, 0.225]
+ BACKBONE:
+ TYPE: vit
+ PRETRAINED_WEIGHTS: hamer_training_data/vitpose_backbone.pth
+ MANO_HEAD:
+ TYPE: transformer_decoder
+ IN_CHANNELS: 2048
+ TRANSFORMER_DECODER:
+ depth: 6
+ heads: 8
+ mlp_dim: 1024
+ dim_head: 64
+ dropout: 0.0
+ emb_dropout: 0.0
+ norm: layer
+ context_dim: 1280 # from vitpose-H
+
+LOSS_WEIGHTS:
+ KEYPOINTS_3D: 0.05
+ KEYPOINTS_2D: 0.01
+ GLOBAL_ORIENT: 0.001
+ HAND_POSE: 0.001
+ BETAS: 0.0005
+ ADVERSARIAL: 0.0005
diff --git a/third_party/hamer/hamer/configs_hydra/extras/default.yaml b/third_party/hamer/hamer/configs_hydra/extras/default.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..b9c6b622283a647fbc513166fc14f016cc3ed8a0
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/extras/default.yaml
@@ -0,0 +1,8 @@
+# disable python warnings if they annoy you
+ignore_warnings: False
+
+# ask user for tags if none are provided in the config
+enforce_tags: True
+
+# pretty print config tree at the start of the run using Rich library
+print_config: True
diff --git a/third_party/hamer/hamer/configs_hydra/hydra/default.yaml b/third_party/hamer/hamer/configs_hydra/hydra/default.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..c30c188f4e68b205ec0f1e5679345626fe187164
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/hydra/default.yaml
@@ -0,0 +1,26 @@
+# @package _global_
+# https://hydra.cc/docs/configure_hydra/intro/
+
+# enable color logging
+defaults:
+ - override /hydra/hydra_logging: colorlog
+ - override /hydra/job_logging: colorlog
+
+# exp_name: ovrd_${hydra:job.override_dirname}
+exp_name: ${now:%Y-%m-%d}_${now:%H-%M-%S}
+
+hydra:
+ run:
+ dir: ${paths.log_dir}/${task_name}/runs/${exp_name}
+ sweep:
+ dir: ${paths.log_dir}/${task_name}/multiruns/${exp_name}
+ subdir: ${hydra.job.num}
+ job:
+ config:
+ override_dirname:
+ exclude_keys:
+ - trainer
+ - trainer.devices
+ - trainer.num_nodes
+ - callbacks
+ - debug
diff --git a/third_party/hamer/hamer/configs_hydra/launcher/local.yaml b/third_party/hamer/hamer/configs_hydra/launcher/local.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..da87047acd416fe6d03bc81a74ab62b449b4ac35
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/launcher/local.yaml
@@ -0,0 +1,13 @@
+# @package _global_
+
+defaults:
+ - override /hydra/launcher: submitit_local
+
+hydra:
+ launcher:
+ timeout_min: 10_080 # 7 days
+ nodes: 1
+ tasks_per_node: ${trainer.devices}
+ cpus_per_task: 6
+ gpus_per_node: ${trainer.devices}
+ name: hamer
diff --git a/third_party/hamer/hamer/configs_hydra/launcher/slurm.yaml b/third_party/hamer/hamer/configs_hydra/launcher/slurm.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..f30ccce9069210830270c665bd31294c9d1799b7
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/launcher/slurm.yaml
@@ -0,0 +1,22 @@
+# @package _global_
+
+defaults:
+ - override /hydra/launcher: submitit_slurm
+
+hydra:
+ launcher:
+ timeout_min: 10_080 # 7 days
+ max_num_timeout: 3
+ partition: g40
+ qos: idle
+ nodes: 1
+ tasks_per_node: ${trainer.devices}
+ gpus_per_task: null
+ cpus_per_task: 12
+ gpus_per_node: ${trainer.devices}
+ cpus_per_gpu: null
+ comment: laion
+ name: hamer
+ setup:
+ - module load cuda openmpi libfabric-aws
+ - export CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7
diff --git a/third_party/hamer/hamer/configs_hydra/paths/default.yaml b/third_party/hamer/hamer/configs_hydra/paths/default.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..b2afd22a65d1b34d881943cb48ee4ce3ff37d165
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/paths/default.yaml
@@ -0,0 +1,18 @@
+# path to root directory
+# this requires PROJECT_ROOT environment variable to exist
+# PROJECT_ROOT is inferred and set by pyrootutils package in `train.py` and `eval.py`
+root_dir: ${oc.env:PROJECT_ROOT}
+
+# path to data directory
+data_dir: ${paths.root_dir}/data/
+
+# path to logging directory
+log_dir: logs/
+
+# path to output directory, created dynamically by hydra
+# path generation pattern is specified in `configs/hydra/default.yaml`
+# use it to store all files generated during the run, like ckpts and metrics
+output_dir: ${hydra:runtime.output_dir}
+
+# path to working directory
+work_dir: ${hydra:runtime.cwd}
diff --git a/third_party/hamer/hamer/configs_hydra/train.yaml b/third_party/hamer/hamer/configs_hydra/train.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..5021b4c156fc5738aee3d7d2fbd9395a2b3bb987
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/train.yaml
@@ -0,0 +1,47 @@
+# @package _global_
+
+# specify here default configuration
+# order of defaults determines the order in which configs override each other
+defaults:
+ - _self_
+ - data: mix_all.yaml
+ - trainer: ddp.yaml
+ - paths: default.yaml
+ - extras: default.yaml
+ - hydra: default.yaml
+
+ # experiment configs allow for version control of specific hyperparameters
+ # e.g. best hyperparameters for given model and datamodule
+ - experiment: null
+ - texture_exp: null
+
+ # optional local config for machine/user specific settings
+ # it's optional since it doesn't need to exist and is excluded from version control
+ - optional launcher: local.yaml
+ # - optional launcher: slurm.yaml
+
+ # debugging config (enable through command line, e.g. `python train.py debug=default)
+ - debug: null
+
+# task name, determines output directory path
+task_name: "train"
+
+# tags to help you identify your experiments
+# you can overwrite this in experiment configs
+# overwrite from command line with `python train.py tags="[first_tag, second_tag]"`
+# appending lists from command line is currently not supported :(
+# https://github.com/facebookresearch/hydra/issues/1547
+tags: ["dev"]
+
+# set False to skip model training
+train: True
+
+# evaluate on test set, using best model weights achieved during training
+# lightning chooses best weights based on the metric specified in checkpoint callback
+test: False
+
+# simply provide checkpoint path to resume training
+ckpt_path: null
+
+# seed for random number generators in pytorch, numpy and python.random
+seed: null
diff --git a/third_party/hamer/hamer/configs_hydra/trainer/cpu.yaml b/third_party/hamer/hamer/configs_hydra/trainer/cpu.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..2464b95ee0d6c03a3dfe202f8a99b0cf04f37031
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/trainer/cpu.yaml
@@ -0,0 +1,6 @@
+defaults:
+ - default.yaml
+ - default_hamer.yaml
+
+accelerator: cpu
+devices: 1
diff --git a/third_party/hamer/hamer/configs_hydra/trainer/ddp.yaml b/third_party/hamer/hamer/configs_hydra/trainer/ddp.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..b365ff6df35d3218970a82895f4f0e27b9647780
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/trainer/ddp.yaml
@@ -0,0 +1,14 @@
+defaults:
+ - default.yaml
+ - default_hamer.yaml
+
+# use "ddp_spawn" instead of "ddp",
+# it's slower but normal "ddp" currently doesn't work ideally with hydra
+# https://github.com/facebookresearch/hydra/issues/2070
+# https://pytorch-lightning.readthedocs.io/en/latest/accelerators/gpu_intermediate.html#distributed-data-parallel-spawn
+strategy: ddp
+
+accelerator: gpu
+devices: 8
+num_nodes: 1
+sync_batchnorm: True
diff --git a/third_party/hamer/hamer/configs_hydra/trainer/default.yaml b/third_party/hamer/hamer/configs_hydra/trainer/default.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..7d444f4671fc77d7cf3f11ec74e638f3f620098f
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/trainer/default.yaml
@@ -0,0 +1,10 @@
+_target_: pytorch_lightning.Trainer
+
+default_root_dir: ${paths.output_dir}
+
+accelerator: cpu
+devices: 1
+
+# set True to to ensure deterministic results
+# makes training slower but gives more reproducibility than just setting seeds
+deterministic: False
diff --git a/third_party/hamer/hamer/configs_hydra/trainer/default_hamer.yaml b/third_party/hamer/hamer/configs_hydra/trainer/default_hamer.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..963b2393c9651ba53f8e0e69256193d635821174
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/trainer/default_hamer.yaml
@@ -0,0 +1,8 @@
+num_sanity_val_steps: 0
+log_every_n_steps: ${GENERAL.LOG_STEPS}
+val_check_interval: ${GENERAL.VAL_STEPS}
+precision: 16
+max_steps: ${GENERAL.TOTAL_STEPS}
+# move_metrics_to_cpu: True
+limit_val_batches: 1
+# track_grad_norm: -1
diff --git a/third_party/hamer/hamer/configs_hydra/trainer/gpu.yaml b/third_party/hamer/hamer/configs_hydra/trainer/gpu.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..6b0c8b9171a83784a1f243d3e4515bfec0a10b1d
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/trainer/gpu.yaml
@@ -0,0 +1,6 @@
+defaults:
+ - default.yaml
+ - default_hamer.yaml
+
+accelerator: gpu
+devices: 1
diff --git a/third_party/hamer/hamer/configs_hydra/trainer/mps.yaml b/third_party/hamer/hamer/configs_hydra/trainer/mps.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..25806bc3cd66c3130ee82c4e14e1700d28b471a0
--- /dev/null
+++ b/third_party/hamer/hamer/configs_hydra/trainer/mps.yaml
@@ -0,0 +1,6 @@
+defaults:
+ - default.yaml
+ - default_hamer.yaml
+
+accelerator: mps
+devices: 1
diff --git a/third_party/hamer/hamer/datasets/__init__.py b/third_party/hamer/hamer/datasets/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..7ddcda25c621ad564053e21d39de85fc47cdb298
--- /dev/null
+++ b/third_party/hamer/hamer/datasets/__init__.py
@@ -0,0 +1,88 @@
+from typing import Dict, Optional
+
+import torch
+import numpy as np
+import pytorch_lightning as pl
+from yacs.config import CfgNode
+
+import webdataset as wds
+from ..configs import to_lower
+from .dataset import Dataset
+from .image_dataset import ImageDataset
+from .mocap_dataset import MoCapDataset
+
+def create_dataset(cfg: CfgNode, dataset_cfg: CfgNode, train: bool = True, **kwargs) -> Dataset:
+ """
+ Instantiate a dataset from a config file.
+ Args:
+ cfg (CfgNode): Model configuration file.
+ dataset_cfg (CfgNode): Dataset configuration info.
+ train (bool): Variable to select between train and val datasets.
+ """
+
+ dataset_type = Dataset.registry[dataset_cfg.TYPE]
+ return dataset_type(cfg, **to_lower(dataset_cfg), train=train, **kwargs)
+
+def create_webdataset(cfg: CfgNode, dataset_cfg: CfgNode, train: bool = True) -> Dataset:
+ """
+ Like `create_dataset` but load data from tars.
+ """
+ dataset_type = Dataset.registry[dataset_cfg.TYPE]
+ return dataset_type.load_tars_as_webdataset(cfg, **to_lower(dataset_cfg), train=train)
+
+
+class MixedWebDataset(wds.WebDataset):
+ def __init__(self, cfg: CfgNode, dataset_cfg: CfgNode, train: bool = True) -> None:
+ super(wds.WebDataset, self).__init__()
+ dataset_list = cfg.DATASETS.TRAIN if train else cfg.DATASETS.VAL
+ datasets = [create_webdataset(cfg, dataset_cfg[dataset], train=train) for dataset, v in dataset_list.items()]
+ weights = np.array([v.WEIGHT for dataset, v in dataset_list.items()])
+ weights = weights / weights.sum() # normalize
+ self.append(wds.RandomMix(datasets, weights))
+
+class HAMERDataModule(pl.LightningDataModule):
+
+ def __init__(self, cfg: CfgNode, dataset_cfg: CfgNode) -> None:
+ """
+ Initialize LightningDataModule for HAMER training
+ Args:
+ cfg (CfgNode): Config file as a yacs CfgNode containing necessary dataset info.
+ dataset_cfg (CfgNode): Dataset configuration file
+ """
+ super().__init__()
+ self.cfg = cfg
+ self.dataset_cfg = dataset_cfg
+ self.train_dataset = None
+ self.val_dataset = None
+ self.test_dataset = None
+ self.mocap_dataset = None
+
+ def setup(self, stage: Optional[str] = None) -> None:
+ """
+ Load datasets necessary for training
+ Args:
+ cfg (CfgNode): Config file as a yacs CfgNode containing necessary dataset info.
+ """
+ if self.train_dataset == None:
+ self.train_dataset = MixedWebDataset(self.cfg, self.dataset_cfg, train=True).with_epoch(100_000).shuffle(4000)
+ self.val_dataset = MixedWebDataset(self.cfg, self.dataset_cfg, train=False).shuffle(4000)
+ self.mocap_dataset = MoCapDataset(**to_lower(self.dataset_cfg[self.cfg.DATASETS.MOCAP]))
+
+ def train_dataloader(self) -> Dict:
+ """
+ Setup training data loader.
+ Returns:
+ Dict: Dictionary containing image and mocap data dataloaders
+ """
+ train_dataloader = torch.utils.data.DataLoader(self.train_dataset, self.cfg.TRAIN.BATCH_SIZE, drop_last=True, num_workers=self.cfg.GENERAL.NUM_WORKERS, prefetch_factor=self.cfg.GENERAL.PREFETCH_FACTOR)
+ mocap_dataloader = torch.utils.data.DataLoader(self.mocap_dataset, self.cfg.TRAIN.NUM_TRAIN_SAMPLES * self.cfg.TRAIN.BATCH_SIZE, shuffle=True, drop_last=True, num_workers=1)
+ return {'img': train_dataloader, 'mocap': mocap_dataloader}
+
+ def val_dataloader(self) -> torch.utils.data.DataLoader:
+ """
+ Setup val data loader.
+ Returns:
+ torch.utils.data.DataLoader: Validation dataloader
+ """
+ val_dataloader = torch.utils.data.DataLoader(self.val_dataset, self.cfg.TRAIN.BATCH_SIZE, drop_last=True, num_workers=self.cfg.GENERAL.NUM_WORKERS)
+ return val_dataloader
diff --git a/third_party/hamer/hamer/datasets/dataset.py b/third_party/hamer/hamer/datasets/dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..22fc5bc5f4a7b75da672bd89859da14823e71aff
--- /dev/null
+++ b/third_party/hamer/hamer/datasets/dataset.py
@@ -0,0 +1,27 @@
+"""
+This file contains the defition of the base Dataset class.
+"""
+
+class DatasetRegistration(type):
+ """
+ Metaclass for registering different datasets
+ """
+ def __init__(cls, name, bases, nmspc):
+ super().__init__(name, bases, nmspc)
+ if not hasattr(cls, 'registry'):
+ cls.registry = dict()
+ cls.registry[name] = cls
+
+ # Metamethods, called on class objects:
+ def __iter__(cls):
+ return iter(cls.registry)
+
+ def __str__(cls):
+ return str(cls.registry)
+
+class Dataset(metaclass=DatasetRegistration):
+ """
+ Base Dataset class
+ """
+ def __init__(self, *args, **kwargs):
+ pass
\ No newline at end of file
diff --git a/third_party/hamer/hamer/datasets/image_dataset.py b/third_party/hamer/hamer/datasets/image_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..03b824fd32bc4622fbfa4a74e5f4092256ae55d7
--- /dev/null
+++ b/third_party/hamer/hamer/datasets/image_dataset.py
@@ -0,0 +1,434 @@
+import copy
+import os
+import numpy as np
+import torch
+from typing import Any, Dict, List
+from yacs.config import CfgNode
+import braceexpand
+import cv2
+
+from .dataset import Dataset
+from .utils import get_example, expand_to_aspect_ratio
+
+def expand(s):
+ return os.path.expanduser(os.path.expandvars(s))
+def expand_urls(urls: str|List[str]):
+ if isinstance(urls, str):
+ urls = [urls]
+ urls = [u for url in urls for u in braceexpand.braceexpand(expand(url))]
+ return urls
+
+FLIP_KEYPOINT_PERMUTATION = [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20]
+
+DEFAULT_MEAN = 255. * np.array([0.485, 0.456, 0.406])
+DEFAULT_STD = 255. * np.array([0.229, 0.224, 0.225])
+DEFAULT_IMG_SIZE = 256
+
+class ImageDataset(Dataset):
+
+ def __init__(self,
+ cfg: CfgNode,
+ dataset_file: str,
+ img_dir: str,
+ train: bool = True,
+ rescale_factor = 2,
+ prune: Dict[str, Any] = {},
+ **kwargs):
+ """
+ Dataset class used for loading images and corresponding annotations.
+ Args:
+ cfg (CfgNode): Model config file.
+ dataset_file (str): Path to npz file containing dataset info.
+ img_dir (str): Path to image folder.
+ train (bool): Whether it is for training or not (enables data augmentation).
+ """
+ super(ImageDataset, self).__init__()
+ self.train = train
+ self.cfg = cfg
+
+ self.img_size = cfg.MODEL.IMAGE_SIZE
+ self.mean = 255. * np.array(self.cfg.MODEL.IMAGE_MEAN)
+ self.std = 255. * np.array(self.cfg.MODEL.IMAGE_STD)
+ self.rescale_factor = rescale_factor
+
+ self.img_dir = img_dir
+ self.data = np.load(dataset_file, allow_pickle=True)
+
+ self.imgname = self.data['imgname']
+ self.personid = np.zeros(len(self.imgname), dtype=np.int32)
+ self.extra_info = self.data.get('extra_info', [{} for _ in range(len(self.imgname))])
+
+ self.flip_keypoint_permutation = copy.copy(FLIP_KEYPOINT_PERMUTATION)
+
+ num_pose = 3 * (self.cfg.MANO.NUM_HAND_JOINTS + 1)
+
+ # Bounding boxes are assumed to be in the center and scale format
+ self.center = self.data['center']
+ self.scale = self.data['scale'].reshape(len(self.center), -1) / 200.0
+ if self.scale.shape[1] == 1:
+ self.scale = np.tile(self.scale, (1, 2))
+ assert self.scale.shape == (len(self.center), 2)
+
+ try:
+ self.right = self.data['right']
+ except KeyError:
+ self.right = np.ones(len(self.imgname), dtype=np.float32)
+
+ # Get gt MANO parameters, if available
+ try:
+ self.hand_pose = self.data['hand_pose'].astype(np.float32)
+ self.has_hand_pose = self.data['has_hand_pose'].astype(np.float32)
+ except KeyError:
+ self.hand_pose = np.zeros((len(self.imgname), num_pose), dtype=np.float32)
+ self.has_hand_pose = np.zeros(len(self.imgname), dtype=np.float32)
+ try:
+ self.betas = self.data['betas'].astype(np.float32)
+ self.has_betas = self.data['has_betas'].astype(np.float32)
+ except KeyError:
+ self.betas = np.zeros((len(self.imgname), 10), dtype=np.float32)
+ self.has_betas = np.zeros(len(self.imgname), dtype=np.float32)
+
+ # Try to get 2d keypoints, if available
+ try:
+ hand_keypoints_2d = self.data['hand_keypoints_2d']
+ except KeyError:
+ hand_keypoints_2d = np.zeros((len(self.center), 21, 3))
+
+ self.keypoints_2d = hand_keypoints_2d
+
+ # Try to get 3d keypoints, if available
+ try:
+ hand_keypoints_3d = self.data['hand_keypoints_3d'].astype(np.float32)
+ except KeyError:
+ hand_keypoints_3d = np.zeros((len(self.center), 21, 4), dtype=np.float32)
+
+ self.keypoints_3d = hand_keypoints_3d
+
+ def __len__(self) -> int:
+ return len(self.scale)
+
+ def __getitem__(self, idx: int) -> Dict:
+ """
+ Returns an example from the dataset.
+ """
+ try:
+ image_file_rel = self.imgname[idx].decode('utf-8')
+ except AttributeError:
+ image_file_rel = self.imgname[idx]
+ image_file = os.path.join(self.img_dir, image_file_rel)
+ keypoints_2d = self.keypoints_2d[idx].copy()
+ keypoints_3d = self.keypoints_3d[idx].copy()
+
+ center = self.center[idx].copy()
+ center_x = center[0]
+ center_y = center[1]
+ scale = self.scale[idx]
+ right = self.right[idx].copy()
+ if self.rescale_factor == -1:
+ BBOX_SHAPE = self.cfg.MODEL.get('BBOX_SHAPE', None)
+ bbox_size = expand_to_aspect_ratio(scale*200, target_aspect_ratio=BBOX_SHAPE).max()
+ bbox_expand_factor = bbox_size / ((scale*200).max())
+ else:
+ bbox_expand_factor = self.rescale_factor
+ bbox_size = bbox_expand_factor*scale.max()*200
+ hand_pose = self.hand_pose[idx].copy().astype(np.float32)
+ betas = self.betas[idx].copy().astype(np.float32)
+
+ has_hand_pose = self.has_hand_pose[idx].copy()
+ has_betas = self.has_betas[idx].copy()
+
+ mano_params = {'global_orient': hand_pose[:3],
+ 'hand_pose': hand_pose[3:],
+ 'betas': betas
+ }
+
+ has_mano_params = {'global_orient': has_hand_pose,
+ 'hand_pose': has_hand_pose,
+ 'betas': has_betas
+ }
+
+ mano_params_is_axis_angle = {'global_orient': True,
+ 'hand_pose': True,
+ 'betas': False
+ }
+
+ augm_config = self.cfg.DATASETS.CONFIG
+ # Crop image and (possibly) perform data augmentation
+ img_patch, keypoints_2d, keypoints_3d, mano_params, has_mano_params, img_size = get_example(image_file,
+ center_x, center_y,
+ bbox_size, bbox_size,
+ keypoints_2d, keypoints_3d,
+ mano_params, has_mano_params,
+ self.flip_keypoint_permutation,
+ self.img_size, self.img_size,
+ self.mean, self.std, self.train, right, augm_config)
+ item = {}
+ # These are the keypoints in the original image coordinates (before cropping)
+ orig_keypoints_2d = self.keypoints_2d[idx].copy()
+
+ item['img'] = img_patch
+ item['keypoints_2d'] = keypoints_2d.astype(np.float32)
+ item['keypoints_3d'] = keypoints_3d.astype(np.float32)
+ item['orig_keypoints_2d'] = orig_keypoints_2d
+ item['box_center'] = self.center[idx].copy()
+ item['box_size'] = bbox_size
+ item['bbox_expand_factor'] = bbox_expand_factor
+ item['img_size'] = 1.0 * img_size[::-1].copy()
+ item['mano_params'] = mano_params
+ item['has_mano_params'] = has_mano_params
+ item['mano_params_is_axis_angle'] = mano_params_is_axis_angle
+ item['imgname'] = image_file
+ item['imgname_rel'] = image_file_rel
+ item['personid'] = int(self.personid[idx])
+ item['extra_info'] = copy.deepcopy(self.extra_info[idx])
+ item['idx'] = idx
+ item['_scale'] = scale
+ item['right'] = self.right[idx].copy()
+ return item
+
+ @staticmethod
+ def load_tars_as_webdataset(cfg: CfgNode, urls: str|List[str], train: bool,
+ resampled=False,
+ epoch_size=None,
+ cache_dir=None,
+ **kwargs) -> Dataset:
+ """
+ Loads the dataset from a webdataset tar file.
+ """
+
+ IMG_SIZE = cfg.MODEL.IMAGE_SIZE
+ BBOX_SHAPE = cfg.MODEL.get('BBOX_SHAPE', None)
+ MEAN = 255. * np.array(cfg.MODEL.IMAGE_MEAN)
+ STD = 255. * np.array(cfg.MODEL.IMAGE_STD)
+
+ def split_data(source):
+ for item in source:
+ datas = item['data.pyd']
+ for data in datas:
+ if 'detection.npz' in item:
+ det_idx = data['extra_info']['detection_npz_idx']
+ mask = item['detection.npz']['masks'][det_idx]
+ else:
+ mask = np.ones_like(item['jpg'][:,:,0], dtype=bool)
+ yield {
+ '__key__': item['__key__'],
+ 'jpg': item['jpg'],
+ 'data.pyd': data,
+ 'mask': mask,
+ }
+
+ def suppress_bad_kps(item, thresh=0.0):
+ if thresh > 0:
+ kp2d = item['data.pyd']['keypoints_2d']
+ kp2d_conf = np.where(kp2d[:, 2] < thresh, 0.0, kp2d[:, 2])
+ item['data.pyd']['keypoints_2d'] = np.concatenate([kp2d[:,:2], kp2d_conf[:,None]], axis=1)
+ return item
+
+ def filter_numkp(item, numkp=4, thresh=0.0):
+ kp_conf = item['data.pyd']['keypoints_2d'][:, 2]
+ return (kp_conf > thresh).sum() > numkp
+
+ def filter_reproj_error(item, thresh=10**4.5):
+ losses = item['data.pyd'].get('extra_info', {}).get('fitting_loss', np.array({})).item()
+ reproj_loss = losses.get('reprojection_loss', None)
+ return reproj_loss is None or reproj_loss < thresh
+
+ def filter_bbox_size(item, thresh=1):
+ bbox_size_min = item['data.pyd']['scale'].min().item() * 200.
+ return bbox_size_min > thresh
+
+ def filter_no_poses(item):
+ return (item['data.pyd']['has_hand_pose'] > 0)
+
+ def supress_bad_betas(item, thresh=3):
+ has_betas = item['data.pyd']['has_betas']
+ if thresh > 0 and has_betas:
+ betas_abs = np.abs(item['data.pyd']['betas'])
+ if (betas_abs > thresh).any():
+ item['data.pyd']['has_betas'] = False
+ return item
+
+ def supress_bad_poses(item):
+ has_hand_pose = item['data.pyd']['has_hand_pose']
+ if has_hand_pose:
+ hand_pose = item['data.pyd']['hand_pose']
+ pose_is_probable = poses_check_probable(torch.from_numpy(hand_pose)[None, 3:], amass_poses_hist100_smooth).item()
+ if not pose_is_probable:
+ item['data.pyd']['has_hand_pose'] = False
+ return item
+
+ def poses_betas_simultaneous(item):
+ # We either have both hand_pose and betas, or neither
+ has_betas = item['data.pyd']['has_betas']
+ has_hand_pose = item['data.pyd']['has_hand_pose']
+ item['data.pyd']['has_betas'] = item['data.pyd']['has_hand_pose'] = np.array(float((has_hand_pose>0) and (has_betas>0)))
+ return item
+
+ def set_betas_for_reg(item):
+ # Always have betas set to true
+ has_betas = item['data.pyd']['has_betas']
+ betas = item['data.pyd']['betas']
+
+ if not (has_betas>0):
+ item['data.pyd']['has_betas'] = np.array(float((True)))
+ item['data.pyd']['betas'] = betas * 0
+ return item
+
+ # Load the dataset
+ if epoch_size is not None:
+ resampled = True
+ #corrupt_filter = lambda sample: (sample['__key__'] not in CORRUPT_KEYS)
+ import webdataset as wds
+ dataset = wds.WebDataset(expand_urls(urls),
+ nodesplitter=wds.split_by_node,
+ shardshuffle=True,
+ resampled=resampled,
+ cache_dir=cache_dir,
+ ) #.select(corrupt_filter)
+ if train:
+ dataset = dataset.shuffle(100)
+ dataset = dataset.decode('rgb8').rename(jpg='jpg;jpeg;png')
+
+ # Process the dataset
+ dataset = dataset.compose(split_data)
+
+ # Filter/clean the dataset
+ SUPPRESS_KP_CONF_THRESH = cfg.DATASETS.get('SUPPRESS_KP_CONF_THRESH', 0.0)
+ SUPPRESS_BETAS_THRESH = cfg.DATASETS.get('SUPPRESS_BETAS_THRESH', 0.0)
+ SUPPRESS_BAD_POSES = cfg.DATASETS.get('SUPPRESS_BAD_POSES', False)
+ POSES_BETAS_SIMULTANEOUS = cfg.DATASETS.get('POSES_BETAS_SIMULTANEOUS', False)
+ BETAS_REG = cfg.DATASETS.get('BETAS_REG', False)
+ FILTER_NO_POSES = cfg.DATASETS.get('FILTER_NO_POSES', False)
+ FILTER_NUM_KP = cfg.DATASETS.get('FILTER_NUM_KP', 4)
+ FILTER_NUM_KP_THRESH = cfg.DATASETS.get('FILTER_NUM_KP_THRESH', 0.0)
+ FILTER_REPROJ_THRESH = cfg.DATASETS.get('FILTER_REPROJ_THRESH', 0.0)
+ FILTER_MIN_BBOX_SIZE = cfg.DATASETS.get('FILTER_MIN_BBOX_SIZE', 0.0)
+ if SUPPRESS_KP_CONF_THRESH > 0:
+ dataset = dataset.map(lambda x: suppress_bad_kps(x, thresh=SUPPRESS_KP_CONF_THRESH))
+ if SUPPRESS_BETAS_THRESH > 0:
+ dataset = dataset.map(lambda x: supress_bad_betas(x, thresh=SUPPRESS_BETAS_THRESH))
+ if SUPPRESS_BAD_POSES:
+ dataset = dataset.map(lambda x: supress_bad_poses(x))
+ if POSES_BETAS_SIMULTANEOUS:
+ dataset = dataset.map(lambda x: poses_betas_simultaneous(x))
+ if FILTER_NO_POSES:
+ dataset = dataset.select(lambda x: filter_no_poses(x))
+ if FILTER_NUM_KP > 0:
+ dataset = dataset.select(lambda x: filter_numkp(x, numkp=FILTER_NUM_KP, thresh=FILTER_NUM_KP_THRESH))
+ if FILTER_REPROJ_THRESH > 0:
+ dataset = dataset.select(lambda x: filter_reproj_error(x, thresh=FILTER_REPROJ_THRESH))
+ if FILTER_MIN_BBOX_SIZE > 0:
+ dataset = dataset.select(lambda x: filter_bbox_size(x, thresh=FILTER_MIN_BBOX_SIZE))
+ if BETAS_REG:
+ dataset = dataset.map(lambda x: set_betas_for_reg(x)) # NOTE: Must be at the end
+
+ use_skimage_antialias = cfg.DATASETS.get('USE_SKIMAGE_ANTIALIAS', False)
+ border_mode = {
+ 'constant': cv2.BORDER_CONSTANT,
+ 'replicate': cv2.BORDER_REPLICATE,
+ }[cfg.DATASETS.get('BORDER_MODE', 'constant')]
+
+ # Process the dataset further
+ dataset = dataset.map(lambda x: ImageDataset.process_webdataset_tar_item(x, train,
+ augm_config=cfg.DATASETS.CONFIG,
+ MEAN=MEAN, STD=STD, IMG_SIZE=IMG_SIZE,
+ BBOX_SHAPE=BBOX_SHAPE,
+ use_skimage_antialias=use_skimage_antialias,
+ border_mode=border_mode,
+ ))
+ if epoch_size is not None:
+ dataset = dataset.with_epoch(epoch_size)
+
+ return dataset
+
+ @staticmethod
+ def process_webdataset_tar_item(item, train,
+ augm_config=None,
+ MEAN=DEFAULT_MEAN,
+ STD=DEFAULT_STD,
+ IMG_SIZE=DEFAULT_IMG_SIZE,
+ BBOX_SHAPE=None,
+ use_skimage_antialias=False,
+ border_mode=cv2.BORDER_CONSTANT,
+ ):
+ # Read data from item
+ key = item['__key__']
+ image = item['jpg']
+ data = item['data.pyd']
+ mask = item['mask']
+
+ keypoints_2d = data['keypoints_2d']
+ keypoints_3d = data['keypoints_3d']
+ center = data['center']
+ scale = data['scale']
+ hand_pose = data['hand_pose']
+ betas = data['betas']
+ right = data['right']
+ has_hand_pose = data['has_hand_pose']
+ has_betas = data['has_betas']
+ # image_file = data['image_file']
+
+ # Process data
+ orig_keypoints_2d = keypoints_2d.copy()
+ center_x = center[0]
+ center_y = center[1]
+ bbox_size = expand_to_aspect_ratio(scale*200, target_aspect_ratio=BBOX_SHAPE).max()
+ if bbox_size < 1:
+ breakpoint()
+
+
+ mano_params = {'global_orient': hand_pose[:3],
+ 'hand_pose': hand_pose[3:],
+ 'betas': betas
+ }
+
+ has_mano_params = {'global_orient': has_hand_pose,
+ 'hand_pose': has_hand_pose,
+ 'betas': has_betas
+ }
+
+ mano_params_is_axis_angle = {'global_orient': True,
+ 'hand_pose': True,
+ 'betas': False
+ }
+
+ augm_config = copy.deepcopy(augm_config)
+ # Crop image and (possibly) perform data augmentation
+ img_rgba = np.concatenate([image, mask.astype(np.uint8)[:,:,None]*255], axis=2)
+ img_patch_rgba, keypoints_2d, keypoints_3d, mano_params, has_mano_params, img_size, trans = get_example(img_rgba,
+ center_x, center_y,
+ bbox_size, bbox_size,
+ keypoints_2d, keypoints_3d,
+ mano_params, has_mano_params,
+ FLIP_KEYPOINT_PERMUTATION,
+ IMG_SIZE, IMG_SIZE,
+ MEAN, STD, train, right, augm_config,
+ is_bgr=False, return_trans=True,
+ use_skimage_antialias=use_skimage_antialias,
+ border_mode=border_mode,
+ )
+ img_patch = img_patch_rgba[:3,:,:]
+ mask_patch = (img_patch_rgba[3,:,:] / 255.0).clip(0,1)
+ if (mask_patch < 0.5).all():
+ mask_patch = np.ones_like(mask_patch)
+
+ item = {}
+
+ item['img'] = img_patch
+ item['mask'] = mask_patch
+ # item['img_og'] = image
+ # item['mask_og'] = mask
+ item['keypoints_2d'] = keypoints_2d.astype(np.float32)
+ item['keypoints_3d'] = keypoints_3d.astype(np.float32)
+ item['orig_keypoints_2d'] = orig_keypoints_2d
+ item['box_center'] = center.copy()
+ item['box_size'] = bbox_size
+ item['img_size'] = 1.0 * img_size[::-1].copy()
+ item['mano_params'] = mano_params
+ item['has_mano_params'] = has_mano_params
+ item['mano_params_is_axis_angle'] = mano_params_is_axis_angle
+ item['_scale'] = scale
+ item['_trans'] = trans
+ item['imgname'] = key
+ # item['idx'] = idx
+ return item
diff --git a/third_party/hamer/hamer/datasets/json_dataset.py b/third_party/hamer/hamer/datasets/json_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..4e258a3e8b84baa386d0edcb75ef45a4770c6301
--- /dev/null
+++ b/third_party/hamer/hamer/datasets/json_dataset.py
@@ -0,0 +1,213 @@
+import copy
+import os
+import json
+import glob
+import numpy as np
+import torch
+from typing import Any, Dict, List
+from yacs.config import CfgNode
+import braceexpand
+import cv2
+
+from .dataset import Dataset
+from .utils import get_example, expand_to_aspect_ratio
+from .smplh_prob_filter import poses_check_probable, load_amass_hist_smooth
+
+def expand(s):
+ return os.path.expanduser(os.path.expandvars(s))
+def expand_urls(urls: str|List[str]):
+ if isinstance(urls, str):
+ urls = [urls]
+ urls = [u for url in urls for u in braceexpand.braceexpand(expand(url))]
+ return urls
+
+AIC_TRAIN_CORRUPT_KEYS = {
+ '0a047f0124ae48f8eee15a9506ce1449ee1ba669',
+ '1a703aa174450c02fbc9cfbf578a5435ef403689',
+ '0394e6dc4df78042929b891dbc24f0fd7ffb6b6d',
+ '5c032b9626e410441544c7669123ecc4ae077058',
+ 'ca018a7b4c5f53494006ebeeff9b4c0917a55f07',
+ '4a77adb695bef75a5d34c04d589baf646fe2ba35',
+ 'a0689017b1065c664daef4ae2d14ea03d543217e',
+ '39596a45cbd21bed4a5f9c2342505532f8ec5cbb',
+ '3d33283b40610d87db660b62982f797d50a7366b',
+}
+CORRUPT_KEYS = {
+ *{f'aic-train/{k}' for k in AIC_TRAIN_CORRUPT_KEYS},
+ *{f'aic-train-vitpose/{k}' for k in AIC_TRAIN_CORRUPT_KEYS},
+}
+
+FLIP_KEYPOINT_PERMUTATION = [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20]
+
+DEFAULT_MEAN = 255. * np.array([0.485, 0.456, 0.406])
+DEFAULT_STD = 255. * np.array([0.229, 0.224, 0.225])
+DEFAULT_IMG_SIZE = 256
+
+class JsonDataset(Dataset):
+
+ def __init__(self,
+ cfg: CfgNode,
+ dataset_file: str,
+ img_dir: str,
+ right: bool,
+ train: bool = False,
+ prune: Dict[str, Any] = {},
+ **kwargs):
+ """
+ Dataset class used for loading images and corresponding annotations.
+ Args:
+ cfg (CfgNode): Model config file.
+ dataset_file (str): Path to npz file containing dataset info.
+ img_dir (str): Path to image folder.
+ train (bool): Whether it is for training or not (enables data augmentation).
+ """
+ super(JsonDataset, self).__init__()
+ self.train = train
+ self.cfg = cfg
+
+ self.img_size = cfg.MODEL.IMAGE_SIZE
+ self.mean = 255. * np.array(self.cfg.MODEL.IMAGE_MEAN)
+ self.std = 255. * np.array(self.cfg.MODEL.IMAGE_STD)
+
+ self.img_dir = img_dir
+ boxes = np.array(json.load(open(dataset_file, 'rb')))
+
+ self.imgname = glob.glob(os.path.join(self.img_dir,'*.jpg'))
+ self.imgname.sort()
+
+ self.flip_keypoint_permutation = copy.copy(FLIP_KEYPOINT_PERMUTATION)
+
+ num_pose = 3 * (self.cfg.MANO.NUM_HAND_JOINTS + 1)
+
+ # Bounding boxes are assumed to be in the center and scale format
+ boxes = boxes.astype(np.float32)
+ self.center = (boxes[:, 2:4] + boxes[:, 0:2]) / 2.0
+ self.scale = 2 * (boxes[:, 2:4] - boxes[:, 0:2]) / 200.0
+ self.personid = np.arange(len(boxes), dtype=np.int32)
+ if right:
+ self.right = np.ones(len(self.imgname), dtype=np.float32)
+ else:
+ self.right = np.zeros(len(self.imgname), dtype=np.float32)
+ assert self.scale.shape == (len(self.center), 2)
+
+ # Get gt SMPLX parameters, if available
+ try:
+ self.hand_pose = self.data['hand_pose'].astype(np.float32)
+ self.has_hand_pose = self.data['has_hand_pose'].astype(np.float32)
+ except:
+ self.hand_pose = np.zeros((len(self.imgname), num_pose), dtype=np.float32)
+ self.has_hand_pose = np.zeros(len(self.imgname), dtype=np.float32)
+ try:
+ self.betas = self.data['betas'].astype(np.float32)
+ self.has_betas = self.data['has_betas'].astype(np.float32)
+ except:
+ self.betas = np.zeros((len(self.imgname), 10), dtype=np.float32)
+ self.has_betas = np.zeros(len(self.imgname), dtype=np.float32)
+
+ # Try to get 2d keypoints, if available
+ try:
+ hand_keypoints_2d = self.data['hand_keypoints_2d']
+ except:
+ hand_keypoints_2d = np.zeros((len(self.center), 21, 3))
+ ## Try to get extra 2d keypoints, if available
+ #try:
+ # extra_keypoints_2d = self.data['extra_keypoints_2d']
+ #except KeyError:
+ # extra_keypoints_2d = np.zeros((len(self.center), 19, 3))
+
+ #self.keypoints_2d = np.concatenate((hand_keypoints_2d, extra_keypoints_2d), axis=1).astype(np.float32)
+ self.keypoints_2d = hand_keypoints_2d
+
+ # Try to get 3d keypoints, if available
+ try:
+ hand_keypoints_3d = self.data['hand_keypoints_3d'].astype(np.float32)
+ except:
+ hand_keypoints_3d = np.zeros((len(self.center), 21, 4), dtype=np.float32)
+ ## Try to get extra 3d keypoints, if available
+ #try:
+ # extra_keypoints_3d = self.data['extra_keypoints_3d'].astype(np.float32)
+ #except KeyError:
+ # extra_keypoints_3d = np.zeros((len(self.center), 19, 4), dtype=np.float32)
+
+ self.keypoints_3d = hand_keypoints_3d
+
+ #body_keypoints_3d[:, [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14], -1] = 0
+
+ #self.keypoints_3d = np.concatenate((body_keypoints_3d, extra_keypoints_3d), axis=1).astype(np.float32)
+
+ def __len__(self) -> int:
+ return len(self.scale)
+
+ def __getitem__(self, idx: int) -> Dict:
+ """
+ Returns an example from the dataset.
+ """
+ try:
+ image_file = self.imgname[idx].decode('utf-8')
+ except AttributeError:
+ image_file = self.imgname[idx]
+ keypoints_2d = self.keypoints_2d[idx].copy()
+ keypoints_3d = self.keypoints_3d[idx].copy()
+
+ center = self.center[idx].copy()
+ center_x = center[0]
+ center_y = center[1]
+ scale = self.scale[idx]
+ right = self.right[idx].copy()
+ BBOX_SHAPE = self.cfg.MODEL.get('BBOX_SHAPE', None)
+ #bbox_size = expand_to_aspect_ratio(scale*200, target_aspect_ratio=BBOX_SHAPE).max()
+ bbox_size = ((scale*200).max())
+ bbox_expand_factor = bbox_size / ((scale*200).max())
+ hand_pose = self.hand_pose[idx].copy().astype(np.float32)
+ betas = self.betas[idx].copy().astype(np.float32)
+
+ has_hand_pose = self.has_hand_pose[idx].copy()
+ has_betas = self.has_betas[idx].copy()
+
+ mano_params = {'global_orient': hand_pose[:3],
+ 'hand_pose': hand_pose[3:],
+ 'betas': betas
+ }
+
+ has_mano_params = {'global_orient': has_hand_pose,
+ 'hand_pose': has_hand_pose,
+ 'betas': has_betas
+ }
+
+ mano_params_is_axis_angle = {'global_orient': True,
+ 'hand_pose': True,
+ 'betas': False
+ }
+
+ augm_config = self.cfg.DATASETS.CONFIG
+ # Crop image and (possibly) perform data augmentation
+ img_patch, keypoints_2d, keypoints_3d, mano_params, has_mano_params, img_size = get_example(image_file,
+ center_x, center_y,
+ bbox_size, bbox_size,
+ keypoints_2d, keypoints_3d,
+ mano_params, has_mano_params,
+ self.flip_keypoint_permutation,
+ self.img_size, self.img_size,
+ self.mean, self.std, self.train, right, augm_config)
+
+ item = {}
+ # These are the keypoints in the original image coordinates (before cropping)
+ orig_keypoints_2d = self.keypoints_2d[idx].copy()
+
+ item['img'] = img_patch
+ item['keypoints_2d'] = keypoints_2d.astype(np.float32)
+ item['keypoints_3d'] = keypoints_3d.astype(np.float32)
+ item['orig_keypoints_2d'] = orig_keypoints_2d
+ item['box_center'] = self.center[idx].copy()
+ item['box_size'] = bbox_size
+ item['bbox_expand_factor'] = bbox_expand_factor
+ item['img_size'] = 1.0 * img_size[::-1].copy()
+ item['mano_params'] = mano_params
+ item['has_mano_params'] = has_mano_params
+ item['mano_params_is_axis_angle'] = mano_params_is_axis_angle
+ item['imgname'] = image_file
+ item['personid'] = int(self.personid[idx])
+ item['idx'] = idx
+ item['_scale'] = scale
+ item['right'] = self.right[idx].copy()
+ return item
diff --git a/third_party/hamer/hamer/datasets/mocap_dataset.py b/third_party/hamer/hamer/datasets/mocap_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..cbf808f83c462646a19eed7e33dea4e50037b512
--- /dev/null
+++ b/third_party/hamer/hamer/datasets/mocap_dataset.py
@@ -0,0 +1,25 @@
+import numpy as np
+from typing import Dict
+
+class MoCapDataset:
+
+ def __init__(self, dataset_file: str):
+ """
+ Dataset class used for loading a dataset of unpaired MANO parameter annotations
+ Args:
+ cfg (CfgNode): Model config file.
+ dataset_file (str): Path to npz file containing dataset info.
+ """
+ data = np.load(dataset_file)
+ self.pose = data['hand_pose'].astype(np.float32)[:, 3:]
+ self.betas = data['betas'].astype(np.float32)
+ self.length = len(self.pose)
+
+ def __getitem__(self, idx: int) -> Dict:
+ pose = self.pose[idx].copy()
+ betas = self.betas[idx].copy()
+ item = {'hand_pose': pose, 'betas': betas}
+ return item
+
+ def __len__(self) -> int:
+ return self.length
diff --git a/third_party/hamer/hamer/datasets/utils.py b/third_party/hamer/hamer/datasets/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..73ded82fcd02ebf95895e3edf6a680f045919d35
--- /dev/null
+++ b/third_party/hamer/hamer/datasets/utils.py
@@ -0,0 +1,993 @@
+"""
+Parts of the code are taken or adapted from
+https://github.com/mkocabas/EpipolarPose/blob/master/lib/utils/img_utils.py
+"""
+import torch
+import numpy as np
+from skimage.transform import rotate, resize
+from skimage.filters import gaussian
+import random
+import cv2
+from typing import List, Dict, Tuple
+from yacs.config import CfgNode
+
+def expand_to_aspect_ratio(input_shape, target_aspect_ratio=None):
+ """Increase the size of the bounding box to match the target shape."""
+ if target_aspect_ratio is None:
+ return input_shape
+
+ try:
+ w , h = input_shape
+ except (ValueError, TypeError):
+ return input_shape
+
+ w_t, h_t = target_aspect_ratio
+ if h / w < h_t / w_t:
+ h_new = max(w * h_t / w_t, h)
+ w_new = w
+ else:
+ h_new = h
+ w_new = max(h * w_t / h_t, w)
+ if h_new < h or w_new < w:
+ breakpoint()
+ return np.array([w_new, h_new])
+
+def do_augmentation(aug_config: CfgNode) -> Tuple:
+ """
+ Compute random augmentation parameters.
+ Args:
+ aug_config (CfgNode): Config containing augmentation parameters.
+ Returns:
+ scale (float): Box rescaling factor.
+ rot (float): Random image rotation.
+ do_flip (bool): Whether to flip image or not.
+ do_extreme_crop (bool): Whether to apply extreme cropping (as proposed in EFT).
+ color_scale (List): Color rescaling factor
+ tx (float): Random translation along the x axis.
+ ty (float): Random translation along the y axis.
+ """
+
+ tx = np.clip(np.random.randn(), -1.0, 1.0) * aug_config.TRANS_FACTOR
+ ty = np.clip(np.random.randn(), -1.0, 1.0) * aug_config.TRANS_FACTOR
+ scale = np.clip(np.random.randn(), -1.0, 1.0) * aug_config.SCALE_FACTOR + 1.0
+ rot = np.clip(np.random.randn(), -2.0,
+ 2.0) * aug_config.ROT_FACTOR if random.random() <= aug_config.ROT_AUG_RATE else 0
+ do_flip = aug_config.DO_FLIP and random.random() <= aug_config.FLIP_AUG_RATE
+ do_extreme_crop = random.random() <= aug_config.EXTREME_CROP_AUG_RATE
+ extreme_crop_lvl = aug_config.get('EXTREME_CROP_AUG_LEVEL', 0)
+ # extreme_crop_lvl = 0
+ c_up = 1.0 + aug_config.COLOR_SCALE
+ c_low = 1.0 - aug_config.COLOR_SCALE
+ color_scale = [random.uniform(c_low, c_up), random.uniform(c_low, c_up), random.uniform(c_low, c_up)]
+ return scale, rot, do_flip, do_extreme_crop, extreme_crop_lvl, color_scale, tx, ty
+
+def rotate_2d(pt_2d: np.array, rot_rad: float) -> np.array:
+ """
+ Rotate a 2D point on the x-y plane.
+ Args:
+ pt_2d (np.array): Input 2D point with shape (2,).
+ rot_rad (float): Rotation angle
+ Returns:
+ np.array: Rotated 2D point.
+ """
+ x = pt_2d[0]
+ y = pt_2d[1]
+ sn, cs = np.sin(rot_rad), np.cos(rot_rad)
+ xx = x * cs - y * sn
+ yy = x * sn + y * cs
+ return np.array([xx, yy], dtype=np.float32)
+
+
+def gen_trans_from_patch_cv(c_x: float, c_y: float,
+ src_width: float, src_height: float,
+ dst_width: float, dst_height: float,
+ scale: float, rot: float) -> np.array:
+ """
+ Create transformation matrix for the bounding box crop.
+ Args:
+ c_x (float): Bounding box center x coordinate in the original image.
+ c_y (float): Bounding box center y coordinate in the original image.
+ src_width (float): Bounding box width.
+ src_height (float): Bounding box height.
+ dst_width (float): Output box width.
+ dst_height (float): Output box height.
+ scale (float): Rescaling factor for the bounding box (augmentation).
+ rot (float): Random rotation applied to the box.
+ Returns:
+ trans (np.array): Target geometric transformation.
+ """
+ # augment size with scale
+ src_w = src_width * scale
+ src_h = src_height * scale
+ src_center = np.zeros(2)
+ src_center[0] = c_x
+ src_center[1] = c_y
+ # augment rotation
+ rot_rad = np.pi * rot / 180
+ src_downdir = rotate_2d(np.array([0, src_h * 0.5], dtype=np.float32), rot_rad)
+ src_rightdir = rotate_2d(np.array([src_w * 0.5, 0], dtype=np.float32), rot_rad)
+
+ dst_w = dst_width
+ dst_h = dst_height
+ dst_center = np.array([dst_w * 0.5, dst_h * 0.5], dtype=np.float32)
+ dst_downdir = np.array([0, dst_h * 0.5], dtype=np.float32)
+ dst_rightdir = np.array([dst_w * 0.5, 0], dtype=np.float32)
+
+ src = np.zeros((3, 2), dtype=np.float32)
+ src[0, :] = src_center
+ src[1, :] = src_center + src_downdir
+ src[2, :] = src_center + src_rightdir
+
+ dst = np.zeros((3, 2), dtype=np.float32)
+ dst[0, :] = dst_center
+ dst[1, :] = dst_center + dst_downdir
+ dst[2, :] = dst_center + dst_rightdir
+
+ trans = cv2.getAffineTransform(np.float32(src), np.float32(dst))
+
+ return trans
+
+
+def trans_point2d(pt_2d: np.array, trans: np.array):
+ """
+ Transform a 2D point using translation matrix trans.
+ Args:
+ pt_2d (np.array): Input 2D point with shape (2,).
+ trans (np.array): Transformation matrix.
+ Returns:
+ np.array: Transformed 2D point.
+ """
+ src_pt = np.array([pt_2d[0], pt_2d[1], 1.]).T
+ dst_pt = np.dot(trans, src_pt)
+ return dst_pt[0:2]
+
+def get_transform(center, scale, res, rot=0):
+ """Generate transformation matrix."""
+ """Taken from PARE: https://github.com/mkocabas/PARE/blob/6e0caca86c6ab49ff80014b661350958e5b72fd8/pare/utils/image_utils.py"""
+ h = 200 * scale
+ t = np.zeros((3, 3))
+ t[0, 0] = float(res[1]) / h
+ t[1, 1] = float(res[0]) / h
+ t[0, 2] = res[1] * (-float(center[0]) / h + .5)
+ t[1, 2] = res[0] * (-float(center[1]) / h + .5)
+ t[2, 2] = 1
+ if not rot == 0:
+ rot = -rot # To match direction of rotation from cropping
+ rot_mat = np.zeros((3, 3))
+ rot_rad = rot * np.pi / 180
+ sn, cs = np.sin(rot_rad), np.cos(rot_rad)
+ rot_mat[0, :2] = [cs, -sn]
+ rot_mat[1, :2] = [sn, cs]
+ rot_mat[2, 2] = 1
+ # Need to rotate around center
+ t_mat = np.eye(3)
+ t_mat[0, 2] = -res[1] / 2
+ t_mat[1, 2] = -res[0] / 2
+ t_inv = t_mat.copy()
+ t_inv[:2, 2] *= -1
+ t = np.dot(t_inv, np.dot(rot_mat, np.dot(t_mat, t)))
+ return t
+
+
+def transform(pt, center, scale, res, invert=0, rot=0, as_int=True):
+ """Transform pixel location to different reference."""
+ """Taken from PARE: https://github.com/mkocabas/PARE/blob/6e0caca86c6ab49ff80014b661350958e5b72fd8/pare/utils/image_utils.py"""
+ t = get_transform(center, scale, res, rot=rot)
+ if invert:
+ t = np.linalg.inv(t)
+ new_pt = np.array([pt[0] - 1, pt[1] - 1, 1.]).T
+ new_pt = np.dot(t, new_pt)
+ if as_int:
+ new_pt = new_pt.astype(int)
+ return new_pt[:2] + 1
+
+def crop_img(img, ul, br, border_mode=cv2.BORDER_CONSTANT, border_value=0):
+ c_x = (ul[0] + br[0])/2
+ c_y = (ul[1] + br[1])/2
+ bb_width = patch_width = br[0] - ul[0]
+ bb_height = patch_height = br[1] - ul[1]
+ trans = gen_trans_from_patch_cv(c_x, c_y, bb_width, bb_height, patch_width, patch_height, 1.0, 0)
+ img_patch = cv2.warpAffine(img, trans, (int(patch_width), int(patch_height)),
+ flags=cv2.INTER_LINEAR,
+ borderMode=border_mode,
+ borderValue=border_value
+ )
+
+ # Force borderValue=cv2.BORDER_CONSTANT for alpha channel
+ if (img.shape[2] == 4) and (border_mode != cv2.BORDER_CONSTANT):
+ img_patch[:,:,3] = cv2.warpAffine(img[:,:,3], trans, (int(patch_width), int(patch_height)),
+ flags=cv2.INTER_LINEAR,
+ borderMode=cv2.BORDER_CONSTANT,
+ )
+
+ return img_patch
+
+def generate_image_patch_skimage(img: np.array, c_x: float, c_y: float,
+ bb_width: float, bb_height: float,
+ patch_width: float, patch_height: float,
+ do_flip: bool, scale: float, rot: float,
+ border_mode=cv2.BORDER_CONSTANT, border_value=0) -> Tuple[np.array, np.array]:
+ """
+ Crop image according to the supplied bounding box.
+ Args:
+ img (np.array): Input image of shape (H, W, 3)
+ c_x (float): Bounding box center x coordinate in the original image.
+ c_y (float): Bounding box center y coordinate in the original image.
+ bb_width (float): Bounding box width.
+ bb_height (float): Bounding box height.
+ patch_width (float): Output box width.
+ patch_height (float): Output box height.
+ do_flip (bool): Whether to flip image or not.
+ scale (float): Rescaling factor for the bounding box (augmentation).
+ rot (float): Random rotation applied to the box.
+ Returns:
+ img_patch (np.array): Cropped image patch of shape (patch_height, patch_height, 3)
+ trans (np.array): Transformation matrix.
+ """
+
+ img_height, img_width, img_channels = img.shape
+ if do_flip:
+ img = img[:, ::-1, :]
+ c_x = img_width - c_x - 1
+
+ trans = gen_trans_from_patch_cv(c_x, c_y, bb_width, bb_height, patch_width, patch_height, scale, rot)
+
+ #img_patch = cv2.warpAffine(img, trans, (int(patch_width), int(patch_height)), flags=cv2.INTER_LINEAR)
+
+ # skimage
+ center = np.zeros(2)
+ center[0] = c_x
+ center[1] = c_y
+ res = np.zeros(2)
+ res[0] = patch_width
+ res[1] = patch_height
+ # assumes bb_width = bb_height
+ # assumes patch_width = patch_height
+ assert bb_width == bb_height, f'{bb_width=} != {bb_height=}'
+ assert patch_width == patch_height, f'{patch_width=} != {patch_height=}'
+ scale1 = scale*bb_width/200.
+
+ # Upper left point
+ ul = np.array(transform([1, 1], center, scale1, res, invert=1, as_int=False)) - 1
+ # Bottom right point
+ br = np.array(transform([res[0] + 1,
+ res[1] + 1], center, scale1, res, invert=1, as_int=False)) - 1
+
+ # Padding so that when rotated proper amount of context is included
+ try:
+ pad = int(np.linalg.norm(br - ul) / 2 - float(br[1] - ul[1]) / 2) + 1
+ except:
+ breakpoint()
+ if not rot == 0:
+ ul -= pad
+ br += pad
+
+
+ if False:
+ # Old way of cropping image
+ ul_int = ul.astype(int)
+ br_int = br.astype(int)
+ new_shape = [br_int[1] - ul_int[1], br_int[0] - ul_int[0]]
+ if len(img.shape) > 2:
+ new_shape += [img.shape[2]]
+ new_img = np.zeros(new_shape)
+
+ # Range to fill new array
+ new_x = max(0, -ul_int[0]), min(br_int[0], len(img[0])) - ul_int[0]
+ new_y = max(0, -ul_int[1]), min(br_int[1], len(img)) - ul_int[1]
+ # Range to sample from original image
+ old_x = max(0, ul_int[0]), min(len(img[0]), br_int[0])
+ old_y = max(0, ul_int[1]), min(len(img), br_int[1])
+ new_img[new_y[0]:new_y[1], new_x[0]:new_x[1]] = img[old_y[0]:old_y[1],
+ old_x[0]:old_x[1]]
+
+ # New way of cropping image
+ new_img = crop_img(img, ul, br, border_mode=border_mode, border_value=border_value).astype(np.float32)
+
+ # print(f'{new_img.shape=}')
+ # print(f'{new_img1.shape=}')
+ # print(f'{np.allclose(new_img, new_img1)=}')
+ # print(f'{img.dtype=}')
+
+
+ if not rot == 0:
+ # Remove padding
+
+ new_img = rotate(new_img, rot) # scipy.misc.imrotate(new_img, rot)
+ new_img = new_img[pad:-pad, pad:-pad]
+
+ if new_img.shape[0] < 1 or new_img.shape[1] < 1:
+ print(f'{img.shape=}')
+ print(f'{new_img.shape=}')
+ print(f'{ul=}')
+ print(f'{br=}')
+ print(f'{pad=}')
+ print(f'{rot=}')
+
+ breakpoint()
+
+ # resize image
+ new_img = resize(new_img, res) # scipy.misc.imresize(new_img, res)
+
+ new_img = np.clip(new_img, 0, 255).astype(np.uint8)
+
+ return new_img, trans
+
+
+def generate_image_patch_cv2(img: np.array, c_x: float, c_y: float,
+ bb_width: float, bb_height: float,
+ patch_width: float, patch_height: float,
+ do_flip: bool, scale: float, rot: float,
+ border_mode=cv2.BORDER_CONSTANT, border_value=0) -> Tuple[np.array, np.array]:
+ """
+ Crop the input image and return the crop and the corresponding transformation matrix.
+ Args:
+ img (np.array): Input image of shape (H, W, 3)
+ c_x (float): Bounding box center x coordinate in the original image.
+ c_y (float): Bounding box center y coordinate in the original image.
+ bb_width (float): Bounding box width.
+ bb_height (float): Bounding box height.
+ patch_width (float): Output box width.
+ patch_height (float): Output box height.
+ do_flip (bool): Whether to flip image or not.
+ scale (float): Rescaling factor for the bounding box (augmentation).
+ rot (float): Random rotation applied to the box.
+ Returns:
+ img_patch (np.array): Cropped image patch of shape (patch_height, patch_height, 3)
+ trans (np.array): Transformation matrix.
+ """
+
+ img_height, img_width, img_channels = img.shape
+ if do_flip:
+ img = img[:, ::-1, :]
+ c_x = img_width - c_x - 1
+
+
+ trans = gen_trans_from_patch_cv(c_x, c_y, bb_width, bb_height, patch_width, patch_height, scale, rot)
+
+ img_patch = cv2.warpAffine(img, trans, (int(patch_width), int(patch_height)),
+ flags=cv2.INTER_LINEAR,
+ borderMode=border_mode,
+ borderValue=border_value,
+ )
+ # Force borderValue=cv2.BORDER_CONSTANT for alpha channel
+ if (img.shape[2] == 4) and (border_mode != cv2.BORDER_CONSTANT):
+ img_patch[:,:,3] = cv2.warpAffine(img[:,:,3], trans, (int(patch_width), int(patch_height)),
+ flags=cv2.INTER_LINEAR,
+ borderMode=cv2.BORDER_CONSTANT,
+ )
+
+ return img_patch, trans
+
+
+def convert_cvimg_to_tensor(cvimg: np.array):
+ """
+ Convert image from HWC to CHW format.
+ Args:
+ cvimg (np.array): Image of shape (H, W, 3) as loaded by OpenCV.
+ Returns:
+ np.array: Output image of shape (3, H, W).
+ """
+ # from h,w,c(OpenCV) to c,h,w
+ img = cvimg.copy()
+ img = np.transpose(img, (2, 0, 1))
+ # from int to float
+ img = img.astype(np.float32)
+ return img
+
+def fliplr_params(mano_params: Dict, has_mano_params: Dict) -> Tuple[Dict, Dict]:
+ """
+ Flip MANO parameters when flipping the image.
+ Args:
+ mano_params (Dict): MANO parameter annotations.
+ has_mano_params (Dict): Whether MANO annotations are valid.
+ Returns:
+ Dict, Dict: Flipped MANO parameters and valid flags.
+ """
+ global_orient = mano_params['global_orient'].copy()
+ hand_pose = mano_params['hand_pose'].copy()
+ betas = mano_params['betas'].copy()
+ has_global_orient = has_mano_params['global_orient'].copy()
+ has_hand_pose = has_mano_params['hand_pose'].copy()
+ has_betas = has_mano_params['betas'].copy()
+
+ global_orient[1::3] *= -1
+ global_orient[2::3] *= -1
+ hand_pose[1::3] *= -1
+ hand_pose[2::3] *= -1
+
+ mano_params = {'global_orient': global_orient.astype(np.float32),
+ 'hand_pose': hand_pose.astype(np.float32),
+ 'betas': betas.astype(np.float32)
+ }
+
+ has_mano_params = {'global_orient': has_global_orient,
+ 'hand_pose': has_hand_pose,
+ 'betas': has_betas
+ }
+
+ return mano_params, has_mano_params
+
+
+def fliplr_keypoints(joints: np.array, width: float, flip_permutation: List[int]) -> np.array:
+ """
+ Flip 2D or 3D keypoints.
+ Args:
+ joints (np.array): Array of shape (N, 3) or (N, 4) containing 2D or 3D keypoint locations and confidence.
+ flip_permutation (List): Permutation to apply after flipping.
+ Returns:
+ np.array: Flipped 2D or 3D keypoints with shape (N, 3) or (N, 4) respectively.
+ """
+ joints = joints.copy()
+ # Flip horizontal
+ joints[:, 0] = width - joints[:, 0] - 1
+ joints = joints[flip_permutation, :]
+
+ return joints
+
+def keypoint_3d_processing(keypoints_3d: np.array, flip_permutation: List[int], rot: float, do_flip: float) -> np.array:
+ """
+ Process 3D keypoints (rotation/flipping).
+ Args:
+ keypoints_3d (np.array): Input array of shape (N, 4) containing the 3D keypoints and confidence.
+ flip_permutation (List): Permutation to apply after flipping.
+ rot (float): Random rotation applied to the keypoints.
+ do_flip (bool): Whether to flip keypoints or not.
+ Returns:
+ np.array: Transformed 3D keypoints with shape (N, 4).
+ """
+ if do_flip:
+ keypoints_3d = fliplr_keypoints(keypoints_3d, 1, flip_permutation)
+ # in-plane rotation
+ rot_mat = np.eye(3)
+ if not rot == 0:
+ rot_rad = -rot * np.pi / 180
+ sn,cs = np.sin(rot_rad), np.cos(rot_rad)
+ rot_mat[0,:2] = [cs, -sn]
+ rot_mat[1,:2] = [sn, cs]
+ keypoints_3d[:, :-1] = np.einsum('ij,kj->ki', rot_mat, keypoints_3d[:, :-1])
+ # flip the x coordinates
+ keypoints_3d = keypoints_3d.astype('float32')
+ return keypoints_3d
+
+def rot_aa(aa: np.array, rot: float) -> np.array:
+ """
+ Rotate axis angle parameters.
+ Args:
+ aa (np.array): Axis-angle vector of shape (3,).
+ rot (np.array): Rotation angle in degrees.
+ Returns:
+ np.array: Rotated axis-angle vector.
+ """
+ # pose parameters
+ R = np.array([[np.cos(np.deg2rad(-rot)), -np.sin(np.deg2rad(-rot)), 0],
+ [np.sin(np.deg2rad(-rot)), np.cos(np.deg2rad(-rot)), 0],
+ [0, 0, 1]])
+ # find the rotation of the hand in camera frame
+ per_rdg, _ = cv2.Rodrigues(aa)
+ # apply the global rotation to the global orientation
+ resrot, _ = cv2.Rodrigues(np.dot(R,per_rdg))
+ aa = (resrot.T)[0]
+ return aa.astype(np.float32)
+
+def mano_param_processing(mano_params: Dict, has_mano_params: Dict, rot: float, do_flip: bool) -> Tuple[Dict, Dict]:
+ """
+ Apply random augmentations to the MANO parameters.
+ Args:
+ mano_params (Dict): MANO parameter annotations.
+ has_mano_params (Dict): Whether mano annotations are valid.
+ rot (float): Random rotation applied to the keypoints.
+ do_flip (bool): Whether to flip keypoints or not.
+ Returns:
+ Dict, Dict: Transformed MANO parameters and valid flags.
+ """
+ if do_flip:
+ mano_params, has_mano_params = fliplr_params(mano_params, has_mano_params)
+ mano_params['global_orient'] = rot_aa(mano_params['global_orient'], rot)
+ return mano_params, has_mano_params
+
+
+
+def get_example(img_path: str|np.ndarray, center_x: float, center_y: float,
+ width: float, height: float,
+ keypoints_2d: np.array, keypoints_3d: np.array,
+ mano_params: Dict, has_mano_params: Dict,
+ flip_kp_permutation: List[int],
+ patch_width: int, patch_height: int,
+ mean: np.array, std: np.array,
+ do_augment: bool, is_right: bool, augm_config: CfgNode,
+ is_bgr: bool = True,
+ use_skimage_antialias: bool = False,
+ border_mode: int = cv2.BORDER_CONSTANT,
+ return_trans: bool = False) -> Tuple:
+ """
+ Get an example from the dataset and (possibly) apply random augmentations.
+ Args:
+ img_path (str): Image filename
+ center_x (float): Bounding box center x coordinate in the original image.
+ center_y (float): Bounding box center y coordinate in the original image.
+ width (float): Bounding box width.
+ height (float): Bounding box height.
+ keypoints_2d (np.array): Array with shape (N,3) containing the 2D keypoints in the original image coordinates.
+ keypoints_3d (np.array): Array with shape (N,4) containing the 3D keypoints.
+ mano_params (Dict): MANO parameter annotations.
+ has_mano_params (Dict): Whether MANO annotations are valid.
+ flip_kp_permutation (List): Permutation to apply to the keypoints after flipping.
+ patch_width (float): Output box width.
+ patch_height (float): Output box height.
+ mean (np.array): Array of shape (3,) containing the mean for normalizing the input image.
+ std (np.array): Array of shape (3,) containing the std for normalizing the input image.
+ do_augment (bool): Whether to apply data augmentation or not.
+ aug_config (CfgNode): Config containing augmentation parameters.
+ Returns:
+ return img_patch, keypoints_2d, keypoints_3d, mano_params, has_mano_params, img_size
+ img_patch (np.array): Cropped image patch of shape (3, patch_height, patch_height)
+ keypoints_2d (np.array): Array with shape (N,3) containing the transformed 2D keypoints.
+ keypoints_3d (np.array): Array with shape (N,4) containing the transformed 3D keypoints.
+ mano_params (Dict): Transformed MANO parameters.
+ has_mano_params (Dict): Valid flag for transformed MANO parameters.
+ img_size (np.array): Image size of the original image.
+ """
+ if isinstance(img_path, str):
+ # 1. load image
+ cvimg = cv2.imread(img_path, cv2.IMREAD_COLOR | cv2.IMREAD_IGNORE_ORIENTATION)
+ if not isinstance(cvimg, np.ndarray):
+ raise IOError("Fail to read %s" % img_path)
+ elif isinstance(img_path, np.ndarray):
+ cvimg = img_path
+ else:
+ raise TypeError('img_path must be either a string or a numpy array')
+ img_height, img_width, img_channels = cvimg.shape
+
+ img_size = np.array([img_height, img_width])
+
+ # 2. get augmentation params
+ if do_augment:
+ scale, rot, do_flip, do_extreme_crop, extreme_crop_lvl, color_scale, tx, ty = do_augmentation(augm_config)
+ else:
+ scale, rot, do_flip, do_extreme_crop, extreme_crop_lvl, color_scale, tx, ty = 1.0, 0, False, False, 0, [1.0, 1.0, 1.0], 0., 0.
+
+ # if it's a left hand, we flip
+ if not is_right:
+ do_flip = True
+
+ if width < 1 or height < 1:
+ breakpoint()
+
+ if do_extreme_crop:
+ if extreme_crop_lvl == 0:
+ center_x1, center_y1, width1, height1 = extreme_cropping(center_x, center_y, width, height, keypoints_2d)
+ elif extreme_crop_lvl == 1:
+ center_x1, center_y1, width1, height1 = extreme_cropping_aggressive(center_x, center_y, width, height, keypoints_2d)
+
+ THRESH = 4
+ if width1 < THRESH or height1 < THRESH:
+ # print(f'{do_extreme_crop=}')
+ # print(f'width: {width}, height: {height}')
+ # print(f'width1: {width1}, height1: {height1}')
+ # print(f'center_x: {center_x}, center_y: {center_y}')
+ # print(f'center_x1: {center_x1}, center_y1: {center_y1}')
+ # print(f'keypoints_2d: {keypoints_2d}')
+ # print(f'\n\n', flush=True)
+ # breakpoint()
+ pass
+ # print(f'skip ==> width1: {width1}, height1: {height1}, width: {width}, height: {height}')
+ else:
+ center_x, center_y, width, height = center_x1, center_y1, width1, height1
+
+ center_x += width * tx
+ center_y += height * ty
+
+ # Process 3D keypoints
+ keypoints_3d = keypoint_3d_processing(keypoints_3d, flip_kp_permutation, rot, do_flip)
+
+ # 3. generate image patch
+ if use_skimage_antialias:
+ # Blur image to avoid aliasing artifacts
+ downsampling_factor = (patch_width / (width*scale))
+ if downsampling_factor > 1.1:
+ cvimg = gaussian(cvimg, sigma=(downsampling_factor-1)/2, channel_axis=2, preserve_range=True, truncate=3.0)
+
+ img_patch_cv, trans = generate_image_patch_cv2(cvimg,
+ center_x, center_y,
+ width, height,
+ patch_width, patch_height,
+ do_flip, scale, rot,
+ border_mode=border_mode)
+ # img_patch_cv, trans = generate_image_patch_skimage(cvimg,
+ # center_x, center_y,
+ # width, height,
+ # patch_width, patch_height,
+ # do_flip, scale, rot,
+ # border_mode=border_mode)
+
+ image = img_patch_cv.copy()
+ if is_bgr:
+ image = image[:, :, ::-1]
+ img_patch_cv = image.copy()
+ img_patch = convert_cvimg_to_tensor(image)
+
+
+ mano_params, has_mano_params = mano_param_processing(mano_params, has_mano_params, rot, do_flip)
+
+ # apply normalization
+ for n_c in range(min(img_channels, 3)):
+ img_patch[n_c, :, :] = np.clip(img_patch[n_c, :, :] * color_scale[n_c], 0, 255)
+ if mean is not None and std is not None:
+ img_patch[n_c, :, :] = (img_patch[n_c, :, :] - mean[n_c]) / std[n_c]
+ if do_flip:
+ keypoints_2d = fliplr_keypoints(keypoints_2d, img_width, flip_kp_permutation)
+
+
+ for n_jt in range(len(keypoints_2d)):
+ keypoints_2d[n_jt, 0:2] = trans_point2d(keypoints_2d[n_jt, 0:2], trans)
+ keypoints_2d[:, :-1] = keypoints_2d[:, :-1] / patch_width - 0.5
+
+ if not return_trans:
+ return img_patch, keypoints_2d, keypoints_3d, mano_params, has_mano_params, img_size
+ else:
+ return img_patch, keypoints_2d, keypoints_3d, mano_params, has_mano_params, img_size, trans
+
+def crop_to_hips(center_x: float, center_y: float, width: float, height: float, keypoints_2d: np.array) -> Tuple:
+ """
+ Extreme cropping: Crop the box up to the hip locations.
+ Args:
+ center_x (float): x coordinate of the bounding box center.
+ center_y (float): y coordinate of the bounding box center.
+ width (float): Bounding box width.
+ height (float): Bounding box height.
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ Returns:
+ center_x (float): x coordinate of the new bounding box center.
+ center_y (float): y coordinate of the new bounding box center.
+ width (float): New bounding box width.
+ height (float): New bounding box height.
+ """
+ keypoints_2d = keypoints_2d.copy()
+ lower_body_keypoints = [10, 11, 13, 14, 19, 20, 21, 22, 23, 24, 25+0, 25+1, 25+4, 25+5]
+ keypoints_2d[lower_body_keypoints, :] = 0
+ if keypoints_2d[:, -1].sum() > 1:
+ center, scale = get_bbox(keypoints_2d)
+ center_x = center[0]
+ center_y = center[1]
+ width = 1.1 * scale[0]
+ height = 1.1 * scale[1]
+ return center_x, center_y, width, height
+
+
+def crop_to_shoulders(center_x: float, center_y: float, width: float, height: float, keypoints_2d: np.array):
+ """
+ Extreme cropping: Crop the box up to the shoulder locations.
+ Args:
+ center_x (float): x coordinate of the bounding box center.
+ center_y (float): y coordinate of the bounding box center.
+ width (float): Bounding box width.
+ height (float): Bounding box height.
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ Returns:
+ center_x (float): x coordinate of the new bounding box center.
+ center_y (float): y coordinate of the new bounding box center.
+ width (float): New bounding box width.
+ height (float): New bounding box height.
+ """
+ keypoints_2d = keypoints_2d.copy()
+ lower_body_keypoints = [3, 4, 6, 7, 8, 9, 10, 11, 12, 13, 14, 19, 20, 21, 22, 23, 24] + [25 + i for i in [0, 1, 2, 3, 4, 5, 6, 7, 10, 11, 14, 15, 16]]
+ keypoints_2d[lower_body_keypoints, :] = 0
+ center, scale = get_bbox(keypoints_2d)
+ if keypoints_2d[:, -1].sum() > 1:
+ center, scale = get_bbox(keypoints_2d)
+ center_x = center[0]
+ center_y = center[1]
+ width = 1.2 * scale[0]
+ height = 1.2 * scale[1]
+ return center_x, center_y, width, height
+
+def crop_to_head(center_x: float, center_y: float, width: float, height: float, keypoints_2d: np.array):
+ """
+ Extreme cropping: Crop the box and keep on only the head.
+ Args:
+ center_x (float): x coordinate of the bounding box center.
+ center_y (float): y coordinate of the bounding box center.
+ width (float): Bounding box width.
+ height (float): Bounding box height.
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ Returns:
+ center_x (float): x coordinate of the new bounding box center.
+ center_y (float): y coordinate of the new bounding box center.
+ width (float): New bounding box width.
+ height (float): New bounding box height.
+ """
+ keypoints_2d = keypoints_2d.copy()
+ lower_body_keypoints = [3, 4, 6, 7, 8, 9, 10, 11, 12, 13, 14, 19, 20, 21, 22, 23, 24] + [25 + i for i in [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 14, 15, 16]]
+ keypoints_2d[lower_body_keypoints, :] = 0
+ if keypoints_2d[:, -1].sum() > 1:
+ center, scale = get_bbox(keypoints_2d)
+ center_x = center[0]
+ center_y = center[1]
+ width = 1.3 * scale[0]
+ height = 1.3 * scale[1]
+ return center_x, center_y, width, height
+
+def crop_torso_only(center_x: float, center_y: float, width: float, height: float, keypoints_2d: np.array):
+ """
+ Extreme cropping: Crop the box and keep on only the torso.
+ Args:
+ center_x (float): x coordinate of the bounding box center.
+ center_y (float): y coordinate of the bounding box center.
+ width (float): Bounding box width.
+ height (float): Bounding box height.
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ Returns:
+ center_x (float): x coordinate of the new bounding box center.
+ center_y (float): y coordinate of the new bounding box center.
+ width (float): New bounding box width.
+ height (float): New bounding box height.
+ """
+ keypoints_2d = keypoints_2d.copy()
+ nontorso_body_keypoints = [0, 3, 4, 6, 7, 10, 11, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24] + [25 + i for i in [0, 1, 4, 5, 6, 7, 10, 11, 13, 17, 18]]
+ keypoints_2d[nontorso_body_keypoints, :] = 0
+ if keypoints_2d[:, -1].sum() > 1:
+ center, scale = get_bbox(keypoints_2d)
+ center_x = center[0]
+ center_y = center[1]
+ width = 1.1 * scale[0]
+ height = 1.1 * scale[1]
+ return center_x, center_y, width, height
+
+def crop_rightarm_only(center_x: float, center_y: float, width: float, height: float, keypoints_2d: np.array):
+ """
+ Extreme cropping: Crop the box and keep on only the right arm.
+ Args:
+ center_x (float): x coordinate of the bounding box center.
+ center_y (float): y coordinate of the bounding box center.
+ width (float): Bounding box width.
+ height (float): Bounding box height.
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ Returns:
+ center_x (float): x coordinate of the new bounding box center.
+ center_y (float): y coordinate of the new bounding box center.
+ width (float): New bounding box width.
+ height (float): New bounding box height.
+ """
+ keypoints_2d = keypoints_2d.copy()
+ nonrightarm_body_keypoints = [0, 1, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24] + [25 + i for i in [0, 1, 2, 3, 4, 5, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18]]
+ keypoints_2d[nonrightarm_body_keypoints, :] = 0
+ if keypoints_2d[:, -1].sum() > 1:
+ center, scale = get_bbox(keypoints_2d)
+ center_x = center[0]
+ center_y = center[1]
+ width = 1.1 * scale[0]
+ height = 1.1 * scale[1]
+ return center_x, center_y, width, height
+
+def crop_leftarm_only(center_x: float, center_y: float, width: float, height: float, keypoints_2d: np.array):
+ """
+ Extreme cropping: Crop the box and keep on only the left arm.
+ Args:
+ center_x (float): x coordinate of the bounding box center.
+ center_y (float): y coordinate of the bounding box center.
+ width (float): Bounding box width.
+ height (float): Bounding box height.
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ Returns:
+ center_x (float): x coordinate of the new bounding box center.
+ center_y (float): y coordinate of the new bounding box center.
+ width (float): New bounding box width.
+ height (float): New bounding box height.
+ """
+ keypoints_2d = keypoints_2d.copy()
+ nonleftarm_body_keypoints = [0, 1, 2, 3, 4, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24] + [25 + i for i in [0, 1, 2, 3, 4, 5, 6, 7, 8, 12, 13, 14, 15, 16, 17, 18]]
+ keypoints_2d[nonleftarm_body_keypoints, :] = 0
+ if keypoints_2d[:, -1].sum() > 1:
+ center, scale = get_bbox(keypoints_2d)
+ center_x = center[0]
+ center_y = center[1]
+ width = 1.1 * scale[0]
+ height = 1.1 * scale[1]
+ return center_x, center_y, width, height
+
+def crop_legs_only(center_x: float, center_y: float, width: float, height: float, keypoints_2d: np.array):
+ """
+ Extreme cropping: Crop the box and keep on only the legs.
+ Args:
+ center_x (float): x coordinate of the bounding box center.
+ center_y (float): y coordinate of the bounding box center.
+ width (float): Bounding box width.
+ height (float): Bounding box height.
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ Returns:
+ center_x (float): x coordinate of the new bounding box center.
+ center_y (float): y coordinate of the new bounding box center.
+ width (float): New bounding box width.
+ height (float): New bounding box height.
+ """
+ keypoints_2d = keypoints_2d.copy()
+ nonlegs_body_keypoints = [0, 1, 2, 3, 4, 5, 6, 7, 15, 16, 17, 18] + [25 + i for i in [6, 7, 8, 9, 10, 11, 12, 13, 15, 16, 17, 18]]
+ keypoints_2d[nonlegs_body_keypoints, :] = 0
+ if keypoints_2d[:, -1].sum() > 1:
+ center, scale = get_bbox(keypoints_2d)
+ center_x = center[0]
+ center_y = center[1]
+ width = 1.1 * scale[0]
+ height = 1.1 * scale[1]
+ return center_x, center_y, width, height
+
+def crop_rightleg_only(center_x: float, center_y: float, width: float, height: float, keypoints_2d: np.array):
+ """
+ Extreme cropping: Crop the box and keep on only the right leg.
+ Args:
+ center_x (float): x coordinate of the bounding box center.
+ center_y (float): y coordinate of the bounding box center.
+ width (float): Bounding box width.
+ height (float): Bounding box height.
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ Returns:
+ center_x (float): x coordinate of the new bounding box center.
+ center_y (float): y coordinate of the new bounding box center.
+ width (float): New bounding box width.
+ height (float): New bounding box height.
+ """
+ keypoints_2d = keypoints_2d.copy()
+ nonrightleg_body_keypoints = [0, 1, 2, 3, 4, 5, 6, 7, 8, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21] + [25 + i for i in [3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18]]
+ keypoints_2d[nonrightleg_body_keypoints, :] = 0
+ if keypoints_2d[:, -1].sum() > 1:
+ center, scale = get_bbox(keypoints_2d)
+ center_x = center[0]
+ center_y = center[1]
+ width = 1.1 * scale[0]
+ height = 1.1 * scale[1]
+ return center_x, center_y, width, height
+
+def crop_leftleg_only(center_x: float, center_y: float, width: float, height: float, keypoints_2d: np.array):
+ """
+ Extreme cropping: Crop the box and keep on only the left leg.
+ Args:
+ center_x (float): x coordinate of the bounding box center.
+ center_y (float): y coordinate of the bounding box center.
+ width (float): Bounding box width.
+ height (float): Bounding box height.
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ Returns:
+ center_x (float): x coordinate of the new bounding box center.
+ center_y (float): y coordinate of the new bounding box center.
+ width (float): New bounding box width.
+ height (float): New bounding box height.
+ """
+ keypoints_2d = keypoints_2d.copy()
+ nonleftleg_body_keypoints = [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 15, 16, 17, 18, 22, 23, 24] + [25 + i for i in [0, 1, 2, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18]]
+ keypoints_2d[nonleftleg_body_keypoints, :] = 0
+ if keypoints_2d[:, -1].sum() > 1:
+ center, scale = get_bbox(keypoints_2d)
+ center_x = center[0]
+ center_y = center[1]
+ width = 1.1 * scale[0]
+ height = 1.1 * scale[1]
+ return center_x, center_y, width, height
+
+def full_body(keypoints_2d: np.array) -> bool:
+ """
+ Check if all main body joints are visible.
+ Args:
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ Returns:
+ bool: True if all main body joints are visible.
+ """
+
+ body_keypoints_openpose = [2, 3, 4, 5, 6, 7, 10, 11, 13, 14]
+ body_keypoints = [25 + i for i in [8, 7, 6, 9, 10, 11, 1, 0, 4, 5]]
+ return (np.maximum(keypoints_2d[body_keypoints, -1], keypoints_2d[body_keypoints_openpose, -1]) > 0).sum() == len(body_keypoints)
+
+def upper_body(keypoints_2d: np.array):
+ """
+ Check if all upper body joints are visible.
+ Args:
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ Returns:
+ bool: True if all main body joints are visible.
+ """
+ lower_body_keypoints_openpose = [10, 11, 13, 14]
+ lower_body_keypoints = [25 + i for i in [1, 0, 4, 5]]
+ upper_body_keypoints_openpose = [0, 1, 15, 16, 17, 18]
+ upper_body_keypoints = [25+8, 25+9, 25+12, 25+13, 25+17, 25+18]
+ return ((keypoints_2d[lower_body_keypoints + lower_body_keypoints_openpose, -1] > 0).sum() == 0)\
+ and ((keypoints_2d[upper_body_keypoints + upper_body_keypoints_openpose, -1] > 0).sum() >= 2)
+
+def get_bbox(keypoints_2d: np.array, rescale: float = 1.2) -> Tuple:
+ """
+ Get center and scale for bounding box from openpose detections.
+ Args:
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ rescale (float): Scale factor to rescale bounding boxes computed from the keypoints.
+ Returns:
+ center (np.array): Array of shape (2,) containing the new bounding box center.
+ scale (float): New bounding box scale.
+ """
+ valid = keypoints_2d[:,-1] > 0
+ valid_keypoints = keypoints_2d[valid][:,:-1]
+ center = 0.5 * (valid_keypoints.max(axis=0) + valid_keypoints.min(axis=0))
+ bbox_size = (valid_keypoints.max(axis=0) - valid_keypoints.min(axis=0))
+ # adjust bounding box tightness
+ scale = bbox_size
+ scale *= rescale
+ return center, scale
+
+def extreme_cropping(center_x: float, center_y: float, width: float, height: float, keypoints_2d: np.array) -> Tuple:
+ """
+ Perform extreme cropping
+ Args:
+ center_x (float): x coordinate of bounding box center.
+ center_y (float): y coordinate of bounding box center.
+ width (float): bounding box width.
+ height (float): bounding box height.
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ rescale (float): Scale factor to rescale bounding boxes computed from the keypoints.
+ Returns:
+ center_x (float): x coordinate of bounding box center.
+ center_y (float): y coordinate of bounding box center.
+ width (float): bounding box width.
+ height (float): bounding box height.
+ """
+ p = torch.rand(1).item()
+ if full_body(keypoints_2d):
+ if p < 0.7:
+ center_x, center_y, width, height = crop_to_hips(center_x, center_y, width, height, keypoints_2d)
+ elif p < 0.9:
+ center_x, center_y, width, height = crop_to_shoulders(center_x, center_y, width, height, keypoints_2d)
+ else:
+ center_x, center_y, width, height = crop_to_head(center_x, center_y, width, height, keypoints_2d)
+ elif upper_body(keypoints_2d):
+ if p < 0.9:
+ center_x, center_y, width, height = crop_to_shoulders(center_x, center_y, width, height, keypoints_2d)
+ else:
+ center_x, center_y, width, height = crop_to_head(center_x, center_y, width, height, keypoints_2d)
+
+ return center_x, center_y, max(width, height), max(width, height)
+
+def extreme_cropping_aggressive(center_x: float, center_y: float, width: float, height: float, keypoints_2d: np.array) -> Tuple:
+ """
+ Perform aggressive extreme cropping
+ Args:
+ center_x (float): x coordinate of bounding box center.
+ center_y (float): y coordinate of bounding box center.
+ width (float): bounding box width.
+ height (float): bounding box height.
+ keypoints_2d (np.array): Array of shape (N, 3) containing 2D keypoint locations.
+ rescale (float): Scale factor to rescale bounding boxes computed from the keypoints.
+ Returns:
+ center_x (float): x coordinate of bounding box center.
+ center_y (float): y coordinate of bounding box center.
+ width (float): bounding box width.
+ height (float): bounding box height.
+ """
+ p = torch.rand(1).item()
+ if full_body(keypoints_2d):
+ if p < 0.2:
+ center_x, center_y, width, height = crop_to_hips(center_x, center_y, width, height, keypoints_2d)
+ elif p < 0.3:
+ center_x, center_y, width, height = crop_to_shoulders(center_x, center_y, width, height, keypoints_2d)
+ elif p < 0.4:
+ center_x, center_y, width, height = crop_to_head(center_x, center_y, width, height, keypoints_2d)
+ elif p < 0.5:
+ center_x, center_y, width, height = crop_torso_only(center_x, center_y, width, height, keypoints_2d)
+ elif p < 0.6:
+ center_x, center_y, width, height = crop_rightarm_only(center_x, center_y, width, height, keypoints_2d)
+ elif p < 0.7:
+ center_x, center_y, width, height = crop_leftarm_only(center_x, center_y, width, height, keypoints_2d)
+ elif p < 0.8:
+ center_x, center_y, width, height = crop_legs_only(center_x, center_y, width, height, keypoints_2d)
+ elif p < 0.9:
+ center_x, center_y, width, height = crop_rightleg_only(center_x, center_y, width, height, keypoints_2d)
+ else:
+ center_x, center_y, width, height = crop_leftleg_only(center_x, center_y, width, height, keypoints_2d)
+ elif upper_body(keypoints_2d):
+ if p < 0.2:
+ center_x, center_y, width, height = crop_to_shoulders(center_x, center_y, width, height, keypoints_2d)
+ elif p < 0.4:
+ center_x, center_y, width, height = crop_to_head(center_x, center_y, width, height, keypoints_2d)
+ elif p < 0.6:
+ center_x, center_y, width, height = crop_torso_only(center_x, center_y, width, height, keypoints_2d)
+ elif p < 0.8:
+ center_x, center_y, width, height = crop_rightarm_only(center_x, center_y, width, height, keypoints_2d)
+ else:
+ center_x, center_y, width, height = crop_leftarm_only(center_x, center_y, width, height, keypoints_2d)
+ return center_x, center_y, max(width, height), max(width, height)
diff --git a/third_party/hamer/hamer/datasets/vitdet_dataset.py b/third_party/hamer/hamer/datasets/vitdet_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..3414ff2d1c42557af5bdf8792faabb92c45318a1
--- /dev/null
+++ b/third_party/hamer/hamer/datasets/vitdet_dataset.py
@@ -0,0 +1,95 @@
+from typing import Dict
+
+import cv2
+import numpy as np
+from skimage.filters import gaussian
+from yacs.config import CfgNode
+import torch
+
+from .utils import (convert_cvimg_to_tensor,
+ expand_to_aspect_ratio,
+ generate_image_patch_cv2)
+
+DEFAULT_MEAN = 255. * np.array([0.485, 0.456, 0.406])
+DEFAULT_STD = 255. * np.array([0.229, 0.224, 0.225])
+
+class ViTDetDataset(torch.utils.data.Dataset):
+
+ def __init__(self,
+ cfg: CfgNode,
+ img_cv2: np.array,
+ boxes: np.array,
+ right: np.array,
+ rescale_factor=2.5,
+ train: bool = False,
+ **kwargs):
+ super().__init__()
+ self.cfg = cfg
+ self.img_cv2 = img_cv2
+ # self.boxes = boxes
+
+ assert train == False, "ViTDetDataset is only for inference"
+ self.train = train
+ self.img_size = cfg.MODEL.IMAGE_SIZE
+ self.mean = 255. * np.array(self.cfg.MODEL.IMAGE_MEAN)
+ self.std = 255. * np.array(self.cfg.MODEL.IMAGE_STD)
+
+ # Preprocess annotations
+ boxes = boxes.astype(np.float32)
+ self.center = (boxes[:, 2:4] + boxes[:, 0:2]) / 2.0
+ self.scale = rescale_factor * (boxes[:, 2:4] - boxes[:, 0:2]) / 200.0
+ self.personid = np.arange(len(boxes), dtype=np.int32)
+ self.right = right.astype(np.float32)
+
+ def __len__(self) -> int:
+ return len(self.personid)
+
+ def __getitem__(self, idx: int) -> Dict[str, np.array]:
+
+ center = self.center[idx].copy()
+ center_x = center[0]
+ center_y = center[1]
+
+ scale = self.scale[idx]
+ BBOX_SHAPE = self.cfg.MODEL.get('BBOX_SHAPE', None)
+ bbox_size = expand_to_aspect_ratio(scale*200, target_aspect_ratio=BBOX_SHAPE).max()
+
+ patch_width = patch_height = self.img_size
+
+ right = self.right[idx].copy()
+ flip = right == 0
+
+ # 3. generate image patch
+ # if use_skimage_antialias:
+ cvimg = self.img_cv2.copy()
+ # Note: Gaussian blur disabled due to compatibility issues with skimage
+ # if True:
+ # # Blur image to avoid aliasing artifacts
+ # downsampling_factor = ((bbox_size*1.0) / patch_width)
+ # downsampling_factor = downsampling_factor / 2.0
+ # if downsampling_factor > 1.1:
+ # cvimg = gaussian(cvimg, sigma=(downsampling_factor-1)/2, channel_axis=2, preserve_range=True)
+
+
+ img_patch_cv, trans = generate_image_patch_cv2(cvimg,
+ center_x, center_y,
+ bbox_size, bbox_size,
+ patch_width, patch_height,
+ flip, 1.0, 0,
+ border_mode=cv2.BORDER_CONSTANT)
+ img_patch_cv = img_patch_cv[:, :, ::-1]
+ img_patch = convert_cvimg_to_tensor(img_patch_cv)
+
+ # apply normalization
+ for n_c in range(min(self.img_cv2.shape[2], 3)):
+ img_patch[n_c, :, :] = (img_patch[n_c, :, :] - self.mean[n_c]) / self.std[n_c]
+
+ item = {
+ 'img': img_patch,
+ 'personid': int(self.personid[idx]),
+ }
+ item['box_center'] = self.center[idx].copy()
+ item['box_size'] = bbox_size
+ item['img_size'] = 1.0 * np.array([cvimg.shape[1], cvimg.shape[0]])
+ item['right'] = self.right[idx].copy()
+ return item
diff --git a/third_party/hamer/hamer/models/__init__.py b/third_party/hamer/hamer/models/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..7261b0a6e84d43e9a7e9675a06af163e6f65e75f
--- /dev/null
+++ b/third_party/hamer/hamer/models/__init__.py
@@ -0,0 +1,52 @@
+from .mano_wrapper import MANO
+from .hamer import HAMER
+from .discriminator import Discriminator
+
+from ..utils.download import cache_url
+from ..configs import CACHE_DIR_HAMER
+
+
+def download_models(folder=CACHE_DIR_HAMER):
+ """Download checkpoints and files for running inference.
+ """
+ import os
+ os.makedirs(folder, exist_ok=True)
+ download_files = {
+ "hamer_demo_data.tar.gz" : ["https://www.cs.utexas.edu/~pavlakos/hamer/data/hamer_demo_data.tar.gz", folder],
+ }
+
+ for file_name, url in download_files.items():
+ output_path = os.path.join(url[1], file_name)
+ if not os.path.exists(output_path):
+ print("Downloading file: " + file_name)
+ # output = gdown.cached_download(url[0], output_path, fuzzy=True)
+ output = cache_url(url[0], output_path)
+ assert os.path.exists(output_path), f"{output} does not exist"
+
+ # if ends with tar.gz, tar -xzf
+ if file_name.endswith(".tar.gz"):
+ print("Extracting file: " + file_name)
+ os.system("tar -xvf " + output_path)
+
+DEFAULT_CHECKPOINT=f'{CACHE_DIR_HAMER}/hamer_ckpts/checkpoints/hamer.ckpt'
+def load_hamer(checkpoint_path=DEFAULT_CHECKPOINT):
+ from pathlib import Path
+ from ..configs import get_config
+ model_cfg = str(Path(checkpoint_path).parent.parent / 'model_config.yaml')
+ model_cfg = get_config(model_cfg, update_cachedir=True)
+
+ # Override some config values, to crop bbox correctly
+ if (model_cfg.MODEL.BACKBONE.TYPE == 'vit') and ('BBOX_SHAPE' not in model_cfg.MODEL):
+ model_cfg.defrost()
+ assert model_cfg.MODEL.IMAGE_SIZE == 256, f"MODEL.IMAGE_SIZE ({model_cfg.MODEL.IMAGE_SIZE}) should be 256 for ViT backbone"
+ model_cfg.MODEL.BBOX_SHAPE = [192,256]
+ model_cfg.freeze()
+
+ # Update config to be compatible with demo
+ if ('PRETRAINED_WEIGHTS' in model_cfg.MODEL.BACKBONE):
+ model_cfg.defrost()
+ model_cfg.MODEL.BACKBONE.pop('PRETRAINED_WEIGHTS')
+ model_cfg.freeze()
+
+ model = HAMER.load_from_checkpoint(checkpoint_path, strict=False, cfg=model_cfg)
+ return model, model_cfg
diff --git a/third_party/hamer/hamer/models/backbones/__init__.py b/third_party/hamer/hamer/models/backbones/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..d2b217b0e624dc5612dcc405c450fa4b43039dff
--- /dev/null
+++ b/third_party/hamer/hamer/models/backbones/__init__.py
@@ -0,0 +1,7 @@
+from .vit import vit
+
+def create_backbone(cfg):
+ if cfg.MODEL.BACKBONE.TYPE == 'vit':
+ return vit(cfg)
+ else:
+ raise NotImplementedError('Backbone type is not implemented')
diff --git a/third_party/hamer/hamer/models/backbones/vit.py b/third_party/hamer/hamer/models/backbones/vit.py
new file mode 100644
index 0000000000000000000000000000000000000000..c56c71889cd441294f57ad687d0678d2443d1eed
--- /dev/null
+++ b/third_party/hamer/hamer/models/backbones/vit.py
@@ -0,0 +1,348 @@
+# Copyright (c) OpenMMLab. All rights reserved.
+import math
+
+import torch
+from functools import partial
+import torch.nn as nn
+import torch.nn.functional as F
+import torch.utils.checkpoint as checkpoint
+
+from timm.models.layers import drop_path, to_2tuple, trunc_normal_
+
+def vit(cfg):
+ return ViT(
+ img_size=(256, 192),
+ patch_size=16,
+ embed_dim=1280,
+ depth=32,
+ num_heads=16,
+ ratio=1,
+ use_checkpoint=False,
+ mlp_ratio=4,
+ qkv_bias=True,
+ drop_path_rate=0.55,
+ )
+
+def get_abs_pos(abs_pos, h, w, ori_h, ori_w, has_cls_token=True):
+ """
+ Calculate absolute positional embeddings. If needed, resize embeddings and remove cls_token
+ dimension for the original embeddings.
+ Args:
+ abs_pos (Tensor): absolute positional embeddings with (1, num_position, C).
+ has_cls_token (bool): If true, has 1 embedding in abs_pos for cls token.
+ hw (Tuple): size of input image tokens.
+
+ Returns:
+ Absolute positional embeddings after processing with shape (1, H, W, C)
+ """
+ cls_token = None
+ B, L, C = abs_pos.shape
+ if has_cls_token:
+ cls_token = abs_pos[:, 0:1]
+ abs_pos = abs_pos[:, 1:]
+
+ if ori_h != h or ori_w != w:
+ new_abs_pos = F.interpolate(
+ abs_pos.reshape(1, ori_h, ori_w, -1).permute(0, 3, 1, 2),
+ size=(h, w),
+ mode="bicubic",
+ align_corners=False,
+ ).permute(0, 2, 3, 1).reshape(B, -1, C)
+
+ else:
+ new_abs_pos = abs_pos
+
+ if cls_token is not None:
+ new_abs_pos = torch.cat([cls_token, new_abs_pos], dim=1)
+ return new_abs_pos
+
+class DropPath(nn.Module):
+ """Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks).
+ """
+ def __init__(self, drop_prob=None):
+ super(DropPath, self).__init__()
+ self.drop_prob = drop_prob
+
+ def forward(self, x):
+ return drop_path(x, self.drop_prob, self.training)
+
+ def extra_repr(self):
+ return 'p={}'.format(self.drop_prob)
+
+class Mlp(nn.Module):
+ def __init__(self, in_features, hidden_features=None, out_features=None, act_layer=nn.GELU, drop=0.):
+ super().__init__()
+ out_features = out_features or in_features
+ hidden_features = hidden_features or in_features
+ self.fc1 = nn.Linear(in_features, hidden_features)
+ self.act = act_layer()
+ self.fc2 = nn.Linear(hidden_features, out_features)
+ self.drop = nn.Dropout(drop)
+
+ def forward(self, x):
+ x = self.fc1(x)
+ x = self.act(x)
+ x = self.fc2(x)
+ x = self.drop(x)
+ return x
+
+class Attention(nn.Module):
+ def __init__(
+ self, dim, num_heads=8, qkv_bias=False, qk_scale=None, attn_drop=0.,
+ proj_drop=0., attn_head_dim=None,):
+ super().__init__()
+ self.num_heads = num_heads
+ head_dim = dim // num_heads
+ self.dim = dim
+
+ if attn_head_dim is not None:
+ head_dim = attn_head_dim
+ all_head_dim = head_dim * self.num_heads
+
+ self.scale = qk_scale or head_dim ** -0.5
+
+ self.qkv = nn.Linear(dim, all_head_dim * 3, bias=qkv_bias)
+
+ self.attn_drop = nn.Dropout(attn_drop)
+ self.proj = nn.Linear(all_head_dim, dim)
+ self.proj_drop = nn.Dropout(proj_drop)
+
+ def forward(self, x):
+ B, N, C = x.shape
+ qkv = self.qkv(x)
+ qkv = qkv.reshape(B, N, 3, self.num_heads, -1).permute(2, 0, 3, 1, 4)
+ q, k, v = qkv[0], qkv[1], qkv[2] # make torchscript happy (cannot use tensor as tuple)
+
+ q = q * self.scale
+ attn = (q @ k.transpose(-2, -1))
+
+ attn = attn.softmax(dim=-1)
+ attn = self.attn_drop(attn)
+
+ x = (attn @ v).transpose(1, 2).reshape(B, N, -1)
+ x = self.proj(x)
+ x = self.proj_drop(x)
+
+ return x
+
+class Block(nn.Module):
+
+ def __init__(self, dim, num_heads, mlp_ratio=4., qkv_bias=False, qk_scale=None,
+ drop=0., attn_drop=0., drop_path=0., act_layer=nn.GELU,
+ norm_layer=nn.LayerNorm, attn_head_dim=None
+ ):
+ super().__init__()
+
+ self.norm1 = norm_layer(dim)
+ self.attn = Attention(
+ dim, num_heads=num_heads, qkv_bias=qkv_bias, qk_scale=qk_scale,
+ attn_drop=attn_drop, proj_drop=drop, attn_head_dim=attn_head_dim
+ )
+
+ # NOTE: drop path for stochastic depth, we shall see if this is better than dropout here
+ self.drop_path = DropPath(drop_path) if drop_path > 0. else nn.Identity()
+ self.norm2 = norm_layer(dim)
+ mlp_hidden_dim = int(dim * mlp_ratio)
+ self.mlp = Mlp(in_features=dim, hidden_features=mlp_hidden_dim, act_layer=act_layer, drop=drop)
+
+ def forward(self, x):
+ x = x + self.drop_path(self.attn(self.norm1(x)))
+ x = x + self.drop_path(self.mlp(self.norm2(x)))
+ return x
+
+
+class PatchEmbed(nn.Module):
+ """ Image to Patch Embedding
+ """
+ def __init__(self, img_size=224, patch_size=16, in_chans=3, embed_dim=768, ratio=1):
+ super().__init__()
+ img_size = to_2tuple(img_size)
+ patch_size = to_2tuple(patch_size)
+ num_patches = (img_size[1] // patch_size[1]) * (img_size[0] // patch_size[0]) * (ratio ** 2)
+ self.patch_shape = (int(img_size[0] // patch_size[0] * ratio), int(img_size[1] // patch_size[1] * ratio))
+ self.origin_patch_shape = (int(img_size[0] // patch_size[0]), int(img_size[1] // patch_size[1]))
+ self.img_size = img_size
+ self.patch_size = patch_size
+ self.num_patches = num_patches
+
+ self.proj = nn.Conv2d(in_chans, embed_dim, kernel_size=patch_size, stride=(patch_size[0] // ratio), padding=4 + 2 * (ratio//2-1))
+
+ def forward(self, x, **kwargs):
+ B, C, H, W = x.shape
+ x = self.proj(x)
+ Hp, Wp = x.shape[2], x.shape[3]
+
+ x = x.flatten(2).transpose(1, 2)
+ return x, (Hp, Wp)
+
+
+class HybridEmbed(nn.Module):
+ """ CNN Feature Map Embedding
+ Extract feature map from CNN, flatten, project to embedding dim.
+ """
+ def __init__(self, backbone, img_size=224, feature_size=None, in_chans=3, embed_dim=768):
+ super().__init__()
+ assert isinstance(backbone, nn.Module)
+ img_size = to_2tuple(img_size)
+ self.img_size = img_size
+ self.backbone = backbone
+ if feature_size is None:
+ with torch.no_grad():
+ training = backbone.training
+ if training:
+ backbone.eval()
+ o = self.backbone(torch.zeros(1, in_chans, img_size[0], img_size[1]))[-1]
+ feature_size = o.shape[-2:]
+ feature_dim = o.shape[1]
+ backbone.train(training)
+ else:
+ feature_size = to_2tuple(feature_size)
+ feature_dim = self.backbone.feature_info.channels()[-1]
+ self.num_patches = feature_size[0] * feature_size[1]
+ self.proj = nn.Linear(feature_dim, embed_dim)
+
+ def forward(self, x):
+ x = self.backbone(x)[-1]
+ x = x.flatten(2).transpose(1, 2)
+ x = self.proj(x)
+ return x
+
+
+class ViT(nn.Module):
+
+ def __init__(self,
+ img_size=224, patch_size=16, in_chans=3, num_classes=80, embed_dim=768, depth=12,
+ num_heads=12, mlp_ratio=4., qkv_bias=False, qk_scale=None, drop_rate=0., attn_drop_rate=0.,
+ drop_path_rate=0., hybrid_backbone=None, norm_layer=None, use_checkpoint=False,
+ frozen_stages=-1, ratio=1, last_norm=True,
+ patch_padding='pad', freeze_attn=False, freeze_ffn=False,
+ ):
+ # Protect mutable default arguments
+ super(ViT, self).__init__()
+ norm_layer = norm_layer or partial(nn.LayerNorm, eps=1e-6)
+ self.num_classes = num_classes
+ self.num_features = self.embed_dim = embed_dim # num_features for consistency with other models
+ self.frozen_stages = frozen_stages
+ self.use_checkpoint = use_checkpoint
+ self.patch_padding = patch_padding
+ self.freeze_attn = freeze_attn
+ self.freeze_ffn = freeze_ffn
+ self.depth = depth
+
+ if hybrid_backbone is not None:
+ self.patch_embed = HybridEmbed(
+ hybrid_backbone, img_size=img_size, in_chans=in_chans, embed_dim=embed_dim)
+ else:
+ self.patch_embed = PatchEmbed(
+ img_size=img_size, patch_size=patch_size, in_chans=in_chans, embed_dim=embed_dim, ratio=ratio)
+ num_patches = self.patch_embed.num_patches
+
+ # since the pretraining model has class token
+ self.pos_embed = nn.Parameter(torch.zeros(1, num_patches + 1, embed_dim))
+
+ dpr = [x.item() for x in torch.linspace(0, drop_path_rate, depth)] # stochastic depth decay rule
+
+ self.blocks = nn.ModuleList([
+ Block(
+ dim=embed_dim, num_heads=num_heads, mlp_ratio=mlp_ratio, qkv_bias=qkv_bias, qk_scale=qk_scale,
+ drop=drop_rate, attn_drop=attn_drop_rate, drop_path=dpr[i], norm_layer=norm_layer,
+ )
+ for i in range(depth)])
+
+ self.last_norm = norm_layer(embed_dim) if last_norm else nn.Identity()
+
+ if self.pos_embed is not None:
+ trunc_normal_(self.pos_embed, std=.02)
+
+ self._freeze_stages()
+
+ def _freeze_stages(self):
+ """Freeze parameters."""
+ if self.frozen_stages >= 0:
+ self.patch_embed.eval()
+ for param in self.patch_embed.parameters():
+ param.requires_grad = False
+
+ for i in range(1, self.frozen_stages + 1):
+ m = self.blocks[i]
+ m.eval()
+ for param in m.parameters():
+ param.requires_grad = False
+
+ if self.freeze_attn:
+ for i in range(0, self.depth):
+ m = self.blocks[i]
+ m.attn.eval()
+ m.norm1.eval()
+ for param in m.attn.parameters():
+ param.requires_grad = False
+ for param in m.norm1.parameters():
+ param.requires_grad = False
+
+ if self.freeze_ffn:
+ self.pos_embed.requires_grad = False
+ self.patch_embed.eval()
+ for param in self.patch_embed.parameters():
+ param.requires_grad = False
+ for i in range(0, self.depth):
+ m = self.blocks[i]
+ m.mlp.eval()
+ m.norm2.eval()
+ for param in m.mlp.parameters():
+ param.requires_grad = False
+ for param in m.norm2.parameters():
+ param.requires_grad = False
+
+ def init_weights(self):
+ """Initialize the weights in backbone.
+ Args:
+ pretrained (str, optional): Path to pre-trained weights.
+ Defaults to None.
+ """
+ def _init_weights(m):
+ if isinstance(m, nn.Linear):
+ trunc_normal_(m.weight, std=.02)
+ if isinstance(m, nn.Linear) and m.bias is not None:
+ nn.init.constant_(m.bias, 0)
+ elif isinstance(m, nn.LayerNorm):
+ nn.init.constant_(m.bias, 0)
+ nn.init.constant_(m.weight, 1.0)
+
+ self.apply(_init_weights)
+
+ def get_num_layers(self):
+ return len(self.blocks)
+
+ @torch.jit.ignore
+ def no_weight_decay(self):
+ return {'pos_embed', 'cls_token'}
+
+ def forward_features(self, x):
+ B, C, H, W = x.shape
+ x, (Hp, Wp) = self.patch_embed(x)
+
+ if self.pos_embed is not None:
+ # fit for multiple GPU training
+ # since the first element for pos embed (sin-cos manner) is zero, it will cause no difference
+ x = x + self.pos_embed[:, 1:] + self.pos_embed[:, :1]
+
+ for blk in self.blocks:
+ if self.use_checkpoint:
+ x = checkpoint.checkpoint(blk, x)
+ else:
+ x = blk(x)
+
+ x = self.last_norm(x)
+
+ xp = x.permute(0, 2, 1).reshape(B, -1, Hp, Wp).contiguous()
+
+ return xp
+
+ def forward(self, x):
+ x = self.forward_features(x)
+ return x
+
+ def train(self, mode=True):
+ """Convert the model into training mode."""
+ super().train(mode)
+ self._freeze_stages()
diff --git a/third_party/hamer/hamer/models/components/__init__.py b/third_party/hamer/hamer/models/components/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391
diff --git a/third_party/hamer/hamer/models/components/pose_transformer.py b/third_party/hamer/hamer/models/components/pose_transformer.py
new file mode 100644
index 0000000000000000000000000000000000000000..ac04971407cb59637490cc4842f048b9bc4758be
--- /dev/null
+++ b/third_party/hamer/hamer/models/components/pose_transformer.py
@@ -0,0 +1,358 @@
+from inspect import isfunction
+from typing import Callable, Optional
+
+import torch
+from einops import rearrange
+from einops.layers.torch import Rearrange
+from torch import nn
+
+from .t_cond_mlp import (
+ AdaptiveLayerNorm1D,
+ FrequencyEmbedder,
+ normalization_layer,
+)
+# from .vit import Attention, FeedForward
+
+
+def exists(val):
+ return val is not None
+
+
+def default(val, d):
+ if exists(val):
+ return val
+ return d() if isfunction(d) else d
+
+
+class PreNorm(nn.Module):
+ def __init__(self, dim: int, fn: Callable, norm: str = "layer", norm_cond_dim: int = -1):
+ super().__init__()
+ self.norm = normalization_layer(norm, dim, norm_cond_dim)
+ self.fn = fn
+
+ def forward(self, x: torch.Tensor, *args, **kwargs):
+ if isinstance(self.norm, AdaptiveLayerNorm1D):
+ return self.fn(self.norm(x, *args), **kwargs)
+ else:
+ return self.fn(self.norm(x), **kwargs)
+
+
+class FeedForward(nn.Module):
+ def __init__(self, dim, hidden_dim, dropout=0.0):
+ super().__init__()
+ self.net = nn.Sequential(
+ nn.Linear(dim, hidden_dim),
+ nn.GELU(),
+ nn.Dropout(dropout),
+ nn.Linear(hidden_dim, dim),
+ nn.Dropout(dropout),
+ )
+
+ def forward(self, x):
+ return self.net(x)
+
+
+class Attention(nn.Module):
+ def __init__(self, dim, heads=8, dim_head=64, dropout=0.0):
+ super().__init__()
+ inner_dim = dim_head * heads
+ project_out = not (heads == 1 and dim_head == dim)
+
+ self.heads = heads
+ self.scale = dim_head**-0.5
+
+ self.attend = nn.Softmax(dim=-1)
+ self.dropout = nn.Dropout(dropout)
+
+ self.to_qkv = nn.Linear(dim, inner_dim * 3, bias=False)
+
+ self.to_out = (
+ nn.Sequential(nn.Linear(inner_dim, dim), nn.Dropout(dropout))
+ if project_out
+ else nn.Identity()
+ )
+
+ def forward(self, x):
+ qkv = self.to_qkv(x).chunk(3, dim=-1)
+ q, k, v = map(lambda t: rearrange(t, "b n (h d) -> b h n d", h=self.heads), qkv)
+
+ dots = torch.matmul(q, k.transpose(-1, -2)) * self.scale
+
+ attn = self.attend(dots)
+ attn = self.dropout(attn)
+
+ out = torch.matmul(attn, v)
+ out = rearrange(out, "b h n d -> b n (h d)")
+ return self.to_out(out)
+
+
+class CrossAttention(nn.Module):
+ def __init__(self, dim, context_dim=None, heads=8, dim_head=64, dropout=0.0):
+ super().__init__()
+ inner_dim = dim_head * heads
+ project_out = not (heads == 1 and dim_head == dim)
+
+ self.heads = heads
+ self.scale = dim_head**-0.5
+
+ self.attend = nn.Softmax(dim=-1)
+ self.dropout = nn.Dropout(dropout)
+
+ context_dim = default(context_dim, dim)
+ self.to_kv = nn.Linear(context_dim, inner_dim * 2, bias=False)
+ self.to_q = nn.Linear(dim, inner_dim, bias=False)
+
+ self.to_out = (
+ nn.Sequential(nn.Linear(inner_dim, dim), nn.Dropout(dropout))
+ if project_out
+ else nn.Identity()
+ )
+
+ def forward(self, x, context=None):
+ context = default(context, x)
+ k, v = self.to_kv(context).chunk(2, dim=-1)
+ q = self.to_q(x)
+ q, k, v = map(lambda t: rearrange(t, "b n (h d) -> b h n d", h=self.heads), [q, k, v])
+
+ dots = torch.matmul(q, k.transpose(-1, -2)) * self.scale
+
+ attn = self.attend(dots)
+ attn = self.dropout(attn)
+
+ out = torch.matmul(attn, v)
+ out = rearrange(out, "b h n d -> b n (h d)")
+ return self.to_out(out)
+
+
+class Transformer(nn.Module):
+ def __init__(
+ self,
+ dim: int,
+ depth: int,
+ heads: int,
+ dim_head: int,
+ mlp_dim: int,
+ dropout: float = 0.0,
+ norm: str = "layer",
+ norm_cond_dim: int = -1,
+ ):
+ super().__init__()
+ self.layers = nn.ModuleList([])
+ for _ in range(depth):
+ sa = Attention(dim, heads=heads, dim_head=dim_head, dropout=dropout)
+ ff = FeedForward(dim, mlp_dim, dropout=dropout)
+ self.layers.append(
+ nn.ModuleList(
+ [
+ PreNorm(dim, sa, norm=norm, norm_cond_dim=norm_cond_dim),
+ PreNorm(dim, ff, norm=norm, norm_cond_dim=norm_cond_dim),
+ ]
+ )
+ )
+
+ def forward(self, x: torch.Tensor, *args):
+ for attn, ff in self.layers:
+ x = attn(x, *args) + x
+ x = ff(x, *args) + x
+ return x
+
+
+class TransformerCrossAttn(nn.Module):
+ def __init__(
+ self,
+ dim: int,
+ depth: int,
+ heads: int,
+ dim_head: int,
+ mlp_dim: int,
+ dropout: float = 0.0,
+ norm: str = "layer",
+ norm_cond_dim: int = -1,
+ context_dim: Optional[int] = None,
+ ):
+ super().__init__()
+ self.layers = nn.ModuleList([])
+ for _ in range(depth):
+ sa = Attention(dim, heads=heads, dim_head=dim_head, dropout=dropout)
+ ca = CrossAttention(
+ dim, context_dim=context_dim, heads=heads, dim_head=dim_head, dropout=dropout
+ )
+ ff = FeedForward(dim, mlp_dim, dropout=dropout)
+ self.layers.append(
+ nn.ModuleList(
+ [
+ PreNorm(dim, sa, norm=norm, norm_cond_dim=norm_cond_dim),
+ PreNorm(dim, ca, norm=norm, norm_cond_dim=norm_cond_dim),
+ PreNorm(dim, ff, norm=norm, norm_cond_dim=norm_cond_dim),
+ ]
+ )
+ )
+
+ def forward(self, x: torch.Tensor, *args, context=None, context_list=None):
+ if context_list is None:
+ context_list = [context] * len(self.layers)
+ if len(context_list) != len(self.layers):
+ raise ValueError(f"len(context_list) != len(self.layers) ({len(context_list)} != {len(self.layers)})")
+
+ for i, (self_attn, cross_attn, ff) in enumerate(self.layers):
+ x = self_attn(x, *args) + x
+ x = cross_attn(x, *args, context=context_list[i]) + x
+ x = ff(x, *args) + x
+ return x
+
+
+class DropTokenDropout(nn.Module):
+ def __init__(self, p: float = 0.1):
+ super().__init__()
+ if p < 0 or p > 1:
+ raise ValueError(
+ "dropout probability has to be between 0 and 1, " "but got {}".format(p)
+ )
+ self.p = p
+
+ def forward(self, x: torch.Tensor):
+ # x: (batch_size, seq_len, dim)
+ if self.training and self.p > 0:
+ zero_mask = torch.full_like(x[0, :, 0], self.p).bernoulli().bool()
+ # TODO: permutation idx for each batch using torch.argsort
+ if zero_mask.any():
+ x = x[:, ~zero_mask, :]
+ return x
+
+
+class ZeroTokenDropout(nn.Module):
+ def __init__(self, p: float = 0.1):
+ super().__init__()
+ if p < 0 or p > 1:
+ raise ValueError(
+ "dropout probability has to be between 0 and 1, " "but got {}".format(p)
+ )
+ self.p = p
+
+ def forward(self, x: torch.Tensor):
+ # x: (batch_size, seq_len, dim)
+ if self.training and self.p > 0:
+ zero_mask = torch.full_like(x[:, :, 0], self.p).bernoulli().bool()
+ # Zero-out the masked tokens
+ x[zero_mask, :] = 0
+ return x
+
+
+class TransformerEncoder(nn.Module):
+ def __init__(
+ self,
+ num_tokens: int,
+ token_dim: int,
+ dim: int,
+ depth: int,
+ heads: int,
+ mlp_dim: int,
+ dim_head: int = 64,
+ dropout: float = 0.0,
+ emb_dropout: float = 0.0,
+ emb_dropout_type: str = "drop",
+ emb_dropout_loc: str = "token",
+ norm: str = "layer",
+ norm_cond_dim: int = -1,
+ token_pe_numfreq: int = -1,
+ ):
+ super().__init__()
+ if token_pe_numfreq > 0:
+ token_dim_new = token_dim * (2 * token_pe_numfreq + 1)
+ self.to_token_embedding = nn.Sequential(
+ Rearrange("b n d -> (b n) d", n=num_tokens, d=token_dim),
+ FrequencyEmbedder(token_pe_numfreq, token_pe_numfreq - 1),
+ Rearrange("(b n) d -> b n d", n=num_tokens, d=token_dim_new),
+ nn.Linear(token_dim_new, dim),
+ )
+ else:
+ self.to_token_embedding = nn.Linear(token_dim, dim)
+ self.pos_embedding = nn.Parameter(torch.randn(1, num_tokens, dim))
+ if emb_dropout_type == "drop":
+ self.dropout = DropTokenDropout(emb_dropout)
+ elif emb_dropout_type == "zero":
+ self.dropout = ZeroTokenDropout(emb_dropout)
+ else:
+ raise ValueError(f"Unknown emb_dropout_type: {emb_dropout_type}")
+ self.emb_dropout_loc = emb_dropout_loc
+
+ self.transformer = Transformer(
+ dim, depth, heads, dim_head, mlp_dim, dropout, norm=norm, norm_cond_dim=norm_cond_dim
+ )
+
+ def forward(self, inp: torch.Tensor, *args, **kwargs):
+ x = inp
+
+ if self.emb_dropout_loc == "input":
+ x = self.dropout(x)
+ x = self.to_token_embedding(x)
+
+ if self.emb_dropout_loc == "token":
+ x = self.dropout(x)
+ b, n, _ = x.shape
+ x += self.pos_embedding[:, :n]
+
+ if self.emb_dropout_loc == "token_afterpos":
+ x = self.dropout(x)
+ x = self.transformer(x, *args)
+ return x
+
+
+class TransformerDecoder(nn.Module):
+ def __init__(
+ self,
+ num_tokens: int,
+ token_dim: int,
+ dim: int,
+ depth: int,
+ heads: int,
+ mlp_dim: int,
+ dim_head: int = 64,
+ dropout: float = 0.0,
+ emb_dropout: float = 0.0,
+ emb_dropout_type: str = 'drop',
+ norm: str = "layer",
+ norm_cond_dim: int = -1,
+ context_dim: Optional[int] = None,
+ skip_token_embedding: bool = False,
+ ):
+ super().__init__()
+ if not skip_token_embedding:
+ self.to_token_embedding = nn.Linear(token_dim, dim)
+ else:
+ self.to_token_embedding = nn.Identity()
+ if token_dim != dim:
+ raise ValueError(
+ f"token_dim ({token_dim}) != dim ({dim}) when skip_token_embedding is True"
+ )
+
+ self.pos_embedding = nn.Parameter(torch.randn(1, num_tokens, dim))
+ if emb_dropout_type == "drop":
+ self.dropout = DropTokenDropout(emb_dropout)
+ elif emb_dropout_type == "zero":
+ self.dropout = ZeroTokenDropout(emb_dropout)
+ elif emb_dropout_type == "normal":
+ self.dropout = nn.Dropout(emb_dropout)
+
+ self.transformer = TransformerCrossAttn(
+ dim,
+ depth,
+ heads,
+ dim_head,
+ mlp_dim,
+ dropout,
+ norm=norm,
+ norm_cond_dim=norm_cond_dim,
+ context_dim=context_dim,
+ )
+
+ def forward(self, inp: torch.Tensor, *args, context=None, context_list=None):
+ x = self.to_token_embedding(inp)
+ b, n, _ = x.shape
+
+ x = self.dropout(x)
+ x += self.pos_embedding[:, :n]
+
+ x = self.transformer(x, *args, context=context, context_list=context_list)
+ return x
+
diff --git a/third_party/hamer/hamer/models/components/t_cond_mlp.py b/third_party/hamer/hamer/models/components/t_cond_mlp.py
new file mode 100644
index 0000000000000000000000000000000000000000..44d5a09bf54f67712a69953039b7b5af41c3f029
--- /dev/null
+++ b/third_party/hamer/hamer/models/components/t_cond_mlp.py
@@ -0,0 +1,199 @@
+import copy
+from typing import List, Optional
+
+import torch
+
+
+class AdaptiveLayerNorm1D(torch.nn.Module):
+ def __init__(self, data_dim: int, norm_cond_dim: int):
+ super().__init__()
+ if data_dim <= 0:
+ raise ValueError(f"data_dim must be positive, but got {data_dim}")
+ if norm_cond_dim <= 0:
+ raise ValueError(f"norm_cond_dim must be positive, but got {norm_cond_dim}")
+ self.norm = torch.nn.LayerNorm(
+ data_dim
+ ) # TODO: Check if elementwise_affine=True is correct
+ self.linear = torch.nn.Linear(norm_cond_dim, 2 * data_dim)
+ torch.nn.init.zeros_(self.linear.weight)
+ torch.nn.init.zeros_(self.linear.bias)
+
+ def forward(self, x: torch.Tensor, t: torch.Tensor) -> torch.Tensor:
+ # x: (batch, ..., data_dim)
+ # t: (batch, norm_cond_dim)
+ # return: (batch, data_dim)
+ x = self.norm(x)
+ alpha, beta = self.linear(t).chunk(2, dim=-1)
+
+ # Add singleton dimensions to alpha and beta
+ if x.dim() > 2:
+ alpha = alpha.view(alpha.shape[0], *([1] * (x.dim() - 2)), alpha.shape[1])
+ beta = beta.view(beta.shape[0], *([1] * (x.dim() - 2)), beta.shape[1])
+
+ return x * (1 + alpha) + beta
+
+
+class SequentialCond(torch.nn.Sequential):
+ def forward(self, input, *args, **kwargs):
+ for module in self:
+ if isinstance(module, (AdaptiveLayerNorm1D, SequentialCond, ResidualMLPBlock)):
+ # print(f'Passing on args to {module}', [a.shape for a in args])
+ input = module(input, *args, **kwargs)
+ else:
+ # print(f'Skipping passing args to {module}', [a.shape for a in args])
+ input = module(input)
+ return input
+
+
+def normalization_layer(norm: Optional[str], dim: int, norm_cond_dim: int = -1):
+ if norm == "batch":
+ return torch.nn.BatchNorm1d(dim)
+ elif norm == "layer":
+ return torch.nn.LayerNorm(dim)
+ elif norm == "ada":
+ assert norm_cond_dim > 0, f"norm_cond_dim must be positive, got {norm_cond_dim}"
+ return AdaptiveLayerNorm1D(dim, norm_cond_dim)
+ elif norm is None:
+ return torch.nn.Identity()
+ else:
+ raise ValueError(f"Unknown norm: {norm}")
+
+
+def linear_norm_activ_dropout(
+ input_dim: int,
+ output_dim: int,
+ activation: torch.nn.Module = torch.nn.ReLU(),
+ bias: bool = True,
+ norm: Optional[str] = "layer", # Options: ada/batch/layer
+ dropout: float = 0.0,
+ norm_cond_dim: int = -1,
+) -> SequentialCond:
+ layers = []
+ layers.append(torch.nn.Linear(input_dim, output_dim, bias=bias))
+ if norm is not None:
+ layers.append(normalization_layer(norm, output_dim, norm_cond_dim))
+ layers.append(copy.deepcopy(activation))
+ if dropout > 0.0:
+ layers.append(torch.nn.Dropout(dropout))
+ return SequentialCond(*layers)
+
+
+def create_simple_mlp(
+ input_dim: int,
+ hidden_dims: List[int],
+ output_dim: int,
+ activation: torch.nn.Module = torch.nn.ReLU(),
+ bias: bool = True,
+ norm: Optional[str] = "layer", # Options: ada/batch/layer
+ dropout: float = 0.0,
+ norm_cond_dim: int = -1,
+) -> SequentialCond:
+ layers = []
+ prev_dim = input_dim
+ for hidden_dim in hidden_dims:
+ layers.extend(
+ linear_norm_activ_dropout(
+ prev_dim, hidden_dim, activation, bias, norm, dropout, norm_cond_dim
+ )
+ )
+ prev_dim = hidden_dim
+ layers.append(torch.nn.Linear(prev_dim, output_dim, bias=bias))
+ return SequentialCond(*layers)
+
+
+class ResidualMLPBlock(torch.nn.Module):
+ def __init__(
+ self,
+ input_dim: int,
+ hidden_dim: int,
+ num_hidden_layers: int,
+ output_dim: int,
+ activation: torch.nn.Module = torch.nn.ReLU(),
+ bias: bool = True,
+ norm: Optional[str] = "layer", # Options: ada/batch/layer
+ dropout: float = 0.0,
+ norm_cond_dim: int = -1,
+ ):
+ super().__init__()
+ if not (input_dim == output_dim == hidden_dim):
+ raise NotImplementedError(
+ f"input_dim {input_dim} != output_dim {output_dim} is not implemented"
+ )
+
+ layers = []
+ prev_dim = input_dim
+ for i in range(num_hidden_layers):
+ layers.append(
+ linear_norm_activ_dropout(
+ prev_dim, hidden_dim, activation, bias, norm, dropout, norm_cond_dim
+ )
+ )
+ prev_dim = hidden_dim
+ self.model = SequentialCond(*layers)
+ self.skip = torch.nn.Identity()
+
+ def forward(self, x: torch.Tensor, *args, **kwargs) -> torch.Tensor:
+ return x + self.model(x, *args, **kwargs)
+
+
+class ResidualMLP(torch.nn.Module):
+ def __init__(
+ self,
+ input_dim: int,
+ hidden_dim: int,
+ num_hidden_layers: int,
+ output_dim: int,
+ activation: torch.nn.Module = torch.nn.ReLU(),
+ bias: bool = True,
+ norm: Optional[str] = "layer", # Options: ada/batch/layer
+ dropout: float = 0.0,
+ num_blocks: int = 1,
+ norm_cond_dim: int = -1,
+ ):
+ super().__init__()
+ self.input_dim = input_dim
+ self.model = SequentialCond(
+ linear_norm_activ_dropout(
+ input_dim, hidden_dim, activation, bias, norm, dropout, norm_cond_dim
+ ),
+ *[
+ ResidualMLPBlock(
+ hidden_dim,
+ hidden_dim,
+ num_hidden_layers,
+ hidden_dim,
+ activation,
+ bias,
+ norm,
+ dropout,
+ norm_cond_dim,
+ )
+ for _ in range(num_blocks)
+ ],
+ torch.nn.Linear(hidden_dim, output_dim, bias=bias),
+ )
+
+ def forward(self, x: torch.Tensor, *args, **kwargs) -> torch.Tensor:
+ return self.model(x, *args, **kwargs)
+
+
+class FrequencyEmbedder(torch.nn.Module):
+ def __init__(self, num_frequencies, max_freq_log2):
+ super().__init__()
+ frequencies = 2 ** torch.linspace(0, max_freq_log2, steps=num_frequencies)
+ self.register_buffer("frequencies", frequencies)
+
+ def forward(self, x):
+ # x should be of size (N,) or (N, D)
+ N = x.size(0)
+ if x.dim() == 1: # (N,)
+ x = x.unsqueeze(1) # (N, D) where D=1
+ x_unsqueezed = x.unsqueeze(-1) # (N, D, 1)
+ scaled = self.frequencies.view(1, 1, -1) * x_unsqueezed # (N, D, num_frequencies)
+ s = torch.sin(scaled)
+ c = torch.cos(scaled)
+ embedded = torch.cat([s, c, x_unsqueezed], dim=-1).view(
+ N, -1
+ ) # (N, D * 2 * num_frequencies + D)
+ return embedded
+
diff --git a/third_party/hamer/hamer/models/discriminator.py b/third_party/hamer/hamer/models/discriminator.py
new file mode 100644
index 0000000000000000000000000000000000000000..f1cb2d1a21fbab47e8fa10dcc603b3d2012686a7
--- /dev/null
+++ b/third_party/hamer/hamer/models/discriminator.py
@@ -0,0 +1,98 @@
+import torch
+import torch.nn as nn
+
+class Discriminator(nn.Module):
+
+ def __init__(self):
+ """
+ Pose + Shape discriminator proposed in HMR
+ """
+ super(Discriminator, self).__init__()
+
+ self.num_joints = 15
+ # poses_alone
+ self.D_conv1 = nn.Conv2d(9, 32, kernel_size=1)
+ nn.init.xavier_uniform_(self.D_conv1.weight)
+ nn.init.zeros_(self.D_conv1.bias)
+ self.relu = nn.ReLU(inplace=True)
+ self.D_conv2 = nn.Conv2d(32, 32, kernel_size=1)
+ nn.init.xavier_uniform_(self.D_conv2.weight)
+ nn.init.zeros_(self.D_conv2.bias)
+ pose_out = []
+ for i in range(self.num_joints):
+ pose_out_temp = nn.Linear(32, 1)
+ nn.init.xavier_uniform_(pose_out_temp.weight)
+ nn.init.zeros_(pose_out_temp.bias)
+ pose_out.append(pose_out_temp)
+ self.pose_out = nn.ModuleList(pose_out)
+
+ # betas
+ self.betas_fc1 = nn.Linear(10, 10)
+ nn.init.xavier_uniform_(self.betas_fc1.weight)
+ nn.init.zeros_(self.betas_fc1.bias)
+ self.betas_fc2 = nn.Linear(10, 5)
+ nn.init.xavier_uniform_(self.betas_fc2.weight)
+ nn.init.zeros_(self.betas_fc2.bias)
+ self.betas_out = nn.Linear(5, 1)
+ nn.init.xavier_uniform_(self.betas_out.weight)
+ nn.init.zeros_(self.betas_out.bias)
+
+ # poses_joint
+ self.D_alljoints_fc1 = nn.Linear(32*self.num_joints, 1024)
+ nn.init.xavier_uniform_(self.D_alljoints_fc1.weight)
+ nn.init.zeros_(self.D_alljoints_fc1.bias)
+ self.D_alljoints_fc2 = nn.Linear(1024, 1024)
+ nn.init.xavier_uniform_(self.D_alljoints_fc2.weight)
+ nn.init.zeros_(self.D_alljoints_fc2.bias)
+ self.D_alljoints_out = nn.Linear(1024, 1)
+ nn.init.xavier_uniform_(self.D_alljoints_out.weight)
+ nn.init.zeros_(self.D_alljoints_out.bias)
+
+
+ def forward(self, poses: torch.Tensor, betas: torch.Tensor) -> torch.Tensor:
+ """
+ Forward pass of the discriminator.
+ Args:
+ poses (torch.Tensor): Tensor of shape (B, 23, 3, 3) containing a batch of MANO hand poses (excluding the global orientation).
+ betas (torch.Tensor): Tensor of shape (B, 10) containign a batch of MANO beta coefficients.
+ Returns:
+ torch.Tensor: Discriminator output with shape (B, 25)
+ """
+ #bn = poses.shape[0]
+ # poses B x 207
+ #poses = poses.reshape(bn, -1)
+ # poses B x num_joints x 1 x 9
+ poses = poses.reshape(-1, self.num_joints, 1, 9)
+ bn = poses.shape[0]
+ # poses B x 9 x num_joints x 1
+ poses = poses.permute(0, 3, 1, 2).contiguous()
+
+ # poses_alone
+ poses = self.D_conv1(poses)
+ poses = self.relu(poses)
+ poses = self.D_conv2(poses)
+ poses = self.relu(poses)
+
+ poses_out = []
+ for i in range(self.num_joints):
+ poses_out_ = self.pose_out[i](poses[:, :, i, 0])
+ poses_out.append(poses_out_)
+ poses_out = torch.cat(poses_out, dim=1)
+
+ # betas
+ betas = self.betas_fc1(betas)
+ betas = self.relu(betas)
+ betas = self.betas_fc2(betas)
+ betas = self.relu(betas)
+ betas_out = self.betas_out(betas)
+
+ # poses_joint
+ poses = poses.reshape(bn,-1)
+ poses_all = self.D_alljoints_fc1(poses)
+ poses_all = self.relu(poses_all)
+ poses_all = self.D_alljoints_fc2(poses_all)
+ poses_all = self.relu(poses_all)
+ poses_all_out = self.D_alljoints_out(poses_all)
+
+ disc_out = torch.cat((poses_out, betas_out, poses_all_out), 1)
+ return disc_out
diff --git a/third_party/hamer/hamer/models/hamer.py b/third_party/hamer/hamer/models/hamer.py
new file mode 100644
index 0000000000000000000000000000000000000000..8a79e4d52b4399d1114ec18598a58dc10ce266c0
--- /dev/null
+++ b/third_party/hamer/hamer/models/hamer.py
@@ -0,0 +1,355 @@
+import torch
+import pytorch_lightning as pl
+from typing import Any, Dict, Mapping, Tuple
+
+from yacs.config import CfgNode
+
+from ..utils import SkeletonRenderer, MeshRenderer
+from ..utils.geometry import aa_to_rotmat, perspective_projection
+from ..utils.pylogger import get_pylogger
+from .backbones import create_backbone
+from .heads import build_mano_head
+from .discriminator import Discriminator
+from .losses import Keypoint3DLoss, Keypoint2DLoss, ParameterLoss
+from . import MANO
+
+log = get_pylogger(__name__)
+
+class HAMER(pl.LightningModule):
+
+ def __init__(self, cfg: CfgNode, init_renderer: bool = True):
+ """
+ Setup HAMER model
+ Args:
+ cfg (CfgNode): Config file as a yacs CfgNode
+ """
+ super().__init__()
+
+ # Save hyperparameters
+ self.save_hyperparameters(logger=False, ignore=['init_renderer'])
+
+ self.cfg = cfg
+ # Create backbone feature extractor
+ self.backbone = create_backbone(cfg)
+ if cfg.MODEL.BACKBONE.get('PRETRAINED_WEIGHTS', None):
+ log.info(f'Loading backbone weights from {cfg.MODEL.BACKBONE.PRETRAINED_WEIGHTS}')
+ self.backbone.load_state_dict(torch.load(cfg.MODEL.BACKBONE.PRETRAINED_WEIGHTS, map_location='cpu')['state_dict'])
+
+ # Create MANO head
+ self.mano_head = build_mano_head(cfg)
+
+ # Create discriminator
+ if self.cfg.LOSS_WEIGHTS.ADVERSARIAL > 0:
+ self.discriminator = Discriminator()
+
+ # Define loss functions
+ self.keypoint_3d_loss = Keypoint3DLoss(loss_type='l1')
+ self.keypoint_2d_loss = Keypoint2DLoss(loss_type='l1')
+ self.mano_parameter_loss = ParameterLoss()
+
+ # Instantiate MANO model
+ mano_cfg = {k.lower(): v for k,v in dict(cfg.MANO).items()}
+ self.mano = MANO(**mano_cfg)
+
+ # Buffer that shows whetheer we need to initialize ActNorm layers
+ self.register_buffer('initialized', torch.tensor(False))
+ # Setup renderer for visualization
+ if init_renderer:
+ self.renderer = SkeletonRenderer(self.cfg)
+ self.mesh_renderer = MeshRenderer(self.cfg, faces=self.mano.faces)
+ else:
+ self.renderer = None
+ self.mesh_renderer = None
+
+ # Disable automatic optimization since we use adversarial training
+ self.automatic_optimization = False
+
+ def get_parameters(self):
+ all_params = list(self.mano_head.parameters())
+ all_params += list(self.backbone.parameters())
+ return all_params
+
+ def configure_optimizers(self) -> Tuple[torch.optim.Optimizer, torch.optim.Optimizer]:
+ """
+ Setup model and distriminator Optimizers
+ Returns:
+ Tuple[torch.optim.Optimizer, torch.optim.Optimizer]: Model and discriminator optimizers
+ """
+ param_groups = [{'params': filter(lambda p: p.requires_grad, self.get_parameters()), 'lr': self.cfg.TRAIN.LR}]
+
+ optimizer = torch.optim.AdamW(params=param_groups,
+ # lr=self.cfg.TRAIN.LR,
+ weight_decay=self.cfg.TRAIN.WEIGHT_DECAY)
+ optimizer_disc = torch.optim.AdamW(params=self.discriminator.parameters(),
+ lr=self.cfg.TRAIN.LR,
+ weight_decay=self.cfg.TRAIN.WEIGHT_DECAY)
+
+ return optimizer, optimizer_disc
+
+ def forward_step(self, batch: Dict, train: bool = False) -> Dict:
+ """
+ Run a forward step of the network
+ Args:
+ batch (Dict): Dictionary containing batch data
+ train (bool): Flag indicating whether it is training or validation mode
+ Returns:
+ Dict: Dictionary containing the regression output
+ """
+
+ # Use RGB image as input
+ x = batch['img']
+ batch_size = x.shape[0]
+
+ # Compute conditioning features using the backbone
+ # if using ViT backbone, we need to use a different aspect ratio
+ conditioning_feats = self.backbone(x[:,:,:,32:-32])
+
+ pred_mano_params, pred_cam, _ = self.mano_head(conditioning_feats)
+
+ # Store useful regression outputs to the output dict
+ output = {}
+ output['pred_cam'] = pred_cam
+ output['pred_mano_params'] = {k: v.clone() for k,v in pred_mano_params.items()}
+
+ # Compute camera translation
+ device = pred_mano_params['hand_pose'].device
+ dtype = pred_mano_params['hand_pose'].dtype
+ focal_length = self.cfg.EXTRA.FOCAL_LENGTH * torch.ones(batch_size, 2, device=device, dtype=dtype)
+ pred_cam_t = torch.stack([pred_cam[:, 1],
+ pred_cam[:, 2],
+ 2*focal_length[:, 0]/(self.cfg.MODEL.IMAGE_SIZE * pred_cam[:, 0] +1e-9)],dim=-1)
+ output['pred_cam_t'] = pred_cam_t
+ output['focal_length'] = focal_length
+
+ # Compute model vertices, joints and the projected joints
+ pred_mano_params['global_orient'] = pred_mano_params['global_orient'].reshape(batch_size, -1, 3, 3)
+ pred_mano_params['hand_pose'] = pred_mano_params['hand_pose'].reshape(batch_size, -1, 3, 3)
+ pred_mano_params['betas'] = pred_mano_params['betas'].reshape(batch_size, -1)
+ mano_output = self.mano(**{k: v.float() for k,v in pred_mano_params.items()}, pose2rot=False)
+ pred_keypoints_3d = mano_output.joints
+ pred_vertices = mano_output.vertices
+ output['pred_keypoints_3d'] = pred_keypoints_3d.reshape(batch_size, -1, 3)
+ output['pred_vertices'] = pred_vertices.reshape(batch_size, -1, 3)
+ pred_cam_t = pred_cam_t.reshape(-1, 3)
+ focal_length = focal_length.reshape(-1, 2)
+ pred_keypoints_2d = perspective_projection(pred_keypoints_3d,
+ translation=pred_cam_t,
+ focal_length=focal_length / self.cfg.MODEL.IMAGE_SIZE)
+
+ output['pred_keypoints_2d'] = pred_keypoints_2d.reshape(batch_size, -1, 2)
+ return output
+
+ def compute_loss(self, batch: Dict, output: Dict, train: bool = True) -> torch.Tensor:
+ """
+ Compute losses given the input batch and the regression output
+ Args:
+ batch (Dict): Dictionary containing batch data
+ output (Dict): Dictionary containing the regression output
+ train (bool): Flag indicating whether it is training or validation mode
+ Returns:
+ torch.Tensor : Total loss for current batch
+ """
+
+ pred_mano_params = output['pred_mano_params']
+ pred_keypoints_2d = output['pred_keypoints_2d']
+ pred_keypoints_3d = output['pred_keypoints_3d']
+
+
+ batch_size = pred_mano_params['hand_pose'].shape[0]
+ device = pred_mano_params['hand_pose'].device
+ dtype = pred_mano_params['hand_pose'].dtype
+
+ # Get annotations
+ gt_keypoints_2d = batch['keypoints_2d']
+ gt_keypoints_3d = batch['keypoints_3d']
+ gt_mano_params = batch['mano_params']
+ has_mano_params = batch['has_mano_params']
+ is_axis_angle = batch['mano_params_is_axis_angle']
+
+ # Compute 3D keypoint loss
+ loss_keypoints_2d = self.keypoint_2d_loss(pred_keypoints_2d, gt_keypoints_2d)
+ loss_keypoints_3d = self.keypoint_3d_loss(pred_keypoints_3d, gt_keypoints_3d, pelvis_id=0)
+
+ # Compute loss on MANO parameters
+ loss_mano_params = {}
+ for k, pred in pred_mano_params.items():
+ gt = gt_mano_params[k].view(batch_size, -1)
+ if is_axis_angle[k].all():
+ gt = aa_to_rotmat(gt.reshape(-1, 3)).view(batch_size, -1, 3, 3)
+ has_gt = has_mano_params[k]
+ loss_mano_params[k] = self.mano_parameter_loss(pred.reshape(batch_size, -1), gt.reshape(batch_size, -1), has_gt)
+
+ loss = self.cfg.LOSS_WEIGHTS['KEYPOINTS_3D'] * loss_keypoints_3d+\
+ self.cfg.LOSS_WEIGHTS['KEYPOINTS_2D'] * loss_keypoints_2d+\
+ sum([loss_mano_params[k] * self.cfg.LOSS_WEIGHTS[k.upper()] for k in loss_mano_params])
+
+ losses = dict(loss=loss.detach(),
+ loss_keypoints_2d=loss_keypoints_2d.detach(),
+ loss_keypoints_3d=loss_keypoints_3d.detach())
+
+ for k, v in loss_mano_params.items():
+ losses['loss_' + k] = v.detach()
+
+ output['losses'] = losses
+
+ return loss
+
+ # Tensoroboard logging should run from first rank only
+ @pl.utilities.rank_zero.rank_zero_only
+ def tensorboard_logging(self, batch: Dict, output: Dict, step_count: int, train: bool = True, write_to_summary_writer: bool = True) -> None:
+ """
+ Log results to Tensorboard
+ Args:
+ batch (Dict): Dictionary containing batch data
+ output (Dict): Dictionary containing the regression output
+ step_count (int): Global training step count
+ train (bool): Flag indicating whether it is training or validation mode
+ """
+
+ mode = 'train' if train else 'val'
+ batch_size = batch['keypoints_2d'].shape[0]
+ images = batch['img']
+ images = images * torch.tensor([0.229, 0.224, 0.225], device=images.device).reshape(1,3,1,1)
+ images = images + torch.tensor([0.485, 0.456, 0.406], device=images.device).reshape(1,3,1,1)
+ #images = 255*images.permute(0, 2, 3, 1).cpu().numpy()
+
+ pred_keypoints_3d = output['pred_keypoints_3d'].detach().reshape(batch_size, -1, 3)
+ pred_vertices = output['pred_vertices'].detach().reshape(batch_size, -1, 3)
+ focal_length = output['focal_length'].detach().reshape(batch_size, 2)
+ gt_keypoints_3d = batch['keypoints_3d']
+ gt_keypoints_2d = batch['keypoints_2d']
+ losses = output['losses']
+ pred_cam_t = output['pred_cam_t'].detach().reshape(batch_size, 3)
+ pred_keypoints_2d = output['pred_keypoints_2d'].detach().reshape(batch_size, -1, 2)
+
+ if write_to_summary_writer:
+ summary_writer = self.logger.experiment
+ for loss_name, val in losses.items():
+ summary_writer.add_scalar(mode +'/' + loss_name, val.detach().item(), step_count)
+ num_images = min(batch_size, self.cfg.EXTRA.NUM_LOG_IMAGES)
+
+ gt_keypoints_3d = batch['keypoints_3d']
+ pred_keypoints_3d = output['pred_keypoints_3d'].detach().reshape(batch_size, -1, 3)
+
+ # We render the skeletons instead of the full mesh because rendering a lot of meshes will make the training slow.
+ #predictions = self.renderer(pred_keypoints_3d[:num_images],
+ # gt_keypoints_3d[:num_images],
+ # 2 * gt_keypoints_2d[:num_images],
+ # images=images[:num_images],
+ # camera_translation=pred_cam_t[:num_images])
+ predictions = self.mesh_renderer.visualize_tensorboard(pred_vertices[:num_images].cpu().numpy(),
+ pred_cam_t[:num_images].cpu().numpy(),
+ images[:num_images].cpu().numpy(),
+ pred_keypoints_2d[:num_images].cpu().numpy(),
+ gt_keypoints_2d[:num_images].cpu().numpy(),
+ focal_length=focal_length[:num_images].cpu().numpy())
+ if write_to_summary_writer:
+ summary_writer.add_image('%s/predictions' % mode, predictions, step_count)
+
+ return predictions
+
+ def forward(self, batch: Dict) -> Dict:
+ """
+ Run a forward step of the network in val mode
+ Args:
+ batch (Dict): Dictionary containing batch data
+ Returns:
+ Dict: Dictionary containing the regression output
+ """
+ return self.forward_step(batch, train=False)
+
+ def training_step_discriminator(self, batch: Dict,
+ hand_pose: torch.Tensor,
+ betas: torch.Tensor,
+ optimizer: torch.optim.Optimizer) -> torch.Tensor:
+ """
+ Run a discriminator training step
+ Args:
+ batch (Dict): Dictionary containing mocap batch data
+ hand_pose (torch.Tensor): Regressed hand pose from current step
+ betas (torch.Tensor): Regressed betas from current step
+ optimizer (torch.optim.Optimizer): Discriminator optimizer
+ Returns:
+ torch.Tensor: Discriminator loss
+ """
+ batch_size = hand_pose.shape[0]
+ gt_hand_pose = batch['hand_pose']
+ gt_betas = batch['betas']
+ gt_rotmat = aa_to_rotmat(gt_hand_pose.view(-1,3)).view(batch_size, -1, 3, 3)
+ disc_fake_out = self.discriminator(hand_pose.detach(), betas.detach())
+ loss_fake = ((disc_fake_out - 0.0) ** 2).sum() / batch_size
+ disc_real_out = self.discriminator(gt_rotmat, gt_betas)
+ loss_real = ((disc_real_out - 1.0) ** 2).sum() / batch_size
+ loss_disc = loss_fake + loss_real
+ loss = self.cfg.LOSS_WEIGHTS.ADVERSARIAL * loss_disc
+ optimizer.zero_grad()
+ self.manual_backward(loss)
+ optimizer.step()
+ return loss_disc.detach()
+
+ def training_step(self, joint_batch: Dict, batch_idx: int) -> Dict:
+ """
+ Run a full training step
+ Args:
+ joint_batch (Dict): Dictionary containing image and mocap batch data
+ batch_idx (int): Unused.
+ batch_idx (torch.Tensor): Unused.
+ Returns:
+ Dict: Dictionary containing regression output.
+ """
+ batch = joint_batch['img']
+ mocap_batch = joint_batch['mocap']
+ optimizer = self.optimizers(use_pl_optimizer=True)
+ if self.cfg.LOSS_WEIGHTS.ADVERSARIAL > 0:
+ optimizer, optimizer_disc = optimizer
+
+ batch_size = batch['img'].shape[0]
+ output = self.forward_step(batch, train=True)
+ pred_mano_params = output['pred_mano_params']
+ if self.cfg.get('UPDATE_GT_SPIN', False):
+ self.update_batch_gt_spin(batch, output)
+ loss = self.compute_loss(batch, output, train=True)
+ if self.cfg.LOSS_WEIGHTS.ADVERSARIAL > 0:
+ disc_out = self.discriminator(pred_mano_params['hand_pose'].reshape(batch_size, -1), pred_mano_params['betas'].reshape(batch_size, -1))
+ loss_adv = ((disc_out - 1.0) ** 2).sum() / batch_size
+ loss = loss + self.cfg.LOSS_WEIGHTS.ADVERSARIAL * loss_adv
+
+ # Error if Nan
+ if torch.isnan(loss):
+ raise ValueError('Loss is NaN')
+
+ optimizer.zero_grad()
+ self.manual_backward(loss)
+ # Clip gradient
+ if self.cfg.TRAIN.get('GRAD_CLIP_VAL', 0) > 0:
+ gn = torch.nn.utils.clip_grad_norm_(self.get_parameters(), self.cfg.TRAIN.GRAD_CLIP_VAL, error_if_nonfinite=True)
+ self.log('train/grad_norm', gn, on_step=True, on_epoch=True, prog_bar=True, logger=True)
+ optimizer.step()
+ if self.cfg.LOSS_WEIGHTS.ADVERSARIAL > 0:
+ loss_disc = self.training_step_discriminator(mocap_batch, pred_mano_params['hand_pose'].reshape(batch_size, -1), pred_mano_params['betas'].reshape(batch_size, -1), optimizer_disc)
+ output['losses']['loss_gen'] = loss_adv
+ output['losses']['loss_disc'] = loss_disc
+
+ if self.global_step > 0 and self.global_step % self.cfg.GENERAL.LOG_STEPS == 0:
+ self.tensorboard_logging(batch, output, self.global_step, train=True)
+
+ self.log('train/loss', output['losses']['loss'], on_step=True, on_epoch=True, prog_bar=True, logger=False)
+
+ return output
+
+ def validation_step(self, batch: Dict, batch_idx: int, dataloader_idx=0) -> Dict:
+ """
+ Run a validation step and log to Tensorboard
+ Args:
+ batch (Dict): Dictionary containing batch data
+ batch_idx (int): Unused.
+ Returns:
+ Dict: Dictionary containing regression output.
+ """
+ # batch_size = batch['img'].shape[0]
+ output = self.forward_step(batch, train=False)
+ loss = self.compute_loss(batch, output, train=False)
+ output['loss'] = loss
+ self.tensorboard_logging(batch, output, self.global_step, train=False)
+
+ return output
diff --git a/third_party/hamer/hamer/models/heads/__init__.py b/third_party/hamer/hamer/models/heads/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..27e24ee70c20d9979a880a149efc9bc617f65e74
--- /dev/null
+++ b/third_party/hamer/hamer/models/heads/__init__.py
@@ -0,0 +1 @@
+from .mano_head import build_mano_head
diff --git a/third_party/hamer/hamer/models/heads/mano_head.py b/third_party/hamer/hamer/models/heads/mano_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..c58487305d4816597d958017415033337f9100f2
--- /dev/null
+++ b/third_party/hamer/hamer/models/heads/mano_head.py
@@ -0,0 +1,111 @@
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import numpy as np
+import einops
+
+from ...utils.geometry import rot6d_to_rotmat, aa_to_rotmat
+from ..components.pose_transformer import TransformerDecoder
+
+def build_mano_head(cfg):
+ mano_head_type = cfg.MODEL.MANO_HEAD.get('TYPE', 'hamer')
+ if mano_head_type == 'transformer_decoder':
+ return MANOTransformerDecoderHead(cfg)
+ else:
+ raise ValueError('Unknown MANO head type: {}'.format(mano_head_type))
+
+class MANOTransformerDecoderHead(nn.Module):
+ """ Cross-attention based MANO Transformer decoder
+ """
+
+ def __init__(self, cfg):
+ super().__init__()
+ self.cfg = cfg
+ self.joint_rep_type = cfg.MODEL.MANO_HEAD.get('JOINT_REP', '6d')
+ self.joint_rep_dim = {'6d': 6, 'aa': 3}[self.joint_rep_type]
+ npose = self.joint_rep_dim * (cfg.MANO.NUM_HAND_JOINTS + 1)
+ self.npose = npose
+ self.input_is_mean_shape = cfg.MODEL.MANO_HEAD.get('TRANSFORMER_INPUT', 'zero') == 'mean_shape'
+ transformer_args = dict(
+ num_tokens=1,
+ token_dim=(npose + 10 + 3) if self.input_is_mean_shape else 1,
+ dim=1024,
+ )
+ transformer_args = (transformer_args | dict(cfg.MODEL.MANO_HEAD.TRANSFORMER_DECODER))
+ self.transformer = TransformerDecoder(
+ **transformer_args
+ )
+ dim=transformer_args['dim']
+ self.decpose = nn.Linear(dim, npose)
+ self.decshape = nn.Linear(dim, 10)
+ self.deccam = nn.Linear(dim, 3)
+
+ if cfg.MODEL.MANO_HEAD.get('INIT_DECODER_XAVIER', False):
+ # True by default in MLP. False by default in Transformer
+ nn.init.xavier_uniform_(self.decpose.weight, gain=0.01)
+ nn.init.xavier_uniform_(self.decshape.weight, gain=0.01)
+ nn.init.xavier_uniform_(self.deccam.weight, gain=0.01)
+
+ mean_params = np.load(cfg.MANO.MEAN_PARAMS)
+ init_hand_pose = torch.from_numpy(mean_params['pose'].astype(np.float32)).unsqueeze(0)
+ init_betas = torch.from_numpy(mean_params['shape'].astype('float32')).unsqueeze(0)
+ init_cam = torch.from_numpy(mean_params['cam'].astype(np.float32)).unsqueeze(0)
+ self.register_buffer('init_hand_pose', init_hand_pose)
+ self.register_buffer('init_betas', init_betas)
+ self.register_buffer('init_cam', init_cam)
+
+ def forward(self, x, **kwargs):
+
+ batch_size = x.shape[0]
+ # vit pretrained backbone is channel-first. Change to token-first
+ x = einops.rearrange(x, 'b c h w -> b (h w) c')
+
+ init_hand_pose = self.init_hand_pose.expand(batch_size, -1)
+ init_betas = self.init_betas.expand(batch_size, -1)
+ init_cam = self.init_cam.expand(batch_size, -1)
+
+ # TODO: Convert init_hand_pose to aa rep if needed
+ if self.joint_rep_type == 'aa':
+ raise NotImplementedError
+
+ pred_hand_pose = init_hand_pose
+ pred_betas = init_betas
+ pred_cam = init_cam
+ pred_hand_pose_list = []
+ pred_betas_list = []
+ pred_cam_list = []
+ for i in range(self.cfg.MODEL.MANO_HEAD.get('IEF_ITERS', 1)):
+ # Input token to transformer is zero token
+ if self.input_is_mean_shape:
+ token = torch.cat([pred_hand_pose, pred_betas, pred_cam], dim=1)[:,None,:]
+ else:
+ token = torch.zeros(batch_size, 1, 1).to(x.device)
+
+ # Pass through transformer
+ token_out = self.transformer(token, context=x)
+ token_out = token_out.squeeze(1) # (B, C)
+
+ # Readout from token_out
+ pred_hand_pose = self.decpose(token_out) + pred_hand_pose
+ pred_betas = self.decshape(token_out) + pred_betas
+ pred_cam = self.deccam(token_out) + pred_cam
+ pred_hand_pose_list.append(pred_hand_pose)
+ pred_betas_list.append(pred_betas)
+ pred_cam_list.append(pred_cam)
+
+ # Convert self.joint_rep_type -> rotmat
+ joint_conversion_fn = {
+ '6d': rot6d_to_rotmat,
+ 'aa': lambda x: aa_to_rotmat(x.view(-1, 3).contiguous())
+ }[self.joint_rep_type]
+
+ pred_mano_params_list = {}
+ pred_mano_params_list['hand_pose'] = torch.cat([joint_conversion_fn(pbp).view(batch_size, -1, 3, 3)[:, 1:, :, :] for pbp in pred_hand_pose_list], dim=0)
+ pred_mano_params_list['betas'] = torch.cat(pred_betas_list, dim=0)
+ pred_mano_params_list['cam'] = torch.cat(pred_cam_list, dim=0)
+ pred_hand_pose = joint_conversion_fn(pred_hand_pose).view(batch_size, self.cfg.MANO.NUM_HAND_JOINTS+1, 3, 3)
+
+ pred_mano_params = {'global_orient': pred_hand_pose[:, [0]],
+ 'hand_pose': pred_hand_pose[:, 1:],
+ 'betas': pred_betas}
+ return pred_mano_params, pred_cam, pred_mano_params_list
diff --git a/third_party/hamer/hamer/models/losses.py b/third_party/hamer/hamer/models/losses.py
new file mode 100644
index 0000000000000000000000000000000000000000..d6e493c081a4d99b97b5641e85152c4d56072a58
--- /dev/null
+++ b/third_party/hamer/hamer/models/losses.py
@@ -0,0 +1,92 @@
+import torch
+import torch.nn as nn
+
+class Keypoint2DLoss(nn.Module):
+
+ def __init__(self, loss_type: str = 'l1'):
+ """
+ 2D keypoint loss module.
+ Args:
+ loss_type (str): Choose between l1 and l2 losses.
+ """
+ super(Keypoint2DLoss, self).__init__()
+ if loss_type == 'l1':
+ self.loss_fn = nn.L1Loss(reduction='none')
+ elif loss_type == 'l2':
+ self.loss_fn = nn.MSELoss(reduction='none')
+ else:
+ raise NotImplementedError('Unsupported loss function')
+
+ def forward(self, pred_keypoints_2d: torch.Tensor, gt_keypoints_2d: torch.Tensor) -> torch.Tensor:
+ """
+ Compute 2D reprojection loss on the keypoints.
+ Args:
+ pred_keypoints_2d (torch.Tensor): Tensor of shape [B, S, N, 2] containing projected 2D keypoints (B: batch_size, S: num_samples, N: num_keypoints)
+ gt_keypoints_2d (torch.Tensor): Tensor of shape [B, S, N, 3] containing the ground truth 2D keypoints and confidence.
+ Returns:
+ torch.Tensor: 2D keypoint loss.
+ """
+ conf = gt_keypoints_2d[:, :, -1].unsqueeze(-1).clone()
+ batch_size = conf.shape[0]
+ loss = (conf * self.loss_fn(pred_keypoints_2d, gt_keypoints_2d[:, :, :-1])).sum(dim=(1,2))
+ return loss.sum()
+
+
+class Keypoint3DLoss(nn.Module):
+
+ def __init__(self, loss_type: str = 'l1'):
+ """
+ 3D keypoint loss module.
+ Args:
+ loss_type (str): Choose between l1 and l2 losses.
+ """
+ super(Keypoint3DLoss, self).__init__()
+ if loss_type == 'l1':
+ self.loss_fn = nn.L1Loss(reduction='none')
+ elif loss_type == 'l2':
+ self.loss_fn = nn.MSELoss(reduction='none')
+ else:
+ raise NotImplementedError('Unsupported loss function')
+
+ def forward(self, pred_keypoints_3d: torch.Tensor, gt_keypoints_3d: torch.Tensor, pelvis_id: int = 0):
+ """
+ Compute 3D keypoint loss.
+ Args:
+ pred_keypoints_3d (torch.Tensor): Tensor of shape [B, S, N, 3] containing the predicted 3D keypoints (B: batch_size, S: num_samples, N: num_keypoints)
+ gt_keypoints_3d (torch.Tensor): Tensor of shape [B, S, N, 4] containing the ground truth 3D keypoints and confidence.
+ Returns:
+ torch.Tensor: 3D keypoint loss.
+ """
+ batch_size = pred_keypoints_3d.shape[0]
+ gt_keypoints_3d = gt_keypoints_3d.clone()
+ pred_keypoints_3d = pred_keypoints_3d - pred_keypoints_3d[:, pelvis_id, :].unsqueeze(dim=1)
+ gt_keypoints_3d[:, :, :-1] = gt_keypoints_3d[:, :, :-1] - gt_keypoints_3d[:, pelvis_id, :-1].unsqueeze(dim=1)
+ conf = gt_keypoints_3d[:, :, -1].unsqueeze(-1).clone()
+ gt_keypoints_3d = gt_keypoints_3d[:, :, :-1]
+ loss = (conf * self.loss_fn(pred_keypoints_3d, gt_keypoints_3d)).sum(dim=(1,2))
+ return loss.sum()
+
+class ParameterLoss(nn.Module):
+
+ def __init__(self):
+ """
+ MANO parameter loss module.
+ """
+ super(ParameterLoss, self).__init__()
+ self.loss_fn = nn.MSELoss(reduction='none')
+
+ def forward(self, pred_param: torch.Tensor, gt_param: torch.Tensor, has_param: torch.Tensor):
+ """
+ Compute MANO parameter loss.
+ Args:
+ pred_param (torch.Tensor): Tensor of shape [B, S, ...] containing the predicted parameters (body pose / global orientation / betas)
+ gt_param (torch.Tensor): Tensor of shape [B, S, ...] containing the ground truth MANO parameters.
+ Returns:
+ torch.Tensor: L2 parameter loss loss.
+ """
+ batch_size = pred_param.shape[0]
+ num_dims = len(pred_param.shape)
+ mask_dimension = [batch_size] + [1] * (num_dims-1)
+ has_param = has_param.type(pred_param.type()).view(*mask_dimension)
+ loss_param = (has_param * self.loss_fn(pred_param, gt_param))
+ return loss_param.sum()
diff --git a/third_party/hamer/hamer/models/mano_wrapper.py b/third_party/hamer/hamer/models/mano_wrapper.py
new file mode 100644
index 0000000000000000000000000000000000000000..f6f0cc336098e9303d2514c571307c56baf3bc86
--- /dev/null
+++ b/third_party/hamer/hamer/models/mano_wrapper.py
@@ -0,0 +1,40 @@
+import torch
+import numpy as np
+import pickle
+from typing import Optional
+import smplx
+from smplx.lbs import vertices2joints
+from smplx.utils import MANOOutput, to_tensor
+from smplx.vertex_ids import vertex_ids
+
+
+class MANO(smplx.MANOLayer):
+ def __init__(self, *args, joint_regressor_extra: Optional[str] = None, **kwargs):
+ """
+ Extension of the official MANO implementation to support more joints.
+ Args:
+ Same as MANOLayer.
+ joint_regressor_extra (str): Path to extra joint regressor.
+ """
+ super(MANO, self).__init__(*args, **kwargs)
+ mano_to_openpose = [0, 13, 14, 15, 16, 1, 2, 3, 17, 4, 5, 6, 18, 10, 11, 12, 19, 7, 8, 9, 20]
+
+ #2, 3, 5, 4, 1
+ if joint_regressor_extra is not None:
+ self.register_buffer('joint_regressor_extra', torch.tensor(pickle.load(open(joint_regressor_extra, 'rb'), encoding='latin1'), dtype=torch.float32))
+ self.register_buffer('extra_joints_idxs', to_tensor(list(vertex_ids['mano'].values()), dtype=torch.long))
+ self.register_buffer('joint_map', torch.tensor(mano_to_openpose, dtype=torch.long))
+
+ def forward(self, *args, **kwargs) -> MANOOutput:
+ """
+ Run forward pass. Same as MANO and also append an extra set of joints if joint_regressor_extra is specified.
+ """
+ mano_output = super(MANO, self).forward(*args, **kwargs)
+ extra_joints = torch.index_select(mano_output.vertices, 1, self.extra_joints_idxs)
+ joints = torch.cat([mano_output.joints, extra_joints], dim=1)
+ joints = joints[:, self.joint_map, :]
+ if hasattr(self, 'joint_regressor_extra'):
+ extra_joints = vertices2joints(self.joint_regressor_extra, mano_output.vertices)
+ joints = torch.cat([joints, extra_joints], dim=1)
+ mano_output.joints = joints
+ return mano_output
diff --git a/third_party/hamer/hamer/utils/__init__.py b/third_party/hamer/hamer/utils/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..09e47cdf8cdb303432d64902fbe58b256273f88a
--- /dev/null
+++ b/third_party/hamer/hamer/utils/__init__.py
@@ -0,0 +1,25 @@
+import torch
+from typing import Any
+
+from .renderer import Renderer
+from .mesh_renderer import MeshRenderer
+from .skeleton_renderer import SkeletonRenderer
+from .pose_utils import eval_pose, Evaluator
+
+def recursive_to(x: Any, target: torch.device):
+ """
+ Recursively transfer a batch of data to the target device
+ Args:
+ x (Any): Batch of data.
+ target (torch.device): Target device.
+ Returns:
+ Batch of data where all tensors are transfered to the target device.
+ """
+ if isinstance(x, dict):
+ return {k: recursive_to(v, target) for k, v in x.items()}
+ elif isinstance(x, torch.Tensor):
+ return x.to(target)
+ elif isinstance(x, list):
+ return [recursive_to(i, target) for i in x]
+ else:
+ return x
diff --git a/third_party/hamer/hamer/utils/download.py b/third_party/hamer/hamer/utils/download.py
new file mode 100644
index 0000000000000000000000000000000000000000..84d9b34a4546aa8f456e9ceae2276ecbe1f60fb6
--- /dev/null
+++ b/third_party/hamer/hamer/utils/download.py
@@ -0,0 +1,66 @@
+import os
+import re
+import sys
+from urllib import request as urlrequest
+
+
+def _progress_bar(count, total):
+ """Report download progress. Credit:
+ https://stackoverflow.com/questions/3173320/text-progress-bar-in-the-console/27871113
+ """
+ bar_len = 60
+ filled_len = int(round(bar_len * count / float(total)))
+ percents = round(100.0 * count / float(total), 1)
+ bar = "=" * filled_len + "-" * (bar_len - filled_len)
+ sys.stdout.write(
+ " [{}] {}% of {:.1f}MB file \r".format(bar, percents, total / 1024 / 1024)
+ )
+ sys.stdout.flush()
+ if count >= total:
+ sys.stdout.write("\n")
+
+
+def download_url(url, dst_file_path, chunk_size=8192, progress_hook=_progress_bar):
+ """Download url and write it to dst_file_path. Credit:
+ https://stackoverflow.com/questions/2028517/python-urllib2-progress-hook
+ """
+ # url = url + "?dl=1" if "dropbox" in url else url
+ req = urlrequest.Request(url)
+ response = urlrequest.urlopen(req)
+ total_size = response.info().get("Content-Length")
+ if total_size is None:
+ raise ValueError("Cannot determine size of download from {}".format(url))
+ total_size = int(total_size.strip())
+ bytes_so_far = 0
+
+ with open(dst_file_path, "wb") as f:
+ while 1:
+ chunk = response.read(chunk_size)
+ bytes_so_far += len(chunk)
+ if not chunk:
+ break
+
+ if progress_hook:
+ progress_hook(bytes_so_far, total_size)
+
+ f.write(chunk)
+ return bytes_so_far
+
+
+def cache_url(url_or_file, cache_file_path, download=True):
+ """Download the file specified by the URL to the cache_dir and return the path to
+ the cached file. If the argument is not a URL, simply return it as is.
+ """
+ is_url = re.match(r"^(?:http)s?://", url_or_file, re.IGNORECASE) is not None
+ if not is_url:
+ return url_or_file
+ url = url_or_file
+ if os.path.exists(cache_file_path):
+ return cache_file_path
+ cache_file_dir = os.path.dirname(cache_file_path)
+ if not os.path.exists(cache_file_dir):
+ os.makedirs(cache_file_dir)
+ if download:
+ print("Downloading remote file {} to {}".format(url, cache_file_path))
+ download_url(url, cache_file_path)
+ return cache_file_path
diff --git a/third_party/hamer/hamer/utils/geometry.py b/third_party/hamer/hamer/utils/geometry.py
new file mode 100644
index 0000000000000000000000000000000000000000..7929ef52608618a4682788487008e73c5736101b
--- /dev/null
+++ b/third_party/hamer/hamer/utils/geometry.py
@@ -0,0 +1,102 @@
+from typing import Optional
+import torch
+from torch.nn import functional as F
+
+def aa_to_rotmat(theta: torch.Tensor):
+ """
+ Convert axis-angle representation to rotation matrix.
+ Works by first converting it to a quaternion.
+ Args:
+ theta (torch.Tensor): Tensor of shape (B, 3) containing axis-angle representations.
+ Returns:
+ torch.Tensor: Corresponding rotation matrices with shape (B, 3, 3).
+ """
+ norm = torch.norm(theta + 1e-8, p = 2, dim = 1)
+ angle = torch.unsqueeze(norm, -1)
+ normalized = torch.div(theta, angle)
+ angle = angle * 0.5
+ v_cos = torch.cos(angle)
+ v_sin = torch.sin(angle)
+ quat = torch.cat([v_cos, v_sin * normalized], dim = 1)
+ return quat_to_rotmat(quat)
+
+def quat_to_rotmat(quat: torch.Tensor) -> torch.Tensor:
+ """
+ Convert quaternion representation to rotation matrix.
+ Args:
+ quat (torch.Tensor) of shape (B, 4); 4 <===> (w, x, y, z).
+ Returns:
+ torch.Tensor: Corresponding rotation matrices with shape (B, 3, 3).
+ """
+ norm_quat = quat
+ norm_quat = norm_quat/norm_quat.norm(p=2, dim=1, keepdim=True)
+ w, x, y, z = norm_quat[:,0], norm_quat[:,1], norm_quat[:,2], norm_quat[:,3]
+
+ B = quat.size(0)
+
+ w2, x2, y2, z2 = w.pow(2), x.pow(2), y.pow(2), z.pow(2)
+ wx, wy, wz = w*x, w*y, w*z
+ xy, xz, yz = x*y, x*z, y*z
+
+ rotMat = torch.stack([w2 + x2 - y2 - z2, 2*xy - 2*wz, 2*wy + 2*xz,
+ 2*wz + 2*xy, w2 - x2 + y2 - z2, 2*yz - 2*wx,
+ 2*xz - 2*wy, 2*wx + 2*yz, w2 - x2 - y2 + z2], dim=1).view(B, 3, 3)
+ return rotMat
+
+
+def rot6d_to_rotmat(x: torch.Tensor) -> torch.Tensor:
+ """
+ Convert 6D rotation representation to 3x3 rotation matrix.
+ Based on Zhou et al., "On the Continuity of Rotation Representations in Neural Networks", CVPR 2019
+ Args:
+ x (torch.Tensor): (B,6) Batch of 6-D rotation representations.
+ Returns:
+ torch.Tensor: Batch of corresponding rotation matrices with shape (B,3,3).
+ """
+ x = x.reshape(-1,2,3).permute(0, 2, 1).contiguous()
+ a1 = x[:, :, 0]
+ a2 = x[:, :, 1]
+ b1 = F.normalize(a1)
+ b2 = F.normalize(a2 - torch.einsum('bi,bi->b', b1, a2).unsqueeze(-1) * b1)
+ b3 = torch.cross(b1, b2)
+ return torch.stack((b1, b2, b3), dim=-1)
+
+def perspective_projection(points: torch.Tensor,
+ translation: torch.Tensor,
+ focal_length: torch.Tensor,
+ camera_center: Optional[torch.Tensor] = None,
+ rotation: Optional[torch.Tensor] = None) -> torch.Tensor:
+ """
+ Computes the perspective projection of a set of 3D points.
+ Args:
+ points (torch.Tensor): Tensor of shape (B, N, 3) containing the input 3D points.
+ translation (torch.Tensor): Tensor of shape (B, 3) containing the 3D camera translation.
+ focal_length (torch.Tensor): Tensor of shape (B, 2) containing the focal length in pixels.
+ camera_center (torch.Tensor): Tensor of shape (B, 2) containing the camera center in pixels.
+ rotation (torch.Tensor): Tensor of shape (B, 3, 3) containing the camera rotation.
+ Returns:
+ torch.Tensor: Tensor of shape (B, N, 2) containing the projection of the input points.
+ """
+ batch_size = points.shape[0]
+ if rotation is None:
+ rotation = torch.eye(3, device=points.device, dtype=points.dtype).unsqueeze(0).expand(batch_size, -1, -1)
+ if camera_center is None:
+ camera_center = torch.zeros(batch_size, 2, device=points.device, dtype=points.dtype)
+ # Populate intrinsic camera matrix K.
+ K = torch.zeros([batch_size, 3, 3], device=points.device, dtype=points.dtype)
+ K[:,0,0] = focal_length[:,0]
+ K[:,1,1] = focal_length[:,1]
+ K[:,2,2] = 1.
+ K[:,:-1, -1] = camera_center
+
+ # Transform points
+ points = torch.einsum('bij,bkj->bki', rotation, points)
+ points = points + translation.unsqueeze(1)
+
+ # Apply perspective distortion
+ projected_points = points / points[:,:,-1].unsqueeze(-1)
+
+ # Apply camera intrinsics
+ projected_points = torch.einsum('bij,bkj->bki', K, projected_points)
+
+ return projected_points[:, :, :-1]
\ No newline at end of file
diff --git a/third_party/hamer/hamer/utils/mesh_renderer.py b/third_party/hamer/hamer/utils/mesh_renderer.py
new file mode 100644
index 0000000000000000000000000000000000000000..bb3e8ed2e9aed8157ec852d06d5f13e8f4ff7c54
--- /dev/null
+++ b/third_party/hamer/hamer/utils/mesh_renderer.py
@@ -0,0 +1,149 @@
+import os
+if 'PYOPENGL_PLATFORM' not in os.environ:
+ os.environ['PYOPENGL_PLATFORM'] = 'egl'
+import torch
+from torchvision.utils import make_grid
+import numpy as np
+import pyrender
+import trimesh
+import cv2
+import torch.nn.functional as F
+
+from .render_openpose import render_openpose
+
+def create_raymond_lights():
+ import pyrender
+ thetas = np.pi * np.array([1.0 / 6.0, 1.0 / 6.0, 1.0 / 6.0])
+ phis = np.pi * np.array([0.0, 2.0 / 3.0, 4.0 / 3.0])
+
+ nodes = []
+
+ for phi, theta in zip(phis, thetas):
+ xp = np.sin(theta) * np.cos(phi)
+ yp = np.sin(theta) * np.sin(phi)
+ zp = np.cos(theta)
+
+ z = np.array([xp, yp, zp])
+ z = z / np.linalg.norm(z)
+ x = np.array([-z[1], z[0], 0.0])
+ if np.linalg.norm(x) == 0:
+ x = np.array([1.0, 0.0, 0.0])
+ x = x / np.linalg.norm(x)
+ y = np.cross(z, x)
+
+ matrix = np.eye(4)
+ matrix[:3,:3] = np.c_[x,y,z]
+ nodes.append(pyrender.Node(
+ light=pyrender.DirectionalLight(color=np.ones(3), intensity=1.0),
+ matrix=matrix
+ ))
+
+ return nodes
+
+class MeshRenderer:
+
+ def __init__(self, cfg, faces=None):
+ self.cfg = cfg
+ self.focal_length = cfg.EXTRA.FOCAL_LENGTH
+ self.img_res = cfg.MODEL.IMAGE_SIZE
+ self.renderer = pyrender.OffscreenRenderer(viewport_width=self.img_res,
+ viewport_height=self.img_res,
+ point_size=1.0)
+
+ self.camera_center = [self.img_res // 2, self.img_res // 2]
+ self.faces = faces
+
+ def visualize(self, vertices, camera_translation, images, focal_length=None, nrow=3, padding=2):
+ images_np = np.transpose(images, (0,2,3,1))
+ rend_imgs = []
+ for i in range(vertices.shape[0]):
+ fl = self.focal_length
+ rend_img = torch.from_numpy(np.transpose(self.__call__(vertices[i], camera_translation[i], images_np[i], focal_length=fl, side_view=False), (2,0,1))).float()
+ rend_img_side = torch.from_numpy(np.transpose(self.__call__(vertices[i], camera_translation[i], images_np[i], focal_length=fl, side_view=True), (2,0,1))).float()
+ rend_imgs.append(torch.from_numpy(images[i]))
+ rend_imgs.append(rend_img)
+ rend_imgs.append(rend_img_side)
+ rend_imgs = make_grid(rend_imgs, nrow=nrow, padding=padding)
+ return rend_imgs
+
+ def visualize_tensorboard(self, vertices, camera_translation, images, pred_keypoints, gt_keypoints, focal_length=None, nrow=5, padding=2):
+ images_np = np.transpose(images, (0,2,3,1))
+ rend_imgs = []
+ pred_keypoints = np.concatenate((pred_keypoints, np.ones_like(pred_keypoints)[:, :, [0]]), axis=-1)
+ pred_keypoints = self.img_res * (pred_keypoints + 0.5)
+ gt_keypoints[:, :, :-1] = self.img_res * (gt_keypoints[:, :, :-1] + 0.5)
+ #keypoint_matches = [(1, 12), (2, 8), (3, 7), (4, 6), (5, 9), (6, 10), (7, 11), (8, 14), (9, 2), (10, 1), (11, 0), (12, 3), (13, 4), (14, 5)]
+ for i in range(vertices.shape[0]):
+ fl = self.focal_length
+ rend_img = torch.from_numpy(np.transpose(self.__call__(vertices[i], camera_translation[i], images_np[i], focal_length=fl, side_view=False), (2,0,1))).float()
+ rend_img_side = torch.from_numpy(np.transpose(self.__call__(vertices[i], camera_translation[i], images_np[i], focal_length=fl, side_view=True), (2,0,1))).float()
+ hand_keypoints = pred_keypoints[i, :21]
+ #extra_keypoints = pred_keypoints[i, -19:]
+ #for pair in keypoint_matches:
+ # hand_keypoints[pair[0], :] = extra_keypoints[pair[1], :]
+ pred_keypoints_img = render_openpose(255 * images_np[i].copy(), hand_keypoints) / 255
+ hand_keypoints = gt_keypoints[i, :21]
+ #extra_keypoints = gt_keypoints[i, -19:]
+ #for pair in keypoint_matches:
+ # if extra_keypoints[pair[1], -1] > 0 and hand_keypoints[pair[0], -1] == 0:
+ # hand_keypoints[pair[0], :] = extra_keypoints[pair[1], :]
+ gt_keypoints_img = render_openpose(255*images_np[i].copy(), hand_keypoints) / 255
+ rend_imgs.append(torch.from_numpy(images[i]))
+ rend_imgs.append(rend_img)
+ rend_imgs.append(rend_img_side)
+ rend_imgs.append(torch.from_numpy(pred_keypoints_img).permute(2,0,1))
+ rend_imgs.append(torch.from_numpy(gt_keypoints_img).permute(2,0,1))
+ rend_imgs = make_grid(rend_imgs, nrow=nrow, padding=padding)
+ return rend_imgs
+
+ def __call__(self, vertices, camera_translation, image, focal_length=5000, text=None, resize=None, side_view=False, baseColorFactor=(1.0, 1.0, 0.9, 1.0), rot_angle=90):
+ renderer = pyrender.OffscreenRenderer(viewport_width=image.shape[1],
+ viewport_height=image.shape[0],
+ point_size=1.0)
+ material = pyrender.MetallicRoughnessMaterial(
+ metallicFactor=0.0,
+ alphaMode='OPAQUE',
+ baseColorFactor=baseColorFactor)
+
+ camera_translation[0] *= -1.
+
+ mesh = trimesh.Trimesh(vertices.copy(), self.faces.copy())
+ if side_view:
+ rot = trimesh.transformations.rotation_matrix(
+ np.radians(rot_angle), [0, 1, 0])
+ mesh.apply_transform(rot)
+ rot = trimesh.transformations.rotation_matrix(
+ np.radians(180), [1, 0, 0])
+ mesh.apply_transform(rot)
+ mesh = pyrender.Mesh.from_trimesh(mesh, material=material)
+
+ scene = pyrender.Scene(bg_color=[0.0, 0.0, 0.0, 0.0],
+ ambient_light=(0.3, 0.3, 0.3))
+ scene.add(mesh, 'mesh')
+
+ camera_pose = np.eye(4)
+ camera_pose[:3, 3] = camera_translation
+ camera_center = [image.shape[1] / 2., image.shape[0] / 2.]
+ camera = pyrender.IntrinsicsCamera(fx=focal_length, fy=focal_length,
+ cx=camera_center[0], cy=camera_center[1])
+ scene.add(camera, pose=camera_pose)
+
+
+ light_nodes = create_raymond_lights()
+ for node in light_nodes:
+ scene.add_node(node)
+
+ color, rend_depth = renderer.render(scene, flags=pyrender.RenderFlags.RGBA)
+ color = color.astype(np.float32) / 255.0
+ valid_mask = (color[:, :, -1] > 0)[:, :, np.newaxis]
+ if not side_view:
+ output_img = (color[:, :, :3] * valid_mask +
+ (1 - valid_mask) * image)
+ else:
+ output_img = color[:, :, :3]
+ if resize is not None:
+ output_img = cv2.resize(output_img, resize)
+
+ output_img = output_img.astype(np.float32)
+ renderer.delete()
+ return output_img
diff --git a/third_party/hamer/hamer/utils/misc.py b/third_party/hamer/hamer/utils/misc.py
new file mode 100644
index 0000000000000000000000000000000000000000..ffcfe784872b305c264ce6ef67fd0a9e9ad3390f
--- /dev/null
+++ b/third_party/hamer/hamer/utils/misc.py
@@ -0,0 +1,203 @@
+import time
+import warnings
+from importlib.util import find_spec
+from pathlib import Path
+from typing import Callable, List
+
+import hydra
+from omegaconf import DictConfig, OmegaConf
+from pytorch_lightning import Callback
+from pytorch_lightning.loggers import Logger
+from pytorch_lightning.utilities import rank_zero_only
+
+from . import pylogger, rich_utils
+
+log = pylogger.get_pylogger(__name__)
+
+
+def task_wrapper(task_func: Callable) -> Callable:
+ """Optional decorator that wraps the task function in extra utilities.
+
+ Makes multirun more resistant to failure.
+
+ Utilities:
+ - Calling the `utils.extras()` before the task is started
+ - Calling the `utils.close_loggers()` after the task is finished
+ - Logging the exception if occurs
+ - Logging the task total execution time
+ - Logging the output dir
+ """
+
+ def wrap(cfg: DictConfig):
+
+ # apply extra utilities
+ extras(cfg)
+
+ # execute the task
+ try:
+ start_time = time.time()
+ ret = task_func(cfg=cfg)
+ except Exception as ex:
+ log.exception("") # save exception to `.log` file
+ raise ex
+ finally:
+ path = Path(cfg.paths.output_dir, "exec_time.log")
+ content = f"'{cfg.task_name}' execution time: {time.time() - start_time} (s)"
+ save_file(path, content) # save task execution time (even if exception occurs)
+ close_loggers() # close loggers (even if exception occurs so multirun won't fail)
+
+ log.info(f"Output dir: {cfg.paths.output_dir}")
+
+ return ret
+
+ return wrap
+
+
+def extras(cfg: DictConfig) -> None:
+ """Applies optional utilities before the task is started.
+
+ Utilities:
+ - Ignoring python warnings
+ - Setting tags from command line
+ - Rich config printing
+ """
+
+ # return if no `extras` config
+ if not cfg.get("extras"):
+ log.warning("Extras config not found! ")
+ return
+
+ # disable python warnings
+ if cfg.extras.get("ignore_warnings"):
+ log.info("Disabling python warnings! ")
+ warnings.filterwarnings("ignore")
+
+ # prompt user to input tags from command line if none are provided in the config
+ if cfg.extras.get("enforce_tags"):
+ log.info("Enforcing tags! ")
+ rich_utils.enforce_tags(cfg, save_to_file=True)
+
+ # pretty print config tree using Rich library
+ if cfg.extras.get("print_config"):
+ log.info("Printing config tree with Rich! ")
+ rich_utils.print_config_tree(cfg, resolve=True, save_to_file=True)
+
+
+@rank_zero_only
+def save_file(path: str, content: str) -> None:
+ """Save file in rank zero mode (only on one process in multi-GPU setup)."""
+ with open(path, "w+") as file:
+ file.write(content)
+
+
+def instantiate_callbacks(callbacks_cfg: DictConfig) -> List[Callback]:
+ """Instantiates callbacks from config."""
+ callbacks: List[Callback] = []
+
+ if not callbacks_cfg:
+ log.warning("Callbacks config is empty.")
+ return callbacks
+
+ if not isinstance(callbacks_cfg, DictConfig):
+ raise TypeError("Callbacks config must be a DictConfig!")
+
+ for _, cb_conf in callbacks_cfg.items():
+ if isinstance(cb_conf, DictConfig) and "_target_" in cb_conf:
+ log.info(f"Instantiating callback <{cb_conf._target_}>")
+ callbacks.append(hydra.utils.instantiate(cb_conf))
+
+ return callbacks
+
+
+def instantiate_loggers(logger_cfg: DictConfig) -> List[Logger]:
+ """Instantiates loggers from config."""
+ logger: List[Logger] = []
+
+ if not logger_cfg:
+ log.warning("Logger config is empty.")
+ return logger
+
+ if not isinstance(logger_cfg, DictConfig):
+ raise TypeError("Logger config must be a DictConfig!")
+
+ for _, lg_conf in logger_cfg.items():
+ if isinstance(lg_conf, DictConfig) and "_target_" in lg_conf:
+ log.info(f"Instantiating logger <{lg_conf._target_}>")
+ logger.append(hydra.utils.instantiate(lg_conf))
+
+ return logger
+
+
+@rank_zero_only
+def log_hyperparameters(object_dict: dict) -> None:
+ """Controls which config parts are saved by lightning loggers.
+
+ Additionally saves:
+ - Number of model parameters
+ """
+
+ hparams = {}
+
+ cfg = object_dict["cfg"]
+ model = object_dict["model"]
+ trainer = object_dict["trainer"]
+
+ if not trainer.logger:
+ log.warning("Logger not found! Skipping hyperparameter logging...")
+ return
+
+ # save number of model parameters
+ hparams["model/params/total"] = sum(p.numel() for p in model.parameters())
+ hparams["model/params/trainable"] = sum(
+ p.numel() for p in model.parameters() if p.requires_grad
+ )
+ hparams["model/params/non_trainable"] = sum(
+ p.numel() for p in model.parameters() if not p.requires_grad
+ )
+
+ for k in cfg.keys():
+ hparams[k] = cfg.get(k)
+
+ # Resolve all interpolations
+ def _resolve(_cfg):
+ if isinstance(_cfg, DictConfig):
+ _cfg = OmegaConf.to_container(_cfg, resolve=True)
+ return _cfg
+
+ hparams = {k: _resolve(v) for k, v in hparams.items()}
+
+ # send hparams to all loggers
+ trainer.logger.log_hyperparams(hparams)
+
+
+def get_metric_value(metric_dict: dict, metric_name: str) -> float:
+ """Safely retrieves value of the metric logged in LightningModule."""
+
+ if not metric_name:
+ log.info("Metric name is None! Skipping metric value retrieval...")
+ return None
+
+ if metric_name not in metric_dict:
+ raise Exception(
+ f"Metric value not found! \n"
+ "Make sure metric name logged in LightningModule is correct!\n"
+ "Make sure `optimized_metric` name in `hparams_search` config is correct!"
+ )
+
+ metric_value = metric_dict[metric_name].item()
+ log.info(f"Retrieved metric value! <{metric_name}={metric_value}>")
+
+ return metric_value
+
+
+def close_loggers() -> None:
+ """Makes sure all loggers closed properly (prevents logging failure during multirun)."""
+
+ log.info("Closing loggers...")
+
+ if find_spec("wandb"): # if wandb is installed
+ import wandb
+
+ if wandb.run:
+ log.info("Closing wandb!")
+ wandb.finish()
diff --git a/third_party/hamer/hamer/utils/pose_utils.py b/third_party/hamer/hamer/utils/pose_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..f66061035f27098d27e4903cbd928032aad5205b
--- /dev/null
+++ b/third_party/hamer/hamer/utils/pose_utils.py
@@ -0,0 +1,352 @@
+"""
+Code adapted from: https://github.com/akanazawa/hmr/blob/master/src/benchmark/eval_util.py
+"""
+
+import torch
+import numpy as np
+from typing import Optional, Dict, List, Tuple
+
+def compute_similarity_transform(S1: torch.Tensor, S2: torch.Tensor) -> torch.Tensor:
+ """
+ Computes a similarity transform (sR, t) in a batched way that takes
+ a set of 3D points S1 (B, N, 3) closest to a set of 3D points S2 (B, N, 3),
+ where R is a 3x3 rotation matrix, t 3x1 translation, s scale.
+ i.e. solves the orthogonal Procrutes problem.
+ Args:
+ S1 (torch.Tensor): First set of points of shape (B, N, 3).
+ S2 (torch.Tensor): Second set of points of shape (B, N, 3).
+ Returns:
+ (torch.Tensor): The first set of points after applying the similarity transformation.
+ """
+
+ batch_size = S1.shape[0]
+ S1 = S1.permute(0, 2, 1)
+ S2 = S2.permute(0, 2, 1)
+ # 1. Remove mean.
+ mu1 = S1.mean(dim=2, keepdim=True)
+ mu2 = S2.mean(dim=2, keepdim=True)
+ X1 = S1 - mu1
+ X2 = S2 - mu2
+
+ # 2. Compute variance of X1 used for scale.
+ var1 = (X1**2).sum(dim=(1,2))
+
+ # 3. The outer product of X1 and X2.
+ K = torch.matmul(X1, X2.permute(0, 2, 1))
+
+ # 4. Solution that Maximizes trace(R'K) is R=U*V', where U, V are singular vectors of K.
+ U, s, V = torch.svd(K)
+ Vh = V.permute(0, 2, 1)
+
+ # Construct Z that fixes the orientation of R to get det(R)=1.
+ Z = torch.eye(U.shape[1], device=U.device).unsqueeze(0).repeat(batch_size, 1, 1)
+ Z[:, -1, -1] *= torch.sign(torch.linalg.det(torch.matmul(U, Vh)))
+
+ # Construct R.
+ R = torch.matmul(torch.matmul(V, Z), U.permute(0, 2, 1))
+
+ # 5. Recover scale.
+ trace = torch.matmul(R, K).diagonal(offset=0, dim1=-1, dim2=-2).sum(dim=-1)
+ scale = (trace / var1).unsqueeze(dim=-1).unsqueeze(dim=-1)
+
+ # 6. Recover translation.
+ t = mu2 - scale*torch.matmul(R, mu1)
+
+ # 7. Error:
+ S1_hat = scale*torch.matmul(R, S1) + t
+
+ return S1_hat.permute(0, 2, 1)
+
+def reconstruction_error(S1, S2) -> np.array:
+ """
+ Computes the mean Euclidean distance of 2 set of points S1, S2 after performing Procrustes alignment.
+ Args:
+ S1 (torch.Tensor): First set of points of shape (B, N, 3).
+ S2 (torch.Tensor): Second set of points of shape (B, N, 3).
+ Returns:
+ (np.array): Reconstruction error.
+ """
+ S1_hat = compute_similarity_transform(S1, S2)
+ re = torch.sqrt( ((S1_hat - S2)** 2).sum(dim=-1)).mean(dim=-1)
+ return re
+
+def eval_pose(pred_joints, gt_joints) -> Tuple[np.array, np.array]:
+ """
+ Compute joint errors in mm before and after Procrustes alignment.
+ Args:
+ pred_joints (torch.Tensor): Predicted 3D joints of shape (B, N, 3).
+ gt_joints (torch.Tensor): Ground truth 3D joints of shape (B, N, 3).
+ Returns:
+ Tuple[np.array, np.array]: Joint errors in mm before and after alignment.
+ """
+ # Absolute error (MPJPE)
+ mpjpe = torch.sqrt(((pred_joints - gt_joints) ** 2).sum(dim=-1)).mean(dim=-1).cpu().numpy()
+
+ # Reconstruction_error
+ r_error = reconstruction_error(pred_joints, gt_joints).cpu().numpy()
+ return 1000 * mpjpe, 1000 * r_error
+
+class Evaluator:
+
+ def __init__(self,
+ dataset_length: int,
+ dataset: str,
+ keypoint_list: List,
+ pelvis_ind: int,
+ metrics: List = ['mode_mpjpe', 'mode_re', 'min_mpjpe', 'min_re'],
+ preds: List = ['vertices', 'keypoints_3d'],
+ pck_thresholds: Optional[List] = None):
+ """
+ Class used for evaluating trained models on different 3D pose datasets.
+ Args:
+ dataset_length (int): Total dataset length.
+ keypoint_list [List]: List of keypoints used for evaluation.
+ pelvis_ind (int): Index of pelvis keypoint; used for aligning the predictions and ground truth.
+ metrics [List]: List of evaluation metrics to record.
+ """
+ self.dataset_length = dataset_length
+ self.dataset = dataset
+ self.keypoint_list = keypoint_list
+ self.pelvis_ind = pelvis_ind
+ self.metrics = metrics
+ self.preds = preds
+ if self.metrics is not None:
+ for metric in self.metrics:
+ setattr(self, metric, np.zeros((dataset_length,)))
+ if self.preds is not None:
+ for pred in self.preds:
+ if pred == 'vertices':
+ self.vertices = np.zeros((dataset_length, 778, 3))
+ if pred == 'keypoints_3d':
+ self.keypoints_3d = np.zeros((dataset_length, 21, 3))
+ self.counter = 0
+ if pck_thresholds is None:
+ self.pck_evaluator = None
+ else:
+ self.pck_evaluator = EvaluatorPCK(pck_thresholds)
+
+ def log(self):
+ """
+ Print current evaluation metrics
+ """
+ if self.counter == 0:
+ print('Evaluation has not started')
+ return
+ print(f'{self.counter} / {self.dataset_length} samples')
+ if self.pck_evaluator is not None:
+ self.pck_evaluator.log()
+ if self.metrics is not None:
+ for metric in self.metrics:
+ if metric in ['mode_mpjpe', 'mode_re', 'min_mpjpe', 'min_re']:
+ unit = 'mm'
+ else:
+ unit = ''
+ print(f'{metric}: {getattr(self, metric)[:self.counter].mean()} {unit}')
+ print('***')
+
+ def get_metrics_dict(self) -> Dict:
+ """
+ Returns:
+ Dict: Dictionary of evaluation metrics.
+ """
+ d1 = {metric: getattr(self, metric)[:self.counter].mean() for metric in self.metrics}
+ if self.pck_evaluator is not None:
+ d2 = self.pck_evaluator.get_metrics_dict()
+ d1.update(d2)
+ return d1
+
+ def get_preds_dict(self) -> Dict:
+ """
+ Returns:
+ Dict: Dictionary of evaluation preds.
+ """
+ d1 = {pred: getattr(self, pred)[:self.counter] for pred in self.preds}
+ return d1
+
+ def __call__(self, output: Dict, batch: Dict, opt_output: Optional[Dict] = None):
+ """
+ Evaluate current batch.
+ Args:
+ output (Dict): Regression output.
+ batch (Dict): Dictionary containing images and their corresponding annotations.
+ opt_output (Dict): Optimization output.
+ """
+ if self.pck_evaluator is not None:
+ self.pck_evaluator(output, batch, opt_output)
+
+ pred_keypoints_3d = output['pred_keypoints_3d'].detach()
+ pred_keypoints_3d = pred_keypoints_3d[:,None,:,:]
+ batch_size = pred_keypoints_3d.shape[0]
+ num_samples = pred_keypoints_3d.shape[1]
+ gt_keypoints_3d = batch['keypoints_3d'][:, :, :-1].unsqueeze(1).repeat(1, num_samples, 1, 1)
+ pred_vertices = output['pred_vertices'].detach()
+
+ # Align predictions and ground truth such that the pelvis location is at the origin
+ pred_keypoints_3d -= pred_keypoints_3d[:, :, [self.pelvis_ind]]
+ gt_keypoints_3d -= gt_keypoints_3d[:, :, [self.pelvis_ind]]
+
+ # Compute joint errors
+ mpjpe, re = eval_pose(pred_keypoints_3d.reshape(batch_size * num_samples, -1, 3)[:, self.keypoint_list], gt_keypoints_3d.reshape(batch_size * num_samples, -1 ,3)[:, self.keypoint_list])
+ mpjpe = mpjpe.reshape(batch_size, num_samples)
+ re = re.reshape(batch_size, num_samples)
+
+ # Compute 2d keypoint errors
+ bbox_expand_factor = batch['bbox_expand_factor'][:,None,None,None].detach()
+ pred_keypoints_2d = output['pred_keypoints_2d'].detach()
+ pred_keypoints_2d = pred_keypoints_2d[:,None,:,:]*bbox_expand_factor
+ gt_keypoints_2d = batch['keypoints_2d'][:,None,:,:].repeat(1, num_samples, 1, 1)*bbox_expand_factor
+ conf = gt_keypoints_2d[:, :, :, -1].clone()
+ kp_err = torch.nn.functional.mse_loss(
+ pred_keypoints_2d,
+ gt_keypoints_2d[:, :, :, :-1],
+ reduction='none'
+ ).sum(dim=3)
+ kp_l2_loss = (conf * kp_err).mean(dim=2)
+ kp_l2_loss = kp_l2_loss.detach().cpu().numpy()
+
+ # Compute joint errors after optimization, if available.
+ if opt_output is not None:
+ opt_keypoints_3d = opt_output['model_joints']
+ opt_keypoints_3d -= opt_keypoints_3d[:, [self.pelvis_ind]]
+ opt_mpjpe, opt_re = eval_pose(opt_keypoints_3d[:, self.keypoint_list], gt_keypoints_3d[:, 0, self.keypoint_list])
+
+ # The 0-th sample always corresponds to the mode
+ if hasattr(self, 'mode_mpjpe'):
+ mode_mpjpe = mpjpe[:, 0]
+ self.mode_mpjpe[self.counter:self.counter+batch_size] = mode_mpjpe
+ if hasattr(self, 'mode_re'):
+ mode_re = re[:, 0]
+ self.mode_re[self.counter:self.counter+batch_size] = mode_re
+ if hasattr(self, 'mode_kpl2'):
+ mode_kpl2 = kp_l2_loss[:, 0]
+ self.mode_kpl2[self.counter:self.counter+batch_size] = mode_kpl2
+ if hasattr(self, 'min_mpjpe'):
+ min_mpjpe = mpjpe.min(axis=-1)
+ self.min_mpjpe[self.counter:self.counter+batch_size] = min_mpjpe
+ if hasattr(self, 'min_re'):
+ min_re = re.min(axis=-1)
+ self.min_re[self.counter:self.counter+batch_size] = min_re
+ if hasattr(self, 'min_kpl2'):
+ min_kpl2 = kp_l2_loss.min(axis=-1)
+ self.min_kpl2[self.counter:self.counter+batch_size] = min_kpl2
+ if hasattr(self, 'opt_mpjpe'):
+ self.opt_mpjpe[self.counter:self.counter+batch_size] = opt_mpjpe
+ if hasattr(self, 'opt_re'):
+ self.opt_re[self.counter:self.counter+batch_size] = opt_re
+ if hasattr(self, 'vertices'):
+ self.vertices[self.counter:self.counter+batch_size] = pred_vertices.cpu().numpy()
+ if hasattr(self, 'keypoints_3d'):
+ if self.dataset == 'HO3D-VAL':
+ pred_keypoints_3d = pred_keypoints_3d[:,:,[0,5,6,7,9,10,11,17,18,19,13,14,15,1,2,3,4,8,12,16,20]]
+ self.keypoints_3d[self.counter:self.counter+batch_size] = pred_keypoints_3d.squeeze().cpu().numpy()
+
+ self.counter += batch_size
+
+ if hasattr(self, 'mode_mpjpe') and hasattr(self, 'mode_re'):
+ return {
+ 'mode_mpjpe': mode_mpjpe,
+ 'mode_re': mode_re,
+ }
+ else:
+ return {}
+
+
+class EvaluatorPCK:
+
+ def __init__(self, thresholds: List = [0.05, 0.1, 0.2, 0.3, 0.4, 0.5],):
+ """
+ Class used for evaluating trained models on different 3D pose datasets.
+ Args:
+ thresholds [List]: List of PCK thresholds to evaluate.
+ metrics [List]: List of evaluation metrics to record.
+ """
+ self.thresholds = thresholds
+ self.pred_kp_2d = []
+ self.gt_kp_2d = []
+ self.gt_conf_2d = []
+ self.scale = []
+ self.counter = 0
+
+ def log(self):
+ """
+ Print current evaluation metrics
+ """
+ if self.counter == 0:
+ print('Evaluation has not started')
+ return
+ print(f'{self.counter} samples')
+ metrics_dict = self.get_metrics_dict()
+ for metric in metrics_dict:
+ print(f'{metric}: {metrics_dict[metric]}')
+ print('***')
+
+ def get_metrics_dict(self) -> Dict:
+ """
+ Returns:
+ Dict: Dictionary of evaluation metrics.
+ """
+ pcks = self.compute_pcks()
+ metrics = {}
+ for thr, (acc,avg_acc,cnt) in zip(self.thresholds, pcks):
+ metrics.update({f'kp{i}_pck_{thr}': float(a) for i, a in enumerate(acc) if a>=0})
+ metrics.update({f'kpAvg_pck_{thr}': float(avg_acc)})
+ return metrics
+
+ def compute_pcks(self):
+ pred_kp_2d = np.concatenate(self.pred_kp_2d, axis=0)
+ gt_kp_2d = np.concatenate(self.gt_kp_2d, axis=0)
+ gt_conf_2d = np.concatenate(self.gt_conf_2d, axis=0)
+ scale = np.concatenate(self.scale, axis=0)
+ assert pred_kp_2d.shape == gt_kp_2d.shape
+ assert pred_kp_2d[..., 0].shape == gt_conf_2d.shape
+ assert pred_kp_2d.shape[1] == 1 # num_samples
+ assert scale.shape[0] == gt_conf_2d.shape[0] # num_samples
+
+ pcks = [
+ self.keypoint_pck_accuracy(
+ pred_kp_2d[:, 0, :, :],
+ gt_kp_2d[:, 0, :, :],
+ gt_conf_2d[:, 0, :]>0.5,
+ thr=thr,
+ scale = scale[:,None]
+ )
+ for thr in self.thresholds
+ ]
+ return pcks
+
+ def keypoint_pck_accuracy(self, pred, gt, conf, thr, scale):
+ dist = np.sqrt(np.sum((pred-gt)**2, axis=2))
+ all_joints = conf>0.5
+ correct_joints = np.logical_and(dist<=scale*thr, all_joints)
+ pck = correct_joints.sum(axis=0)/all_joints.sum(axis=0)
+ return pck, pck.mean(), pck.shape[0]
+
+ def __call__(self, output: Dict, batch: Dict, opt_output: Optional[Dict] = None):
+ """
+ Evaluate current batch.
+ Args:
+ output (Dict): Regression output.
+ batch (Dict): Dictionary containing images and their corresponding annotations.
+ opt_output (Dict): Optimization output.
+ """
+ pred_keypoints_2d = output['pred_keypoints_2d'].detach()
+ num_samples = 1
+ batch_size = pred_keypoints_2d.shape[0]
+
+ right = batch['right'].detach()
+ pred_keypoints_2d[:,:,0] = (2*right[:,None]-1)*pred_keypoints_2d[:,:,0]
+ box_size = batch['box_size'].detach()
+ box_center = batch['box_center'].detach()
+ bbox_expand_factor = batch['bbox_expand_factor'].detach()
+ scale = box_size/bbox_expand_factor
+ bbox_expand_factor = bbox_expand_factor[:,None,None,None]
+ pred_keypoints_2d = pred_keypoints_2d*box_size[:,None,None]+box_center[:,None]
+ pred_keypoints_2d = pred_keypoints_2d[:,None,:,:]
+ gt_keypoints_2d = batch['orig_keypoints_2d'][:,None,:,:].repeat(1, num_samples, 1, 1)
+
+ self.pred_kp_2d.append(pred_keypoints_2d[:, :, :, :2].detach().cpu().numpy())
+ self.gt_conf_2d.append(gt_keypoints_2d[:, :, :, -1].detach().cpu().numpy())
+ self.gt_kp_2d.append(gt_keypoints_2d[:, :, :, :2].detach().cpu().numpy())
+ self.scale.append(scale.detach().cpu().numpy())
+
+ self.counter += batch_size
diff --git a/third_party/hamer/hamer/utils/pylogger.py b/third_party/hamer/hamer/utils/pylogger.py
new file mode 100644
index 0000000000000000000000000000000000000000..92ffa71893ec20acde65e44d899334a38d8d1333
--- /dev/null
+++ b/third_party/hamer/hamer/utils/pylogger.py
@@ -0,0 +1,17 @@
+import logging
+
+from pytorch_lightning.utilities import rank_zero_only
+
+
+def get_pylogger(name=__name__) -> logging.Logger:
+ """Initializes multi-GPU-friendly python command line logger."""
+
+ logger = logging.getLogger(name)
+
+ # this ensures all logging levels get marked with the rank zero decorator
+ # otherwise logs would get multiplied for each GPU process in multi-GPU setup
+ logging_levels = ("debug", "info", "warning", "error", "exception", "fatal", "critical")
+ for level in logging_levels:
+ setattr(logger, level, rank_zero_only(getattr(logger, level)))
+
+ return logger
diff --git a/third_party/hamer/hamer/utils/render_openpose.py b/third_party/hamer/hamer/utils/render_openpose.py
new file mode 100644
index 0000000000000000000000000000000000000000..9eb7784e125d40e57eca4cf1e470f43e39654dae
--- /dev/null
+++ b/third_party/hamer/hamer/utils/render_openpose.py
@@ -0,0 +1,191 @@
+"""
+Render OpenPose keypoints.
+Code was ported to Python from the official C++ implementation https://github.com/CMU-Perceptual-Computing-Lab/openpose/blob/master/src/openpose/utilities/keypoint.cpp
+"""
+import cv2
+import math
+import numpy as np
+from typing import List, Tuple
+
+def get_keypoints_rectangle(keypoints: np.array, threshold: float) -> Tuple[float, float, float]:
+ """
+ Compute rectangle enclosing keypoints above the threshold.
+ Args:
+ keypoints (np.array): Keypoint array of shape (N, 3).
+ threshold (float): Confidence visualization threshold.
+ Returns:
+ Tuple[float, float, float]: Rectangle width, height and area.
+ """
+ valid_ind = keypoints[:, -1] > threshold
+ if valid_ind.sum() > 0:
+ valid_keypoints = keypoints[valid_ind][:, :-1]
+ max_x = valid_keypoints[:,0].max()
+ max_y = valid_keypoints[:,1].max()
+ min_x = valid_keypoints[:,0].min()
+ min_y = valid_keypoints[:,1].min()
+ width = max_x - min_x
+ height = max_y - min_y
+ area = width * height
+ return width, height, area
+ else:
+ return 0,0,0
+
+def render_keypoints(img: np.array,
+ keypoints: np.array,
+ pairs: List,
+ colors: List,
+ thickness_circle_ratio: float,
+ thickness_line_ratio_wrt_circle: float,
+ pose_scales: List,
+ threshold: float = 0.1,
+ alpha: float = 1.0) -> np.array:
+ """
+ Render keypoints on input image.
+ Args:
+ img (np.array): Input image of shape (H, W, 3) with pixel values in the [0,255] range.
+ keypoints (np.array): Keypoint array of shape (N, 3).
+ pairs (List): List of keypoint pairs per limb.
+ colors: (List): List of colors per keypoint.
+ thickness_circle_ratio (float): Circle thickness ratio.
+ thickness_line_ratio_wrt_circle (float): Line thickness ratio wrt the circle.
+ pose_scales (List): List of pose scales.
+ threshold (float): Only visualize keypoints with confidence above the threshold.
+ Returns:
+ (np.array): Image of shape (H, W, 3) with keypoints drawn on top of the original image.
+ """
+ img_orig = img.copy()
+ width, height = img.shape[1], img.shape[2]
+ area = width * height
+
+ lineType = 8
+ shift = 0
+ numberColors = len(colors)
+ thresholdRectangle = 0.1
+
+ person_width, person_height, person_area = get_keypoints_rectangle(keypoints, thresholdRectangle)
+ if person_area > 0:
+ ratioAreas = min(1, max(person_width / width, person_height / height))
+ thicknessRatio = np.maximum(np.round(math.sqrt(area) * thickness_circle_ratio * ratioAreas), 2)
+ thicknessCircle = np.maximum(1, thicknessRatio if ratioAreas > 0.05 else -np.ones_like(thicknessRatio))
+ thicknessLine = np.maximum(1, np.round(thicknessRatio * thickness_line_ratio_wrt_circle))
+ radius = thicknessRatio / 2
+
+ img = np.ascontiguousarray(img.copy())
+ for i, pair in enumerate(pairs):
+ index1, index2 = pair
+ if keypoints[index1, -1] > threshold and keypoints[index2, -1] > threshold:
+ thicknessLineScaled = int(round(min(thicknessLine[index1], thicknessLine[index2]) * pose_scales[0]))
+ colorIndex = index2
+ color = colors[colorIndex % numberColors]
+ keypoint1 = keypoints[index1, :-1].astype(np.int32)
+ keypoint2 = keypoints[index2, :-1].astype(np.int32)
+ cv2.line(img, tuple(keypoint1.tolist()), tuple(keypoint2.tolist()), tuple(color.tolist()), thicknessLineScaled, lineType, shift)
+ for part in range(len(keypoints)):
+ faceIndex = part
+ if keypoints[faceIndex, -1] > threshold:
+ radiusScaled = int(round(radius[faceIndex] * pose_scales[0]))
+ thicknessCircleScaled = int(round(thicknessCircle[faceIndex] * pose_scales[0]))
+ colorIndex = part
+ color = colors[colorIndex % numberColors]
+ center = keypoints[faceIndex, :-1].astype(np.int32)
+ cv2.circle(img, tuple(center.tolist()), radiusScaled, tuple(color.tolist()), thicknessCircleScaled, lineType, shift)
+ return img
+
+def render_hand_keypoints(img, right_hand_keypoints, threshold=0.1, use_confidence=False, map_fn=lambda x: np.ones_like(x), alpha=1.0):
+ if use_confidence and map_fn is not None:
+ #thicknessCircleRatioLeft = 1./50 * map_fn(left_hand_keypoints[:, -1])
+ thicknessCircleRatioRight = 1./50 * map_fn(right_hand_keypoints[:, -1])
+ else:
+ #thicknessCircleRatioLeft = 1./50 * np.ones(left_hand_keypoints.shape[0])
+ thicknessCircleRatioRight = 1./50 * np.ones(right_hand_keypoints.shape[0])
+ thicknessLineRatioWRTCircle = 0.75
+ pairs = [0,1, 1,2, 2,3, 3,4, 0,5, 5,6, 6,7, 7,8, 0,9, 9,10, 10,11, 11,12, 0,13, 13,14, 14,15, 15,16, 0,17, 17,18, 18,19, 19,20]
+ pairs = np.array(pairs).reshape(-1,2)
+
+ colors = [100., 100., 100.,
+ 100., 0., 0.,
+ 150., 0., 0.,
+ 200., 0., 0.,
+ 255., 0., 0.,
+ 100., 100., 0.,
+ 150., 150., 0.,
+ 200., 200., 0.,
+ 255., 255., 0.,
+ 0., 100., 50.,
+ 0., 150., 75.,
+ 0., 200., 100.,
+ 0., 255., 125.,
+ 0., 50., 100.,
+ 0., 75., 150.,
+ 0., 100., 200.,
+ 0., 125., 255.,
+ 100., 0., 100.,
+ 150., 0., 150.,
+ 200., 0., 200.,
+ 255., 0., 255.]
+ colors = np.array(colors).reshape(-1,3)
+ #colors = np.zeros_like(colors)
+ poseScales = [1]
+ #img = render_keypoints(img, left_hand_keypoints, pairs, colors, thicknessCircleRatioLeft, thicknessLineRatioWRTCircle, poseScales, threshold, alpha=alpha)
+ img = render_keypoints(img, right_hand_keypoints, pairs, colors, thicknessCircleRatioRight, thicknessLineRatioWRTCircle, poseScales, threshold, alpha=alpha)
+ #img = render_keypoints(img, right_hand_keypoints, pairs, colors, thickness_circle_ratio, thickness_line_ratio_wrt_circle, pose_scales, 0.1)
+ return img
+
+def render_body_keypoints(img: np.array,
+ body_keypoints: np.array) -> np.array:
+ """
+ Render OpenPose body keypoints on input image.
+ Args:
+ img (np.array): Input image of shape (H, W, 3) with pixel values in the [0,255] range.
+ body_keypoints (np.array): Keypoint array of shape (N, 3); 3 <====> (x, y, confidence).
+ Returns:
+ (np.array): Image of shape (H, W, 3) with keypoints drawn on top of the original image.
+ """
+
+ thickness_circle_ratio = 1./75. * np.ones(body_keypoints.shape[0])
+ thickness_line_ratio_wrt_circle = 0.75
+ pairs = []
+ pairs = [1,8,1,2,1,5,2,3,3,4,5,6,6,7,8,9,9,10,10,11,8,12,12,13,13,14,1,0,0,15,15,17,0,16,16,18,14,19,19,20,14,21,11,22,22,23,11,24]
+ pairs = np.array(pairs).reshape(-1,2)
+ colors = [255., 0., 85.,
+ 255., 0., 0.,
+ 255., 85., 0.,
+ 255., 170., 0.,
+ 255., 255., 0.,
+ 170., 255., 0.,
+ 85., 255., 0.,
+ 0., 255., 0.,
+ 255., 0., 0.,
+ 0., 255., 85.,
+ 0., 255., 170.,
+ 0., 255., 255.,
+ 0., 170., 255.,
+ 0., 85., 255.,
+ 0., 0., 255.,
+ 255., 0., 170.,
+ 170., 0., 255.,
+ 255., 0., 255.,
+ 85., 0., 255.,
+ 0., 0., 255.,
+ 0., 0., 255.,
+ 0., 0., 255.,
+ 0., 255., 255.,
+ 0., 255., 255.,
+ 0., 255., 255.]
+ colors = np.array(colors).reshape(-1,3)
+ pose_scales = [1]
+ return render_keypoints(img, body_keypoints, pairs, colors, thickness_circle_ratio, thickness_line_ratio_wrt_circle, pose_scales, 0.1)
+
+def render_openpose(img: np.array,
+ hand_keypoints: np.array) -> np.array:
+ """
+ Render keypoints in the OpenPose format on input image.
+ Args:
+ img (np.array): Input image of shape (H, W, 3) with pixel values in the [0,255] range.
+ body_keypoints (np.array): Keypoint array of shape (N, 3); 3 <====> (x, y, confidence).
+ Returns:
+ (np.array): Image of shape (H, W, 3) with keypoints drawn on top of the original image.
+ """
+ #img = render_body_keypoints(img, body_keypoints)
+ img = render_hand_keypoints(img, hand_keypoints)
+ return img
diff --git a/third_party/hamer/hamer/utils/renderer.py b/third_party/hamer/hamer/utils/renderer.py
new file mode 100644
index 0000000000000000000000000000000000000000..0e161bb05921e52a684427e3eb87c4f8739a5d89
--- /dev/null
+++ b/third_party/hamer/hamer/utils/renderer.py
@@ -0,0 +1,423 @@
+import os
+if 'PYOPENGL_PLATFORM' not in os.environ:
+ os.environ['PYOPENGL_PLATFORM'] = 'egl'
+import torch
+import numpy as np
+import pyrender
+import trimesh
+import cv2
+from yacs.config import CfgNode
+from typing import List, Optional
+
+def cam_crop_to_full(cam_bbox, box_center, box_size, img_size, focal_length=5000.):
+ # Convert cam_bbox to full image
+ img_w, img_h = img_size[:, 0], img_size[:, 1]
+ cx, cy, b = box_center[:, 0], box_center[:, 1], box_size
+ w_2, h_2 = img_w / 2., img_h / 2.
+ bs = b * cam_bbox[:, 0] + 1e-9
+ tz = 2 * focal_length / bs
+ tx = (2 * (cx - w_2) / bs) + cam_bbox[:, 1]
+ ty = (2 * (cy - h_2) / bs) + cam_bbox[:, 2]
+ full_cam = torch.stack([tx, ty, tz], dim=-1)
+ return full_cam
+
+def get_light_poses(n_lights=5, elevation=np.pi / 3, dist=12):
+ # get lights in a circle around origin at elevation
+ thetas = elevation * np.ones(n_lights)
+ phis = 2 * np.pi * np.arange(n_lights) / n_lights
+ poses = []
+ trans = make_translation(torch.tensor([0, 0, dist]))
+ for phi, theta in zip(phis, thetas):
+ rot = make_rotation(rx=-theta, ry=phi, order="xyz")
+ poses.append((rot @ trans).numpy())
+ return poses
+
+def make_translation(t):
+ return make_4x4_pose(torch.eye(3), t)
+
+def make_rotation(rx=0, ry=0, rz=0, order="xyz"):
+ Rx = rotx(rx)
+ Ry = roty(ry)
+ Rz = rotz(rz)
+ if order == "xyz":
+ R = Rz @ Ry @ Rx
+ elif order == "xzy":
+ R = Ry @ Rz @ Rx
+ elif order == "yxz":
+ R = Rz @ Rx @ Ry
+ elif order == "yzx":
+ R = Rx @ Rz @ Ry
+ elif order == "zyx":
+ R = Rx @ Ry @ Rz
+ elif order == "zxy":
+ R = Ry @ Rx @ Rz
+ return make_4x4_pose(R, torch.zeros(3))
+
+def make_4x4_pose(R, t):
+ """
+ :param R (*, 3, 3)
+ :param t (*, 3)
+ return (*, 4, 4)
+ """
+ dims = R.shape[:-2]
+ pose_3x4 = torch.cat([R, t.view(*dims, 3, 1)], dim=-1)
+ bottom = (
+ torch.tensor([0, 0, 0, 1], device=R.device)
+ .reshape(*(1,) * len(dims), 1, 4)
+ .expand(*dims, 1, 4)
+ )
+ return torch.cat([pose_3x4, bottom], dim=-2)
+
+
+def rotx(theta):
+ return torch.tensor(
+ [
+ [1, 0, 0],
+ [0, np.cos(theta), -np.sin(theta)],
+ [0, np.sin(theta), np.cos(theta)],
+ ],
+ dtype=torch.float32,
+ )
+
+
+def roty(theta):
+ return torch.tensor(
+ [
+ [np.cos(theta), 0, np.sin(theta)],
+ [0, 1, 0],
+ [-np.sin(theta), 0, np.cos(theta)],
+ ],
+ dtype=torch.float32,
+ )
+
+
+def rotz(theta):
+ return torch.tensor(
+ [
+ [np.cos(theta), -np.sin(theta), 0],
+ [np.sin(theta), np.cos(theta), 0],
+ [0, 0, 1],
+ ],
+ dtype=torch.float32,
+ )
+
+
+def create_raymond_lights() -> List[pyrender.Node]:
+ """
+ Return raymond light nodes for the scene.
+ """
+ thetas = np.pi * np.array([1.0 / 6.0, 1.0 / 6.0, 1.0 / 6.0])
+ phis = np.pi * np.array([0.0, 2.0 / 3.0, 4.0 / 3.0])
+
+ nodes = []
+
+ for phi, theta in zip(phis, thetas):
+ xp = np.sin(theta) * np.cos(phi)
+ yp = np.sin(theta) * np.sin(phi)
+ zp = np.cos(theta)
+
+ z = np.array([xp, yp, zp])
+ z = z / np.linalg.norm(z)
+ x = np.array([-z[1], z[0], 0.0])
+ if np.linalg.norm(x) == 0:
+ x = np.array([1.0, 0.0, 0.0])
+ x = x / np.linalg.norm(x)
+ y = np.cross(z, x)
+
+ matrix = np.eye(4)
+ matrix[:3,:3] = np.c_[x,y,z]
+ nodes.append(pyrender.Node(
+ light=pyrender.DirectionalLight(color=np.ones(3), intensity=1.0),
+ matrix=matrix
+ ))
+
+ return nodes
+
+class Renderer:
+
+ def __init__(self, cfg: CfgNode, faces: np.array):
+ """
+ Wrapper around the pyrender renderer to render MANO meshes.
+ Args:
+ cfg (CfgNode): Model config file.
+ faces (np.array): Array of shape (F, 3) containing the mesh faces.
+ """
+ self.cfg = cfg
+ self.focal_length = cfg.EXTRA.FOCAL_LENGTH
+ self.img_res = cfg.MODEL.IMAGE_SIZE
+
+ # add faces that make the hand mesh watertight
+ faces_new = np.array([[92, 38, 234],
+ [234, 38, 239],
+ [38, 122, 239],
+ [239, 122, 279],
+ [122, 118, 279],
+ [279, 118, 215],
+ [118, 117, 215],
+ [215, 117, 214],
+ [117, 119, 214],
+ [214, 119, 121],
+ [119, 120, 121],
+ [121, 120, 78],
+ [120, 108, 78],
+ [78, 108, 79]])
+ faces = np.concatenate([faces, faces_new], axis=0)
+
+ self.camera_center = [self.img_res // 2, self.img_res // 2]
+ self.faces = faces
+ self.faces_left = self.faces[:,[0,2,1]]
+
+ def __call__(self,
+ vertices: np.array,
+ camera_translation: np.array,
+ image: torch.Tensor,
+ full_frame: bool = False,
+ imgname: Optional[str] = None,
+ side_view=False, rot_angle=90,
+ mesh_base_color=(1.0, 1.0, 0.9),
+ scene_bg_color=(0,0,0),
+ return_rgba=False,
+ ) -> np.array:
+ """
+ Render meshes on input image
+ Args:
+ vertices (np.array): Array of shape (V, 3) containing the mesh vertices.
+ camera_translation (np.array): Array of shape (3,) with the camera translation.
+ image (torch.Tensor): Tensor of shape (3, H, W) containing the image crop with normalized pixel values.
+ full_frame (bool): If True, then render on the full image.
+ imgname (Optional[str]): Contains the original image filenamee. Used only if full_frame == True.
+ """
+
+ if full_frame:
+ image = cv2.imread(imgname).astype(np.float32)[:, :, ::-1] / 255.
+ else:
+ image = image.clone() * torch.tensor(self.cfg.MODEL.IMAGE_STD, device=image.device).reshape(3,1,1)
+ image = image + torch.tensor(self.cfg.MODEL.IMAGE_MEAN, device=image.device).reshape(3,1,1)
+ image = image.permute(1, 2, 0).cpu().numpy()
+
+ renderer = pyrender.OffscreenRenderer(viewport_width=image.shape[1],
+ viewport_height=image.shape[0],
+ point_size=1.0)
+ material = pyrender.MetallicRoughnessMaterial(
+ metallicFactor=0.0,
+ alphaMode='OPAQUE',
+ baseColorFactor=(*mesh_base_color, 1.0))
+
+ camera_translation[0] *= -1.
+
+ mesh = trimesh.Trimesh(vertices.copy(), self.faces.copy())
+ if side_view:
+ rot = trimesh.transformations.rotation_matrix(
+ np.radians(rot_angle), [0, 1, 0])
+ mesh.apply_transform(rot)
+ rot = trimesh.transformations.rotation_matrix(
+ np.radians(180), [1, 0, 0])
+ mesh.apply_transform(rot)
+ mesh = pyrender.Mesh.from_trimesh(mesh, material=material)
+
+ scene = pyrender.Scene(bg_color=[*scene_bg_color, 0.0],
+ ambient_light=(0.3, 0.3, 0.3))
+ scene.add(mesh, 'mesh')
+
+ camera_pose = np.eye(4)
+ camera_pose[:3, 3] = camera_translation
+ camera_center = [image.shape[1] / 2., image.shape[0] / 2.]
+ camera = pyrender.IntrinsicsCamera(fx=self.focal_length, fy=self.focal_length,
+ cx=camera_center[0], cy=camera_center[1], zfar=1e12)
+ scene.add(camera, pose=camera_pose)
+
+
+ light_nodes = create_raymond_lights()
+ for node in light_nodes:
+ scene.add_node(node)
+
+ color, rend_depth = renderer.render(scene, flags=pyrender.RenderFlags.RGBA)
+ color = color.astype(np.float32) / 255.0
+ renderer.delete()
+
+ if return_rgba:
+ return color
+
+ valid_mask = (color[:, :, -1])[:, :, np.newaxis]
+ if not side_view:
+ output_img = (color[:, :, :3] * valid_mask + (1 - valid_mask) * image)
+ else:
+ output_img = color[:, :, :3]
+
+ output_img = output_img.astype(np.float32)
+ return output_img
+
+ def vertices_to_trimesh(self, vertices, camera_translation, mesh_base_color=(1.0, 1.0, 0.9),
+ rot_axis=[1,0,0], rot_angle=0, is_right=1):
+ # material = pyrender.MetallicRoughnessMaterial(
+ # metallicFactor=0.0,
+ # alphaMode='OPAQUE',
+ # baseColorFactor=(*mesh_base_color, 1.0))
+ vertex_colors = np.array([(*mesh_base_color, 1.0)] * vertices.shape[0])
+ if is_right:
+ mesh = trimesh.Trimesh(vertices.copy() + camera_translation, self.faces.copy(), vertex_colors=vertex_colors)
+ else:
+ mesh = trimesh.Trimesh(vertices.copy() + camera_translation, self.faces_left.copy(), vertex_colors=vertex_colors)
+ # mesh = trimesh.Trimesh(vertices.copy(), self.faces.copy())
+
+ rot = trimesh.transformations.rotation_matrix(
+ np.radians(rot_angle), rot_axis)
+ mesh.apply_transform(rot)
+
+ rot = trimesh.transformations.rotation_matrix(
+ np.radians(180), [1, 0, 0])
+ mesh.apply_transform(rot)
+ return mesh
+
+ def render_rgba(
+ self,
+ vertices: np.array,
+ cam_t = None,
+ rot=None,
+ rot_axis=[1,0,0],
+ rot_angle=0,
+ camera_z=3,
+ # camera_translation: np.array,
+ mesh_base_color=(1.0, 1.0, 0.9),
+ scene_bg_color=(0,0,0),
+ render_res=[256, 256],
+ focal_length=None,
+ is_right=None,
+ ):
+
+ renderer = pyrender.OffscreenRenderer(viewport_width=render_res[0],
+ viewport_height=render_res[1],
+ point_size=1.0)
+ # material = pyrender.MetallicRoughnessMaterial(
+ # metallicFactor=0.0,
+ # alphaMode='OPAQUE',
+ # baseColorFactor=(*mesh_base_color, 1.0))
+
+ focal_length = focal_length if focal_length is not None else self.focal_length
+
+ if cam_t is not None:
+ camera_translation = cam_t.copy()
+ camera_translation[0] *= -1.
+ else:
+ camera_translation = np.array([0, 0, camera_z * focal_length/render_res[1]])
+
+ mesh = self.vertices_to_trimesh(vertices, np.array([0, 0, 0]), mesh_base_color, rot_axis, rot_angle, is_right=is_right)
+ mesh = pyrender.Mesh.from_trimesh(mesh)
+ # mesh = pyrender.Mesh.from_trimesh(mesh, material=material)
+
+ scene = pyrender.Scene(bg_color=[*scene_bg_color, 0.0],
+ ambient_light=(0.3, 0.3, 0.3))
+ scene.add(mesh, 'mesh')
+
+ camera_pose = np.eye(4)
+ camera_pose[:3, 3] = camera_translation
+ camera_center = [render_res[0] / 2., render_res[1] / 2.]
+ camera = pyrender.IntrinsicsCamera(fx=focal_length, fy=focal_length,
+ cx=camera_center[0], cy=camera_center[1], zfar=1e12)
+
+ # Create camera node and add it to pyRender scene
+ camera_node = pyrender.Node(camera=camera, matrix=camera_pose)
+ scene.add_node(camera_node)
+ self.add_point_lighting(scene, camera_node)
+ self.add_lighting(scene, camera_node)
+
+ light_nodes = create_raymond_lights()
+ for node in light_nodes:
+ scene.add_node(node)
+
+ color, rend_depth = renderer.render(scene, flags=pyrender.RenderFlags.RGBA)
+ color = color.astype(np.float32) / 255.0
+ renderer.delete()
+
+ return color
+
+ def render_rgba_multiple(
+ self,
+ vertices: List[np.array],
+ cam_t: List[np.array],
+ rot_axis=[1,0,0],
+ rot_angle=0,
+ mesh_base_color=(1.0, 1.0, 0.9),
+ scene_bg_color=(0,0,0),
+ render_res=[256, 256],
+ focal_length=None,
+ is_right=None,
+ ):
+
+ renderer = pyrender.OffscreenRenderer(viewport_width=render_res[0],
+ viewport_height=render_res[1],
+ point_size=1.0)
+ # material = pyrender.MetallicRoughnessMaterial(
+ # metallicFactor=0.0,
+ # alphaMode='OPAQUE',
+ # baseColorFactor=(*mesh_base_color, 1.0))
+
+ if is_right is None:
+ is_right = [1 for _ in range(len(vertices))]
+
+ mesh_list = [pyrender.Mesh.from_trimesh(self.vertices_to_trimesh(vvv, ttt.copy(), mesh_base_color, rot_axis, rot_angle, is_right=sss)) for vvv,ttt,sss in zip(vertices, cam_t, is_right)]
+
+ scene = pyrender.Scene(bg_color=[*scene_bg_color, 0.0],
+ ambient_light=(0.3, 0.3, 0.3))
+ for i,mesh in enumerate(mesh_list):
+ scene.add(mesh, f'mesh_{i}')
+
+ camera_pose = np.eye(4)
+ # camera_pose[:3, 3] = camera_translation
+ camera_center = [render_res[0] / 2., render_res[1] / 2.]
+ focal_length = focal_length if focal_length is not None else self.focal_length
+ camera = pyrender.IntrinsicsCamera(fx=focal_length, fy=focal_length,
+ cx=camera_center[0], cy=camera_center[1], zfar=1e12)
+
+ # Create camera node and add it to pyRender scene
+ camera_node = pyrender.Node(camera=camera, matrix=camera_pose)
+ scene.add_node(camera_node)
+ self.add_point_lighting(scene, camera_node)
+ self.add_lighting(scene, camera_node)
+
+ light_nodes = create_raymond_lights()
+ for node in light_nodes:
+ scene.add_node(node)
+
+ color, rend_depth = renderer.render(scene, flags=pyrender.RenderFlags.RGBA)
+ color = color.astype(np.float32) / 255.0
+ renderer.delete()
+
+ return color
+
+ def add_lighting(self, scene, cam_node, color=np.ones(3), intensity=1.0):
+ # from phalp.visualize.py_renderer import get_light_poses
+ light_poses = get_light_poses()
+ light_poses.append(np.eye(4))
+ cam_pose = scene.get_pose(cam_node)
+ for i, pose in enumerate(light_poses):
+ matrix = cam_pose @ pose
+ node = pyrender.Node(
+ name=f"light-{i:02d}",
+ light=pyrender.DirectionalLight(color=color, intensity=intensity),
+ matrix=matrix,
+ )
+ if scene.has_node(node):
+ continue
+ scene.add_node(node)
+
+ def add_point_lighting(self, scene, cam_node, color=np.ones(3), intensity=1.0):
+ # from phalp.visualize.py_renderer import get_light_poses
+ light_poses = get_light_poses(dist=0.5)
+ light_poses.append(np.eye(4))
+ cam_pose = scene.get_pose(cam_node)
+ for i, pose in enumerate(light_poses):
+ matrix = cam_pose @ pose
+ # node = pyrender.Node(
+ # name=f"light-{i:02d}",
+ # light=pyrender.DirectionalLight(color=color, intensity=intensity),
+ # matrix=matrix,
+ # )
+ node = pyrender.Node(
+ name=f"plight-{i:02d}",
+ light=pyrender.PointLight(color=color, intensity=intensity),
+ matrix=matrix,
+ )
+ if scene.has_node(node):
+ continue
+ scene.add_node(node)
diff --git a/third_party/hamer/hamer/utils/rich_utils.py b/third_party/hamer/hamer/utils/rich_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..19f97494ed2958ec2c3d75c772360b5367f2dc7b
--- /dev/null
+++ b/third_party/hamer/hamer/utils/rich_utils.py
@@ -0,0 +1,105 @@
+from pathlib import Path
+from typing import Sequence
+
+import rich
+import rich.syntax
+import rich.tree
+from hydra.core.hydra_config import HydraConfig
+from omegaconf import DictConfig, OmegaConf, open_dict
+from pytorch_lightning.utilities import rank_zero_only
+from rich.prompt import Prompt
+
+from . import pylogger
+
+log = pylogger.get_pylogger(__name__)
+
+
+@rank_zero_only
+def print_config_tree(
+ cfg: DictConfig,
+ print_order: Sequence[str] = (
+ "datamodule",
+ "model",
+ "callbacks",
+ "logger",
+ "trainer",
+ "paths",
+ "extras",
+ ),
+ resolve: bool = False,
+ save_to_file: bool = False,
+) -> None:
+ """Prints content of DictConfig using Rich library and its tree structure.
+
+ Args:
+ cfg (DictConfig): Configuration composed by Hydra.
+ print_order (Sequence[str], optional): Determines in what order config components are printed.
+ resolve (bool, optional): Whether to resolve reference fields of DictConfig.
+ save_to_file (bool, optional): Whether to export config to the hydra output folder.
+ """
+
+ style = "dim"
+ tree = rich.tree.Tree("CONFIG", style=style, guide_style=style)
+
+ queue = []
+
+ # add fields from `print_order` to queue
+ for field in print_order:
+ queue.append(field) if field in cfg else log.warning(
+ f"Field '{field}' not found in config. Skipping '{field}' config printing..."
+ )
+
+ # add all the other fields to queue (not specified in `print_order`)
+ for field in cfg:
+ if field not in queue:
+ queue.append(field)
+
+ # generate config tree from queue
+ for field in queue:
+ branch = tree.add(field, style=style, guide_style=style)
+
+ config_group = cfg[field]
+ if isinstance(config_group, DictConfig):
+ branch_content = OmegaConf.to_yaml(config_group, resolve=resolve)
+ else:
+ branch_content = str(config_group)
+
+ branch.add(rich.syntax.Syntax(branch_content, "yaml"))
+
+ # print config tree
+ rich.print(tree)
+
+ # save config tree to file
+ if save_to_file:
+ with open(Path(cfg.paths.output_dir, "config_tree.log"), "w") as file:
+ rich.print(tree, file=file)
+
+
+@rank_zero_only
+def enforce_tags(cfg: DictConfig, save_to_file: bool = False) -> None:
+ """Prompts user to input tags from command line if no tags are provided in config."""
+
+ if not cfg.get("tags"):
+ if "id" in HydraConfig().cfg.hydra.job:
+ raise ValueError("Specify tags before launching a multirun!")
+
+ log.warning("No tags provided in config. Prompting user to input tags...")
+ tags = Prompt.ask("Enter a list of comma separated tags", default="dev")
+ tags = [t.strip() for t in tags.split(",") if t != ""]
+
+ with open_dict(cfg):
+ cfg.tags = tags
+
+ log.info(f"Tags: {cfg.tags}")
+
+ if save_to_file:
+ with open(Path(cfg.paths.output_dir, "tags.log"), "w") as file:
+ rich.print(cfg.tags, file=file)
+
+
+if __name__ == "__main__":
+ from hydra import compose, initialize
+
+ with initialize(version_base="1.2", config_path="../../configs"):
+ cfg = compose(config_name="train.yaml", return_hydra_config=False, overrides=[])
+ print_config_tree(cfg, resolve=False, save_to_file=False)
diff --git a/third_party/hamer/hamer/utils/skeleton_renderer.py b/third_party/hamer/hamer/utils/skeleton_renderer.py
new file mode 100644
index 0000000000000000000000000000000000000000..46a5df75bff887eab00984eeb5be3c1f6e752960
--- /dev/null
+++ b/third_party/hamer/hamer/utils/skeleton_renderer.py
@@ -0,0 +1,124 @@
+import torch
+import numpy as np
+import trimesh
+from typing import Optional
+from yacs.config import CfgNode
+
+from .geometry import perspective_projection
+from .render_openpose import render_openpose
+
+class SkeletonRenderer:
+
+ def __init__(self, cfg: CfgNode):
+ """
+ Object used to render 3D keypoints. Faster for use during training.
+ Args:
+ cfg (CfgNode): Model config file.
+ """
+ self.cfg = cfg
+
+ def __call__(self,
+ pred_keypoints_3d: torch.Tensor,
+ gt_keypoints_3d: torch.Tensor,
+ gt_keypoints_2d: torch.Tensor,
+ images: Optional[np.array] = None,
+ camera_translation: Optional[torch.Tensor] = None) -> np.array:
+ """
+ Render batch of 3D keypoints.
+ Args:
+ pred_keypoints_3d (torch.Tensor): Tensor of shape (B, S, N, 3) containing a batch of predicted 3D keypoints, with S samples per image.
+ gt_keypoints_3d (torch.Tensor): Tensor of shape (B, N, 4) containing corresponding ground truth 3D keypoints; last value is the confidence.
+ gt_keypoints_2d (torch.Tensor): Tensor of shape (B, N, 3) containing corresponding ground truth 2D keypoints.
+ images (torch.Tensor): Tensor of shape (B, H, W, 3) containing images with values in the [0,255] range.
+ camera_translation (torch.Tensor): Tensor of shape (B, 3) containing the camera translation.
+ Returns:
+ np.array : Image with the following layout. Each row contains the a) input image,
+ b) image with gt 2D keypoints,
+ c) image with projected gt 3D keypoints,
+ d_1, ... , d_S) image with projected predicted 3D keypoints,
+ e) gt 3D keypoints rendered from a side view,
+ f_1, ... , f_S) predicted 3D keypoints frorm a side view
+ """
+ batch_size = pred_keypoints_3d.shape[0]
+# num_samples = pred_keypoints_3d.shape[1]
+ pred_keypoints_3d = pred_keypoints_3d.clone().cpu().float()
+ gt_keypoints_3d = gt_keypoints_3d.clone().cpu().float()
+ gt_keypoints_3d[:, :, :-1] = gt_keypoints_3d[:, :, :-1] - gt_keypoints_3d[:, [0], :-1] + pred_keypoints_3d[:, [0]]
+ gt_keypoints_2d = gt_keypoints_2d.clone().cpu().float().numpy()
+ gt_keypoints_2d[:, :, :-1] = self.cfg.MODEL.IMAGE_SIZE * (gt_keypoints_2d[:, :, :-1] + 1.0) / 2.0
+
+ #openpose_indices = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14]
+ #gt_indices = [12, 8, 7, 6, 9, 10, 11, 14, 2, 1, 0, 3, 4, 5]
+ #gt_indices = [25 + i for i in gt_indices]
+ openpose_indices = [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20]
+ gt_indices = openpose_indices
+ keypoints_to_render = torch.ones(batch_size, gt_keypoints_3d.shape[1], 1)
+ rotation = torch.eye(3).unsqueeze(0)
+ if camera_translation is None:
+ camera_translation = torch.tensor([0.0, 0.0, 2 * self.cfg.EXTRA.FOCAL_LENGTH / (0.8 * self.cfg.MODEL.IMAGE_SIZE)]).unsqueeze(0).repeat(batch_size, 1)
+ else:
+ camera_translation = camera_translation.cpu()
+
+ if images is None:
+ images = np.zeros((batch_size, self.cfg.MODEL.IMAGE_SIZE, self.cfg.MODEL.IMAGE_SIZE, 3))
+ focal_length = torch.tensor([self.cfg.EXTRA.FOCAL_LENGTH, self.cfg.EXTRA.FOCAL_LENGTH]).reshape(1, 2)
+ camera_center = torch.tensor([self.cfg.MODEL.IMAGE_SIZE, self.cfg.MODEL.IMAGE_SIZE], dtype=torch.float).reshape(1, 2) / 2.
+ gt_keypoints_3d_proj = perspective_projection(gt_keypoints_3d[:, :, :-1], rotation=rotation.repeat(batch_size, 1, 1), translation=camera_translation[:, :], focal_length=focal_length.repeat(batch_size, 1), camera_center=camera_center.repeat(batch_size, 1))
+ pred_keypoints_3d_proj = perspective_projection(pred_keypoints_3d.reshape(batch_size, -1, 3), rotation=rotation.repeat(batch_size, 1, 1), translation=camera_translation.reshape(batch_size, -1), focal_length=focal_length.repeat(batch_size, 1), camera_center=camera_center.repeat(batch_size, 1)).reshape(batch_size, -1, 2)
+ gt_keypoints_3d_proj = torch.cat([gt_keypoints_3d_proj, gt_keypoints_3d[:, :, [-1]]], dim=-1).cpu().numpy()
+ pred_keypoints_3d_proj = torch.cat([pred_keypoints_3d_proj, keypoints_to_render.reshape(batch_size, -1, 1)], dim=-1).cpu().numpy()
+ rows = []
+ # Rotate keypoints to visualize side view
+ R = torch.tensor(trimesh.transformations.rotation_matrix(np.radians(90), [0, 1, 0])[:3, :3]).float()
+ gt_keypoints_3d_side = gt_keypoints_3d.clone()
+ gt_keypoints_3d_side[:, :, :-1] = torch.einsum('bni,ij->bnj', gt_keypoints_3d_side[:, :, :-1], R)
+ pred_keypoints_3d_side = pred_keypoints_3d.clone()
+ pred_keypoints_3d_side = torch.einsum('bni,ij->bnj', pred_keypoints_3d_side, R)
+ gt_keypoints_3d_proj_side = perspective_projection(gt_keypoints_3d_side[:, :, :-1], rotation=rotation.repeat(batch_size, 1, 1), translation=camera_translation[:, :], focal_length=focal_length.repeat(batch_size, 1), camera_center=camera_center.repeat(batch_size, 1))
+ pred_keypoints_3d_proj_side = perspective_projection(pred_keypoints_3d_side.reshape(batch_size, -1, 3), rotation=rotation.repeat(batch_size, 1, 1), translation=camera_translation.reshape(batch_size, -1), focal_length=focal_length.repeat(batch_size, 1), camera_center=camera_center.repeat(batch_size, 1)).reshape(batch_size, -1, 2)
+ gt_keypoints_3d_proj_side = torch.cat([gt_keypoints_3d_proj_side, gt_keypoints_3d_side[:, :, [-1]]], dim=-1).cpu().numpy()
+ pred_keypoints_3d_proj_side = torch.cat([pred_keypoints_3d_proj_side, keypoints_to_render.reshape(batch_size, -1, 1)], dim=-1).cpu().numpy()
+ for i in range(batch_size):
+ img = images[i]
+ side_img = np.zeros((self.cfg.MODEL.IMAGE_SIZE, self.cfg.MODEL.IMAGE_SIZE, 3))
+ # gt 2D keypoints
+ body_keypoints_2d = gt_keypoints_2d[i, :21].copy()
+ for op, gt in zip(openpose_indices, gt_indices):
+ if gt_keypoints_2d[i, gt, -1] > body_keypoints_2d[op, -1]:
+ body_keypoints_2d[op] = gt_keypoints_2d[i, gt]
+ gt_keypoints_img = render_openpose(img, body_keypoints_2d) / 255.
+ # gt 3D keypoints
+ body_keypoints_3d_proj = gt_keypoints_3d_proj[i, :21].copy()
+ for op, gt in zip(openpose_indices, gt_indices):
+ if gt_keypoints_3d_proj[i, gt, -1] > body_keypoints_3d_proj[op, -1]:
+ body_keypoints_3d_proj[op] = gt_keypoints_3d_proj[i, gt]
+ gt_keypoints_3d_proj_img = render_openpose(img, body_keypoints_3d_proj) / 255.
+ # gt 3D keypoints from the side
+ body_keypoints_3d_proj = gt_keypoints_3d_proj_side[i, :21].copy()
+ for op, gt in zip(openpose_indices, gt_indices):
+ if gt_keypoints_3d_proj_side[i, gt, -1] > body_keypoints_3d_proj[op, -1]:
+ body_keypoints_3d_proj[op] = gt_keypoints_3d_proj_side[i, gt]
+ gt_keypoints_3d_proj_img_side = render_openpose(side_img, body_keypoints_3d_proj) / 255.
+ # pred 3D keypoints
+ pred_keypoints_3d_proj_imgs = []
+ body_keypoints_3d_proj = pred_keypoints_3d_proj[i, :21].copy()
+ for op, gt in zip(openpose_indices, gt_indices):
+ if pred_keypoints_3d_proj[i, gt, -1] >= body_keypoints_3d_proj[op, -1]:
+ body_keypoints_3d_proj[op] = pred_keypoints_3d_proj[i, gt]
+ pred_keypoints_3d_proj_imgs.append(render_openpose(img, body_keypoints_3d_proj) / 255.)
+ pred_keypoints_3d_proj_img = np.concatenate(pred_keypoints_3d_proj_imgs, axis=1)
+ # gt 3D keypoints from the side
+ pred_keypoints_3d_proj_imgs_side = []
+ body_keypoints_3d_proj = pred_keypoints_3d_proj_side[i, :21].copy()
+ for op, gt in zip(openpose_indices, gt_indices):
+ if pred_keypoints_3d_proj_side[i, gt, -1] >= body_keypoints_3d_proj[op, -1]:
+ body_keypoints_3d_proj[op] = pred_keypoints_3d_proj_side[i, gt]
+ pred_keypoints_3d_proj_imgs_side.append(render_openpose(side_img, body_keypoints_3d_proj) / 255.)
+ pred_keypoints_3d_proj_img_side = np.concatenate(pred_keypoints_3d_proj_imgs_side, axis=1)
+ rows.append(np.concatenate((gt_keypoints_img, gt_keypoints_3d_proj_img, pred_keypoints_3d_proj_img, gt_keypoints_3d_proj_img_side, pred_keypoints_3d_proj_img_side), axis=1))
+ # Concatenate images
+ img = np.concatenate(rows, axis=0)
+ img[:, ::self.cfg.MODEL.IMAGE_SIZE, :] = 1.0
+ img[::self.cfg.MODEL.IMAGE_SIZE, :, :] = 1.0
+ img[:, (1+1+1)*self.cfg.MODEL.IMAGE_SIZE, :] = 0.5
+ return img
diff --git a/third_party/hamer/hamer/utils/utils_detectron2.py b/third_party/hamer/hamer/utils/utils_detectron2.py
new file mode 100644
index 0000000000000000000000000000000000000000..fe01e02f8edbcbd5d545c6f3cb65aeb688a1dff4
--- /dev/null
+++ b/third_party/hamer/hamer/utils/utils_detectron2.py
@@ -0,0 +1,93 @@
+import detectron2.data.transforms as T
+import torch
+from detectron2.checkpoint import DetectionCheckpointer
+from detectron2.config import CfgNode, instantiate
+from detectron2.data import MetadataCatalog
+from omegaconf import OmegaConf
+
+
+class DefaultPredictor_Lazy:
+ """Create a simple end-to-end predictor with the given config that runs on single device for a
+ single input image.
+
+ Compared to using the model directly, this class does the following additions:
+
+ 1. Load checkpoint from the weights specified in config (cfg.MODEL.WEIGHTS).
+ 2. Always take BGR image as the input and apply format conversion internally.
+ 3. Apply resizing defined by the config (`cfg.INPUT.{MIN,MAX}_SIZE_TEST`).
+ 4. Take one input image and produce a single output, instead of a batch.
+
+ This is meant for simple demo purposes, so it does the above steps automatically.
+ This is not meant for benchmarks or running complicated inference logic.
+ If you'd like to do anything more complicated, please refer to its source code as
+ examples to build and use the model manually.
+
+ Attributes:
+ metadata (Metadata): the metadata of the underlying dataset, obtained from
+ test dataset name in the config.
+
+
+ Examples:
+ ::
+ pred = DefaultPredictor(cfg)
+ inputs = cv2.imread("input.jpg")
+ outputs = pred(inputs)
+ """
+
+ def __init__(self, cfg):
+ """
+ Args:
+ cfg: a yacs CfgNode or a omegaconf dict object.
+ """
+ if isinstance(cfg, CfgNode):
+ self.cfg = cfg.clone() # cfg can be modified by model
+ self.model = build_model(self.cfg) # noqa: F821
+ if len(cfg.DATASETS.TEST):
+ test_dataset = cfg.DATASETS.TEST[0]
+
+ checkpointer = DetectionCheckpointer(self.model)
+ checkpointer.load(cfg.MODEL.WEIGHTS)
+
+ self.aug = T.ResizeShortestEdge(
+ [cfg.INPUT.MIN_SIZE_TEST, cfg.INPUT.MIN_SIZE_TEST], cfg.INPUT.MAX_SIZE_TEST
+ )
+
+ self.input_format = cfg.INPUT.FORMAT
+ else: # new LazyConfig
+ self.cfg = cfg
+ self.model = instantiate(cfg.model)
+ test_dataset = OmegaConf.select(cfg, "dataloader.test.dataset.names", default=None)
+ if isinstance(test_dataset, (list, tuple)):
+ test_dataset = test_dataset[0]
+
+ checkpointer = DetectionCheckpointer(self.model)
+ checkpointer.load(OmegaConf.select(cfg, "train.init_checkpoint", default=""))
+
+ mapper = instantiate(cfg.dataloader.test.mapper)
+ self.aug = mapper.augmentations
+ self.input_format = mapper.image_format
+
+ self.model.eval().cuda()
+ if test_dataset:
+ self.metadata = MetadataCatalog.get(test_dataset)
+ assert self.input_format in ["RGB", "BGR"], self.input_format
+
+ def __call__(self, original_image):
+ """
+ Args:
+ original_image (np.ndarray): an image of shape (H, W, C) (in BGR order).
+
+ Returns:
+ predictions (dict):
+ the output of the model for one image only.
+ See :doc:`/tutorials/models` for details about the format.
+ """
+ with torch.no_grad():
+ if self.input_format == "RGB":
+ original_image = original_image[:, :, ::-1]
+ height, width = original_image.shape[:2]
+ image = self.aug(T.AugInput(original_image)).apply_image(original_image)
+ image = torch.as_tensor(image.astype("float32").transpose(2, 0, 1))
+ inputs = {"image": image, "height": height, "width": width}
+ predictions = self.model([inputs])[0]
+ return predictions
diff --git a/third_party/hamer/setup.py b/third_party/hamer/setup.py
new file mode 100644
index 0000000000000000000000000000000000000000..59b1cc571c9b22a2b7fea30a1dab3efc42f05197
--- /dev/null
+++ b/third_party/hamer/setup.py
@@ -0,0 +1,37 @@
+from setuptools import setup, find_packages
+
+print('Found packages:', find_packages())
+setup(
+ description='HaMeR as a package',
+ name='hamer',
+ packages=find_packages(),
+ install_requires=[
+ 'gdown',
+ 'numpy',
+ 'opencv-python',
+ 'pyrender',
+ 'pytorch-lightning',
+ 'scikit-image',
+ 'smplx==0.1.28',
+ 'torch',
+ 'torchvision',
+ 'yacs',
+ 'detectron2 @ git+https://github.com/facebookresearch/detectron2',
+ 'chumpy @ git+https://github.com/mattloper/chumpy',
+ 'mmcv==1.3.9',
+ 'timm',
+ 'einops',
+ 'xtcocotools',
+ 'pandas',
+ ],
+ extras_require={
+ 'all': [
+ 'hydra-core',
+ 'hydra-submitit-launcher',
+ 'hydra-colorlog',
+ 'pyrootutils',
+ 'rich',
+ 'webdataset',
+ ],
+ },
+)
diff --git a/third_party/hamer/train.py b/third_party/hamer/train.py
new file mode 100644
index 0000000000000000000000000000000000000000..329e40bb19f3fed4ba42fd9fff1abaefa22ff287
--- /dev/null
+++ b/third_party/hamer/train.py
@@ -0,0 +1,113 @@
+from typing import Optional, Tuple
+import pyrootutils
+
+root = pyrootutils.setup_root(
+ search_from=__file__,
+ indicator=[".git", "pyproject.toml"],
+ pythonpath=True,
+ dotenv=True,
+)
+
+import os
+from pathlib import Path
+
+import hydra
+import pytorch_lightning as pl
+from omegaconf import DictConfig, OmegaConf
+from pytorch_lightning import Trainer
+from pytorch_lightning.loggers import TensorBoardLogger
+from pytorch_lightning.plugins.environments import SLURMEnvironment
+#from pytorch_lightning.trainingtype import DDPPlugin
+
+from yacs.config import CfgNode
+from hamer.configs import dataset_config
+from hamer.datasets import HAMERDataModule
+from hamer.models.hamer import HAMER
+from hamer.utils.pylogger import get_pylogger
+from hamer.utils.misc import task_wrapper, log_hyperparameters
+
+# HACK reset the signal handling so the lightning is free to set it
+# Based on https://github.com/facebookincubator/submitit/issues/1709#issuecomment-1246758283
+import signal
+signal.signal(signal.SIGUSR1, signal.SIG_DFL)
+
+log = get_pylogger(__name__)
+
+
+@pl.utilities.rank_zero.rank_zero_only
+def save_configs(model_cfg: CfgNode, dataset_cfg: CfgNode, rootdir: str):
+ """Save config files to rootdir."""
+ Path(rootdir).mkdir(parents=True, exist_ok=True)
+ OmegaConf.save(config=model_cfg, f=os.path.join(rootdir, 'model_config.yaml'))
+ with open(os.path.join(rootdir, 'dataset_config.yaml'), 'w') as f:
+ f.write(dataset_cfg.dump())
+
+@task_wrapper
+def train(cfg: DictConfig) -> Tuple[dict, dict]:
+
+ # Load dataset config
+ dataset_cfg = dataset_config()
+
+ # Save configs
+ save_configs(cfg, dataset_cfg, cfg.paths.output_dir)
+
+ # Setup training and validation datasets
+ datamodule = HAMERDataModule(cfg, dataset_cfg)
+
+ # Setup model
+ model = HAMER(cfg)
+
+ # Setup Tensorboard logger
+ logger = TensorBoardLogger(os.path.join(cfg.paths.output_dir, 'tensorboard'), name='', version='', default_hp_metric=False)
+ loggers = [logger]
+
+ # Setup checkpoint saving
+ checkpoint_callback = pl.callbacks.ModelCheckpoint(
+ dirpath=os.path.join(cfg.paths.output_dir, 'checkpoints'),
+ every_n_train_steps=cfg.GENERAL.CHECKPOINT_STEPS,
+ save_last=True,
+ save_top_k=cfg.GENERAL.CHECKPOINT_SAVE_TOP_K,
+ )
+ rich_callback = pl.callbacks.RichProgressBar()
+ lr_monitor = pl.callbacks.LearningRateMonitor(logging_interval='step')
+ callbacks = [
+ checkpoint_callback,
+ lr_monitor,
+ # rich_callback
+ ]
+
+ log.info(f"Instantiating trainer <{cfg.trainer._target_}>")
+ trainer: Trainer = hydra.utils.instantiate(
+ cfg.trainer,
+ callbacks=callbacks,
+ logger=loggers,
+ #plugins=(SLURMEnvironment(requeue_signal=signal.SIGUSR2) if (cfg.get('launcher',None) is not None) else DDPPlugin(find_unused_parameters=False)), # Submitit uses SIGUSR2
+ plugins=(SLURMEnvironment(requeue_signal=signal.SIGUSR2) if (cfg.get('launcher',None) is not None) else None), # Submitit uses SIGUSR2
+ )
+
+ object_dict = {
+ "cfg": cfg,
+ "datamodule": datamodule,
+ "model": model,
+ "callbacks": callbacks,
+ "logger": logger,
+ "trainer": trainer,
+ }
+
+ if logger:
+ log.info("Logging hyperparameters!")
+ log_hyperparameters(object_dict)
+
+ # Train the model
+ trainer.fit(model, datamodule=datamodule, ckpt_path='last')
+ log.info("Fitting done")
+
+
+@hydra.main(version_base="1.2", config_path=str(root/"hamer/configs_hydra"), config_name="train.yaml")
+def main(cfg: DictConfig) -> Optional[float]:
+ # train the model
+ train(cfg)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/third_party/hamer/vitpose_model.py b/third_party/hamer/vitpose_model.py
new file mode 100644
index 0000000000000000000000000000000000000000..861ba5c1e1388613c8620130b0279febc0251d8b
--- /dev/null
+++ b/third_party/hamer/vitpose_model.py
@@ -0,0 +1,87 @@
+from __future__ import annotations
+
+import os
+
+import numpy as np
+import torch
+import torch.nn as nn
+
+from mmpose.apis import inference_top_down_pose_model, init_pose_model, process_mmdet_results, vis_pose_result
+
+os.environ["PYOPENGL_PLATFORM"] = "egl"
+
+# project root directory
+ROOT_DIR = "./"
+VIT_DIR = os.path.join(ROOT_DIR, "third-party/ViTPose")
+
+class ViTPoseModel(object):
+ MODEL_DICT = {
+ 'ViTPose+-G (multi-task train, COCO)': {
+ 'config': f'{VIT_DIR}/configs/wholebody/2d_kpt_sview_rgb_img/topdown_heatmap/coco-wholebody/ViTPose_huge_wholebody_256x192.py',
+ 'model': f'{ROOT_DIR}/_DATA/vitpose_ckpts/vitpose+_huge/wholebody.pth',
+ },
+ }
+
+ def __init__(self, device: str | torch.device):
+ self.device = torch.device(device)
+ self.model_name = 'ViTPose+-G (multi-task train, COCO)'
+ self.model = self._load_model(self.model_name)
+
+ def _load_all_models_once(self) -> None:
+ for name in self.MODEL_DICT:
+ self._load_model(name)
+
+ def _load_model(self, name: str) -> nn.Module:
+ dic = self.MODEL_DICT[name]
+ ckpt_path = dic['model']
+ model = init_pose_model(dic['config'], ckpt_path, device=self.device)
+ return model
+
+ def set_model(self, name: str) -> None:
+ if name == self.model_name:
+ return
+ self.model_name = name
+ self.model = self._load_model(name)
+
+ def predict_pose_and_visualize(
+ self,
+ image: np.ndarray,
+ det_results: list[np.ndarray],
+ box_score_threshold: float,
+ kpt_score_threshold: float,
+ vis_dot_radius: int,
+ vis_line_thickness: int,
+ ) -> tuple[list[dict[str, np.ndarray]], np.ndarray]:
+ out = self.predict_pose(image, det_results, box_score_threshold)
+ vis = self.visualize_pose_results(image, out, kpt_score_threshold,
+ vis_dot_radius, vis_line_thickness)
+ return out, vis
+
+ def predict_pose(
+ self,
+ image: np.ndarray,
+ det_results: list[np.ndarray],
+ box_score_threshold: float = 0.5) -> list[dict[str, np.ndarray]]:
+ image = image[:, :, ::-1] # RGB -> BGR
+ person_results = process_mmdet_results(det_results, 1)
+ out, _ = inference_top_down_pose_model(self.model,
+ image,
+ person_results=person_results,
+ bbox_thr=box_score_threshold,
+ format='xyxy')
+ return out
+
+ def visualize_pose_results(self,
+ image: np.ndarray,
+ pose_results: list[np.ndarray],
+ kpt_score_threshold: float = 0.3,
+ vis_dot_radius: int = 4,
+ vis_line_thickness: int = 1) -> np.ndarray:
+ image = image[:, :, ::-1] # RGB -> BGR
+ vis = vis_pose_result(self.model,
+ image,
+ pose_results,
+ kpt_score_thr=kpt_score_threshold,
+ radius=vis_dot_radius,
+ thickness=vis_line_thickness)
+ return vis[:, :, ::-1] # BGR -> RGB
diff --git a/third_party/hamer/wget-log b/third_party/hamer/wget-log
new file mode 100644
index 0000000000000000000000000000000000000000..459bceac43da879373781dacf95eabed2111f12e
--- /dev/null
+++ b/third_party/hamer/wget-log
@@ -0,0 +1,11 @@
+--2026-01-09 09:56:21-- https://www.cs.utexas.edu/~pavlakos/hamer/data/hamer_demo_data.tar.gz
+Resolving www.cs.utexas.edu (www.cs.utexas.edu)... 128.83.139.191
+Connecting to www.cs.utexas.edu (www.cs.utexas.edu)|128.83.139.191|:443... connected.
+HTTP request sent, awaiting response... 200 OK
+Length: 6037554929 (5.6G) [application/x-gzip]
+Saving to: ‘hamer_demo_data.tar.gz’
+
+
hamer_demo_data.tar.gz 83%[====================================================================> ] 4.72G 14.5MB/s eta 78s
hamer_demo_data.tar.gz 84%[====================================================================> ] 4.72G 14.5MB/s eta 78s
hamer_demo_data.tar.gz 84%[====================================================================> ] 4.73G 14.6MB/s eta 78s
hamer_demo_data.tar.gz 84%[====================================================================> ] 4.73G 14.9MB/s eta 78s
hamer_demo_data.tar.gz 84%[====================================================================> ] 4.73G 15.1MB/s eta 77s
hamer_demo_data.tar.gz 84%[====================================================================> ] 4.74G 15.2MB/s eta 77s
hamer_demo_data.tar.gz 84%[====================================================================> ] 4.74G 15.3MB/s eta 77s
hamer_demo_data.tar.gz 84%[=====================================================================> ] 4.74G 15.3MB/s eta 77s
hamer_demo_data.tar.gz 84%[=====================================================================> ] 4.75G 15.6MB/s eta 77s
hamer_demo_data.tar.gz 84%[=====================================================================> ] 4.75G 15.6MB/s eta 75s
hamer_demo_data.tar.gz 84%[=====================================================================> ] 4.75G 15.8MB/s eta 75s
hamer_demo_data.tar.gz 84%[=====================================================================> ] 4.76G 16.2MB/s eta 75s
hamer_demo_data.tar.gz 84%[=====================================================================> ] 4.76G 16.4MB/s eta 75s
hamer_demo_data.tar.gz 84%[=====================================================================> ] 4.77G 16.5MB/s eta 75s
hamer_demo_data.tar.gz 84%[=====================================================================> ] 4.77G 16.7MB/s eta 74s
hamer_demo_data.tar.gz 84%[=====================================================================> ] 4.77G 16.9MB/s eta 74s
hamer_demo_data.tar.gz 84%[=====================================================================> ] 4.78G 17.0MB/s eta 74s
hamer_demo_data.tar.gz 85%[=====================================================================> ] 4.78G 17.1MB/s eta 74s
hamer_demo_data.tar.gz 85%[=====================================================================> ] 4.79G 17.5MB/s eta 74s
hamer_demo_data.tar.gz 85%[=====================================================================> ] 4.79G 17.9MB/s eta 72s
hamer_demo_data.tar.gz 85%[=====================================================================> ] 4.79G 18.3MB/s eta 72s
hamer_demo_data.tar.gz 85%[=====================================================================> ] 4.80G 18.5MB/s eta 72s
hamer_demo_data.tar.gz 85%[=====================================================================> ] 4.80G 18.4MB/s eta 72s
hamer_demo_data.tar.gz 85%[=====================================================================> ] 4.81G 18.1MB/s eta 72s
hamer_demo_data.tar.gz 85%[======================================================================> ] 4.81G 18.5MB/s eta 70s
hamer_demo_data.tar.gz 85%[======================================================================> ] 4.82G 18.2MB/s eta 70s
hamer_demo_data.tar.gz 85%[======================================================================> ] 4.82G 18.2MB/s eta 70s
hamer_demo_data.tar.gz 85%[======================================================================> ] 4.82G 18.4MB/s eta 70s
hamer_demo_data.tar.gz 85%[======================================================================> ] 4.82G 18.1MB/s eta 70s
hamer_demo_data.tar.gz 85%[======================================================================> ] 4.83G 18.1MB/s eta 68s
hamer_demo_data.tar.gz 85%[======================================================================> ] 4.83G 17.8MB/s eta 68s
hamer_demo_data.tar.gz 86%[======================================================================> ] 4.84G 17.8MB/s eta 68s
hamer_demo_data.tar.gz 86%[======================================================================> ] 4.84G 17.5MB/s eta 68s
hamer_demo_data.tar.gz 86%[======================================================================> ] 4.84G 17.6MB/s eta 68s
hamer_demo_data.tar.gz 86%[======================================================================> ] 4.85G 17.4MB/s eta 67s
hamer_demo_data.tar.gz 86%[======================================================================> ] 4.85G 17.3MB/s eta 67s
hamer_demo_data.tar.gz 86%[======================================================================> ] 4.85G 17.1MB/s eta 67s
hamer_demo_data.tar.gz 86%[======================================================================> ] 4.86G 17.1MB/s eta 67s
hamer_demo_data.tar.gz 86%[======================================================================> ] 4.86G 17.6MB/s eta 67s
hamer_demo_data.tar.gz 86%[======================================================================> ] 4.87G 17.6MB/s eta 65s
hamer_demo_data.tar.gz 86%[======================================================================> ] 4.87G 17.7MB/s eta 65s
hamer_demo_data.tar.gz 86%[======================================================================> ] 4.88G 17.8MB/s eta 65s
hamer_demo_data.tar.gz 86%[=======================================================================> ] 4.88G 17.9MB/s eta 65s
hamer_demo_data.tar.gz 86%[=======================================================================> ] 4.88G 18.3MB/s eta 65s
hamer_demo_data.tar.gz 86%[=======================================================================> ] 4.89G 18.3MB/s eta 63s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.89G 18.6MB/s eta 63s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.90G 18.8MB/s eta 63s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.90G 19.1MB/s eta 63s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.90G 19.3MB/s eta 63s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.91G 19.4MB/s eta 61s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.91G 19.4MB/s eta 61s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.92G 19.4MB/s eta 61s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.92G 19.6MB/s eta 61s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.92G 19.8MB/s eta 61s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.93G 19.8MB/s eta 59s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.93G 19.7MB/s eta 59s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.94G 20.2MB/s eta 59s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.94G 20.2MB/s eta 59s
hamer_demo_data.tar.gz 87%[=======================================================================> ] 4.94G 20.2MB/s eta 59s
hamer_demo_data.tar.gz 87%[========================================================================> ] 4.95G 19.9MB/s eta 57s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.95G 19.5MB/s eta 57s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.95G 19.1MB/s eta 57s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.96G 18.9MB/s eta 57s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.96G 18.5MB/s eta 57s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.96G 17.8MB/s eta 56s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.97G 17.6MB/s eta 56s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.97G 17.3MB/s eta 56s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.97G 16.5MB/s eta 56s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.97G 16.0MB/s eta 56s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.97G 15.3MB/s eta 55s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.98G 15.3MB/s eta 55s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.98G 15.0MB/s eta 55s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.98G 14.5MB/s eta 55s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.99G 13.9MB/s eta 55s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.99G 13.5MB/s eta 54s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.99G 13.3MB/s eta 54s
hamer_demo_data.tar.gz 88%[========================================================================> ] 4.99G 12.9MB/s eta 54s
hamer_demo_data.tar.gz 88%[========================================================================> ] 5.00G 12.8MB/s eta 54s
hamer_demo_data.tar.gz 88%[========================================================================> ] 5.00G 12.5MB/s eta 54s
hamer_demo_data.tar.gz 88%[========================================================================> ] 5.00G 12.4MB/s eta 53s
hamer_demo_data.tar.gz 89%[========================================================================> ] 5.00G 12.3MB/s eta 53s
hamer_demo_data.tar.gz 89%[========================================================================> ] 5.01G 12.7MB/s eta 53s
hamer_demo_data.tar.gz 89%[========================================================================> ] 5.01G 12.4MB/s eta 53s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.01G 12.6MB/s eta 53s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.02G 13.2MB/s eta 52s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.02G 12.8MB/s eta 52s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.02G 12.8MB/s eta 52s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.03G 12.9MB/s eta 52s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.03G 12.8MB/s eta 52s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.03G 13.2MB/s eta 50s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.03G 13.3MB/s eta 50s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.04G 13.3MB/s eta 50s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.04G 13.4MB/s eta 50s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.04G 13.4MB/s eta 50s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.04G 13.4MB/s eta 49s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.05G 13.4MB/s eta 49s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.05G 13.4MB/s eta 49s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.05G 13.7MB/s eta 49s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.06G 13.7MB/s eta 49s
hamer_demo_data.tar.gz 89%[=========================================================================> ] 5.06G 13.6MB/s eta 48s
hamer_demo_data.tar.gz 90%[=========================================================================> ] 5.06G 13.7MB/s eta 48s
hamer_demo_data.tar.gz 90%[=========================================================================> ] 5.07G 13.7MB/s eta 48s
hamer_demo_data.tar.gz 90%[=========================================================================> ] 5.07G 13.6MB/s eta 48s
hamer_demo_data.tar.gz 90%[=========================================================================> ] 5.07G 13.6MB/s eta 48s
hamer_demo_data.tar.gz 90%[=========================================================================> ] 5.07G 13.8MB/s eta 47s
hamer_demo_data.tar.gz 90%[=========================================================================> ] 5.08G 13.8MB/s eta 47s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.08G 13.8MB/s eta 47s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.08G 13.7MB/s eta 47s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.09G 13.8MB/s eta 47s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.09G 13.8MB/s eta 45s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.09G 13.6MB/s eta 45s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.09G 13.1MB/s eta 45s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.10G 13.3MB/s eta 45s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.10G 13.3MB/s eta 45s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.10G 13.0MB/s eta 44s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.10G 12.8MB/s eta 44s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.11G 12.6MB/s eta 44s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.11G 12.4MB/s eta 44s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.11G 12.2MB/s eta 44s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.11G 11.4MB/s eta 43s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.11G 11.6MB/s eta 43s
hamer_demo_data.tar.gz 90%[==========================================================================> ] 5.12G 11.2MB/s eta 43s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.12G 10.9MB/s eta 43s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.12G 10.6MB/s eta 43s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.12G 10.3MB/s eta 43s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.12G 9.94MB/s eta 43s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.12G 10.1MB/s eta 43s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.13G 9.53MB/s eta 43s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.13G 9.51MB/s eta 43s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.13G 9.34MB/s eta 42s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.13G 9.27MB/s eta 42s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.13G 9.16MB/s eta 42s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.14G 9.17MB/s eta 42s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.14G 8.83MB/s eta 42s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.14G 9.28MB/s eta 41s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.14G 8.78MB/s eta 41s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.14G 8.99MB/s eta 41s
hamer_demo_data.tar.gz 91%[==========================================================================> ] 5.15G 8.92MB/s eta 41s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.15G 9.13MB/s eta 41s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.15G 9.13MB/s eta 40s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.15G 9.25MB/s eta 40s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.15G 9.19MB/s eta 40s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.16G 9.35MB/s eta 40s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.16G 9.33MB/s eta 40s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.16G 9.38MB/s eta 39s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.16G 9.40MB/s eta 39s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.17G 9.61MB/s eta 39s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.17G 9.48MB/s eta 39s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.17G 9.53MB/s eta 39s
hamer_demo_data.tar.gz 91%[===========================================================================> ] 5.17G 9.50MB/s eta 39s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.17G 9.56MB/s eta 39s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.17G 9.55MB/s eta 39s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.18G 9.64MB/s eta 39s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.18G 9.61MB/s eta 39s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.18G 9.60MB/s eta 38s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.18G 9.56MB/s eta 38s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.19G 9.64MB/s eta 38s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.19G 9.53MB/s eta 38s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.19G 9.66MB/s eta 38s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.19G 9.65MB/s eta 37s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.19G 9.67MB/s eta 37s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.20G 9.65MB/s eta 37s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.20G 9.60MB/s eta 37s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.20G 9.57MB/s eta 37s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.20G 9.52MB/s eta 36s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.20G 9.58MB/s eta 36s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.21G 9.64MB/s eta 36s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.21G 9.56MB/s eta 36s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.21G 9.70MB/s eta 36s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.21G 9.55MB/s eta 35s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.21G 9.63MB/s eta 35s
hamer_demo_data.tar.gz 92%[===========================================================================> ] 5.22G 9.61MB/s eta 35s
hamer_demo_data.tar.gz 92%[============================================================================> ] 5.22G 9.62MB/s eta 35s
hamer_demo_data.tar.gz 92%[============================================================================> ] 5.22G 9.53MB/s eta 35s
hamer_demo_data.tar.gz 92%[============================================================================> ] 5.22G 9.66MB/s eta 34s
hamer_demo_data.tar.gz 92%[============================================================================> ] 5.22G 9.72MB/s eta 34s
hamer_demo_data.tar.gz 92%[============================================================================> ] 5.23G 9.76MB/s eta 34s
hamer_demo_data.tar.gz 92%[============================================================================> ] 5.23G 9.79MB/s eta 34s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.23G 9.75MB/s eta 34s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.23G 9.73MB/s eta 33s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.24G 9.74MB/s eta 33s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.24G 9.95MB/s eta 33s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.24G 9.91MB/s eta 33s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.24G 9.98MB/s eta 33s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.24G 9.96MB/s eta 32s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.25G 10.1MB/s eta 32s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.25G 10.0MB/s eta 32s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.25G 10.2MB/s eta 32s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.25G 10.2MB/s eta 32s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.25G 10.3MB/s eta 31s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.26G 10.4MB/s eta 31s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.26G 10.6MB/s eta 31s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.26G 10.5MB/s eta 31s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.26G 10.6MB/s eta 31s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.27G 10.7MB/s eta 30s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.27G 10.8MB/s eta 30s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.27G 10.9MB/s eta 30s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.28G 11.2MB/s eta 30s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.28G 11.3MB/s eta 30s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.28G 11.4MB/s eta 29s
hamer_demo_data.tar.gz 93%[============================================================================> ] 5.28G 11.5MB/s eta 29s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.29G 11.8MB/s eta 29s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.29G 11.8MB/s eta 29s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.29G 12.1MB/s eta 29s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.29G 12.2MB/s eta 28s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.30G 12.5MB/s eta 28s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.30G 12.5MB/s eta 28s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.30G 12.7MB/s eta 28s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.31G 13.0MB/s eta 28s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.31G 13.2MB/s eta 27s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.31G 13.5MB/s eta 27s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.31G 11.8MB/s eta 27s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.32G 12.5MB/s eta 26s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.32G 11.7MB/s eta 26s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.32G 11.1MB/s eta 26s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.33G 11.0MB/s eta 26s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.33G 10.7MB/s eta 25s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.33G 10.6MB/s eta 25s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.33G 10.2MB/s eta 25s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.33G 10.1MB/s eta 25s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.33G 9.80MB/s eta 25s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.34G 9.50MB/s eta 24s
hamer_demo_data.tar.gz 94%[=============================================================================> ] 5.34G 9.10MB/s eta 24s
hamer_demo_data.tar.gz 95%[=============================================================================> ] 5.34G 8.86MB/s eta 24s
hamer_demo_data.tar.gz 95%[=============================================================================> ] 5.34G 8.46MB/s eta 24s
hamer_demo_data.tar.gz 95%[=============================================================================> ] 5.35G 8.39MB/s eta 24s
hamer_demo_data.tar.gz 95%[=============================================================================> ] 5.35G 8.10MB/s eta 24s
hamer_demo_data.tar.gz 95%[=============================================================================> ] 5.35G 8.80MB/s eta 24s
hamer_demo_data.tar.gz 95%[=============================================================================> ] 5.35G 8.66MB/s eta 24s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.35G 9.36MB/s eta 24s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.36G 8.87MB/s eta 24s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.36G 9.39MB/s eta 23s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.36G 9.21MB/s eta 23s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.36G 9.30MB/s eta 23s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.36G 9.26MB/s eta 23s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.37G 9.65MB/s eta 23s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.37G 9.40MB/s eta 22s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.37G 9.62MB/s eta 22s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.37G 9.52MB/s eta 22s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.38G 9.69MB/s eta 22s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.38G 9.60MB/s eta 22s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.38G 9.97MB/s eta 21s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.38G 9.69MB/s eta 21s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.38G 9.91MB/s eta 21s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.38G 9.76MB/s eta 21s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.39G 9.88MB/s eta 21s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.39G 9.77MB/s eta 20s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.39G 10.1MB/s eta 20s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.39G 9.85MB/s eta 20s
hamer_demo_data.tar.gz 95%[==============================================================================> ] 5.40G 10.0MB/s eta 20s
hamer_demo_data.tar.gz 96%[==============================================================================> ] 5.40G 9.85MB/s eta 20s
hamer_demo_data.tar.gz 96%[==============================================================================> ] 5.40G 10.0MB/s eta 19s
hamer_demo_data.tar.gz 96%[==============================================================================> ] 5.40G 9.85MB/s eta 19s
hamer_demo_data.tar.gz 96%[==============================================================================> ] 5.41G 10.2MB/s eta 19s
hamer_demo_data.tar.gz 96%[==============================================================================> ] 5.41G 9.96MB/s eta 19s
hamer_demo_data.tar.gz 96%[==============================================================================> ] 5.41G 10.0MB/s eta 19s
hamer_demo_data.tar.gz 96%[==============================================================================> ] 5.41G 9.88MB/s eta 18s
hamer_demo_data.tar.gz 96%[==============================================================================> ] 5.41G 10.0MB/s eta 18s
hamer_demo_data.tar.gz 96%[==============================================================================> ] 5.42G 9.79MB/s eta 18s
hamer_demo_data.tar.gz 96%[==============================================================================> ] 5.42G 10.3MB/s eta 18s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.42G 10.0MB/s eta 18s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.42G 10.2MB/s eta 17s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.42G 9.87MB/s eta 17s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.43G 10.1MB/s eta 17s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.43G 9.85MB/s eta 17s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.43G 10.3MB/s eta 17s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.43G 10.0MB/s eta 16s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.44G 10.2MB/s eta 16s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.44G 9.91MB/s eta 16s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.44G 10.1MB/s eta 16s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.44G 9.90MB/s eta 16s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.44G 10.4MB/s eta 15s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.45G 10.1MB/s eta 15s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.45G 10.3MB/s eta 15s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.45G 9.99MB/s eta 15s
hamer_demo_data.tar.gz 96%[===============================================================================> ] 5.45G 10.2MB/s eta 15s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.45G 10.1MB/s eta 14s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.46G 10.5MB/s eta 14s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.46G 10.3MB/s eta 14s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.46G 10.4MB/s eta 14s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.46G 10.2MB/s eta 14s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.47G 10.5MB/s eta 13s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.47G 10.5MB/s eta 13s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.47G 10.7MB/s eta 13s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.47G 10.5MB/s eta 13s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.48G 10.8MB/s eta 13s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.48G 10.6MB/s eta 12s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.48G 10.9MB/s eta 12s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.48G 11.0MB/s eta 12s
hamer_demo_data.tar.gz 97%[===============================================================================> ] 5.49G 11.3MB/s eta 12s
hamer_demo_data.tar.gz 97%[================================================================================> ] 5.49G 11.0MB/s eta 12s
hamer_demo_data.tar.gz 97%[================================================================================> ] 5.49G 11.4MB/s eta 11s
hamer_demo_data.tar.gz 97%[================================================================================> ] 5.49G 11.2MB/s eta 11s
hamer_demo_data.tar.gz 97%[================================================================================> ] 5.50G 11.6MB/s eta 11s
hamer_demo_data.tar.gz 97%[================================================================================> ] 5.50G 11.7MB/s eta 11s
hamer_demo_data.tar.gz 97%[================================================================================> ] 5.50G 12.0MB/s eta 11s
hamer_demo_data.tar.gz 97%[================================================================================> ] 5.50G 11.8MB/s eta 10s
hamer_demo_data.tar.gz 97%[================================================================================> ] 5.51G 12.2MB/s eta 10s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.51G 12.1MB/s eta 10s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.51G 12.5MB/s eta 10s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.52G 12.7MB/s eta 10s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.52G 13.1MB/s eta 9s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.52G 13.0MB/s eta 9s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.53G 13.4MB/s eta 9s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.53G 13.2MB/s eta 9s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.53G 13.8MB/s eta 9s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.54G 14.0MB/s eta 7s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.54G 14.6MB/s eta 7s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.54G 14.5MB/s eta 7s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.55G 14.7MB/s eta 7s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.55G 15.0MB/s eta 7s
hamer_demo_data.tar.gz 98%[================================================================================> ] 5.55G 15.1MB/s eta 6s
hamer_demo_data.tar.gz 98%[=================================================================================> ] 5.56G 15.4MB/s eta 6s
hamer_demo_data.tar.gz 98%[=================================================================================> ] 5.56G 15.7MB/s eta 6s
hamer_demo_data.tar.gz 98%[=================================================================================> ] 5.57G 16.1MB/s eta 6s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.57G 17.0MB/s eta 6s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.57G 17.0MB/s eta 4s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.58G 16.7MB/s eta 4s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.58G 16.9MB/s eta 4s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.59G 16.8MB/s eta 4s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.59G 16.7MB/s eta 3s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.60G 16.6MB/s eta 3s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.60G 16.5MB/s eta 3s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.60G 16.8MB/s eta 3s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.61G 16.7MB/s eta 1s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.61G 16.5MB/s eta 1s
hamer_demo_data.tar.gz 99%[=================================================================================> ] 5.62G 16.5MB/s eta 1s
hamer_demo_data.tar.gz 100%[==================================================================================>] 5.62G 16.5MB/s in 8m 1s
+
+2026-01-09 10:04:23 (12.0 MB/s) - ‘hamer_demo_data.tar.gz’ saved [6037554929/6037554929]
+
diff --git a/train.log b/train.log
new file mode 100644
index 0000000000000000000000000000000000000000..0d9a43a6e4012ecc7730faacbcd1da8d47d71e7a
--- /dev/null
+++ b/train.log
@@ -0,0 +1,8695 @@
+[12/28 07:27:29][INFO] [Exp Name]: finetune_
+[12/28 07:27:29][INFO] [GPU x Batch] = 1 x 32
+[12/28 07:27:38][INFO] [Exp Name]: finetune_
+[12/28 07:27:38][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:25:12][INFO] [Exp Name]: finetune_
+[12/28 18:25:12][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:26:34][INFO] [Exp Name]: finetune_
+[12/28 18:26:34][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:26:41][INFO] [Exp Name]: finetune_
+[12/28 18:26:41][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:29:59][INFO] [Exp Name]: finetune_
+[12/28 18:29:59][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:30:02][INFO] [AMASS] Loading from inputs/AMASS/hmr4d_support/smplxpose_v2.pth ...
+[12/28 18:32:12][INFO] [Exp Name]: finetune_
+[12/28 18:32:12][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:32:14][INFO] [AMASS] Loading from inputs/AMASS/hmr4d_support/smplxpose_v2.pth ...
+[12/28 18:35:21][INFO] [Exp Name]: finetune_
+[12/28 18:35:21][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:35:23][INFO] [AMASS] Loading from inputs/AMASS/hmr4d_support/smplxpose_v2.pth ...
+[12/28 18:35:23][WARNING] [Train Dataset] Skipping amass_train_v11 due to error: Error in call to target 'genmo.datasets.pure_motion.amass.AmassDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.train.amass_train_v11
+[12/28 18:35:24][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 18:35:24][WARNING] [Train Dataset] Skipping humanml3d_static_train due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.train.humanml3d_static_train
+[12/28 18:35:25][INFO] [BEDLAM] Loading from inputs/BEDLAM/hmr4d_support
+[12/28 18:35:25][WARNING] [Train Dataset] Skipping bedlam_v2 due to error: Error in call to target 'genmo.datasets.bedlam.bedlam.BedlamDatasetV2':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.train.bedlam_v2
+[12/28 18:35:25][INFO] [H36M] Loading from inputs/H36M/hmr4d_support/smplxpose_v1.pt ...
+[12/28 18:35:25][WARNING] [Train Dataset] Skipping h36m_v1 due to error: Error in call to target 'genmo.datasets.h36m.h36m.H36mSmplDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.train.h36m_v1
+[12/28 18:35:25][WARNING] [Train Dataset] Skipping 3dpw_v1 due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_train.ThreedpwSmplDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.train.3dpw_v1
+[12/28 18:35:25][WARNING] [Train Dataset] Skipping 3dpw_occ_v1 due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_train.ThreedpwOccSmplDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.train.3dpw_occ_v1
+[12/28 18:35:25][WARNING] [Train Dataset] Skipping aistpp_train due to error: Error locating target 'genmo.datasets.aistplusplus.aistplusplus.AISTPlusPlusSmplDataset', set env var HYDRA_FULL_ERROR=1 to see chained exception.
+full_key: dataset_opts.train.aistpp_train
+[12/28 18:35:25][WARNING] [Train Dataset] Skipping beat2_static_train due to error: Error locating target 'genmo.datasets.beat2.beat2.BEAT2SmplDataset', set env var HYDRA_FULL_ERROR=1 to see chained exception.
+full_key: dataset_opts.train.beat2_static_train
+[12/28 18:35:25][WARNING] [Train Dataset] Skipping unity due to error: Error locating target 'genmo.datasets.unity_dataset.UnityDataset', set env var HYDRA_FULL_ERROR=1 to see chained exception.
+full_key: dataset_opts.train.unity
+[12/28 18:38:07][INFO] [Exp Name]: finetune_
+[12/28 18:38:07][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:38:07][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:38:07][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:38:07][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 18:38:07][INFO]
+[12/28 18:38:10][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 18:38:10][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 18:38:10][INFO] [EMDB] Full sequence, split=1
+[12/28 18:38:10][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 18:38:10][INFO] [EMDB] Full sequence, split=2
+[12/28 18:38:10][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 18:38:10][INFO] [RICH] Full sequence, Test
+[12/28 18:38:10][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 18:38:10][INFO] [3DPW] Full sequence
+[12/28 18:38:10][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 18:38:10][INFO] [3DPW_OCC] Full sequence
+[12/28 18:38:10][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 18:38:10][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:38:10][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:38:10][INFO]
+[12/28 18:38:14][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 18:38:37][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_0/checkpoints'
+[12/28 18:40:27][INFO] [Exp Name]: finetune_
+[12/28 18:40:27][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:40:27][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:40:27][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:40:27][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 18:40:27][INFO]
+[12/28 18:40:30][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 18:40:30][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 18:40:30][INFO] [EMDB] Full sequence, split=1
+[12/28 18:40:30][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 18:40:30][INFO] [EMDB] Full sequence, split=2
+[12/28 18:40:30][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 18:40:30][INFO] [RICH] Full sequence, Test
+[12/28 18:40:30][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 18:40:30][INFO] [3DPW] Full sequence
+[12/28 18:40:30][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 18:40:30][INFO] [3DPW_OCC] Full sequence
+[12/28 18:40:30][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 18:40:30][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:40:30][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:40:30][INFO]
+[12/28 18:40:34][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 18:40:58][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_1/checkpoints'
+[12/28 18:46:06][INFO] [Exp Name]: finetune_
+[12/28 18:46:06][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:46:06][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:46:06][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:46:06][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 18:46:06][INFO]
+[12/28 18:46:09][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 18:46:09][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 18:46:09][INFO] [EMDB] Full sequence, split=1
+[12/28 18:46:09][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 18:46:09][INFO] [EMDB] Full sequence, split=2
+[12/28 18:46:09][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 18:46:09][INFO] [RICH] Full sequence, Test
+[12/28 18:46:09][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 18:46:09][INFO] [3DPW] Full sequence
+[12/28 18:46:09][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 18:46:09][INFO] [3DPW_OCC] Full sequence
+[12/28 18:46:09][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 18:46:09][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:46:09][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:46:09][INFO]
+[12/28 18:46:13][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 18:46:26][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_2/checkpoints'
+[12/28 18:47:12][INFO] [Exp Name]: finetune_
+[12/28 18:47:12][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:47:12][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:47:12][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:47:12][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 18:47:12][INFO]
+[12/28 18:47:15][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 18:47:15][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 18:47:15][INFO] [EMDB] Full sequence, split=1
+[12/28 18:47:15][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 18:47:15][INFO] [EMDB] Full sequence, split=2
+[12/28 18:47:15][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 18:47:15][INFO] [RICH] Full sequence, Test
+[12/28 18:47:15][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 18:47:15][INFO] [3DPW] Full sequence
+[12/28 18:47:15][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 18:47:15][INFO] [3DPW_OCC] Full sequence
+[12/28 18:47:15][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 18:47:15][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:47:15][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:47:15][INFO]
+[12/28 18:47:19][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 18:47:25][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_3/checkpoints'
+[12/28 18:49:40][INFO] [Exp Name]: finetune_
+[12/28 18:49:40][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:49:40][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:49:40][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:49:40][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 18:49:40][INFO]
+[12/28 18:49:42][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 18:49:42][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 18:49:42][INFO] [EMDB] Full sequence, split=1
+[12/28 18:49:42][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 18:49:42][INFO] [EMDB] Full sequence, split=2
+[12/28 18:49:42][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 18:49:42][INFO] [RICH] Full sequence, Test
+[12/28 18:49:42][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 18:49:42][INFO] [3DPW] Full sequence
+[12/28 18:49:42][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 18:49:42][INFO] [3DPW_OCC] Full sequence
+[12/28 18:49:42][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 18:49:42][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:49:42][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:49:42][INFO]
+[12/28 18:49:47][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 18:49:53][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_4/checkpoints'
+[12/28 18:55:06][INFO] [Exp Name]: finetune_
+[12/28 18:55:06][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:55:06][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:55:06][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:55:06][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 18:55:06][INFO]
+[12/28 18:55:09][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 18:55:09][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 18:55:09][INFO] [EMDB] Full sequence, split=1
+[12/28 18:55:09][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 18:55:09][INFO] [EMDB] Full sequence, split=2
+[12/28 18:55:09][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 18:55:09][INFO] [RICH] Full sequence, Test
+[12/28 18:55:09][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 18:55:09][INFO] [3DPW] Full sequence
+[12/28 18:55:09][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 18:55:09][INFO] [3DPW_OCC] Full sequence
+[12/28 18:55:09][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 18:55:09][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:55:09][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:55:09][INFO]
+[12/28 18:55:13][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 18:55:27][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_5/checkpoints'
+[12/28 18:58:14][INFO] [Exp Name]: finetune_
+[12/28 18:58:14][INFO] [GPU x Batch] = 1 x 32
+[12/28 18:58:14][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:58:14][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:58:14][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 18:58:14][INFO]
+[12/28 18:58:16][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 18:58:16][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 18:58:16][INFO] [EMDB] Full sequence, split=1
+[12/28 18:58:16][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 18:58:16][INFO] [EMDB] Full sequence, split=2
+[12/28 18:58:16][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 18:58:16][INFO] [RICH] Full sequence, Test
+[12/28 18:58:16][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 18:58:16][INFO] [3DPW] Full sequence
+[12/28 18:58:16][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 18:58:16][INFO] [3DPW_OCC] Full sequence
+[12/28 18:58:16][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 18:58:16][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 18:58:16][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 18:58:16][INFO]
+[12/28 18:58:21][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 18:58:30][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_6/checkpoints'
+[12/28 19:00:20][INFO] [Exp Name]: finetune_
+[12/28 19:00:20][INFO] [GPU x Batch] = 1 x 32
+[12/28 19:00:20][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:00:20][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:00:20][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 19:00:20][INFO]
+[12/28 19:00:23][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:00:23][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:00:23][INFO] [EMDB] Full sequence, split=1
+[12/28 19:00:23][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:00:23][INFO] [EMDB] Full sequence, split=2
+[12/28 19:00:23][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:00:23][INFO] [RICH] Full sequence, Test
+[12/28 19:00:23][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:00:23][INFO] [3DPW] Full sequence
+[12/28 19:00:23][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:00:23][INFO] [3DPW_OCC] Full sequence
+[12/28 19:00:23][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:00:23][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:00:23][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:00:23][INFO]
+[12/28 19:00:28][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:00:41][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_7/checkpoints'
+[12/28 19:05:28][INFO] [Exp Name]: finetune_
+[12/28 19:05:28][INFO] [GPU x Batch] = 1 x 32
+[12/28 19:05:28][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:05:28][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:05:28][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 19:05:28][INFO]
+[12/28 19:05:31][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:05:31][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:05:31][INFO] [EMDB] Full sequence, split=1
+[12/28 19:05:31][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:05:31][INFO] [EMDB] Full sequence, split=2
+[12/28 19:05:31][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:05:31][INFO] [RICH] Full sequence, Test
+[12/28 19:05:31][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:05:31][INFO] [3DPW] Full sequence
+[12/28 19:05:31][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:05:31][INFO] [3DPW_OCC] Full sequence
+[12/28 19:05:31][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:05:31][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:05:31][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:05:31][INFO]
+[12/28 19:05:35][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:05:53][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_8/checkpoints'
+[12/28 19:07:04][INFO] [Exp Name]: finetune_
+[12/28 19:07:04][INFO] [GPU x Batch] = 1 x 32
+[12/28 19:07:04][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:07:04][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:07:04][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 19:07:04][INFO]
+[12/28 19:07:07][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:07:07][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:07:08][INFO] [EMDB] Full sequence, split=1
+[12/28 19:07:08][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:07:08][INFO] [EMDB] Full sequence, split=2
+[12/28 19:07:08][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:07:08][INFO] [RICH] Full sequence, Test
+[12/28 19:07:08][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:07:08][INFO] [3DPW] Full sequence
+[12/28 19:07:08][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:07:08][INFO] [3DPW_OCC] Full sequence
+[12/28 19:07:08][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:07:08][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:07:08][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:07:08][INFO]
+[12/28 19:07:13][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:07:28][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_9/checkpoints'
+[12/28 19:11:11][INFO] [Exp Name]: finetune_
+[12/28 19:11:11][INFO] [GPU x Batch] = 1 x 32
+[12/28 19:11:11][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:11:11][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:11:11][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 19:11:11][INFO]
+[12/28 19:11:14][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:11:14][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:11:14][INFO] [EMDB] Full sequence, split=1
+[12/28 19:11:14][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:11:14][INFO] [EMDB] Full sequence, split=2
+[12/28 19:11:14][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:11:14][INFO] [RICH] Full sequence, Test
+[12/28 19:11:14][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:11:14][INFO] [3DPW] Full sequence
+[12/28 19:11:14][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:11:14][INFO] [3DPW_OCC] Full sequence
+[12/28 19:11:14][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:11:14][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:11:14][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:11:14][INFO]
+[12/28 19:11:20][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:11:31][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_10/checkpoints'
+[12/28 19:13:44][INFO] [Exp Name]: finetune_
+[12/28 19:13:44][INFO] [GPU x Batch] = 1 x 32
+[12/28 19:13:44][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:13:44][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:13:44][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 19:13:44][INFO]
+[12/28 19:13:47][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:13:47][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:13:47][INFO] [EMDB] Full sequence, split=1
+[12/28 19:13:47][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:13:47][INFO] [EMDB] Full sequence, split=2
+[12/28 19:13:47][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:13:47][INFO] [RICH] Full sequence, Test
+[12/28 19:13:47][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:13:47][INFO] [3DPW] Full sequence
+[12/28 19:13:47][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:13:47][INFO] [3DPW_OCC] Full sequence
+[12/28 19:13:47][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:13:47][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:13:47][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:13:47][INFO]
+[12/28 19:13:52][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:14:11][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_11/checkpoints'
+[12/28 19:16:44][INFO] [Exp Name]: finetune_
+[12/28 19:16:44][INFO] [GPU x Batch] = 1 x 32
+[12/28 19:16:44][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:16:44][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:16:44][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 19:16:44][INFO]
+[12/28 19:16:46][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:16:46][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:16:46][INFO] [EMDB] Full sequence, split=1
+[12/28 19:16:46][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:16:46][INFO] [EMDB] Full sequence, split=2
+[12/28 19:16:46][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:16:46][INFO] [RICH] Full sequence, Test
+[12/28 19:16:46][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:16:46][INFO] [3DPW] Full sequence
+[12/28 19:16:46][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:16:46][INFO] [3DPW_OCC] Full sequence
+[12/28 19:16:46][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:16:46][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:16:46][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:16:46][INFO]
+[12/28 19:16:50][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:17:07][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_12/checkpoints'
+[12/28 19:18:17][INFO] [Exp Name]: finetune_
+[12/28 19:18:17][INFO] [GPU x Batch] = 1 x 32
+[12/28 19:18:17][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:18:17][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:18:17][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 19:18:17][INFO]
+[12/28 19:18:20][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:18:20][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:18:20][INFO] [EMDB] Full sequence, split=1
+[12/28 19:18:20][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:18:20][INFO] [EMDB] Full sequence, split=2
+[12/28 19:18:20][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:18:20][INFO] [RICH] Full sequence, Test
+[12/28 19:18:20][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:18:20][INFO] [3DPW] Full sequence
+[12/28 19:18:20][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:18:20][INFO] [3DPW_OCC] Full sequence
+[12/28 19:18:20][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:18:20][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:18:20][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:18:20][INFO]
+[12/28 19:18:24][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:18:41][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_13/checkpoints'
+[12/28 19:19:09][INFO] Start Fitting...
+[12/28 19:19:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 19:19:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/data.py:106: Total length of `DataLoader` across ranks is zero. Please make sure this was your intention.
+
+[12/28 19:19:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/data.py:106: Total length of `CombinedLoader` across ranks is zero. Please make sure this was your intention.
+
+[12/28 19:19:21][INFO] End of script.
+[12/28 19:20:37][INFO] [Exp Name]: finetune_
+[12/28 19:20:37][INFO] [GPU x Batch] = 1 x 1
+[12/28 19:20:37][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:20:37][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:20:37][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 19:20:37][INFO]
+[12/28 19:20:41][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:20:41][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:20:41][INFO] [EMDB] Full sequence, split=1
+[12/28 19:20:41][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:20:41][INFO] [EMDB] Full sequence, split=2
+[12/28 19:20:41][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:20:41][INFO] [RICH] Full sequence, Test
+[12/28 19:20:41][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:20:41][INFO] [3DPW] Full sequence
+[12/28 19:20:41][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:20:41][INFO] [3DPW_OCC] Full sequence
+[12/28 19:20:41][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:20:41][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:20:41][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:20:41][INFO]
+[12/28 19:20:46][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:20:59][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_14/checkpoints'
+[12/28 19:21:27][INFO] Start Fitting...
+[12/28 19:21:50][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 19:23:35][INFO] [Exp Name]: finetune_
+[12/28 19:23:35][INFO] [GPU x Batch] = 1 x 1
+[12/28 19:23:35][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:23:35][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:23:35][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 19:23:35][INFO]
+[12/28 19:23:38][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:23:38][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:23:38][INFO] [EMDB] Full sequence, split=1
+[12/28 19:23:38][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:23:38][INFO] [EMDB] Full sequence, split=2
+[12/28 19:23:38][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:23:38][INFO] [RICH] Full sequence, Test
+[12/28 19:23:38][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:23:38][INFO] [3DPW] Full sequence
+[12/28 19:23:38][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:23:38][INFO] [3DPW_OCC] Full sequence
+[12/28 19:23:38][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:23:38][INFO] [UnityDataset] Initialized with root=/root/miko/puni/train/GVHMR/processed_dataset, split=train
+[12/28 19:23:38][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:23:38][INFO]
+[12/28 19:23:43][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:23:59][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_15/checkpoints'
+[12/28 19:24:26][INFO] Start Fitting...
+[12/28 19:24:38][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 19:24:38][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 19:30:56][INFO] [Exp Name]: finetune_
+[12/28 19:30:56][INFO] [GPU x Batch] = 1 x 1
+[12/28 19:30:56][INFO] [UnityDataset] Initialized with 1 sequences from root=/root/miko/puni/train/GVHMR/processed_dataset/gvhmr, split=train
+[12/28 19:30:56][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:30:56][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/28 19:30:56][INFO]
+[12/28 19:30:59][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:30:59][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:30:59][INFO] [EMDB] Full sequence, split=1
+[12/28 19:30:59][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:30:59][INFO] [EMDB] Full sequence, split=2
+[12/28 19:30:59][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:30:59][INFO] [RICH] Full sequence, Test
+[12/28 19:30:59][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:30:59][INFO] [3DPW] Full sequence
+[12/28 19:30:59][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:30:59][INFO] [3DPW_OCC] Full sequence
+[12/28 19:30:59][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:30:59][INFO] [UnityDataset] Initialized with 1 sequences from root=/root/miko/puni/train/GVHMR/processed_dataset/gvhmr, split=train
+[12/28 19:30:59][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:30:59][INFO]
+[12/28 19:31:04][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:31:24][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_16/checkpoints'
+[12/28 19:31:51][INFO] Start Fitting...
+[12/28 19:32:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 19:32:00][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 19:43:06][INFO] [Exp Name]: finetune_
+[12/28 19:43:06][INFO] [GPU x Batch] = 1 x 1
+[12/28 19:43:06][INFO] [UnityDataset] Found 5 sequences.
+[12/28 19:43:06][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:43:06][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 19:43:06][INFO]
+[12/28 19:43:09][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:43:09][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:43:09][INFO] [EMDB] Full sequence, split=1
+[12/28 19:43:09][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:43:09][INFO] [EMDB] Full sequence, split=2
+[12/28 19:43:09][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:43:09][INFO] [RICH] Full sequence, Test
+[12/28 19:43:09][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:43:09][INFO] [3DPW] Full sequence
+[12/28 19:43:09][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:43:09][INFO] [3DPW_OCC] Full sequence
+[12/28 19:43:09][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:43:09][INFO] [UnityDataset] Found 5 sequences.
+[12/28 19:43:09][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:43:09][INFO]
+[12/28 19:43:15][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:43:35][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_17/checkpoints'
+[12/28 19:43:58][INFO] Start Fitting...
+[12/28 19:44:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 19:44:11][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 19:53:21][INFO] [Exp Name]: finetune_
+[12/28 19:53:21][INFO] [GPU x Batch] = 1 x 1
+[12/28 19:53:21][INFO] [UnityDataset] Found 5 sequences.
+[12/28 19:53:21][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:53:21][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 19:53:21][INFO]
+[12/28 19:53:24][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:53:24][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:53:24][INFO] [EMDB] Full sequence, split=1
+[12/28 19:53:24][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:53:24][INFO] [EMDB] Full sequence, split=2
+[12/28 19:53:24][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:53:24][INFO] [RICH] Full sequence, Test
+[12/28 19:53:24][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:53:24][INFO] [3DPW] Full sequence
+[12/28 19:53:24][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:53:24][INFO] [3DPW_OCC] Full sequence
+[12/28 19:53:24][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:53:24][INFO] [UnityDataset] Found 5 sequences.
+[12/28 19:53:24][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:53:24][INFO]
+[12/28 19:53:31][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:53:49][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_18/checkpoints'
+[12/28 19:54:14][INFO] Start Fitting...
+[12/28 19:54:25][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 19:54:25][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 19:54:47][INFO] [Exp Name]: finetune_
+[12/28 19:54:47][INFO] [GPU x Batch] = 1 x 1
+[12/28 19:54:48][INFO] [UnityDataset] Found 5 sequences.
+[12/28 19:54:48][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:54:48][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 19:54:48][INFO]
+[12/28 19:54:50][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:54:50][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:54:50][INFO] [EMDB] Full sequence, split=1
+[12/28 19:54:50][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:54:50][INFO] [EMDB] Full sequence, split=2
+[12/28 19:54:50][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:54:50][INFO] [RICH] Full sequence, Test
+[12/28 19:54:50][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:54:50][INFO] [3DPW] Full sequence
+[12/28 19:54:50][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:54:50][INFO] [3DPW_OCC] Full sequence
+[12/28 19:54:50][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:54:50][INFO] [UnityDataset] Found 5 sequences.
+[12/28 19:54:50][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:54:50][INFO]
+[12/28 19:54:55][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:55:15][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_19/checkpoints'
+[12/28 19:55:39][INFO] Start Fitting...
+[12/28 19:55:51][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 19:55:51][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 19:58:54][INFO] [Exp Name]: finetune_
+[12/28 19:58:54][INFO] [GPU x Batch] = 1 x 1
+[12/28 19:58:54][INFO] [UnityDataset] Found 5 sequences.
+[12/28 19:58:54][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:58:54][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 19:58:54][INFO]
+[12/28 19:58:57][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 19:58:57][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 19:58:57][INFO] [EMDB] Full sequence, split=1
+[12/28 19:58:57][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 19:58:57][INFO] [EMDB] Full sequence, split=2
+[12/28 19:58:57][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 19:58:57][INFO] [RICH] Full sequence, Test
+[12/28 19:58:57][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 19:58:57][INFO] [3DPW] Full sequence
+[12/28 19:58:57][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 19:58:57][INFO] [3DPW_OCC] Full sequence
+[12/28 19:58:57][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 19:58:57][INFO] [UnityDataset] Found 5 sequences.
+[12/28 19:58:57][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 19:58:57][INFO]
+[12/28 19:59:02][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 19:59:20][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_20/checkpoints'
+[12/28 19:59:43][INFO] Start Fitting...
+[12/28 19:59:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 19:59:55][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 20:00:16][INFO] [Exp Name]: finetune_
+[12/28 20:00:16][INFO] [GPU x Batch] = 1 x 1
+[12/28 20:00:16][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:00:16][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:00:16][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 20:00:16][INFO]
+[12/28 20:00:19][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 20:00:19][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 20:00:19][INFO] [EMDB] Full sequence, split=1
+[12/28 20:00:19][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 20:00:19][INFO] [EMDB] Full sequence, split=2
+[12/28 20:00:19][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 20:00:19][INFO] [RICH] Full sequence, Test
+[12/28 20:00:19][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 20:00:19][INFO] [3DPW] Full sequence
+[12/28 20:00:19][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 20:00:19][INFO] [3DPW_OCC] Full sequence
+[12/28 20:00:19][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 20:00:19][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:00:19][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:00:19][INFO]
+[12/28 20:00:24][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 20:00:40][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_21/checkpoints'
+[12/28 20:01:09][INFO] Start Fitting...
+[12/28 20:01:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 20:01:21][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 20:04:52][INFO] [Exp Name]: finetune_
+[12/28 20:04:52][INFO] [GPU x Batch] = 1 x 1
+[12/28 20:04:52][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:04:52][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:04:52][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 20:04:52][INFO]
+[12/28 20:04:55][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 20:04:55][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 20:04:55][INFO] [EMDB] Full sequence, split=1
+[12/28 20:04:55][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 20:04:55][INFO] [EMDB] Full sequence, split=2
+[12/28 20:04:55][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 20:04:55][INFO] [RICH] Full sequence, Test
+[12/28 20:04:55][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 20:04:55][INFO] [3DPW] Full sequence
+[12/28 20:04:55][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 20:04:55][INFO] [3DPW_OCC] Full sequence
+[12/28 20:04:55][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 20:04:55][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:04:55][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:04:55][INFO]
+[12/28 20:05:00][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 20:05:15][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_22/checkpoints'
+[12/28 20:05:38][INFO] Start Fitting...
+[12/28 20:05:51][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 20:05:51][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 20:07:48][INFO] [Exp Name]: finetune_
+[12/28 20:07:48][INFO] [GPU x Batch] = 1 x 1
+[12/28 20:07:48][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:07:48][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:07:48][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 20:07:48][INFO]
+[12/28 20:07:51][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 20:07:51][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 20:07:51][INFO] [EMDB] Full sequence, split=1
+[12/28 20:07:51][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 20:07:51][INFO] [EMDB] Full sequence, split=2
+[12/28 20:07:51][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 20:07:51][INFO] [RICH] Full sequence, Test
+[12/28 20:07:51][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 20:07:51][INFO] [3DPW] Full sequence
+[12/28 20:07:51][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 20:07:51][INFO] [3DPW_OCC] Full sequence
+[12/28 20:07:51][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 20:07:51][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:07:51][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:07:51][INFO]
+[12/28 20:07:56][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 20:08:15][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_23/checkpoints'
+[12/28 20:08:40][INFO] Start Fitting...
+[12/28 20:08:50][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 20:08:50][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 20:08:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 20:08:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 20:08:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 20:08:58][INFO] ✅[FIT][Epoch 0] finished! 00:08→14:29 | loss_epoch=132
+[12/28 20:08:58][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[12/28 20:09:02][INFO] ✅[FIT][Epoch 1] finished! 00:12→10:31 | loss_epoch=125
+[12/28 20:09:02][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[12/28 20:09:07][INFO] ✅[FIT][Epoch 2] finished! 00:17→09:34 | loss_epoch=445
+[12/28 20:09:07][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[12/28 20:09:14][INFO] ✅[FIT][Epoch 3] finished! 00:25→10:09 | loss_epoch=52.7
+[12/28 20:09:14][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[12/28 20:09:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 20:28:58][INFO] [Exp Name]: finetune_
+[12/28 20:28:58][INFO] [GPU x Batch] = 1 x 1
+[12/28 20:28:58][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:28:58][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:28:58][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 20:28:58][INFO]
+[12/28 20:29:02][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 20:29:02][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 20:29:02][INFO] [EMDB] Full sequence, split=1
+[12/28 20:29:02][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 20:29:02][INFO] [EMDB] Full sequence, split=2
+[12/28 20:29:02][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 20:29:02][INFO] [RICH] Full sequence, Test
+[12/28 20:29:02][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 20:29:02][INFO] [3DPW] Full sequence
+[12/28 20:29:02][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 20:29:02][INFO] [3DPW_OCC] Full sequence
+[12/28 20:29:02][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 20:29:02][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:29:02][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:29:02][INFO]
+[12/28 20:29:07][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 20:29:32][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_24/checkpoints'
+[12/28 20:29:56][INFO] Start Fitting...
+[12/28 20:30:07][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 20:30:07][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 20:30:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 20:30:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 20:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 20:30:14][INFO] ✅[FIT][Epoch 0] finished! 00:07→12:32 | loss_epoch=132
+[12/28 20:30:14][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[12/28 20:30:18][INFO] ✅[FIT][Epoch 1] finished! 00:12→10:06 | loss_epoch=125
+[12/28 20:30:18][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[12/28 20:30:24][INFO] ✅[FIT][Epoch 2] finished! 00:18→09:46 | loss_epoch=445
+[12/28 20:30:24][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[12/28 20:30:30][INFO] ✅[FIT][Epoch 3] finished! 00:23→09:26 | loss_epoch=52.7
+[12/28 20:30:30][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[12/28 20:30:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 20:34:09][INFO] [Exp Name]: finetune_
+[12/28 20:34:09][INFO] [GPU x Batch] = 1 x 1
+[12/28 20:34:09][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:34:09][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:34:09][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 20:34:09][INFO]
+[12/28 20:34:13][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 20:34:13][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 20:34:13][INFO] [EMDB] Full sequence, split=1
+[12/28 20:34:13][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 20:34:13][INFO] [EMDB] Full sequence, split=2
+[12/28 20:34:13][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 20:34:13][INFO] [RICH] Full sequence, Test
+[12/28 20:34:13][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 20:34:13][INFO] [3DPW] Full sequence
+[12/28 20:34:13][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 20:34:13][INFO] [3DPW_OCC] Full sequence
+[12/28 20:34:13][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 20:34:13][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:34:13][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:34:13][INFO]
+[12/28 20:34:18][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 20:34:43][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_0/checkpoints'
+[12/28 20:35:08][INFO] Start Fitting...
+[12/28 20:35:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 20:35:21][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 20:35:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 20:35:25][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 20:35:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 20:35:29][INFO] ✅[FIT][Epoch 0] finished! 00:08→14:20 | loss_epoch=132
+[12/28 20:35:29][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[12/28 20:35:33][INFO] ✅[FIT][Epoch 1] finished! 00:13→10:52 | loss_epoch=125
+[12/28 20:35:33][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[12/28 20:35:39][INFO] ✅[FIT][Epoch 2] finished! 00:19→10:27 | loss_epoch=445
+[12/28 20:35:39][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[12/28 20:35:45][INFO] ✅[FIT][Epoch 3] finished! 00:24→09:58 | loss_epoch=52.7
+[12/28 20:35:45][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[12/28 20:35:50][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 20:36:16][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:36:16][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:36:16][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:36:16][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:36:16][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:36:16][INFO] ✅[FIT][Epoch 4] finished! 00:55→17:40 | loss_epoch=53.2
+[12/28 20:36:16][INFO] 🚀[FIT][Epoch 5] Data: unity Experiment: finetune_
+[12/28 20:36:21][INFO] ✅[FIT][Epoch 5] finished! 01:01→15:58 | loss_epoch=24.8
+[12/28 20:36:21][INFO] 🚀[FIT][Epoch 6] Data: unity Experiment: finetune_
+[12/28 20:36:27][INFO] ✅[FIT][Epoch 6] finished! 01:06→14:46 | loss_epoch=25
+[12/28 20:36:27][INFO] 🚀[FIT][Epoch 7] Data: unity Experiment: finetune_
+[12/28 20:36:32][INFO] ✅[FIT][Epoch 7] finished! 01:11→13:47 | loss_epoch=35.1
+[12/28 20:36:32][INFO] 🚀[FIT][Epoch 8] Data: unity Experiment: finetune_
+[12/28 20:36:38][INFO] ✅[FIT][Epoch 8] finished! 01:18→13:09 | loss_epoch=19.4
+[12/28 20:36:38][INFO] 🚀[FIT][Epoch 9] Data: unity Experiment: finetune_
+[12/28 20:37:10][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:37:10][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:37:10][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:37:10][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:37:10][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:37:10][INFO] ✅[FIT][Epoch 9] finished! 01:49→16:27 | loss_epoch=19.2
+[12/28 20:37:10][INFO] 🚀[FIT][Epoch 10] Data: unity Experiment: finetune_
+[12/28 20:37:16][INFO] ✅[FIT][Epoch 10] finished! 01:55→15:36 | loss_epoch=23.1
+[12/28 20:37:16][INFO] 🚀[FIT][Epoch 11] Data: unity Experiment: finetune_
+[12/28 20:37:22][INFO] ✅[FIT][Epoch 11] finished! 02:01→14:53 | loss_epoch=20.9
+[12/28 20:37:22][INFO] 🚀[FIT][Epoch 12] Data: unity Experiment: finetune_
+[12/28 20:37:28][INFO] ✅[FIT][Epoch 12] finished! 02:07→14:16 | loss_epoch=23.8
+[12/28 20:37:28][INFO] 🚀[FIT][Epoch 13] Data: unity Experiment: finetune_
+[12/28 20:37:34][INFO] ✅[FIT][Epoch 13] finished! 02:13→13:42 | loss_epoch=20.3
+[12/28 20:37:34][INFO] 🚀[FIT][Epoch 14] Data: unity Experiment: finetune_
+[12/28 20:38:04][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:38:04][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:38:04][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:38:04][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:38:04][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:38:04][INFO] ✅[FIT][Epoch 14] finished! 02:44→15:31 | loss_epoch=24.1
+[12/28 20:38:04][INFO] 🚀[FIT][Epoch 15] Data: unity Experiment: finetune_
+[12/28 20:38:11][INFO] ✅[FIT][Epoch 15] finished! 02:50→14:56 | loss_epoch=19.2
+[12/28 20:38:11][INFO] 🚀[FIT][Epoch 16] Data: unity Experiment: finetune_
+[12/28 20:38:18][INFO] ✅[FIT][Epoch 16] finished! 02:57→14:26 | loss_epoch=15.9
+[12/28 20:38:18][INFO] 🚀[FIT][Epoch 17] Data: unity Experiment: finetune_
+[12/28 20:38:24][INFO] ✅[FIT][Epoch 17] finished! 03:03→13:57 | loss_epoch=30
+[12/28 20:38:24][INFO] 🚀[FIT][Epoch 18] Data: unity Experiment: finetune_
+[12/28 20:38:30][INFO] ✅[FIT][Epoch 18] finished! 03:09→13:29 | loss_epoch=19.6
+[12/28 20:38:30][INFO] 🚀[FIT][Epoch 19] Data: unity Experiment: finetune_
+[12/28 20:39:02][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:39:02][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:39:02][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:39:02][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:39:02][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:39:02][INFO] ✅[FIT][Epoch 19] finished! 03:42→14:49 | loss_epoch=20.5
+[12/28 20:39:02][INFO] 🚀[FIT][Epoch 20] Data: unity Experiment: finetune_
+[12/28 20:39:09][INFO] ✅[FIT][Epoch 20] finished! 03:48→14:20 | loss_epoch=22.6
+[12/28 20:39:09][INFO] 🚀[FIT][Epoch 21] Data: unity Experiment: finetune_
+[12/28 20:39:15][INFO] ✅[FIT][Epoch 21] finished! 03:55→13:54 | loss_epoch=14.1
+[12/28 20:39:15][INFO] 🚀[FIT][Epoch 22] Data: unity Experiment: finetune_
+[12/28 20:39:22][INFO] ✅[FIT][Epoch 22] finished! 04:01→13:28 | loss_epoch=14.6
+[12/28 20:39:22][INFO] 🚀[FIT][Epoch 23] Data: unity Experiment: finetune_
+[12/28 20:39:28][INFO] ✅[FIT][Epoch 23] finished! 04:07→13:03 | loss_epoch=18.5
+[12/28 20:39:28][INFO] 🚀[FIT][Epoch 24] Data: unity Experiment: finetune_
+[12/28 20:40:00][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:40:00][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:40:00][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:40:00][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:40:00][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:40:00][INFO] ✅[FIT][Epoch 24] finished! 04:40→14:00 | loss_epoch=18.3
+[12/28 20:40:00][INFO] 🚀[FIT][Epoch 25] Data: unity Experiment: finetune_
+[12/28 20:40:07][INFO] ✅[FIT][Epoch 25] finished! 04:46→13:36 | loss_epoch=14.8
+[12/28 20:40:07][INFO] 🚀[FIT][Epoch 26] Data: unity Experiment: finetune_
+[12/28 20:40:13][INFO] ✅[FIT][Epoch 26] finished! 04:53→13:13 | loss_epoch=12.4
+[12/28 20:40:13][INFO] 🚀[FIT][Epoch 27] Data: unity Experiment: finetune_
+[12/28 20:40:20][INFO] ✅[FIT][Epoch 27] finished! 04:59→12:50 | loss_epoch=16.6
+[12/28 20:40:20][INFO] 🚀[FIT][Epoch 28] Data: unity Experiment: finetune_
+[12/28 20:40:26][INFO] ✅[FIT][Epoch 28] finished! 05:05→12:28 | loss_epoch=13.8
+[12/28 20:40:26][INFO] 🚀[FIT][Epoch 29] Data: unity Experiment: finetune_
+[12/28 20:41:18][INFO] [Exp Name]: finetune_
+[12/28 20:41:18][INFO] [GPU x Batch] = 1 x 1
+[12/28 20:41:18][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:41:18][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:41:18][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 20:41:18][INFO]
+[12/28 20:41:21][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 20:41:21][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 20:41:21][INFO] [EMDB] Full sequence, split=1
+[12/28 20:41:21][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 20:41:21][INFO] [EMDB] Full sequence, split=2
+[12/28 20:41:21][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 20:41:22][INFO] [RICH] Full sequence, Test
+[12/28 20:41:22][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 20:41:22][INFO] [3DPW] Full sequence
+[12/28 20:41:22][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 20:41:22][INFO] [3DPW_OCC] Full sequence
+[12/28 20:41:22][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 20:41:22][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:41:22][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:41:22][INFO]
+[12/28 20:41:26][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 20:41:50][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_1/checkpoints'
+[12/28 20:42:15][INFO] Start Fitting...
+[12/28 20:42:27][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 20:42:27][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 20:42:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 20:42:31][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 20:42:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 20:42:35][INFO] ✅[FIT][Epoch 0] finished! 00:09→00:36 | loss_epoch=132
+[12/28 20:42:35][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[12/28 20:42:39][INFO] ✅[FIT][Epoch 1] finished! 00:13→00:20 | loss_epoch=125
+[12/28 20:42:39][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[12/28 20:42:44][INFO] ✅[FIT][Epoch 2] finished! 00:18→00:12 | loss_epoch=445
+[12/28 20:42:44][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[12/28 20:42:49][INFO] ✅[FIT][Epoch 3] finished! 00:23→00:05 | loss_epoch=52.7
+[12/28 20:42:49][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[12/28 20:42:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 20:43:22][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:43:22][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:43:22][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:43:22][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:43:22][INFO] 0 sequences evaluated in MetricMocap
+[12/28 20:43:22][INFO] ✅[FIT][Epoch 4] finished! 00:56→00:00 | loss_epoch=53.2
+[12/28 20:43:27][INFO] End of script.
+[12/28 20:51:09][INFO] [Exp Name]: finetune_
+[12/28 20:51:09][INFO] [GPU x Batch] = 1 x 1
+[12/28 20:51:09][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:51:09][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:51:09][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 20:51:09][INFO]
+[12/28 20:51:12][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 20:51:12][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 20:51:12][INFO] [EMDB] Full sequence, split=1
+[12/28 20:51:12][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 20:51:12][INFO] [EMDB] Full sequence, split=2
+[12/28 20:51:12][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 20:51:12][INFO] [RICH] Full sequence, Test
+[12/28 20:51:12][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 20:51:12][INFO] [3DPW] Full sequence
+[12/28 20:51:12][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 20:51:12][INFO] [3DPW_OCC] Full sequence
+[12/28 20:51:12][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 20:51:12][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:51:12][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:51:12][INFO]
+[12/28 20:51:17][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 20:51:42][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_2/checkpoints'
+[12/28 20:52:07][INFO] Start Fitting...
+[12/28 20:52:34][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 20:52:34][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 20:52:34][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 20:52:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 20:52:38][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 20:52:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 20:54:55][INFO] [Exp Name]: finetune_
+[12/28 20:54:55][INFO] [GPU x Batch] = 1 x 1
+[12/28 20:54:55][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:54:55][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:54:55][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 20:54:55][INFO]
+[12/28 20:54:58][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 20:54:58][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 20:54:58][INFO] [EMDB] Full sequence, split=1
+[12/28 20:54:58][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 20:54:58][INFO] [EMDB] Full sequence, split=2
+[12/28 20:54:58][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 20:54:58][INFO] [RICH] Full sequence, Test
+[12/28 20:54:58][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 20:54:58][INFO] [3DPW] Full sequence
+[12/28 20:54:58][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 20:54:58][INFO] [3DPW_OCC] Full sequence
+[12/28 20:54:58][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 20:54:58][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:54:58][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:54:58][INFO]
+[12/28 20:55:03][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 20:55:27][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_3/checkpoints'
+[12/28 20:55:51][INFO] Start Fitting...
+[12/28 20:56:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 20:56:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 20:56:04][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 20:56:07][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 20:56:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 20:56:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 20:56:46][INFO] [Exp Name]: finetune_
+[12/28 20:56:46][INFO] [GPU x Batch] = 1 x 1
+[12/28 20:56:47][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:56:47][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:56:47][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 20:56:47][INFO]
+[12/28 20:56:50][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 20:56:50][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 20:56:50][INFO] [EMDB] Full sequence, split=1
+[12/28 20:56:50][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 20:56:50][INFO] [EMDB] Full sequence, split=2
+[12/28 20:56:50][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 20:56:50][INFO] [RICH] Full sequence, Test
+[12/28 20:56:50][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 20:56:50][INFO] [3DPW] Full sequence
+[12/28 20:56:50][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 20:56:50][INFO] [3DPW_OCC] Full sequence
+[12/28 20:56:50][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 20:56:50][INFO] [UnityDataset] Found 5 sequences.
+[12/28 20:56:50][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 20:56:50][INFO]
+[12/28 20:56:54][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 20:57:18][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_4/checkpoints'
+[12/28 20:57:50][INFO] Start Fitting...
+[12/28 20:58:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 20:58:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 20:58:01][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 20:58:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 20:58:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 20:58:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 21:00:47][INFO] [Exp Name]: finetune_
+[12/28 21:00:47][INFO] [GPU x Batch] = 1 x 1
+[12/28 21:00:47][INFO] [UnityDataset] Found 5 sequences.
+[12/28 21:00:47][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 21:00:47][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 21:00:47][INFO]
+[12/28 21:00:50][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 21:00:50][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 21:00:50][INFO] [EMDB] Full sequence, split=1
+[12/28 21:00:50][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 21:00:50][INFO] [EMDB] Full sequence, split=2
+[12/28 21:00:50][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 21:00:50][INFO] [RICH] Full sequence, Test
+[12/28 21:00:50][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 21:00:50][INFO] [3DPW] Full sequence
+[12/28 21:00:50][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 21:00:50][INFO] [3DPW_OCC] Full sequence
+[12/28 21:00:50][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 21:00:50][INFO] [UnityDataset] Found 5 sequences.
+[12/28 21:00:50][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 21:00:50][INFO]
+[12/28 21:00:56][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 21:01:18][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_5/checkpoints'
+[12/28 21:01:44][INFO] Start Fitting...
+[12/28 21:01:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 21:01:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 21:01:58][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 21:03:57][INFO] [Exp Name]: finetune_
+[12/28 21:03:57][INFO] [GPU x Batch] = 1 x 1
+[12/28 21:03:57][INFO] [UnityDataset] Found 5 sequences.
+[12/28 21:03:57][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 21:03:57][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 21:03:57][INFO]
+[12/28 21:04:00][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 21:04:00][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 21:04:00][INFO] [EMDB] Full sequence, split=1
+[12/28 21:04:00][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 21:04:00][INFO] [EMDB] Full sequence, split=2
+[12/28 21:04:00][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 21:04:00][INFO] [RICH] Full sequence, Test
+[12/28 21:04:00][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 21:04:00][INFO] [3DPW] Full sequence
+[12/28 21:04:00][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 21:04:00][INFO] [3DPW_OCC] Full sequence
+[12/28 21:04:00][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 21:04:00][INFO] [UnityDataset] Found 5 sequences.
+[12/28 21:04:00][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 21:04:00][INFO]
+[12/28 21:04:05][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 21:04:21][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_6/checkpoints'
+[12/28 21:04:46][INFO] Start Fitting...
+[12/28 21:05:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 21:05:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 21:05:02][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 21:05:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 21:05:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 21:05:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 21:05:42][INFO] 5 sequences evaluated in MetricMocap
+[12/28 21:05:42][INFO] monitored metric mpjpe per sequence
+756.7 : 100_biboo_birthday_speech_explosion_1
+719.1 : 105_biboo_birthday_speech_explosion_6
+711.3 : 101_biboo_birthday_speech_explosion_2
+669.2 : 102_biboo_birthday_speech_explosion_3
+619.6 : 103_biboo_birthday_speech_explosion_4
+------
+[12/28 21:05:42][INFO] [Metrics] Unity:
+pa_mpjpe: 234.3
+mpjpe: 695.2
+pve: 812.0
+accel: 8.1
+------
+[12/28 21:05:42][INFO] 5 sequences evaluated in MetricMocap
+[12/28 21:05:42][INFO] monitored metric wa2_mpjpe per sequence
+5629.3 : 100_biboo_birthday_speech_explosion_1
+3964.5 : 102_biboo_birthday_speech_explosion_3
+3530.3 : 101_biboo_birthday_speech_explosion_2
+3050.9 : 103_biboo_birthday_speech_explosion_4
+2628.2 : 105_biboo_birthday_speech_explosion_6
+------
+[12/28 21:05:42][INFO] [Metrics] Unity:
+wa2_mpjpe: 3760.7
+waa_mpjpe: 422.7
+rte: 419.6
+jitter: 960.2
+fs: 249.6
+------
+[12/28 21:05:42][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:05:42][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:05:42][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:05:42][INFO] ✅[FIT][Epoch 0] finished! 00:41→00:00 | loss_epoch=132
+[12/28 21:05:44][INFO] End of script.
+[12/28 21:09:50][INFO] [Exp Name]: finetune_
+[12/28 21:09:50][INFO] [GPU x Batch] = 1 x 1
+[12/28 21:09:50][INFO] [UnityDataset] Found 5 sequences.
+[12/28 21:09:50][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 21:09:50][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 21:09:50][INFO]
+[12/28 21:09:53][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 21:09:53][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 21:09:53][INFO] [EMDB] Full sequence, split=1
+[12/28 21:09:53][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 21:09:53][INFO] [EMDB] Full sequence, split=2
+[12/28 21:09:53][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 21:09:53][INFO] [RICH] Full sequence, Test
+[12/28 21:09:53][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 21:09:53][INFO] [3DPW] Full sequence
+[12/28 21:09:53][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 21:09:53][INFO] [3DPW_OCC] Full sequence
+[12/28 21:09:53][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 21:09:53][INFO] [UnityDataset] Found 5 sequences.
+[12/28 21:09:53][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 21:09:53][INFO]
+[12/28 21:09:59][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 21:10:19][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_7/checkpoints'
+[12/28 21:10:46][INFO] Start Fitting...
+[12/28 21:10:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 21:10:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 21:10:57][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 21:10:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 21:11:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 21:11:05][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 21:11:37][INFO] 5 sequences evaluated in MetricMocap
+[12/28 21:11:37][INFO] monitored metric mpjpe per sequence
+756.7 : 100_biboo_birthday_speech_explosion_1
+719.1 : 105_biboo_birthday_speech_explosion_6
+711.3 : 101_biboo_birthday_speech_explosion_2
+669.2 : 102_biboo_birthday_speech_explosion_3
+619.6 : 103_biboo_birthday_speech_explosion_4
+------
+[12/28 21:11:37][INFO] [Metrics] EMDB_1:
+pa_mpjpe: 234.3
+mpjpe: 695.2
+pve: 812.0
+accel: 8.1
+------
+[12/28 21:11:37][INFO] 5 sequences evaluated in MetricMocap
+[12/28 21:11:37][INFO] monitored metric wa2_mpjpe per sequence
+5629.3 : 100_biboo_birthday_speech_explosion_1
+3964.5 : 102_biboo_birthday_speech_explosion_3
+3530.3 : 101_biboo_birthday_speech_explosion_2
+3050.9 : 103_biboo_birthday_speech_explosion_4
+2628.2 : 105_biboo_birthday_speech_explosion_6
+------
+[12/28 21:11:37][INFO] [Metrics] EMDB_2:
+wa2_mpjpe: 3760.7
+waa_mpjpe: 422.7
+rte: 419.6
+jitter: 960.2
+fs: 249.6
+------
+[12/28 21:11:37][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:11:37][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:11:37][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:11:37][INFO] ✅[FIT][Epoch 0] finished! 00:41→00:00 | loss_epoch=132
+[12/28 21:11:38][INFO] End of script.
+[12/28 21:24:49][INFO] [Exp Name]: finetune_
+[12/28 21:24:49][INFO] [GPU x Batch] = 1 x 1
+[12/28 21:24:49][INFO] [UnityDataset] Found 5 sequences.
+[12/28 21:24:49][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 21:24:49][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 21:24:49][INFO]
+[12/28 21:24:52][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 21:24:52][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 21:24:52][INFO] [EMDB] Full sequence, split=1
+[12/28 21:24:52][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 21:24:52][INFO] [EMDB] Full sequence, split=2
+[12/28 21:24:52][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 21:24:52][INFO] [RICH] Full sequence, Test
+[12/28 21:24:52][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 21:24:52][INFO] [3DPW] Full sequence
+[12/28 21:24:52][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 21:24:52][INFO] [3DPW_OCC] Full sequence
+[12/28 21:24:52][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 21:24:52][INFO] [UnityDataset] Found 5 sequences.
+[12/28 21:24:52][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 21:24:52][INFO]
+[12/28 21:24:57][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 21:25:18][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_8/checkpoints'
+[12/28 21:25:48][INFO] Start Fitting...
+[12/28 21:25:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 21:25:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 21:25:59][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 21:26:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 21:26:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 21:26:07][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 21:26:40][INFO] 5 sequences evaluated in MetricMocap
+[12/28 21:26:40][INFO] monitored metric mpjpe per sequence
+756.7 : 100_biboo_birthday_speech_explosion_1
+719.1 : 105_biboo_birthday_speech_explosion_6
+711.3 : 101_biboo_birthday_speech_explosion_2
+669.2 : 102_biboo_birthday_speech_explosion_3
+619.6 : 103_biboo_birthday_speech_explosion_4
+------
+[12/28 21:26:40][INFO] [Metrics] EMDB_1:
+pa_mpjpe: 234.3
+mpjpe: 695.2
+pve: 812.0
+accel: 8.1
+------
+[12/28 21:26:40][INFO] 5 sequences evaluated in MetricMocap
+[12/28 21:26:40][INFO] monitored metric wa2_mpjpe per sequence
+5629.3 : 100_biboo_birthday_speech_explosion_1
+3964.5 : 102_biboo_birthday_speech_explosion_3
+3530.3 : 101_biboo_birthday_speech_explosion_2
+3050.9 : 103_biboo_birthday_speech_explosion_4
+2628.2 : 105_biboo_birthday_speech_explosion_6
+------
+[12/28 21:26:40][INFO] [Metrics] EMDB_2:
+wa2_mpjpe: 3760.7
+waa_mpjpe: 422.7
+rte: 419.6
+jitter: 960.2
+fs: 249.6
+------
+[12/28 21:26:40][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:26:40][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:26:40][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:26:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 21:26:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 21:26:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 21:26:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 21:26:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/wa2_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 21:26:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/waa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 21:26:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/rte', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 21:26:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/jitter', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 21:26:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/fs', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 21:26:40][INFO] ✅[FIT][Epoch 0] finished! 00:42→00:00 | loss_epoch=132
+[12/28 21:26:42][INFO] End of script.
+[12/28 21:33:06][INFO] [Exp Name]: finetune_
+[12/28 21:33:06][INFO] [GPU x Batch] = 1 x 1
+[12/28 21:33:06][INFO] [UnityDataset] Found 5 sequences.
+[12/28 21:33:06][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 21:33:06][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 21:33:06][INFO]
+[12/28 21:33:10][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 21:33:10][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 21:33:10][INFO] [EMDB] Full sequence, split=1
+[12/28 21:33:10][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 21:33:10][INFO] [EMDB] Full sequence, split=2
+[12/28 21:33:10][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 21:33:10][INFO] [RICH] Full sequence, Test
+[12/28 21:33:10][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 21:33:10][INFO] [3DPW] Full sequence
+[12/28 21:33:10][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 21:33:10][INFO] [3DPW_OCC] Full sequence
+[12/28 21:33:10][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 21:33:10][INFO] [UnityDataset] Found 5 sequences.
+[12/28 21:33:10][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 21:33:10][INFO]
+[12/28 21:33:15][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 21:33:38][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_9/checkpoints'
+[12/28 21:34:07][INFO] Start Fitting...
+[12/28 21:34:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 21:34:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 21:34:42][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 21:34:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 21:34:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 21:34:50][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 21:35:25][INFO] 5 sequences evaluated in MetricMocap
+[12/28 21:35:25][INFO] monitored metric mpjpe per sequence
+756.7 : 100_biboo_birthday_speech_explosion_1
+719.1 : 105_biboo_birthday_speech_explosion_6
+711.3 : 101_biboo_birthday_speech_explosion_2
+669.2 : 102_biboo_birthday_speech_explosion_3
+619.6 : 103_biboo_birthday_speech_explosion_4
+------
+[12/28 21:35:25][INFO] [Metrics] EMDB_1:
+pa_mpjpe: 234.3
+mpjpe: 695.2
+pve: 812.0
+accel: 8.1
+------
+[12/28 21:35:25][INFO] 5 sequences evaluated in MetricMocap
+[12/28 21:35:25][INFO] monitored metric wa2_mpjpe per sequence
+5629.3 : 100_biboo_birthday_speech_explosion_1
+3964.5 : 102_biboo_birthday_speech_explosion_3
+3530.3 : 101_biboo_birthday_speech_explosion_2
+3050.9 : 103_biboo_birthday_speech_explosion_4
+2628.2 : 105_biboo_birthday_speech_explosion_6
+------
+[12/28 21:35:25][INFO] [Metrics] EMDB_2:
+wa2_mpjpe: 3760.7
+waa_mpjpe: 422.7
+rte: 419.6
+jitter: 960.2
+fs: 249.6
+------
+[12/28 21:35:25][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:35:25][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:35:25][INFO] 0 sequences evaluated in MetricMocap
+[12/28 21:35:25][INFO] ✅[FIT][Epoch 0] finished! 00:44→00:00 | loss_epoch=132
+[12/28 21:36:06][INFO] Manually saved checkpoint to /root/miko/puni/train/GENMO/checkpoints/last_manual.ckpt
+[12/28 21:36:07][INFO] End of script.
+[12/28 22:03:16][INFO] [Exp Name]: finetune_
+[12/28 22:03:16][INFO] [GPU x Batch] = 1 x 1
+[12/28 22:03:16][INFO] [UnityDataset] Found 5 sequences.
+[12/28 22:03:16][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 22:03:16][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 22:03:16][INFO]
+[12/28 22:03:19][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 22:03:19][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 22:03:19][INFO] [EMDB] Full sequence, split=1
+[12/28 22:03:19][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 22:03:19][INFO] [EMDB] Full sequence, split=2
+[12/28 22:03:19][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 22:03:19][INFO] [RICH] Full sequence, Test
+[12/28 22:03:19][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 22:03:19][INFO] [3DPW] Full sequence
+[12/28 22:03:19][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 22:03:19][INFO] [3DPW_OCC] Full sequence
+[12/28 22:03:19][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 22:03:19][INFO] [UnityDataset] Found 5 sequences.
+[12/28 22:03:19][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 22:03:19][INFO]
+[12/28 22:03:24][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 22:03:35][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_0/checkpoints'
+[12/28 22:04:02][INFO] Start Fitting...
+[12/28 22:04:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 22:04:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 22:04:13][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 22:04:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 22:04:18][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 22:04:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 22:04:52][INFO] 5 sequences evaluated in MetricMocap
+[12/28 22:04:52][INFO] monitored metric mpjpe per sequence
+756.7 : 100_biboo_birthday_speech_explosion_1
+719.1 : 105_biboo_birthday_speech_explosion_6
+711.3 : 101_biboo_birthday_speech_explosion_2
+669.2 : 102_biboo_birthday_speech_explosion_3
+619.6 : 103_biboo_birthday_speech_explosion_4
+------
+[12/28 22:04:52][INFO] [Metrics] EMDB_1:
+pa_mpjpe: 234.3
+mpjpe: 695.2
+pve: 812.0
+accel: 8.1
+------
+[12/28 22:04:52][INFO] 5 sequences evaluated in MetricMocap
+[12/28 22:04:52][INFO] monitored metric wa2_mpjpe per sequence
+5629.3 : 100_biboo_birthday_speech_explosion_1
+3964.5 : 102_biboo_birthday_speech_explosion_3
+3530.3 : 101_biboo_birthday_speech_explosion_2
+3050.9 : 103_biboo_birthday_speech_explosion_4
+2628.2 : 105_biboo_birthday_speech_explosion_6
+------
+[12/28 22:04:52][INFO] [Metrics] EMDB_2:
+wa2_mpjpe: 3760.7
+waa_mpjpe: 422.7
+rte: 419.6
+jitter: 960.2
+fs: 249.6
+------
+[12/28 22:04:52][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:04:52][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:04:52][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:04:52][INFO] ✅[FIT][Epoch 0] finished! 00:40→00:00 | loss_epoch=132
+[12/28 22:05:00][INFO] Manually saved checkpoint to /root/miko/puni/train/GENMO/checkpoints/last_manual.ckpt
+[12/28 22:05:02][INFO] End of script.
+[12/28 22:14:01][INFO] [Exp Name]: finetune_
+[12/28 22:14:01][INFO] [GPU x Batch] = 1 x 1
+[12/28 22:14:01][INFO] [UnityDataset] Found 5 sequences.
+[12/28 22:14:01][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 22:14:01][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 22:14:01][INFO]
+[12/28 22:14:04][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 22:14:04][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 22:14:04][INFO] [EMDB] Full sequence, split=1
+[12/28 22:14:04][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 22:14:04][INFO] [EMDB] Full sequence, split=2
+[12/28 22:14:04][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 22:14:04][INFO] [RICH] Full sequence, Test
+[12/28 22:14:04][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 22:14:04][INFO] [3DPW] Full sequence
+[12/28 22:14:04][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 22:14:04][INFO] [3DPW_OCC] Full sequence
+[12/28 22:14:04][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 22:14:04][INFO] [UnityDataset] Found 5 sequences.
+[12/28 22:14:04][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 22:14:04][INFO]
+[12/28 22:14:09][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 22:14:21][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_1/checkpoints'
+[12/28 22:14:49][INFO] Start Fitting...
+[12/28 22:15:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 22:15:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 22:15:11][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 22:15:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 22:15:15][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 22:15:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 22:15:50][INFO] 5 sequences evaluated in MetricMocap
+[12/28 22:15:50][INFO] monitored metric mpjpe per sequence
+756.7 : 100_biboo_birthday_speech_explosion_1
+719.1 : 105_biboo_birthday_speech_explosion_6
+711.3 : 101_biboo_birthday_speech_explosion_2
+669.2 : 102_biboo_birthday_speech_explosion_3
+619.6 : 103_biboo_birthday_speech_explosion_4
+------
+[12/28 22:15:50][INFO] [Metrics] EMDB_1:
+pa_mpjpe: 234.3
+mpjpe: 695.2
+pve: 812.0
+accel: 8.1
+------
+[12/28 22:15:50][INFO] 5 sequences evaluated in MetricMocap
+[12/28 22:15:50][INFO] monitored metric wa2_mpjpe per sequence
+5629.3 : 100_biboo_birthday_speech_explosion_1
+3964.5 : 102_biboo_birthday_speech_explosion_3
+3530.3 : 101_biboo_birthday_speech_explosion_2
+3050.9 : 103_biboo_birthday_speech_explosion_4
+2628.2 : 105_biboo_birthday_speech_explosion_6
+------
+[12/28 22:15:50][INFO] [Metrics] EMDB_2:
+wa2_mpjpe: 3760.7
+waa_mpjpe: 422.7
+rte: 419.6
+jitter: 960.2
+fs: 249.6
+------
+[12/28 22:15:50][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:15:50][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:15:50][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:15:50][INFO] ✅[FIT][Epoch 0] finished! 00:40→00:00 | loss_epoch=132
+[12/28 22:17:03][INFO] Manually saved checkpoint to /root/miko/puni/train/GENMO/checkpoints/last_manual.ckpt
+[12/28 22:17:04][INFO] End of script.
+[12/28 22:20:51][INFO] [Exp Name]: finetune_
+[12/28 22:20:51][INFO] [GPU x Batch] = 1 x 1
+[12/28 22:20:51][INFO] [UnityDataset] Found 5 sequences.
+[12/28 22:20:51][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 22:20:51][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 22:20:51][INFO]
+[12/28 22:20:54][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 22:20:54][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 22:20:54][INFO] [EMDB] Full sequence, split=1
+[12/28 22:20:54][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 22:20:54][INFO] [EMDB] Full sequence, split=2
+[12/28 22:20:54][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 22:20:54][INFO] [RICH] Full sequence, Test
+[12/28 22:20:54][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 22:20:54][INFO] [3DPW] Full sequence
+[12/28 22:20:54][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 22:20:54][INFO] [3DPW_OCC] Full sequence
+[12/28 22:20:54][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 22:20:54][INFO] [UnityDataset] Found 5 sequences.
+[12/28 22:20:54][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 22:20:54][INFO]
+[12/28 22:20:59][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 22:21:10][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_2/checkpoints'
+[12/28 22:21:38][INFO] Start Fitting...
+[12/28 22:21:50][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 22:21:51][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 22:21:51][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 22:21:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 22:21:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 22:21:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 22:22:29][INFO] 5 sequences evaluated in MetricMocap
+[12/28 22:22:29][INFO] monitored metric mpjpe per sequence
+756.7 : 100_biboo_birthday_speech_explosion_1
+719.1 : 105_biboo_birthday_speech_explosion_6
+711.3 : 101_biboo_birthday_speech_explosion_2
+669.2 : 102_biboo_birthday_speech_explosion_3
+619.6 : 103_biboo_birthday_speech_explosion_4
+------
+[12/28 22:22:29][INFO] [Metrics] EMDB_1:
+pa_mpjpe: 234.3
+mpjpe: 695.2
+pve: 812.0
+accel: 8.1
+------
+[12/28 22:22:29][INFO] 5 sequences evaluated in MetricMocap
+[12/28 22:22:29][INFO] monitored metric wa2_mpjpe per sequence
+5629.3 : 100_biboo_birthday_speech_explosion_1
+3964.5 : 102_biboo_birthday_speech_explosion_3
+3530.3 : 101_biboo_birthday_speech_explosion_2
+3050.9 : 103_biboo_birthday_speech_explosion_4
+2628.2 : 105_biboo_birthday_speech_explosion_6
+------
+[12/28 22:22:29][INFO] [Metrics] EMDB_2:
+wa2_mpjpe: 3760.7
+waa_mpjpe: 422.7
+rte: 419.6
+jitter: 960.2
+fs: 249.6
+------
+[12/28 22:22:29][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:22:29][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:22:29][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:22:29][INFO] ✅[FIT][Epoch 0] finished! 00:39→00:00 | loss_epoch=132
+[12/28 22:23:27][INFO] Manually saved checkpoint to /root/miko/puni/train/GENMO/checkpoints/manual_epoch_0.ckpt
+[12/28 22:23:29][INFO] End of script.
+[12/28 22:25:04][INFO] [Exp Name]: finetune_
+[12/28 22:25:04][INFO] [GPU x Batch] = 1 x 1
+[12/28 22:25:04][INFO] [UnityDataset] Found 5 sequences.
+[12/28 22:25:04][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 22:25:04][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 22:25:04][INFO]
+[12/28 22:25:08][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 22:25:08][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 22:25:08][INFO] [EMDB] Full sequence, split=1
+[12/28 22:25:08][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 22:25:08][INFO] [EMDB] Full sequence, split=2
+[12/28 22:25:08][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 22:25:08][INFO] [RICH] Full sequence, Test
+[12/28 22:25:08][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 22:25:08][INFO] [3DPW] Full sequence
+[12/28 22:25:08][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 22:25:08][INFO] [3DPW_OCC] Full sequence
+[12/28 22:25:08][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 22:25:08][INFO] [UnityDataset] Found 5 sequences.
+[12/28 22:25:08][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 22:25:08][INFO]
+[12/28 22:25:12][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 22:25:19][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_3/checkpoints'
+[12/28 22:25:48][INFO] Start Fitting...
+[12/28 22:26:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 22:26:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 22:26:14][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 22:26:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 22:26:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 22:26:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 22:26:55][INFO] 5 sequences evaluated in MetricMocap
+[12/28 22:26:55][INFO] monitored metric mpjpe per sequence
+756.7 : 100_biboo_birthday_speech_explosion_1
+719.1 : 105_biboo_birthday_speech_explosion_6
+711.3 : 101_biboo_birthday_speech_explosion_2
+669.2 : 102_biboo_birthday_speech_explosion_3
+619.6 : 103_biboo_birthday_speech_explosion_4
+------
+[12/28 22:26:55][INFO] [Metrics] EMDB_1:
+pa_mpjpe: 234.3
+mpjpe: 695.2
+pve: 812.0
+accel: 8.1
+------
+[12/28 22:26:55][INFO] 5 sequences evaluated in MetricMocap
+[12/28 22:26:55][INFO] monitored metric wa2_mpjpe per sequence
+5629.3 : 100_biboo_birthday_speech_explosion_1
+3964.5 : 102_biboo_birthday_speech_explosion_3
+3530.3 : 101_biboo_birthday_speech_explosion_2
+3050.9 : 103_biboo_birthday_speech_explosion_4
+2628.2 : 105_biboo_birthday_speech_explosion_6
+------
+[12/28 22:26:55][INFO] [Metrics] EMDB_2:
+wa2_mpjpe: 3760.7
+waa_mpjpe: 422.7
+rte: 419.6
+jitter: 960.2
+fs: 249.6
+------
+[12/28 22:26:55][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:26:55][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:26:55][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:26:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:26:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:26:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:26:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:26:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/wa2_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:26:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/waa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:26:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/rte', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:26:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/jitter', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:26:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/fs', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:26:55][INFO] ✅[FIT][Epoch 0] finished! 00:42→00:00 | loss_epoch=132
+[12/28 22:26:59][INFO] End of script.
+[12/28 22:27:36][INFO] [Exp Name]: finetune_
+[12/28 22:27:36][INFO] [GPU x Batch] = 1 x 1
+[12/28 22:27:36][INFO] [UnityDataset] Found 5 sequences.
+[12/28 22:27:36][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 22:27:36][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 22:27:36][INFO]
+[12/28 22:27:39][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 22:27:39][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 22:27:39][INFO] [EMDB] Full sequence, split=1
+[12/28 22:27:39][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 22:27:39][INFO] [EMDB] Full sequence, split=2
+[12/28 22:27:39][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 22:27:39][INFO] [RICH] Full sequence, Test
+[12/28 22:27:39][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 22:27:39][INFO] [3DPW] Full sequence
+[12/28 22:27:39][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 22:27:39][INFO] [3DPW_OCC] Full sequence
+[12/28 22:27:39][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 22:27:39][INFO] [UnityDataset] Found 5 sequences.
+[12/28 22:27:39][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 22:27:39][INFO]
+[12/28 22:27:44][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 22:28:03][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_4/checkpoints'
+[12/28 22:28:31][INFO] Start Fitting...
+[12/28 22:29:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 22:29:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 22:29:35][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 22:29:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 22:29:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 22:29:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 22:30:14][INFO] 5 sequences evaluated in MetricMocap
+[12/28 22:30:14][INFO] monitored metric mpjpe per sequence
+756.7 : 100_biboo_birthday_speech_explosion_1
+719.1 : 105_biboo_birthday_speech_explosion_6
+711.3 : 101_biboo_birthday_speech_explosion_2
+669.2 : 102_biboo_birthday_speech_explosion_3
+619.6 : 103_biboo_birthday_speech_explosion_4
+------
+[12/28 22:30:14][INFO] [Metrics] EMDB_1:
+pa_mpjpe: 234.3
+mpjpe: 695.2
+pve: 812.0
+accel: 8.1
+------
+[12/28 22:30:14][INFO] 5 sequences evaluated in MetricMocap
+[12/28 22:30:14][INFO] monitored metric wa2_mpjpe per sequence
+5629.3 : 100_biboo_birthday_speech_explosion_1
+3964.5 : 102_biboo_birthday_speech_explosion_3
+3530.3 : 101_biboo_birthday_speech_explosion_2
+3050.9 : 103_biboo_birthday_speech_explosion_4
+2628.2 : 105_biboo_birthday_speech_explosion_6
+------
+[12/28 22:30:14][INFO] [Metrics] EMDB_2:
+wa2_mpjpe: 3760.7
+waa_mpjpe: 422.7
+rte: 419.6
+jitter: 960.2
+fs: 249.6
+------
+[12/28 22:30:14][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:30:14][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:30:14][INFO] 0 sequences evaluated in MetricMocap
+[12/28 22:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/wa2_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/waa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/rte', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/jitter', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/fs', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 22:30:14][INFO] ✅[FIT][Epoch 0] finished! 00:40→00:00 | loss_epoch=132
+[12/28 22:31:02][INFO] Manually saved checkpoint to /root/miko/puni/train/GENMO/checkpoints/manual_epoch_0.ckpt
+[12/28 22:31:04][INFO] End of script.
+[12/28 23:12:52][INFO] [Exp Name]: finetune_
+[12/28 23:12:52][INFO] [GPU x Batch] = 1 x 1
+[12/28 23:12:52][INFO] [UnityDataset] Found 5 sequences.
+[12/28 23:12:52][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 23:12:52][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 23:12:52][INFO]
+[12/28 23:12:56][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 23:12:56][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 23:12:56][INFO] [EMDB] Full sequence, split=1
+[12/28 23:12:56][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 23:12:56][INFO] [EMDB] Full sequence, split=2
+[12/28 23:12:56][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 23:12:56][INFO] [RICH] Full sequence, Test
+[12/28 23:12:56][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 23:12:56][INFO] [3DPW] Full sequence
+[12/28 23:12:56][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 23:12:56][INFO] [3DPW_OCC] Full sequence
+[12/28 23:12:56][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 23:12:56][INFO] [UnityDataset] Found 5 sequences.
+[12/28 23:12:56][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 23:12:56][INFO]
+[12/28 23:13:02][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 23:13:35][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_5/checkpoints'
+[12/28 23:14:02][INFO] Start Fitting...
+[12/28 23:14:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/28 23:14:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/28 23:14:13][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/28 23:14:15][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/28 23:14:18][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/28 23:14:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/28 23:14:54][INFO] 5 sequences evaluated in MetricMocap
+[12/28 23:14:54][INFO] monitored metric mpjpe per sequence
+756.7 : 100_biboo_birthday_speech_explosion_1
+719.1 : 105_biboo_birthday_speech_explosion_6
+711.3 : 101_biboo_birthday_speech_explosion_2
+669.2 : 102_biboo_birthday_speech_explosion_3
+619.6 : 103_biboo_birthday_speech_explosion_4
+------
+[12/28 23:14:54][INFO] [Metrics] EMDB_1:
+pa_mpjpe: 234.3
+mpjpe: 695.2
+pve: 812.0
+accel: 8.1
+------
+[12/28 23:14:54][INFO] 5 sequences evaluated in MetricMocap
+[12/28 23:14:54][INFO] monitored metric wa2_mpjpe per sequence
+5629.3 : 100_biboo_birthday_speech_explosion_1
+3964.5 : 102_biboo_birthday_speech_explosion_3
+3530.3 : 101_biboo_birthday_speech_explosion_2
+3050.9 : 103_biboo_birthday_speech_explosion_4
+2628.2 : 105_biboo_birthday_speech_explosion_6
+------
+[12/28 23:14:54][INFO] [Metrics] EMDB_2:
+wa2_mpjpe: 3760.7
+waa_mpjpe: 422.7
+rte: 419.6
+jitter: 960.2
+fs: 249.6
+------
+[12/28 23:14:54][INFO] 0 sequences evaluated in MetricMocap
+[12/28 23:14:54][INFO] 0 sequences evaluated in MetricMocap
+[12/28 23:14:54][INFO] 0 sequences evaluated in MetricMocap
+[12/28 23:14:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 23:14:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 23:14:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 23:14:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_1/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 23:14:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/wa2_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 23:14:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/waa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 23:14:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/rte', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 23:14:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/jitter', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 23:14:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_EMDB_2/fs', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/28 23:14:54][INFO] ✅[FIT][Epoch 0] finished! 00:42→00:00 | loss_epoch=132
+[12/28 23:15:44][INFO] Manually saved checkpoint to /root/miko/puni/train/GENMO/checkpoints/manual_epoch_0.ckpt
+[12/28 23:15:47][INFO] End of script.
+[12/28 23:29:49][INFO] [Exp Name]: finetune_
+[12/28 23:29:49][INFO] [GPU x Batch] = 1 x 1
+[12/28 23:29:49][INFO] [UnityDataset] Found 5 sequences.
+[12/28 23:29:49][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 23:29:49][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 23:29:49][INFO]
+[12/28 23:29:52][INFO] [HumanML3D] Loading from inputs/HumanML3D_SMPL/hmr4d_support/humanml3d_smplhpose_train.pth ...
+[12/28 23:29:52][WARNING] [val] Skipping humanml3d_eval due to error: Error in call to target 'genmo.datasets.pure_motion.humanml3d.Humanml3dDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.humanml3d_eval
+[12/28 23:29:52][INFO] [EMDB] Full sequence, split=1
+[12/28 23:29:52][WARNING] [val] Skipping emdb1_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb1_fliptest
+[12/28 23:29:52][INFO] [EMDB] Full sequence, split=2
+[12/28 23:29:52][WARNING] [val] Skipping emdb2_fliptest due to error: Error in call to target 'genmo.datasets.emdb.emdb_motion_test.EmdbSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.emdb2_fliptest
+[12/28 23:29:52][INFO] [RICH] Full sequence, Test
+[12/28 23:29:52][WARNING] [val] Skipping rich_test due to error: Error in call to target 'genmo.datasets.rich.rich_motion_test.RichSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.rich_test
+[12/28 23:29:52][INFO] [3DPW] Full sequence
+[12/28 23:29:52][WARNING] [val] Skipping 3dpw_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_motion_test.ThreedpwSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_fliptest
+[12/28 23:29:52][INFO] [3DPW_OCC] Full sequence
+[12/28 23:29:52][WARNING] [val] Skipping 3dpw_occ_fliptest due to error: Error in call to target 'genmo.datasets.threedpw.threedpw_occ_motion_test.ThreedpwOccSmplFullSeqDataset':
+FileNotFoundError(2, 'No such file or directory')
+full_key: dataset_opts.val.3dpw_occ_fliptest
+[12/28 23:29:52][INFO] [UnityDataset] Found 5 sequences.
+[12/28 23:29:52][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 23:29:52][INFO]
+[12/28 23:29:57][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 23:30:31][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_6/checkpoints'
+[12/28 23:58:20][INFO] [Exp Name]: finetune_
+[12/28 23:58:20][INFO] [GPU x Batch] = 1 x 128
+[12/28 23:58:20][INFO] [UnityDataset] Found 5 sequences.
+[12/28 23:58:20][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 23:58:20][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/28 23:58:20][INFO]
+[12/28 23:58:20][INFO] [UnityDataset] Found 5 sequences.
+[12/28 23:58:20][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/28 23:58:20][INFO]
+[12/28 23:58:26][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/28 23:58:41][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_7/checkpoints'
+[12/28 23:58:41][INFO] Start Fitting...
+[12/28 23:59:01][INFO] End of script.
+[12/29 00:01:50][INFO] [Exp Name]: finetune_
+[12/29 00:01:50][INFO] [GPU x Batch] = 1 x 1
+[12/29 00:01:50][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:01:50][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:01:50][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/29 00:01:50][INFO]
+[12/29 00:01:50][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:01:50][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:01:50][INFO]
+[12/29 00:01:57][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/29 00:02:07][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_8/checkpoints'
+[12/29 00:02:07][INFO] Start Fitting...
+[12/29 00:02:21][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 00:02:54][INFO] ✅[FIT][Epoch 0] finished! 00:34→00:00 | loss_epoch=132
+[12/29 00:03:41][INFO] Manually saved checkpoint to /root/miko/puni/train/GENMO/checkpoints/manual_epoch_0.ckpt
+[12/29 00:03:44][INFO] End of script.
+[12/29 00:05:19][INFO] [Exp Name]: finetune_
+[12/29 00:05:19][INFO] [GPU x Batch] = 1 x 1
+[12/29 00:05:19][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:05:19][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:05:19][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/29 00:05:19][INFO]
+[12/29 00:05:19][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:05:19][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:05:19][INFO]
+[12/29 00:05:26][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/29 00:05:43][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_9/checkpoints'
+[12/29 00:05:43][INFO] Start Fitting...
+[12/29 00:06:02][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 00:06:34][INFO] ✅[FIT][Epoch 0] finished! 00:34→00:00 | loss_epoch=132
+[12/29 00:06:36][INFO] End of script.
+[12/29 00:08:20][INFO] [Exp Name]: finetune_
+[12/29 00:08:20][INFO] [GPU x Batch] = 1 x 1
+[12/29 00:08:55][INFO] [Exp Name]: finetune_
+[12/29 00:08:55][INFO] [GPU x Batch] = 1 x 1
+[12/29 00:08:55][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:08:55][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:08:55][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/29 00:08:55][INFO]
+[12/29 00:08:55][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:08:55][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:08:55][INFO]
+[12/29 00:09:01][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/29 00:09:16][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_10/checkpoints'
+[12/29 00:09:17][INFO] Start Fitting...
+[12/29 00:09:29][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 00:10:03][INFO] ✅[FIT][Epoch 0] finished! 00:35→00:00 | loss_epoch=132
+[12/29 00:12:38][INFO] End of script.
+[12/29 00:25:51][INFO] [Exp Name]: finetune_
+[12/29 00:25:51][INFO] [GPU x Batch] = 1 x 1
+[12/29 00:25:51][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:25:51][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:25:51][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/29 00:25:51][INFO]
+[12/29 00:25:51][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:25:51][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:25:51][INFO]
+[12/29 00:25:58][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/29 00:26:20][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_0/checkpoints'
+[12/29 00:26:32][INFO] Start Fitting...
+[12/29 00:26:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 00:26:34][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 00:26:34][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 00:26:34][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 00:26:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 00:26:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 00:27:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:27:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:27:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:27:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:27:08][INFO] ✅[FIT][Epoch 0] finished! 00:36→00:00 | loss_epoch=132
+[12/29 00:29:24][INFO] End of script.
+[12/29 00:41:56][INFO] [Exp Name]: finetune_
+[12/29 00:41:56][INFO] [GPU x Batch] = 1 x 1
+[12/29 00:41:56][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:41:56][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:41:56][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/29 00:41:56][INFO]
+[12/29 00:41:56][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:41:56][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:41:56][INFO]
+[12/29 00:42:03][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/29 00:42:26][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_1/checkpoints'
+[12/29 00:42:35][INFO] Start Fitting...
+[12/29 00:42:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 00:42:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 00:42:36][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 00:42:38][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 00:42:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 00:42:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 00:44:35][INFO] [Exp Name]: finetune_
+[12/29 00:44:35][INFO] [GPU x Batch] = 1 x 1
+[12/29 00:44:35][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:44:35][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:44:35][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/29 00:44:35][INFO]
+[12/29 00:44:35][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:44:35][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:44:35][INFO]
+[12/29 00:44:41][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/29 00:45:02][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_2/checkpoints'
+[12/29 00:45:14][INFO] Start Fitting...
+[12/29 00:45:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 00:45:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 00:45:17][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 00:45:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 00:45:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 00:45:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 00:46:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:46:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:46:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:46:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:46:45][INFO] ✅[FIT][Epoch 0] finished! 01:29→00:00 | loss_epoch=132
+[12/29 00:53:28][INFO] [Exp Name]: finetune_
+[12/29 00:53:28][INFO] [GPU x Batch] = 1 x 1
+[12/29 00:53:28][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:53:28][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:53:28][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/29 00:53:28][INFO]
+[12/29 00:53:28][INFO] [UnityDataset] Found 5 sequences.
+[12/29 00:53:28][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 00:53:28][INFO]
+[12/29 00:53:36][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/29 00:53:59][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_3/checkpoints'
+[12/29 00:54:10][INFO] Start Fitting...
+[12/29 00:54:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 00:54:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 00:54:12][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 00:54:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 00:54:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 00:54:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 00:55:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:55:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:55:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:55:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 00:55:39][INFO] ✅[FIT][Epoch 0] finished! 01:28→00:00 | loss_epoch=132
+[12/29 00:57:13][INFO] End of script.
+[12/29 01:03:29][INFO] [Exp Name]: finetune_
+[12/29 01:03:29][INFO] [GPU x Batch] = 1 x 1
+[12/29 01:03:29][INFO] [UnityDataset] Found 5 sequences.
+[12/29 01:03:29][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 01:03:29][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/29 01:03:29][INFO]
+[12/29 01:03:29][INFO] [UnityDataset] Found 5 sequences.
+[12/29 01:03:29][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 01:03:29][INFO]
+[12/29 01:03:37][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/29 01:04:00][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_4/checkpoints'
+[12/29 01:04:09][INFO] Start Fitting...
+[12/29 01:04:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 01:04:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 01:04:10][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 01:04:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 01:04:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 01:04:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 01:05:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 01:05:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 01:05:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 01:05:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 01:05:36][INFO] ✅[FIT][Epoch 0] finished! 01:26→00:00 | loss_epoch=132
+[12/29 01:11:36][INFO] [Exp Name]: finetune_
+[12/29 01:11:36][INFO] [GPU x Batch] = 1 x 1
+[12/29 01:11:36][INFO] [UnityDataset] Found 5 sequences.
+[12/29 01:11:36][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 01:11:36][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/29 01:11:36][INFO]
+[12/29 01:11:36][INFO] [UnityDataset] Found 5 sequences.
+[12/29 01:11:36][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/29 01:11:36][INFO]
+[12/29 01:11:43][INFO] [PL-Trainer] Loading ckpt: /root/miko/puni/train/GENMO/s050000.ckpt
+[12/29 01:12:08][INFO] [Simple Ckpt Saver]: Save to `outputs/unity_finetune_v1/version_5/checkpoints'
+[12/29 01:12:16][INFO] Start Fitting...
+[12/29 01:12:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 01:12:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 01:12:18][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 01:12:18][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 01:12:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 01:12:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 01:13:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 01:13:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 01:13:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 01:13:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 01:13:44][INFO] ✅[FIT][Epoch 0] finished! 01:27→00:00 | loss_epoch=132
+[12/29 01:15:56][INFO] End of script.
+[12/29 19:01:58][INFO] [Exp Name]: finetune_
+[12/29 19:01:58][INFO] [GPU x Batch] = 1 x 1
+[12/29 19:01:58][INFO] [UnityDataset] Found 2 sequences.
+[12/29 19:01:58][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:01:58][INFO] [Train Dataset][All]: ConcatDataset size=2
+[12/29 19:01:58][INFO]
+[12/29 19:01:58][INFO] [UnityDataset] Found 2 sequences.
+[12/29 19:01:58][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:01:58][INFO]
+[12/29 19:02:06][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/29 19:02:27][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[12/29 19:02:40][INFO] Start Fitting...
+[12/29 19:02:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 19:02:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 19:02:42][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 19:02:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 19:02:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 19:02:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 19:02:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:02:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:02:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:02:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:02:58][INFO] ✅[FIT][Epoch 0] finished! 00:16→00:00 | loss_epoch=168
+[12/29 19:04:38][INFO] End of script.
+[12/29 19:10:14][INFO] [Exp Name]: finetune_
+[12/29 19:10:14][INFO] [GPU x Batch] = 1 x 1
+[12/29 19:10:14][INFO] [UnityDataset] Found 2 sequences.
+[12/29 19:10:14][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:10:14][INFO] [Train Dataset][All]: ConcatDataset size=2
+[12/29 19:10:14][INFO]
+[12/29 19:10:14][INFO] [UnityDataset] Found 2 sequences.
+[12/29 19:10:14][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:10:14][INFO]
+[12/29 19:10:20][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/29 19:10:54][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[12/29 19:11:05][INFO] Start Fitting...
+[12/29 19:11:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 19:11:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 19:11:09][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 19:11:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 19:11:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 19:11:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 19:11:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:11:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:11:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:11:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:11:24][INFO] ✅[FIT][Epoch 0] finished! 00:18→00:00 | loss_epoch=168
+[12/29 19:13:38][INFO] End of script.
+[12/29 19:17:28][INFO] [Exp Name]: finetune_
+[12/29 19:17:28][INFO] [GPU x Batch] = 1 x 1
+[12/29 19:17:28][INFO] [UnityDataset] Found 2 sequences.
+[12/29 19:17:28][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:17:28][INFO] [Train Dataset][All]: ConcatDataset size=2
+[12/29 19:17:28][INFO]
+[12/29 19:17:28][INFO] [UnityDataset] Found 2 sequences.
+[12/29 19:17:28][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:17:28][INFO]
+[12/29 19:17:36][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/29 19:18:14][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[12/29 19:18:22][INFO] Start Fitting...
+[12/29 19:18:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 19:18:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 19:18:24][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 19:18:25][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 19:18:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 19:18:27][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 19:19:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:19:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:19:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:19:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:19:32][INFO] ✅[FIT][Epoch 0] finished! 01:09→00:00 | loss_epoch=168
+[12/29 19:20:28][INFO] End of script.
+[12/29 19:27:24][INFO] [Exp Name]: finetune_
+[12/29 19:27:24][INFO] [GPU x Batch] = 1 x 1
+[12/29 19:27:24][INFO] [UnityDataset] Found 2 sequences.
+[12/29 19:27:24][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:27:24][INFO] [Train Dataset][All]: ConcatDataset size=2
+[12/29 19:27:24][INFO]
+[12/29 19:27:24][INFO] [UnityDataset] Found 2 sequences.
+[12/29 19:27:24][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:27:24][INFO]
+[12/29 19:27:31][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/29 19:28:08][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_3/checkpoints'
+[12/29 19:28:17][INFO] Start Fitting...
+[12/29 19:28:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 19:28:18][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 19:28:18][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 19:28:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 19:28:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 19:28:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 19:29:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:29:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:29:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:29:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:29:20][INFO] ✅[FIT][Epoch 0] finished! 01:03→00:00 | loss_epoch=31.9
+[12/29 19:31:32][INFO] End of script.
+[12/29 19:39:33][INFO] [Exp Name]: finetune_
+[12/29 19:39:33][INFO] [GPU x Batch] = 1 x 1
+[12/29 19:39:33][INFO] [UnityDataset] Found 2 sequences.
+[12/29 19:39:33][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:39:33][INFO] [Train Dataset][All]: ConcatDataset size=2
+[12/29 19:39:33][INFO]
+[12/29 19:39:33][INFO] [UnityDataset] Found 2 sequences.
+[12/29 19:39:33][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:39:33][INFO]
+[12/29 19:39:42][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/29 19:40:16][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_4/checkpoints'
+[12/29 19:40:23][INFO] Start Fitting...
+[12/29 19:40:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 19:40:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 19:40:24][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 19:40:25][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 19:40:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 19:40:27][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 19:41:28][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:41:28][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:41:28][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:41:28][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:41:28][INFO] ✅[FIT][Epoch 0] finished! 01:04→00:00 | loss_epoch=31.9
+[12/29 19:42:58][INFO] End of script.
+[12/29 19:47:07][INFO] [Exp Name]: finetune_
+[12/29 19:47:07][INFO] [GPU x Batch] = 1 x 1
+[12/29 19:47:07][INFO] [UnityDataset] Found 1 sequences.
+[12/29 19:47:07][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:47:07][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/29 19:47:07][INFO]
+[12/29 19:47:07][INFO] [UnityDataset] Found 1 sequences.
+[12/29 19:47:07][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/29 19:47:07][INFO]
+[12/29 19:47:14][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/29 19:47:50][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_5/checkpoints'
+[12/29 19:47:58][INFO] Start Fitting...
+[12/29 19:47:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/29 19:48:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/29 19:48:00][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/29 19:48:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/29 19:48:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/29 19:48:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/29 19:49:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:49:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:49:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:49:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/29 19:49:03][INFO] ✅[FIT][Epoch 0] finished! 01:04→00:00 | loss_epoch=31.6
+[12/29 19:50:03][INFO] End of script.
+[12/30 05:08:55][INFO] [Exp Name]: finetune_
+[12/30 05:08:55][INFO] [GPU x Batch] = 1 x 1
+[12/30 05:08:55][INFO] [UnityDataset] Found 3 sequences.
+[12/30 05:08:55][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[12/30 05:08:55][INFO] [Train Dataset][All]: ConcatDataset size=3
+[12/30 05:08:55][INFO]
+[12/30 05:08:55][INFO] [UnityDataset] Found 3 sequences.
+[12/30 05:08:55][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[12/30 05:08:55][INFO]
+[12/30 05:09:01][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 05:09:50][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_6/checkpoints'
+[12/30 05:10:09][INFO] Start Fitting...
+[12/30 05:10:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 05:10:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 05:10:12][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 05:14:42][INFO] [Exp Name]: finetune_
+[12/30 05:14:42][INFO] [GPU x Batch] = 1 x 1
+[12/30 05:14:42][INFO] [UnityDataset] Found 3 sequences.
+[12/30 05:14:42][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[12/30 05:14:42][INFO] [Train Dataset][All]: ConcatDataset size=3
+[12/30 05:14:42][INFO]
+[12/30 05:14:42][INFO] [UnityDataset] Found 3 sequences.
+[12/30 05:14:42][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[12/30 05:14:42][INFO]
+[12/30 05:14:49][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 05:15:30][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_7/checkpoints'
+[12/30 05:15:42][INFO] Start Fitting...
+[12/30 05:15:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 05:15:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 05:15:45][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 05:15:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 05:15:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 05:15:51][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 05:16:05][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9820 pred=+2.3235 delta(pred-gt)=+1.3415
+[12/30 05:17:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 05:17:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 05:17:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 05:17:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 05:17:00][INFO] ✅[FIT][Epoch 0] finished! 01:16→00:00 | loss_epoch=37
+[12/30 05:19:15][INFO] End of script.
+[12/30 05:33:19][INFO] [Exp Name]: finetune_
+[12/30 05:33:19][INFO] [GPU x Batch] = 1 x 1
+[12/30 05:33:19][INFO] [UnityDataset] Found 1 sequences.
+[12/30 05:33:19][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/30 05:33:19][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/30 05:33:19][INFO]
+[12/30 05:33:19][INFO] [UnityDataset] Found 1 sequences.
+[12/30 05:33:19][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/30 05:33:19][INFO]
+[12/30 05:33:26][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 05:34:04][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_8/checkpoints'
+[12/30 05:34:15][INFO] Start Fitting...
+[12/30 05:34:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 05:34:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 05:34:17][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 05:35:55][INFO] [Exp Name]: finetune_
+[12/30 05:35:55][INFO] [GPU x Batch] = 1 x 1
+[12/30 05:35:56][INFO] [UnityDataset] Found 1 sequences.
+[12/30 05:35:56][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/30 05:35:56][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/30 05:35:56][INFO]
+[12/30 05:35:56][INFO] [UnityDataset] Found 1 sequences.
+[12/30 05:35:56][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/30 05:35:56][INFO]
+[12/30 05:36:01][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 05:36:41][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_9/checkpoints'
+[12/30 05:36:54][INFO] Start Fitting...
+[12/30 05:36:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 05:36:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 05:36:56][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 05:36:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 05:37:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 05:37:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 05:37:15][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9816 delta(pred-gt)=-0.0059
+[12/30 05:38:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 05:38:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 05:38:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 05:38:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 05:38:00][INFO] ✅[FIT][Epoch 0] finished! 01:05→00:00 | loss_epoch=14.1
+[12/30 05:39:34][INFO] End of script.
+[12/30 05:57:45][INFO] [Exp Name]: finetune_
+[12/30 05:57:45][INFO] [GPU x Batch] = 1 x 1
+[12/30 05:57:46][INFO] [UnityDataset] Found 5 sequences.
+[12/30 05:57:46][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/30 05:57:46][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/30 05:57:46][INFO]
+[12/30 05:57:46][INFO] [UnityDataset] Found 5 sequences.
+[12/30 05:57:46][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/30 05:57:46][INFO]
+[12/30 05:57:55][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 05:58:35][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_10/checkpoints'
+[12/30 05:58:47][INFO] Start Fitting...
+[12/30 05:58:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 05:58:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 05:58:49][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 05:58:51][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 05:58:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 05:58:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 05:59:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9894 pred=+0.9799 delta(pred-gt)=-0.0094
+[12/30 06:00:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:00:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:00:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:00:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:00:21][INFO] ✅[FIT][Epoch 0] finished! 01:33→00:00 | loss_epoch=38.8
+[12/30 06:01:56][INFO] End of script.
+[12/30 06:03:41][INFO] [Exp Name]: finetune_
+[12/30 06:03:41][INFO] [GPU x Batch] = 1 x 1
+[12/30 06:03:41][INFO] [UnityDataset] Found 5 sequences.
+[12/30 06:03:41][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/30 06:03:41][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/30 06:03:41][INFO]
+[12/30 06:03:41][INFO] [UnityDataset] Found 5 sequences.
+[12/30 06:03:41][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/30 06:03:41][INFO]
+[12/30 06:03:50][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 06:04:17][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_11/checkpoints'
+[12/30 06:04:29][INFO] Start Fitting...
+[12/30 06:04:30][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 06:04:31][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 06:04:31][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 06:04:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 06:04:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 06:04:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 06:04:50][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9894 pred=+0.9799 delta(pred-gt)=-0.0094
+[12/30 06:05:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:05:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:05:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:05:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:05:58][INFO] ✅[FIT][Epoch 0] finished! 01:29→05:56 | loss_epoch=38.8
+[12/30 06:07:32][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[12/30 06:07:48][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9954 pred=+0.8699 delta(pred-gt)=-0.1255
+[12/30 06:09:05][INFO] ✅[FIT][Epoch 1] finished! 04:35→06:53 | loss_epoch=82.5
+[12/30 06:12:43][INFO] [Exp Name]: finetune_
+[12/30 06:12:43][INFO] [GPU x Batch] = 1 x 1
+[12/30 06:12:43][INFO] [UnityDataset] Found 4 sequences.
+[12/30 06:12:43][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[12/30 06:12:43][INFO] [Train Dataset][All]: ConcatDataset size=4
+[12/30 06:12:43][INFO]
+[12/30 06:12:43][INFO] [UnityDataset] Found 4 sequences.
+[12/30 06:12:43][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[12/30 06:12:43][INFO]
+[12/30 06:12:51][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 06:13:16][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_12/checkpoints'
+[12/30 06:13:27][INFO] Start Fitting...
+[12/30 06:13:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 06:13:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 06:13:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 06:13:29][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 06:13:31][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 06:13:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 06:13:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 06:13:48][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9883 pred=+0.5209 delta(pred-gt)=-0.4674
+[12/30 06:14:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:14:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:14:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:14:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 06:14:49][INFO] ✅[FIT][Epoch 0] finished! 01:21→05:24 | loss_epoch=46.4
+[12/30 06:17:01][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[12/30 06:17:14][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9894 pred=+0.9280 delta(pred-gt)=-0.0614
+[12/30 06:18:24][INFO] ✅[FIT][Epoch 1] finished! 04:55→07:23 | loss_epoch=70.8
+[12/30 07:06:34][INFO] [Exp Name]: finetune_
+[12/30 07:06:34][INFO] [GPU x Batch] = 1 x 1
+[12/30 07:06:34][WARNING] [Train Dataset] Skipping unity due to error: Error in call to target 'genmo.datasets.unity_dataset.UnityDataset':
+FileNotFoundError('Feature dir not found: third_party/GVHMR/processed_dataset/genmo_features')
+full_key: dataset_opts.train.unity
+[12/30 07:22:07][INFO] [Exp Name]: finetune_
+[12/30 07:22:07][INFO] [GPU x Batch] = 1 x 1
+[12/30 07:22:07][INFO] [UnityDataset] Found 1 sequences.
+[12/30 07:22:07][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/30 07:22:07][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/30 07:22:07][INFO]
+[12/30 07:22:07][INFO] [UnityDataset] Found 1 sequences.
+[12/30 07:22:07][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/30 07:22:07][INFO]
+[12/30 07:22:13][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 07:22:36][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[12/30 07:22:46][INFO] Start Fitting...
+[12/30 07:22:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 07:22:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 07:22:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 07:22:48][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 07:22:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 07:22:51][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 07:22:51][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 07:23:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9816 delta(pred-gt)=-0.0059
+[12/30 07:23:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+148.05
+[12/30 07:23:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 07:23:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 07:23:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 07:23:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 07:23:47][INFO] ✅[FIT][Epoch 0] finished! 01:00→04:02 | loss_epoch=14.1
+[12/30 07:25:32][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[12/30 07:25:42][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9948 pred=+0.9890 delta(pred-gt)=-0.0058
+[12/30 07:25:42][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+99.30
+[12/30 07:26:26][INFO] ✅[FIT][Epoch 1] finished! 03:39→05:29 | loss_epoch=21.5
+[12/30 07:28:44][INFO] [Exp Name]: finetune_
+[12/30 07:28:44][INFO] [GPU x Batch] = 1 x 1
+[12/30 07:28:44][INFO] [UnityDataset] Found 1 sequences.
+[12/30 07:28:44][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/30 07:28:44][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/30 07:28:44][INFO]
+[12/30 07:28:44][INFO] [UnityDataset] Found 1 sequences.
+[12/30 07:28:44][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/30 07:28:44][INFO]
+[12/30 07:28:51][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 07:29:13][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[12/30 07:29:22][INFO] Start Fitting...
+[12/30 07:29:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 07:29:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 07:29:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 07:29:23][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 07:29:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 07:29:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 07:29:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 07:29:38][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9816 delta(pred-gt)=-0.0059
+[12/30 07:29:38][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.01689476 -0.20703591 0.01797612] global_orient0_aa(pred)=[-0.02222293 -2.7995787 -0.05008147]
+[12/30 07:29:38][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-11.85,+1.07,+0.92) pred=(-160.44,-2.14,+0.54) pred_vs_gt=(-148.56,-3.06,-1.03)
+[12/30 07:29:38][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+148.05
+[12/30 07:30:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 07:30:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 07:30:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 07:30:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 07:30:22][INFO] ✅[FIT][Epoch 0] finished! 01:00→04:00 | loss_epoch=14.1
+[12/30 07:32:42][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[12/30 07:32:51][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9948 pred=+0.9890 delta(pred-gt)=-0.0058
+[12/30 07:32:51][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03199771 -0.16761649 0.02280195] global_orient0_aa(pred)=[-0.01023587 -1.876039 0.0269159 ]
+[12/30 07:32:51][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-9.59,+1.93,+1.15) pred=(-107.49,+0.77,+1.19) pred_vs_gt=(-97.90,-1.15,-0.15)
+[12/30 07:32:51][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+99.30
+[12/30 08:07:02][INFO] [Exp Name]: finetune_
+[12/30 08:07:02][INFO] [GPU x Batch] = 1 x 1
+[12/30 08:07:02][INFO] [UnityDataset] Found 8 sequences.
+[12/30 08:07:02][INFO] [Train Dataset][9/9]: name=unity, size=8, genmo.datasets.unity_dataset.UnityDataset
+[12/30 08:07:02][INFO] [Train Dataset][All]: ConcatDataset size=8
+[12/30 08:07:02][INFO]
+[12/30 08:07:02][INFO] [UnityDataset] Found 8 sequences.
+[12/30 08:07:02][INFO] [Val Dataset][7/7]: name=unity_val, size=8, genmo.datasets.unity_dataset.UnityDataset
+[12/30 08:07:02][INFO]
+[12/30 08:07:09][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 08:07:32][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[12/30 08:07:43][INFO] Start Fitting...
+[12/30 08:07:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 08:07:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 08:07:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 08:07:45][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 08:07:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 08:07:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 08:08:09][INFO] [VisUnityVal] e000_10_biboo_birthday_speech_poke_large_object root_y0: gt=+0.9749 pred=+0.8970 delta(pred-gt)=-0.0778
+[12/30 08:08:09][INFO] [VisUnityVal] e000_10_biboo_birthday_speech_poke_large_object global_orient0_aa(gt)=[ 0.01488011 1.3818389 -0.01262669] global_orient0_aa(pred)=[-0.6875641 2.6614892 -0.57625073]
+[12/30 08:08:09][INFO] [VisUnityVal] e000_10_biboo_birthday_speech_poke_large_object global_orient0_yxz_deg gt=(+79.18,+1.03,-0.01) pred=(+154.73,+17.35,-32.89) pred_vs_gt=(+87.12,-27.10,-24.98)
+[12/30 08:08:09][INFO] [VisUnityVal] e000_10_biboo_birthday_speech_poke_large_object yaw0_deg(pred_vs_gt)=-87.08
+[12/30 08:09:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 08:09:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 08:09:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 08:09:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 08:09:33][INFO] ✅[FIT][Epoch 0] finished! 01:49→07:16 | loss_epoch=52.3
+[12/30 22:16:05][INFO] [Exp Name]: finetune_
+[12/30 22:16:05][INFO] [GPU x Batch] = 1 x 1
+[12/30 22:16:05][INFO] [UnityDataset] Found 5 sequences.
+[12/30 22:16:05][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/30 22:16:05][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/30 22:16:05][INFO]
+[12/30 22:16:05][INFO] [UnityDataset] Found 5 sequences.
+[12/30 22:16:05][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/30 22:16:05][INFO]
+[12/30 22:26:06][INFO] [Exp Name]: finetune_
+[12/30 22:26:06][INFO] [GPU x Batch] = 1 x 1
+[12/30 22:26:06][INFO] [UnityDataset] Found 5 sequences.
+[12/30 22:26:06][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/30 22:26:06][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/30 22:26:06][INFO]
+[12/30 22:26:06][INFO] [UnityDataset] Found 5 sequences.
+[12/30 22:26:06][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/30 22:26:06][INFO]
+[12/30 22:26:11][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 22:26:42][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_3/checkpoints'
+[12/30 22:26:54][INFO] Start Fitting...
+[12/30 22:26:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 22:26:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 22:26:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 22:26:56][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 22:27:28][INFO] [Exp Name]: finetune_
+[12/30 22:27:28][INFO] [GPU x Batch] = 1 x 1
+[12/30 22:27:28][INFO] [UnityDataset] Found 5 sequences.
+[12/30 22:27:28][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/30 22:27:28][INFO] [Train Dataset][All]: ConcatDataset size=5
+[12/30 22:27:28][INFO]
+[12/30 22:27:28][INFO] [UnityDataset] Found 5 sequences.
+[12/30 22:27:28][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[12/30 22:27:28][INFO]
+[12/30 22:27:37][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 22:27:56][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_4/checkpoints'
+[12/30 22:28:08][INFO] Start Fitting...
+[12/30 22:28:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 22:28:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 22:28:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 22:28:11][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 22:29:56][INFO] [Exp Name]: finetune_
+[12/30 22:29:56][INFO] [GPU x Batch] = 1 x 1
+[12/30 22:29:56][INFO] [UnityDataset] Found 2 sequences.
+[12/30 22:29:56][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/30 22:29:56][INFO] [Train Dataset][All]: ConcatDataset size=2
+[12/30 22:29:56][INFO]
+[12/30 22:29:56][INFO] [UnityDataset] Found 2 sequences.
+[12/30 22:29:56][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/30 22:29:56][INFO]
+[12/30 22:30:02][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 22:30:17][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_5/checkpoints'
+[12/30 22:30:30][INFO] Start Fitting...
+[12/30 22:30:31][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 22:30:31][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 22:30:31][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 22:30:31][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 22:56:38][INFO] [Exp Name]: finetune_
+[12/30 22:56:38][INFO] [GPU x Batch] = 1 x 1
+[12/30 22:56:38][INFO] [UnityDataset] Found 6 sequences.
+[12/30 22:56:38][INFO] [Train Dataset][9/9]: name=unity, size=6, genmo.datasets.unity_dataset.UnityDataset
+[12/30 22:56:38][INFO] [Train Dataset][All]: ConcatDataset size=6
+[12/30 22:56:38][INFO]
+[12/30 22:56:38][INFO] [UnityDataset] Found 6 sequences.
+[12/30 22:56:38][INFO] [Val Dataset][7/7]: name=unity_val, size=6, genmo.datasets.unity_dataset.UnityDataset
+[12/30 22:56:38][INFO]
+[12/30 22:56:44][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 22:57:07][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_6/checkpoints'
+[12/30 22:57:27][INFO] Start Fitting...
+[12/30 22:57:30][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 22:57:30][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 22:57:31][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 22:57:31][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 22:57:34][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 22:57:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 22:57:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 22:57:47][WARNING] [VisUnityVal] Failed to read image: third_party/GVHMR/processed_dataset/images/0_biboo_birthday_speech/img_00699.jpg
+[12/30 22:58:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 22:58:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 22:58:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 22:58:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 22:58:14][INFO] ✅[FIT][Epoch 0] finished! 00:46→03:06 | loss_epoch=28
+[12/30 23:01:18][INFO] [Exp Name]: finetune_
+[12/30 23:01:18][INFO] [GPU x Batch] = 1 x 1
+[12/30 23:01:18][INFO] [UnityDataset] Found 6 sequences.
+[12/30 23:01:18][INFO] [Train Dataset][9/9]: name=unity, size=6, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:01:18][INFO] [Train Dataset][All]: ConcatDataset size=6
+[12/30 23:01:18][INFO]
+[12/30 23:01:18][INFO] [UnityDataset] Found 6 sequences.
+[12/30 23:01:18][INFO] [Val Dataset][7/7]: name=unity_val, size=6, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:01:18][INFO]
+[12/30 23:01:26][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 23:01:45][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[12/30 23:01:57][INFO] Start Fitting...
+[12/30 23:01:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 23:01:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:01:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:01:59][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 23:02:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 23:02:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 23:02:07][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 23:02:13][WARNING] [VisUnityVal] Failed to read image: third_party/GVHMR/processed_dataset/images/0_biboo_birthday_speech/img_00699.jpg
+[12/30 23:02:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:02:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:02:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:02:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:02:41][INFO] ✅[FIT][Epoch 0] finished! 00:43→02:53 | loss_epoch=28
+[12/30 23:09:40][INFO] [Exp Name]: finetune_
+[12/30 23:09:40][INFO] [GPU x Batch] = 1 x 1
+[12/30 23:09:41][INFO] [UnityDataset] Found 6 sequences.
+[12/30 23:09:41][INFO] [Train Dataset][9/9]: name=unity, size=6, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:09:41][INFO] [Train Dataset][All]: ConcatDataset size=6
+[12/30 23:09:41][INFO]
+[12/30 23:09:41][INFO] [UnityDataset] Found 6 sequences.
+[12/30 23:09:41][INFO] [Val Dataset][7/7]: name=unity_val, size=6, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:09:41][INFO]
+[12/30 23:09:49][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 23:10:08][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[12/30 23:10:17][INFO] Start Fitting...
+[12/30 23:10:18][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 23:10:18][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:10:18][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:10:18][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 23:10:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 23:10:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 23:10:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 23:10:44][INFO] [Exp Name]: finetune_
+[12/30 23:10:44][INFO] [GPU x Batch] = 1 x 1
+[12/30 23:10:44][INFO] [UnityDataset] Found 6 sequences.
+[12/30 23:10:44][INFO] [Train Dataset][9/9]: name=unity, size=6, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:10:44][INFO] [Train Dataset][All]: ConcatDataset size=6
+[12/30 23:10:44][INFO]
+[12/30 23:10:44][INFO] [UnityDataset] Found 6 sequences.
+[12/30 23:10:44][INFO] [Val Dataset][7/7]: name=unity_val, size=6, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:10:44][INFO]
+[12/30 23:10:52][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 23:11:04][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[12/30 23:11:11][INFO] Start Fitting...
+[12/30 23:11:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 23:11:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:11:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:11:13][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 23:11:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 23:11:15][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 23:11:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 23:11:27][WARNING] [VisUnityVal] Failed to read image: third_party/GVHMR/processed_dataset/images/0_biboo_birthday_speech/img_00699.jpg
+[12/30 23:11:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:11:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:11:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:11:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:11:53][INFO] ✅[FIT][Epoch 0] finished! 00:40→02:43 | loss_epoch=28
+[12/30 23:29:33][INFO] [Exp Name]: finetune_
+[12/30 23:29:33][INFO] [GPU x Batch] = 1 x 1
+[12/30 23:29:33][INFO] [UnityDataset] Found 2 sequences.
+[12/30 23:29:33][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:29:33][INFO] [Train Dataset][All]: ConcatDataset size=2
+[12/30 23:29:33][INFO]
+[12/30 23:29:33][INFO] [UnityDataset] Found 2 sequences.
+[12/30 23:29:33][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:29:33][INFO]
+[12/30 23:29:39][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 23:30:02][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_3/checkpoints'
+[12/30 23:30:13][INFO] Start Fitting...
+[12/30 23:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 23:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:30:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:30:14][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 23:30:15][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 23:30:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 23:30:18][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 23:30:30][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9930 pred=+0.9643 delta(pred-gt)=-0.0287
+[12/30 23:30:30][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03876931 -0.17480041 0.02509396] global_orient0_aa(pred)=[-0.1090048 -1.7763788 -0.15125035]
+[12/30 23:30:30][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-9.99,+2.34,+1.24) pred=(-101.99,-9.31,-0.53) pred_vs_gt=(-91.73,-11.16,-3.80)
+[12/30 23:30:30][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+94.23
+[12/30 23:46:28][INFO] [Exp Name]: finetune_
+[12/30 23:46:28][INFO] [GPU x Batch] = 1 x 1
+[12/30 23:46:28][INFO] [UnityDataset] Found 2 sequences.
+[12/30 23:46:28][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:46:28][INFO] [Train Dataset][All]: ConcatDataset size=2
+[12/30 23:46:28][INFO]
+[12/30 23:46:28][INFO] [UnityDataset] Found 2 sequences.
+[12/30 23:46:28][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:46:28][INFO]
+[12/30 23:46:34][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 23:46:54][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_4/checkpoints'
+[12/30 23:47:05][INFO] Start Fitting...
+[12/30 23:47:07][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 23:47:07][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:47:07][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:47:07][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 23:47:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 23:47:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 23:47:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 23:47:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9930 pred=+0.9643 delta(pred-gt)=-0.0287
+[12/30 23:47:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03876931 -0.17480041 0.02509396] global_orient0_aa(pred)=[-0.1090048 -1.7763788 -0.15125035]
+[12/30 23:47:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-9.99,+2.34,+1.24) pred=(-101.99,-9.31,-0.53) pred_vs_gt=(-91.73,-11.16,-3.80)
+[12/30 23:47:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+94.23
+[12/30 23:51:42][INFO] [Exp Name]: finetune_
+[12/30 23:51:42][INFO] [GPU x Batch] = 1 x 1
+[12/30 23:51:42][INFO] [UnityDataset] Found 2 sequences.
+[12/30 23:51:42][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:51:42][INFO] [Train Dataset][All]: ConcatDataset size=2
+[12/30 23:51:42][INFO]
+[12/30 23:51:42][INFO] [UnityDataset] Found 2 sequences.
+[12/30 23:51:42][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:51:42][INFO]
+[12/30 23:51:48][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 23:52:04][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_5/checkpoints'
+[12/30 23:52:15][INFO] Start Fitting...
+[12/30 23:52:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 23:52:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:52:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:52:16][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 23:52:18][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 23:52:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 23:52:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 23:52:35][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9930 pred=+0.9643 delta(pred-gt)=-0.0287
+[12/30 23:52:35][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03876931 -0.17480041 0.02509396] global_orient0_aa(pred)=[-0.1090048 -1.7763788 -0.15125035]
+[12/30 23:52:35][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-9.99,+2.34,+1.24) pred=(-101.99,-9.31,-0.53) pred_vs_gt=(-91.73,-11.16,-3.80)
+[12/30 23:52:35][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+94.23
+[12/30 23:53:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:53:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:53:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:53:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:53:24][INFO] ✅[FIT][Epoch 0] finished! 01:08→04:34 | loss_epoch=24.5
+[12/30 23:55:59][INFO] [Exp Name]: finetune_
+[12/30 23:55:59][INFO] [GPU x Batch] = 1 x 1
+[12/30 23:55:59][INFO] [UnityDataset] Found 2 sequences.
+[12/30 23:55:59][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:55:59][INFO] [Train Dataset][All]: ConcatDataset size=2
+[12/30 23:55:59][INFO]
+[12/30 23:55:59][INFO] [UnityDataset] Found 2 sequences.
+[12/30 23:55:59][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:55:59][INFO]
+[12/30 23:56:06][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 23:56:23][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_6/checkpoints'
+[12/30 23:56:35][INFO] Start Fitting...
+[12/30 23:56:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 23:56:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:56:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:56:37][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 23:56:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 23:56:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 23:56:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 23:56:54][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9930 pred=+0.9643 delta(pred-gt)=-0.0287
+[12/30 23:56:54][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03876931 -0.17480041 0.02509396] global_orient0_aa(pred)=[-0.1090048 -1.7763788 -0.15125035]
+[12/30 23:56:54][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-9.99,+2.34,+1.24) pred=(-101.99,-9.31,-0.53) pred_vs_gt=(-91.73,-11.16,-3.80)
+[12/30 23:56:54][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+94.23
+[12/30 23:57:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:57:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:57:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:57:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/30 23:57:45][INFO] ✅[FIT][Epoch 0] finished! 01:09→04:38 | loss_epoch=24.5
+[12/30 23:58:35][INFO] [Exp Name]: finetune_
+[12/30 23:58:35][INFO] [GPU x Batch] = 1 x 1
+[12/30 23:58:35][INFO] [UnityDataset] Found 1 sequences.
+[12/30 23:58:35][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:58:35][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/30 23:58:35][INFO]
+[12/30 23:58:35][INFO] [UnityDataset] Found 1 sequences.
+[12/30 23:58:35][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/30 23:58:35][INFO]
+[12/30 23:58:44][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/30 23:59:06][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_7/checkpoints'
+[12/30 23:59:18][INFO] Start Fitting...
+[12/30 23:59:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/30 23:59:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:59:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/30 23:59:20][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/30 23:59:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/30 23:59:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/30 23:59:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/30 23:59:36][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9944 pred=+0.9685 delta(pred-gt)=-0.0259
+[12/30 23:59:36][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[0.02056097 0.18737577 0.01068786] global_orient0_aa(pred)=[ 0.0337113 -2.8594027 -0.01747983]
+[12/30 23:59:36][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(+10.74,+1.11,+0.72) pred=(-163.84,-0.50,-1.42) pred_vs_gt=(-174.54,-1.98,-1.80)
+[12/30 23:59:36][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+174.27
+[12/31 02:50:01][INFO] [Exp Name]: finetune_
+[12/31 02:50:01][INFO] [GPU x Batch] = 1 x 1
+[12/31 02:50:01][INFO] [UnityDataset] Found 1 sequences.
+[12/31 02:50:01][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 02:50:01][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 02:50:01][INFO]
+[12/31 02:50:01][INFO] [UnityDataset] Found 1 sequences.
+[12/31 02:50:01][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 02:50:01][INFO]
+[12/31 02:50:07][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 02:50:28][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_8/checkpoints'
+[12/31 02:50:41][INFO] Start Fitting...
+[12/31 02:50:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 02:50:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 02:50:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 02:50:42][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 02:50:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 02:50:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 02:50:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 02:50:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9726 delta(pred-gt)=-0.0149
+[12/31 02:50:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.01689476 -0.20703591 0.01797612] global_orient0_aa(pred)=[-0.0321125 -2.8486555 -0.07525362]
+[12/31 02:50:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-11.85,+1.07,+0.92) pred=(-163.30,-3.15,+0.83) pred_vs_gt=(-151.41,-4.11,-0.96)
+[12/31 02:50:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+151.99
+[12/31 02:51:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 02:51:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 02:51:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 02:51:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 02:51:41][INFO] ✅[FIT][Epoch 0] finished! 01:00→04:01 | loss_epoch=12.6
+[12/31 03:10:22][INFO] [Exp Name]: finetune_
+[12/31 03:10:22][INFO] [GPU x Batch] = 1 x 1
+[12/31 03:10:22][INFO] [UnityDataset] Found 1 sequences.
+[12/31 03:10:22][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 03:10:22][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 03:10:22][INFO]
+[12/31 03:10:22][INFO] [UnityDataset] Found 1 sequences.
+[12/31 03:10:22][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 03:10:22][INFO]
+[12/31 03:10:28][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 03:10:51][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_9/checkpoints'
+[12/31 03:11:01][INFO] Start Fitting...
+[12/31 03:11:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 03:11:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 03:11:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 03:11:03][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 03:11:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 03:11:05][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 03:11:05][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 03:11:16][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9726 delta(pred-gt)=-0.0149
+[12/31 03:11:16][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.02646996 2.9343371 -0.02487765] global_orient0_aa(pred)=[-0.03202499 -2.848779 -0.0755955 ]
+[12/31 03:11:16][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+168.15,+1.07,+0.92) pred=(-163.31,-3.16,+0.82) pred_vs_gt=(+28.58,+4.12,+0.97)
+[12/31 03:11:16][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-28.01
+[12/31 03:12:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 03:12:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 03:12:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 03:12:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 03:12:02][INFO] ✅[FIT][Epoch 0] finished! 01:00→04:01 | loss_epoch=14.2
+[12/31 03:16:57][INFO] [Exp Name]: finetune_
+[12/31 03:16:57][INFO] [GPU x Batch] = 1 x 1
+[12/31 03:16:57][INFO] [UnityDataset] Found 1 sequences.
+[12/31 03:16:57][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 03:16:57][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 03:16:57][INFO]
+[12/31 03:16:57][INFO] [UnityDataset] Found 1 sequences.
+[12/31 03:16:57][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 03:16:57][INFO]
+[12/31 03:17:04][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 03:17:24][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_10/checkpoints'
+[12/31 03:17:36][INFO] Start Fitting...
+[12/31 03:17:38][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 03:17:38][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 03:17:38][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 03:17:38][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 03:17:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 03:17:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 03:17:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 03:17:52][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9944 pred=+0.9686 delta(pred-gt)=-0.0258
+[12/31 03:17:52][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[-0.01583315 -2.9540217 0.03045931] global_orient0_aa(pred)=[ 0.03382589 -2.8592563 -0.01758517]
+[12/31 03:17:52][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-169.26,+1.11,+0.72) pred=(-163.83,-0.50,-1.43) pred_vs_gt=(+5.47,+1.99,+1.81)
+[12/31 03:17:52][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=-5.75
+[12/31 03:18:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 03:18:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 03:18:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 03:18:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 03:18:36][INFO] ✅[FIT][Epoch 0] finished! 00:58→03:55 | loss_epoch=23
+[12/31 06:06:15][INFO] [Exp Name]: finetune_
+[12/31 06:06:15][INFO] [GPU x Batch] = 1 x 1
+[12/31 06:06:15][INFO] [UnityDataset] Found 3 sequences.
+[12/31 06:06:15][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[12/31 06:06:15][INFO] [Train Dataset][All]: ConcatDataset size=3
+[12/31 06:06:15][INFO]
+[12/31 06:06:15][INFO] [UnityDataset] Found 3 sequences.
+[12/31 06:06:15][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[12/31 06:06:15][INFO]
+[12/31 06:06:21][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 06:06:49][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_11/checkpoints'
+[12/31 06:07:02][INFO] Start Fitting...
+[12/31 06:07:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 06:07:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 06:07:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 06:07:04][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 06:07:07][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 06:07:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 06:07:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 06:07:22][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9820 pred=+0.9698 delta(pred-gt)=-0.0123
+[12/31 06:07:22][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03590106 -0.17807975 0.02012725] global_orient0_aa(pred)=[-0.08420898 -2.6493108 -0.07150012]
+[12/31 06:07:22][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-10.19,+2.15,+0.96) pred=(-151.99,-3.76,+2.70) pred_vs_gt=(-141.82,-6.13,+0.67)
+[12/31 06:07:22][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+144.65
+[12/31 06:08:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:08:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:08:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:08:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:08:17][INFO] ✅[FIT][Epoch 0] finished! 01:14→04:57 | loss_epoch=41.9
+[12/31 06:13:34][INFO] [Exp Name]: finetune_
+[12/31 06:13:34][INFO] [GPU x Batch] = 1 x 1
+[12/31 06:13:34][INFO] [UnityDataset] Found 3 sequences.
+[12/31 06:13:34][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[12/31 06:13:34][INFO] [Train Dataset][All]: ConcatDataset size=3
+[12/31 06:13:34][INFO]
+[12/31 06:13:34][INFO] [UnityDataset] Found 3 sequences.
+[12/31 06:13:34][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[12/31 06:13:34][INFO]
+[12/31 06:13:43][INFO] [Exp Name]: finetune_
+[12/31 06:13:43][INFO] [GPU x Batch] = 1 x 1
+[12/31 06:13:43][INFO] [UnityDataset] Found 1 sequences.
+[12/31 06:13:43][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 06:13:43][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 06:13:43][INFO]
+[12/31 06:13:43][INFO] [UnityDataset] Found 1 sequences.
+[12/31 06:13:43][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 06:13:43][INFO]
+[12/31 06:13:48][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 06:14:11][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_12/checkpoints'
+[12/31 06:14:22][INFO] Start Fitting...
+[12/31 06:14:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 06:14:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 06:14:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 06:14:26][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 06:14:28][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 06:14:30][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 06:14:30][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 06:14:41][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9726 delta(pred-gt)=-0.0149
+[12/31 06:14:41][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.01689476 -0.20703594 0.01797612] global_orient0_aa(pred)=[-0.03202499 -2.848779 -0.0755955 ]
+[12/31 06:14:41][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-11.85,+1.07,+0.92) pred=(-163.31,-3.16,+0.82) pred_vs_gt=(-151.42,-4.12,-0.97)
+[12/31 06:14:41][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+151.99
+[12/31 06:15:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:15:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:15:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:15:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:15:23][INFO] ✅[FIT][Epoch 0] finished! 01:00→04:00 | loss_epoch=14.3
+[12/31 06:19:20][INFO] [Exp Name]: finetune_
+[12/31 06:19:20][INFO] [GPU x Batch] = 1 x 1
+[12/31 06:19:20][INFO] [UnityDataset] Found 1 sequences.
+[12/31 06:19:20][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 06:19:20][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 06:19:20][INFO]
+[12/31 06:19:20][INFO] [UnityDataset] Found 1 sequences.
+[12/31 06:19:20][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 06:19:20][INFO]
+[12/31 06:19:26][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 06:19:48][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_13/checkpoints'
+[12/31 06:19:59][INFO] Start Fitting...
+[12/31 06:20:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 06:20:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 06:20:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 06:20:01][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 06:20:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 06:20:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 06:20:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 06:20:16][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9726 delta(pred-gt)=-0.0149
+[12/31 06:20:16][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.02646996 2.9343371 -0.02487765] global_orient0_aa(pred)=[-0.03202499 -2.848779 -0.0755955 ]
+[12/31 06:20:16][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+168.15,+1.07,+0.92) pred=(-163.31,-3.16,+0.82) pred_vs_gt=(+28.58,+4.12,+0.97)
+[12/31 06:20:16][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-28.01
+[12/31 06:20:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:20:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:20:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:20:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 06:20:59][INFO] ✅[FIT][Epoch 0] finished! 00:59→03:56 | loss_epoch=14.2
+[12/31 17:32:47][INFO] [Exp Name]: finetune_
+[12/31 17:32:47][INFO] [GPU x Batch] = 1 x 1
+[12/31 17:32:47][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:32:47][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:32:47][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 17:32:47][INFO]
+[12/31 17:32:47][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:32:47][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:32:47][INFO]
+[12/31 17:33:00][INFO] [Exp Name]: finetune_
+[12/31 17:33:00][INFO] [GPU x Batch] = 1 x 1
+[12/31 17:33:00][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:33:00][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:33:00][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 17:33:00][INFO]
+[12/31 17:33:00][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:33:00][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:33:00][INFO]
+[12/31 17:33:06][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 17:33:29][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_14/checkpoints'
+[12/31 17:33:41][INFO] Start Fitting...
+[12/31 17:33:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 17:33:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 17:33:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 17:34:11][INFO] [Exp Name]: finetune_
+[12/31 17:34:11][INFO] [GPU x Batch] = 1 x 1
+[12/31 17:34:11][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:34:11][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:34:11][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 17:34:11][INFO]
+[12/31 17:34:11][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:34:11][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:34:11][INFO]
+[12/31 17:34:17][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 17:34:39][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_15/checkpoints'
+[12/31 17:34:51][INFO] Start Fitting...
+[12/31 17:34:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 17:34:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 17:34:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 17:34:52][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 17:34:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 17:34:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 17:34:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 17:35:08][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9747 delta(pred-gt)=-0.0128
+[12/31 17:35:08][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.02646996 2.9343371 -0.02487765] global_orient0_aa(pred)=[ 0.04138837 -2.8503516 -0.16602226]
+[12/31 17:35:08][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+168.15,+1.07,+0.92) pred=(-163.44,-6.29,-2.58) pred_vs_gt=(+28.54,+6.48,+4.95)
+[12/31 17:35:08][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-23.65
+[12/31 17:35:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:35:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:35:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:35:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:35:53][INFO] ✅[FIT][Epoch 0] finished! 01:01→04:05 | loss_epoch=17.9
+[12/31 17:46:55][INFO] [Exp Name]: finetune_
+[12/31 17:46:55][INFO] [GPU x Batch] = 1 x 1
+[12/31 17:46:55][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:46:55][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:46:55][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 17:46:55][INFO]
+[12/31 17:46:55][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:46:55][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:46:55][INFO]
+[12/31 17:47:03][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 17:47:20][INFO] [Exp Name]: finetune_
+[12/31 17:47:20][INFO] [GPU x Batch] = 1 x 1
+[12/31 17:47:20][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:47:20][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:47:20][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 17:47:20][INFO]
+[12/31 17:47:20][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:47:20][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:47:20][INFO]
+[12/31 17:47:27][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 17:47:39][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[12/31 17:47:50][INFO] Start Fitting...
+[12/31 17:47:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 17:47:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 17:47:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 17:47:52][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 17:47:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 17:47:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 17:47:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 17:48:07][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9747 delta(pred-gt)=-0.0128
+[12/31 17:48:07][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.02646996 2.9343371 -0.02487765] global_orient0_aa(pred)=[ 0.04149618 -2.8503237 -0.16659868]
+[12/31 17:48:07][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+168.15,+1.07,+0.92) pred=(-163.44,-6.32,-2.59) pred_vs_gt=(+28.54,+6.50,+4.96)
+[12/31 17:48:07][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-23.65
+[12/31 17:48:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:48:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:48:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:48:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:48:52][INFO] ✅[FIT][Epoch 0] finished! 01:01→04:06 | loss_epoch=17.9
+[12/31 17:54:30][INFO] [Exp Name]: finetune_
+[12/31 17:54:30][INFO] [GPU x Batch] = 1 x 1
+[12/31 17:54:30][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:54:30][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:54:30][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 17:54:30][INFO]
+[12/31 17:54:30][INFO] [UnityDataset] Found 1 sequences.
+[12/31 17:54:30][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 17:54:30][INFO]
+[12/31 17:54:36][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 17:54:56][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[12/31 17:55:08][INFO] Start Fitting...
+[12/31 17:55:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 17:55:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 17:55:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 17:55:10][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 17:55:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 17:55:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 17:55:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 17:55:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9747 delta(pred-gt)=-0.0128
+[12/31 17:55:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.02646996 2.9343371 -0.02487765] global_orient0_aa(pred)=[ 0.04149618 -2.8503237 -0.16659868]
+[12/31 17:55:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+168.15,+1.07,+0.92) pred=(-163.44,-6.32,-2.59) pred_vs_gt=(+28.54,+6.50,+4.96)
+[12/31 17:55:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-23.65
+[12/31 17:56:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:56:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:56:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:56:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 17:56:10][INFO] ✅[FIT][Epoch 0] finished! 01:01→04:07 | loss_epoch=17.9
+[12/31 18:11:41][INFO] [Exp Name]: finetune_
+[12/31 18:11:41][INFO] [GPU x Batch] = 1 x 1
+[12/31 18:11:41][INFO] [UnityDataset] Found 1 sequences.
+[12/31 18:11:41][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 18:11:41][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 18:11:41][INFO]
+[12/31 18:11:41][INFO] [UnityDataset] Found 1 sequences.
+[12/31 18:11:41][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 18:11:41][INFO]
+[12/31 18:11:47][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 18:12:11][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[12/31 18:12:22][INFO] Start Fitting...
+[12/31 18:12:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 18:12:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 18:12:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 18:12:23][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 18:12:25][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 18:12:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 18:12:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 18:12:37][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9747 delta(pred-gt)=-0.0128
+[12/31 18:12:37][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.02646996 2.9343371 -0.02487765] global_orient0_aa(pred)=[ 0.04149618 -2.8503237 -0.16659868]
+[12/31 18:12:37][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+168.15,+1.07,+0.92) pred=(-163.44,-6.32,-2.59) pred_vs_gt=(+28.54,+6.50,+4.96)
+[12/31 18:12:37][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-23.65
+[12/31 18:13:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 18:13:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 18:13:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 18:13:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 18:13:22][INFO] ✅[FIT][Epoch 0] finished! 01:00→04:00 | loss_epoch=17.9
+[12/31 18:39:25][INFO] [Exp Name]: finetune_
+[12/31 18:39:25][INFO] [GPU x Batch] = 1 x 1
+[12/31 18:39:25][INFO] [UnityDataset] Found 1 sequences.
+[12/31 18:39:25][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 18:39:25][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 18:39:25][INFO]
+[12/31 18:39:25][INFO] [UnityDataset] Found 1 sequences.
+[12/31 18:39:25][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 18:39:25][INFO]
+[12/31 18:39:32][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 18:39:57][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_3/checkpoints'
+[12/31 18:40:08][INFO] Start Fitting...
+[12/31 18:40:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 18:40:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 18:40:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 18:40:10][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 18:40:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 18:40:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 18:40:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 18:40:25][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9747 delta(pred-gt)=-0.0128
+[12/31 18:40:25][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.02646996 2.9343371 -0.02487765] global_orient0_aa(pred)=[ 0.0413059 -2.8506136 -0.16606733]
+[12/31 18:40:25][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+168.15,+1.07,+0.92) pred=(-163.45,-6.30,-2.58) pred_vs_gt=(+28.52,+6.48,+4.95)
+[12/31 18:40:25][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-23.63
+[12/31 18:41:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 18:41:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 18:41:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 18:41:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 18:41:09][INFO] ✅[FIT][Epoch 0] finished! 01:00→04:01 | loss_epoch=17.9
+[12/31 19:17:44][INFO] [Exp Name]: finetune_
+[12/31 19:17:44][INFO] [GPU x Batch] = 1 x 8
+[12/31 19:17:44][INFO] [UnityDataset] Found 1 sequences.
+[12/31 19:17:44][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 19:17:44][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 19:17:44][INFO]
+[12/31 19:17:44][INFO] [UnityDataset] Found 1 sequences.
+[12/31 19:17:44][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 19:17:44][INFO]
+[12/31 19:17:52][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 19:18:06][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_4/checkpoints'
+[12/31 19:18:18][INFO] Start Fitting...
+[12/31 19:18:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 19:18:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/data.py:106: Total length of `DataLoader` across ranks is zero. Please make sure this was your intention.
+
+[12/31 19:18:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/data.py:106: Total length of `CombinedLoader` across ranks is zero. Please make sure this was your intention.
+
+[12/31 19:18:21][INFO] End of script.
+[12/31 19:19:43][INFO] [Exp Name]: finetune_
+[12/31 19:19:43][INFO] [GPU x Batch] = 1 x 1
+[12/31 19:19:43][INFO] [UnityDataset] Found 1 sequences.
+[12/31 19:19:43][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 19:19:43][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 19:19:43][INFO]
+[12/31 19:19:43][INFO] [UnityDataset] Found 1 sequences.
+[12/31 19:19:43][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 19:19:43][INFO]
+[12/31 19:19:50][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 19:20:06][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_5/checkpoints'
+[12/31 19:20:19][INFO] Start Fitting...
+[12/31 19:20:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 19:20:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 19:20:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 19:20:48][INFO] [Exp Name]: finetune_
+[12/31 19:20:48][INFO] [GPU x Batch] = 1 x 1
+[12/31 19:20:48][INFO] [UnityDataset] Found 1 sequences.
+[12/31 19:20:48][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 19:20:48][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 19:20:48][INFO]
+[12/31 19:20:48][INFO] [UnityDataset] Found 1 sequences.
+[12/31 19:20:48][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 19:20:48][INFO]
+[12/31 19:20:54][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 19:21:09][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_6/checkpoints'
+[12/31 19:21:20][INFO] Start Fitting...
+[12/31 19:21:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 19:21:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 19:21:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 19:21:22][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 19:21:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 19:21:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 19:21:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 19:21:39][INFO] [VisUnityVal] e000_1_biboo_birthday_speech_showcasin_herself root_y0: gt=+1.0200 pred=+1.0056 delta(pred-gt)=-0.0144
+[12/31 19:21:39][INFO] [VisUnityVal] e000_1_biboo_birthday_speech_showcasin_herself global_orient0_aa(gt)=[ 0.10563142 -0.6420222 -0.1793047 ] global_orient0_aa(pred)=[ 0.08445861 -2.7046905 0.13478287]
+[12/31 19:21:39][INFO] [VisUnityVal] e000_1_biboo_birthday_speech_showcasin_herself global_orient0_yxz_deg gt=(-37.16,+2.43,-11.47) pred=(-155.31,+6.19,-2.22) pred_vs_gt=(-119.06,-2.51,+9.64)
+[12/31 19:21:39][INFO] [VisUnityVal] e000_1_biboo_birthday_speech_showcasin_herself yaw0_deg(pred_vs_gt)=+111.38
+[12/31 19:22:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 19:22:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 19:22:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 19:22:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[12/31 19:22:23][INFO] ✅[FIT][Epoch 0] finished! 01:01→04:07 | loss_epoch=62.3
+[12/31 19:26:19][INFO] [Exp Name]: finetune_
+[12/31 19:26:19][INFO] [GPU x Batch] = 1 x 1
+[12/31 19:26:19][INFO] [UnityDataset] Found 1 sequences.
+[12/31 19:26:19][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 19:26:19][INFO] [Train Dataset][All]: ConcatDataset size=1
+[12/31 19:26:19][INFO]
+[12/31 19:26:19][INFO] [UnityDataset] Found 1 sequences.
+[12/31 19:26:19][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[12/31 19:26:19][INFO]
+[12/31 19:26:27][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[12/31 19:26:49][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_7/checkpoints'
+[12/31 19:27:00][INFO] Start Fitting...
+[12/31 19:27:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[12/31 19:27:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 19:27:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[12/31 19:27:02][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[12/31 19:27:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[12/31 19:27:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[12/31 19:27:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[12/31 19:27:18][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9875 pred=+0.9747 delta(pred-gt)=-0.0128
+[12/31 19:27:18][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.02646996 2.9343371 -0.02487765] global_orient0_aa(pred)=[ 0.0413059 -2.8506136 -0.16606733]
+[12/31 19:27:18][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+168.15,+1.07,+0.92) pred=(-163.45,-6.30,-2.58) pred_vs_gt=(+28.52,+6.48,+4.95)
+[12/31 19:27:18][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-23.63
+[01/07 02:50:23][INFO] [Exp Name]: finetune_
+[01/07 02:50:23][INFO] [GPU x Batch] = 1 x 1
+[01/07 02:50:23][INFO] [UnityDataset] Found 5 sequences.
+[01/07 02:50:23][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 02:50:23][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/07 02:50:23][INFO]
+[01/07 02:50:23][INFO] [UnityDataset] Found 5 sequences.
+[01/07 02:50:23][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 02:50:23][INFO]
+[01/07 02:50:30][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 02:50:53][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[01/07 02:51:05][INFO] Start Fitting...
+[01/07 02:51:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 02:51:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[01/07 02:51:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[01/07 02:51:59][INFO] [Exp Name]: finetune_
+[01/07 02:51:59][INFO] [GPU x Batch] = 1 x 1
+[01/07 02:51:59][INFO] [UnityDataset] Found 5 sequences.
+[01/07 02:51:59][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 02:51:59][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/07 02:51:59][INFO]
+[01/07 02:51:59][INFO] [UnityDataset] Found 5 sequences.
+[01/07 02:51:59][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 02:51:59][INFO]
+[01/07 02:52:05][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 02:52:24][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[01/07 02:52:35][INFO] Start Fitting...
+[01/07 02:52:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 02:52:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[01/07 02:52:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[01/07 02:52:37][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 02:52:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 02:52:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 02:52:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 02:52:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9894 pred=+0.9714 delta(pred-gt)=-0.0180
+[01/07 02:52:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.02677214 3.0618339 -0.0481008 ] global_orient0_aa(pred)=[-0.02097668 -2.9527793 -0.2730169 ]
+[01/07 02:52:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+175.47,+1.84,+0.93) pred=(-169.85,-10.55,-0.13) pred_vs_gt=(+14.81,+12.27,+2.05)
+[01/07 02:52:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-11.38
+[01/07 02:54:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 02:54:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 02:54:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 02:54:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 02:54:02][INFO] ✅[FIT][Epoch 0] finished! 01:26→05:44 | loss_epoch=46.9
+[01/07 02:54:02][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 02:54:25][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9967 pred=+0.9839 delta(pred-gt)=-0.0128
+[01/07 02:54:25][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.01103197 -1.7438734 0.05666695] global_orient0_aa(pred)=[ 0.13036197 -2.5327241 0.523538 ]
+[01/07 02:54:25][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-99.91,+2.54,+1.41) pred=(-147.51,+23.11,+0.93) pred_vs_gt=(-48.03,-3.03,+20.36)
+[01/07 02:54:25][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+60.80
+[01/07 02:55:35][INFO] ✅[FIT][Epoch 1] finished! 02:58→04:28 | loss_epoch=92.4
+[01/07 02:55:35][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 02:56:09][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself root_y0: gt=+1.0001 pred=+0.7999 delta(pred-gt)=-0.2002
+[01/07 02:56:09][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself global_orient0_aa(gt)=[-0.02601463 -3.0444345 -0.07026858] global_orient0_aa(pred)=[ 0.64247894 -2.3704875 0.29896355]
+[01/07 02:56:09][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself global_orient0_yxz_deg gt=(-174.50,-2.69,+0.85) pred=(-144.73,+21.56,-23.40) pred_vs_gt=(+28.69,-21.61,+26.80)
+[01/07 02:56:09][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself yaw0_deg(pred_vs_gt)=-0.97
+[01/07 02:57:06][INFO] ✅[FIT][Epoch 2] finished! 04:29→02:59 | loss_epoch=146
+[01/07 02:57:06][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/07 02:57:19][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+0.9868 pred=+0.9695 delta(pred-gt)=-0.0173
+[01/07 02:57:19][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[-0.04277359 -2.426832 0.06694556] global_orient0_aa(pred)=[-0.03508484 2.960991 -0.09433255]
+[01/07 02:57:19][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-139.05,+2.11,+2.81) pred=(+169.69,+3.50,-1.67) pred_vs_gt=(-50.97,+1.88,+4.29)
+[01/07 02:57:19][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+67.38
+[01/07 03:04:35][INFO] [Exp Name]: finetune_
+[01/07 03:04:35][INFO] [GPU x Batch] = 1 x 1
+[01/07 03:04:35][INFO] [UnityDataset] Found 4 sequences.
+[01/07 03:04:35][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/07 03:04:35][INFO] [Train Dataset][All]: ConcatDataset size=4
+[01/07 03:04:35][INFO]
+[01/07 03:04:35][INFO] [UnityDataset] Found 4 sequences.
+[01/07 03:04:35][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/07 03:04:35][INFO]
+[01/07 03:04:42][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 03:05:02][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[01/07 03:05:12][INFO] Start Fitting...
+[01/07 03:05:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 03:05:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'train_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[01/07 03:05:15][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/data_connector.py:434: The 'val_dataloader' does not have many workers which may be a bottleneck. Consider increasing the value of the `num_workers` argument` to `num_workers=11` in the `DataLoader` to improve performance.
+
+[01/07 03:05:15][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 03:05:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 03:05:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 03:05:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 03:05:34][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9955 pred=+0.9762 delta(pred-gt)=-0.0193
+[01/07 03:05:34][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04243016 3.105979 -0.04786136] global_orient0_aa(pred)=[-0.01791237 2.1086626 -0.0683528 ]
+[01/07 03:05:34][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+178.02,+1.79,+1.53) pred=(+120.81,+2.39,-2.33) pred_vs_gt=(-57.10,-0.74,+3.84)
+[01/07 03:05:34][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+57.25
+[01/07 03:06:38][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 03:06:38][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 03:06:38][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 03:06:38][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 03:06:38][INFO] ✅[FIT][Epoch 0] finished! 01:25→05:40 | loss_epoch=23.8
+[01/07 03:06:38][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 03:06:57][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9962 pred=+0.9363 delta(pred-gt)=-0.0599
+[01/07 03:06:57][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[-0.02580477 -2.9908826 0.04811292] global_orient0_aa(pred)=[ 0.04689924 -2.3753507 0.0421551 ]
+[01/07 03:06:57][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-171.37,+1.76,+1.12) pred=(-136.16,+2.53,-1.24) pred_vs_gt=(+35.30,-0.41,+2.45)
+[01/07 03:06:57][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-39.78
+[01/07 03:09:21][INFO] ✅[FIT][Epoch 1] finished! 04:08→06:12 | loss_epoch=65.5
+[01/07 03:09:21][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 03:10:17][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself root_y0: gt=+0.9937 pred=+1.0131 delta(pred-gt)=+0.0195
+[01/07 03:10:17][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself global_orient0_aa(gt)=[ 0.0582686 -0.8796101 -0.08903502] global_orient0_aa(pred)=[-3.5505800e-04 2.8833494e+00 1.8360559e-02]
+[01/07 03:10:17][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself global_orient0_yxz_deg gt=(-50.49,+0.82,-5.85) pred=(+165.21,-0.72,+0.08) pred_vs_gt=(-144.44,-5.55,+2.59)
+[01/07 03:10:17][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself yaw0_deg(pred_vs_gt)=+124.35
+[01/07 03:12:25][INFO] ✅[FIT][Epoch 2] finished! 07:12→04:48 | loss_epoch=91.4
+[01/07 03:12:25][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/07 03:12:39][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+0.9744 pred=+1.0335 delta(pred-gt)=+0.0592
+[01/07 03:12:39][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03248583 3.0496407 -0.36503553] global_orient0_aa(pred)=[-0.06297354 -2.789346 0.10817138]
+[01/07 03:12:39][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+176.05,+13.68,+0.75) pred=(-159.85,+3.86,+3.27) pred_vs_gt=(+23.55,+9.95,-1.87)
+[01/07 03:12:39][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-41.13
+[01/07 03:14:34][INFO] ✅[FIT][Epoch 3] finished! 09:20→02:20 | loss_epoch=66.8
+[01/07 03:14:34][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[01/07 03:14:56][INFO] [VisUnityVal] e004_10_biboo_birthday_speech_poke_large_object root_y0: gt=+0.9782 pred=+0.9808 delta(pred-gt)=+0.0026
+[01/07 03:14:56][INFO] [VisUnityVal] e004_10_biboo_birthday_speech_poke_large_object global_orient0_aa(gt)=[-0.04247386 -1.5824367 0.00234429] global_orient0_aa(pred)=[-0.08416452 -2.7888622 0.17107579]
+[01/07 03:14:56][INFO] [VisUnityVal] e004_10_biboo_birthday_speech_poke_large_object global_orient0_yxz_deg gt=(-90.70,-1.45,+1.64) pred=(-159.87,+6.22,+4.56) pred_vs_gt=(-69.49,-3.00,+7.64)
+[01/07 03:14:56][INFO] [VisUnityVal] e004_10_biboo_birthday_speech_poke_large_object yaw0_deg(pred_vs_gt)=+53.29
+[01/07 03:18:20][INFO] [Exp Name]: finetune_
+[01/07 03:18:20][INFO] [GPU x Batch] = 1 x 2
+[01/07 03:18:20][INFO] [UnityDataset] Found 4 sequences.
+[01/07 03:18:20][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/07 03:18:20][INFO] [Train Dataset][All]: ConcatDataset size=4
+[01/07 03:18:20][INFO]
+[01/07 03:18:20][INFO] [UnityDataset] Found 4 sequences.
+[01/07 03:18:20][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/07 03:18:20][INFO]
+[01/07 03:18:27][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 03:18:47][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[01/07 03:18:57][INFO] Start Fitting...
+[01/07 03:18:58][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 03:20:13][INFO] [Exp Name]: finetune_
+[01/07 03:20:13][INFO] [GPU x Batch] = 1 x 2
+[01/07 03:20:13][INFO] [UnityDataset] Found 4 sequences.
+[01/07 03:20:13][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/07 03:20:13][INFO] [Train Dataset][All]: ConcatDataset size=4
+[01/07 03:20:13][INFO]
+[01/07 03:20:13][INFO] [UnityDataset] Found 4 sequences.
+[01/07 03:20:13][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/07 03:20:13][INFO]
+[01/07 03:20:19][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 03:20:33][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_3/checkpoints'
+[01/07 03:20:43][INFO] Start Fitting...
+[01/07 03:20:45][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 03:22:10][INFO] [Exp Name]: finetune_
+[01/07 03:22:10][INFO] [GPU x Batch] = 1 x 2
+[01/07 03:22:10][INFO] [UnityDataset] Found 4 sequences.
+[01/07 03:22:10][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/07 03:22:10][INFO] [Train Dataset][All]: ConcatDataset size=4
+[01/07 03:22:10][INFO]
+[01/07 03:22:10][INFO] [UnityDataset] Found 4 sequences.
+[01/07 03:22:10][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/07 03:22:10][INFO]
+[01/07 03:22:17][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 03:22:28][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_4/checkpoints'
+[01/07 03:22:38][INFO] Start Fitting...
+[01/07 03:22:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 03:22:39][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 03:22:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 03:22:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 03:22:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 03:22:55][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 03:22:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9955 pred=-0.3431 delta(pred-gt)=-1.3386
+[01/07 03:22:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04243016 3.105979 -0.04786136] global_orient0_aa(pred)=[-0.01561811 2.0978932 -0.06504466]
+[01/07 03:22:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+178.02,+1.79,+1.53) pred=(+120.19,+2.30,-2.18) pred_vs_gt=(-57.72,-0.64,+3.69)
+[01/07 03:22:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+57.58
+[01/07 03:23:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 03:23:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 03:23:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 03:23:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 03:23:47][INFO] ✅[FIT][Epoch 0] finished! 01:08→21:43 | loss_epoch=30.1
+[01/07 03:23:47][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 03:23:55][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 03:23:55][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9945 pred=-0.3459 delta(pred-gt)=-1.3404
+[01/07 03:23:55][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03128589 3.0756965 -0.05346771] global_orient0_aa(pred)=[-0.0044097 2.4632363 -0.11970556]
+[01/07 03:23:55][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+176.28,+2.03,+1.10) pred=(+141.18,+4.89,-1.93) pred_vs_gt=(-35.00,-3.05,+2.83)
+[01/07 03:23:55][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+34.77
+[01/07 03:24:45][INFO] ✅[FIT][Epoch 1] finished! 02:07→19:06 | loss_epoch=32.4
+[01/07 03:24:45][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 03:25:01][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 03:25:01][INFO] [VisUnityVal] e002_10_biboo_birthday_speech_poke_large_object root_y0: gt=+0.9782 pred=-0.3642 delta(pred-gt)=-1.3424
+[01/07 03:25:01][INFO] [VisUnityVal] e002_10_biboo_birthday_speech_poke_large_object global_orient0_aa(gt)=[-0.04247386 -1.5824367 0.00234429] global_orient0_aa(pred)=[-0.05981582 -2.621395 0.03088558]
+[01/07 03:25:01][INFO] [VisUnityVal] e002_10_biboo_birthday_speech_poke_large_object global_orient0_yxz_deg gt=(-90.70,-1.45,+1.64) pred=(-150.22,+0.61,+2.78) pred_vs_gt=(-59.53,-1.16,+2.05)
+[01/07 03:25:01][INFO] [VisUnityVal] e002_10_biboo_birthday_speech_poke_large_object yaw0_deg(pred_vs_gt)=+54.61
+[01/07 03:25:46][INFO] ✅[FIT][Epoch 2] finished! 03:08→17:45 | loss_epoch=43.6
+[01/07 03:25:46][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/07 03:25:57][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 03:25:57][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+0.9796 pred=-0.3341 delta(pred-gt)=-1.3137
+[01/07 03:25:57][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.02515982 3.0540893 -0.3200977 ] global_orient0_aa(pred)=[-0.07087284 2.750005 -0.07977168]
+[01/07 03:25:57][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+175.98,+11.98,+0.52) pred=(+157.59,+2.64,-3.47) pred_vs_gt=(-17.52,+9.01,+4.69)
+[01/07 03:25:57][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+12.89
+[01/07 03:27:18][INFO] ✅[FIT][Epoch 3] finished! 04:40→18:40 | loss_epoch=29.9
+[01/07 03:27:18][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[01/07 03:27:38][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 03:27:38][INFO] [VisUnityVal] e004_10_biboo_birthday_speech_poke_large_object root_y0: gt=+0.9782 pred=-0.3630 delta(pred-gt)=-1.3411
+[01/07 03:27:38][INFO] [VisUnityVal] e004_10_biboo_birthday_speech_poke_large_object global_orient0_aa(gt)=[-0.04247386 -1.5824367 0.00234429] global_orient0_aa(pred)=[ 0.05911441 -2.8294032 0.11405515]
+[01/07 03:27:38][INFO] [VisUnityVal] e004_10_biboo_birthday_speech_poke_large_object global_orient0_yxz_deg gt=(-90.70,-1.45,+1.64) pred=(-162.33,+4.87,-1.64) pred_vs_gt=(-71.36,+3.19,+6.37)
+[01/07 03:27:38][INFO] [VisUnityVal] e004_10_biboo_birthday_speech_poke_large_object yaw0_deg(pred_vs_gt)=+56.94
+[01/07 03:28:47][INFO] ✅[FIT][Epoch 4] finished! 06:09→18:28 | loss_epoch=159
+[01/07 04:11:49][INFO] [Exp Name]: finetune_
+[01/07 04:11:49][INFO] [GPU x Batch] = 1 x 2
+[01/07 04:11:49][INFO] [UnityDataset] Found 5 sequences.
+[01/07 04:11:49][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 04:11:49][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/07 04:11:49][INFO]
+[01/07 04:11:49][INFO] [UnityDataset] Found 5 sequences.
+[01/07 04:11:49][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 04:11:49][INFO]
+[01/07 04:11:57][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 04:12:25][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_5/checkpoints'
+[01/07 04:12:37][INFO] Start Fitting...
+[01/07 04:12:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 04:12:39][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 04:12:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 04:12:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 04:12:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 04:12:54][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 04:12:54][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9883 pred=-0.3526 delta(pred-gt)=-1.3409
+[01/07 04:12:54][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.01812689 -1.4423912 0.06422435] global_orient0_aa(pred)=[-0.17699353 -2.7341197 -0.30370054]
+[01/07 04:12:54][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-82.62,+2.94,+1.90) pred=(-158.34,-13.57,+4.80) pred_vs_gt=(-74.75,-4.89,-16.05)
+[01/07 04:12:54][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+77.79
+[01/07 04:13:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 04:13:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 04:13:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 04:13:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 04:13:49][INFO] ✅[FIT][Epoch 0] finished! 01:11→22:32 | loss_epoch=19.1
+[01/07 04:13:49][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 04:13:57][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 04:13:57][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9986 pred=-0.3338 delta(pred-gt)=-1.3324
+[01/07 04:13:57][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.07598312 -1.661053 0.08091637] global_orient0_aa(pred)=[-0.06675979 2.044109 0.01869922]
+[01/07 04:13:57][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.26,+5.65,-0.08) pred=(+117.20,-2.43,-2.26) pred_vs_gt=(-147.67,+2.91,-7.85)
+[01/07 04:13:57][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+148.81
+[01/07 04:14:51][INFO] ✅[FIT][Epoch 1] finished! 02:13→20:02 | loss_epoch=16.5
+[01/07 04:14:51][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 04:15:10][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself root_y0: gt=+0.9937 pred=+0.9720 delta(pred-gt)=-0.0216
+[01/07 04:15:10][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself global_orient0_aa(gt)=[ 0.0582686 -0.8796101 -0.08903502] global_orient0_aa(pred)=[-0.07172708 -1.6887778 -0.06577263]
+[01/07 04:15:10][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself global_orient0_yxz_deg gt=(-50.49,+0.82,-5.85) pred=(-96.85,-4.91,+0.51) pred_vs_gt=(-46.11,-8.54,-0.39)
+[01/07 04:15:10][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself yaw0_deg(pred_vs_gt)=+43.18
+[01/07 04:15:55][INFO] ✅[FIT][Epoch 2] finished! 03:17→18:39 | loss_epoch=14.3
+[01/07 04:15:55][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/07 04:16:05][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 04:16:05][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9858 pred=-0.3305 delta(pred-gt)=-1.3163
+[01/07 04:16:05][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[-0.06670296 -1.3308843 0.07193115] global_orient0_aa(pred)=[-0.04272965 -1.9080312 0.04337485]
+[01/07 04:16:05][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-76.33,-0.43,+5.20) pred=(-109.34,+0.52,+2.94) pred_vs_gt=(-33.00,+2.42,+0.39)
+[01/07 04:16:05][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+32.29
+[01/07 04:16:58][INFO] ✅[FIT][Epoch 3] finished! 04:20→17:22 | loss_epoch=26.4
+[01/07 04:16:58][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[01/07 04:17:11][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 04:17:11][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9936 pred=-0.3494 delta(pred-gt)=-1.3430
+[01/07 04:17:11][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[0.12341892 2.9447625 0.03551891] global_orient0_aa(pred)=[0.02076989 2.9911885 0.2396409 ]
+[01/07 04:17:11][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+168.83,-0.90,+4.89) pred=(+171.80,-9.06,+1.45) pred_vs_gt=(+2.98,+7.34,+4.95)
+[01/07 04:17:11][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+6.87
+[01/07 04:17:59][INFO] ✅[FIT][Epoch 4] finished! 05:21→16:03 | loss_epoch=24
+[01/07 04:19:31][INFO] 🚀[FIT][Epoch 5] Data: unity Experiment: finetune_
+[01/07 04:19:44][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 04:19:44][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9924 pred=-0.3388 delta(pred-gt)=-1.3312
+[01/07 04:19:44][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.01128925 -1.6884888 0.06709146] global_orient0_aa(pred)=[-0.11900912 2.416963 -0.14330322]
+[01/07 04:19:44][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-96.73,+2.92,+1.83) pred=(+138.52,+4.08,-7.18) pred_vs_gt=(-124.02,+8.79,+2.26)
+[01/07 04:19:44][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+121.95
+[01/07 04:20:36][INFO] ✅[FIT][Epoch 5] finished! 07:58→18:36 | loss_epoch=34.6
+[01/07 04:20:36][INFO] 🚀[FIT][Epoch 6] Data: unity Experiment: finetune_
+[01/07 04:20:57][INFO] [VisUnityVal] e006_1_biboo_birthday_speech_showcasin_herself root_y0: gt=+1.0210 pred=+1.0024 delta(pred-gt)=-0.0187
+[01/07 04:20:57][INFO] [VisUnityVal] e006_1_biboo_birthday_speech_showcasin_herself global_orient0_aa(gt)=[ 0.05166753 -0.90287507 -0.06808691] global_orient0_aa(pred)=[-0.00331426 -1.8080392 -0.14359376]
+[01/07 04:20:57][INFO] [VisUnityVal] e006_1_biboo_birthday_speech_showcasin_herself global_orient0_yxz_deg gt=(-51.80,+0.93,-4.64) pred=(-103.53,-5.72,-4.30) pred_vs_gt=(-51.52,-4.37,-5.02)
+[01/07 04:20:57][INFO] [VisUnityVal] e006_1_biboo_birthday_speech_showcasin_herself yaw0_deg(pred_vs_gt)=+46.09
+[01/07 04:21:40][INFO] ✅[FIT][Epoch 6] finished! 09:02→16:47 | loss_epoch=213
+[01/07 04:21:40][INFO] 🚀[FIT][Epoch 7] Data: unity Experiment: finetune_
+[01/07 04:21:57][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 04:21:57][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9889 pred=-0.3441 delta(pred-gt)=-1.3330
+[01/07 04:21:57][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[ 0.04285995 3.0026402 -0.01758261] global_orient0_aa(pred)=[-0.06989681 2.380601 -0.01814651]
+[01/07 04:21:57][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+172.07,+0.78,+1.58) pred=(+136.45,-0.41,-3.20) pred_vs_gt=(-35.58,+0.51,+4.90)
+[01/07 04:21:57][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+32.76
+[01/07 04:59:02][INFO] [Exp Name]: finetune_
+[01/07 04:59:02][INFO] [GPU x Batch] = 1 x 2
+[01/07 04:59:02][INFO] [UnityDataset] Found 5 sequences.
+[01/07 04:59:02][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 04:59:02][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/07 04:59:02][INFO]
+[01/07 04:59:02][INFO] [UnityDataset] Found 5 sequences.
+[01/07 04:59:02][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 04:59:02][INFO]
+[01/07 04:59:08][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 04:59:35][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[01/07 04:59:46][INFO] Start Fitting...
+[01/07 04:59:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 04:59:48][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 04:59:50][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 04:59:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 04:59:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 05:00:05][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 05:00:05][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9883 pred=-0.3526 delta(pred-gt)=-1.3409
+[01/07 05:00:05][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.01812689 -1.4423912 0.06422435] global_orient0_aa(pred)=[-0.17683062 -2.7341459 -0.30365485]
+[01/07 05:00:05][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-82.62,+2.94,+1.90) pred=(-158.34,-13.57,+4.79) pred_vs_gt=(-74.76,-4.88,-16.05)
+[01/07 05:00:05][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+77.79
+[01/07 05:01:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 05:01:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 05:01:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 05:01:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 05:01:01][INFO] ✅[FIT][Epoch 0] finished! 01:15→23:45 | loss_epoch=19.1
+[01/07 05:01:01][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 05:01:10][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 05:01:10][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9986 pred=-0.3338 delta(pred-gt)=-1.3324
+[01/07 05:01:10][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.07598312 -1.661053 0.08091637] global_orient0_aa(pred)=[-0.06669246 2.0441105 0.01866209]
+[01/07 05:01:10][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.26,+5.65,-0.08) pred=(+117.20,-2.42,-2.26) pred_vs_gt=(-147.67,+2.90,-7.85)
+[01/07 05:01:10][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+148.80
+[01/07 05:02:07][INFO] ✅[FIT][Epoch 1] finished! 02:20→21:05 | loss_epoch=16.5
+[01/07 05:02:07][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 05:02:26][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself root_y0: gt=+0.9937 pred=+0.9720 delta(pred-gt)=-0.0216
+[01/07 05:02:26][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself global_orient0_aa(gt)=[ 0.0582686 -0.8796101 -0.08903502] global_orient0_aa(pred)=[-0.07172708 -1.6887778 -0.06577263]
+[01/07 05:02:26][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself global_orient0_yxz_deg gt=(-50.49,+0.82,-5.85) pred=(-96.85,-4.91,+0.51) pred_vs_gt=(-46.11,-8.54,-0.39)
+[01/07 05:02:26][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself yaw0_deg(pred_vs_gt)=+43.18
+[01/07 05:03:13][INFO] ✅[FIT][Epoch 2] finished! 03:26→19:29 | loss_epoch=14.3
+[01/07 05:03:13][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/07 05:03:23][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 05:03:23][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9858 pred=-0.3305 delta(pred-gt)=-1.3163
+[01/07 05:03:23][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[-0.06670296 -1.3308843 0.07193115] global_orient0_aa(pred)=[-0.04245484 -1.9080217 0.04370799]
+[01/07 05:03:23][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-76.33,-0.43,+5.20) pred=(-109.34,+0.54,+2.94) pred_vs_gt=(-33.00,+2.43,+0.41)
+[01/07 05:03:23][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+32.29
+[01/07 05:04:17][INFO] ✅[FIT][Epoch 3] finished! 04:31→18:04 | loss_epoch=26.4
+[01/07 05:04:17][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[01/07 05:04:30][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 05:04:30][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9936 pred=-0.3494 delta(pred-gt)=-1.3430
+[01/07 05:04:30][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[0.12341892 2.9447625 0.03551891] global_orient0_aa(pred)=[0.02076989 2.9911885 0.2396409 ]
+[01/07 05:04:30][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+168.83,-0.90,+4.89) pred=(+171.80,-9.06,+1.45) pred_vs_gt=(+2.98,+7.34,+4.95)
+[01/07 05:04:30][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+6.87
+[01/07 05:05:19][INFO] ✅[FIT][Epoch 4] finished! 05:33→16:39 | loss_epoch=24
+[01/07 05:22:08][INFO] [Exp Name]: finetune_
+[01/07 05:22:08][INFO] [GPU x Batch] = 1 x 2
+[01/07 05:22:08][INFO] [UnityDataset] Found 5 sequences.
+[01/07 05:22:08][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 05:22:08][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/07 05:22:08][INFO]
+[01/07 05:22:08][INFO] [UnityDataset] Found 5 sequences.
+[01/07 05:22:08][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 05:22:08][INFO]
+[01/07 05:22:14][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 05:22:36][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[01/07 05:22:45][INFO] Start Fitting...
+[01/07 05:22:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 05:22:46][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 05:22:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 05:22:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 05:22:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 05:23:00][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 05:23:00][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9883 pred=+0.9889 delta(pred-gt)=+0.0006
+[01/07 05:23:00][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.01812689 -1.4423912 0.06422435] global_orient0_aa(pred)=[-0.17699353 -2.7341197 -0.30370054]
+[01/07 05:23:00][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-82.62,+2.94,+1.90) pred=(-158.34,-13.57,+4.80) pred_vs_gt=(-74.75,-4.89,-16.05)
+[01/07 05:23:00][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+77.79
+[01/07 05:23:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 05:23:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 05:23:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 05:23:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 05:23:57][INFO] ✅[FIT][Epoch 0] finished! 01:11→22:39 | loss_epoch=19.1
+[01/07 05:23:57][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 05:24:06][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 05:24:06][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9986 pred=+1.0077 delta(pred-gt)=+0.0091
+[01/07 05:24:06][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.07598312 -1.661053 0.08091637] global_orient0_aa(pred)=[-0.06675979 2.044109 0.01869922]
+[01/07 05:24:06][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.26,+5.65,-0.08) pred=(+117.20,-2.43,-2.26) pred_vs_gt=(-147.67,+2.91,-7.85)
+[01/07 05:24:06][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+148.81
+[01/07 05:25:02][INFO] ✅[FIT][Epoch 1] finished! 02:16→20:28 | loss_epoch=16.5
+[01/07 05:25:02][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 05:25:22][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself root_y0: gt=+0.9937 pred=+2.3135 delta(pred-gt)=+1.3199
+[01/07 05:25:22][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself global_orient0_aa(gt)=[ 0.0582686 -0.8796101 -0.08903502] global_orient0_aa(pred)=[-0.07172708 -1.6887778 -0.06577263]
+[01/07 05:25:22][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself global_orient0_yxz_deg gt=(-50.49,+0.82,-5.85) pred=(-96.85,-4.91,+0.51) pred_vs_gt=(-46.11,-8.54,-0.39)
+[01/07 05:25:22][INFO] [VisUnityVal] e002_1_biboo_birthday_speech_showcasin_herself yaw0_deg(pred_vs_gt)=+43.18
+[01/07 05:26:09][INFO] ✅[FIT][Epoch 2] finished! 03:23→19:14 | loss_epoch=14.3
+[01/07 05:26:09][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/07 05:26:19][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 05:26:19][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9858 pred=+1.0110 delta(pred-gt)=+0.0252
+[01/07 05:26:19][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[-0.06670296 -1.3308843 0.07193115] global_orient0_aa(pred)=[-0.04272965 -1.9080312 0.04337485]
+[01/07 05:26:19][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-76.33,-0.43,+5.20) pred=(-109.34,+0.52,+2.94) pred_vs_gt=(-33.00,+2.42,+0.39)
+[01/07 05:26:19][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+32.29
+[01/07 05:27:16][INFO] ✅[FIT][Epoch 3] finished! 04:30→18:02 | loss_epoch=26.4
+[01/07 05:27:16][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[01/07 05:27:30][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 05:27:30][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9936 pred=+0.9921 delta(pred-gt)=-0.0015
+[01/07 05:27:30][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[0.12341892 2.9447625 0.03551891] global_orient0_aa(pred)=[0.02069596 2.9913292 0.23965281]
+[01/07 05:27:30][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+168.83,-0.90,+4.89) pred=(+171.80,-9.06,+1.44) pred_vs_gt=(+2.99,+7.34,+4.96)
+[01/07 05:27:30][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+6.87
+[01/07 05:28:21][INFO] ✅[FIT][Epoch 4] finished! 05:35→16:47 | loss_epoch=24
+[01/07 05:29:51][INFO] 🚀[FIT][Epoch 5] Data: unity Experiment: finetune_
+[01/07 05:30:06][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 05:30:06][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9924 pred=+1.0027 delta(pred-gt)=+0.0103
+[01/07 05:30:06][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.01128925 -1.6884888 0.06709146] global_orient0_aa(pred)=[-0.11925301 2.416083 -0.14310336]
+[01/07 05:30:06][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-96.73,+2.92,+1.83) pred=(+138.48,+4.06,-7.19) pred_vs_gt=(-124.07,+8.80,+2.24)
+[01/07 05:30:06][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+121.99
+[01/07 05:31:20][INFO] ✅[FIT][Epoch 5] finished! 08:35→20:02 | loss_epoch=34.7
+[01/07 05:31:20][INFO] 🚀[FIT][Epoch 6] Data: unity Experiment: finetune_
+[01/07 05:31:43][INFO] [VisUnityVal] e006_1_biboo_birthday_speech_showcasin_herself root_y0: gt=+1.0210 pred=+2.3440 delta(pred-gt)=+1.3229
+[01/07 05:31:43][INFO] [VisUnityVal] e006_1_biboo_birthday_speech_showcasin_herself global_orient0_aa(gt)=[ 0.05166753 -0.90287507 -0.06808691] global_orient0_aa(pred)=[-0.00288026 -1.8073184 -0.14513534]
+[01/07 05:31:43][INFO] [VisUnityVal] e006_1_biboo_birthday_speech_showcasin_herself global_orient0_yxz_deg gt=(-51.80,+0.93,-4.64) pred=(-103.49,-5.77,-4.37) pred_vs_gt=(-51.48,-4.35,-5.10)
+[01/07 05:31:43][INFO] [VisUnityVal] e006_1_biboo_birthday_speech_showcasin_herself yaw0_deg(pred_vs_gt)=+46.05
+[01/07 05:32:30][INFO] ✅[FIT][Epoch 6] finished! 09:44→18:06 | loss_epoch=213
+[01/07 05:32:30][INFO] 🚀[FIT][Epoch 7] Data: unity Experiment: finetune_
+[01/07 05:32:48][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 05:32:48][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9889 pred=+0.9974 delta(pred-gt)=+0.0085
+[01/07 05:32:48][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[ 0.04285995 3.0026402 -0.01758261] global_orient0_aa(pred)=[-0.06997269 2.379384 -0.01811996]
+[01/07 05:32:48][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+172.07,+0.78,+1.58) pred=(+136.38,-0.41,-3.20) pred_vs_gt=(-35.65,+0.52,+4.90)
+[01/07 05:32:48][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+32.83
+[01/07 05:42:25][INFO] [Exp Name]: finetune_
+[01/07 05:42:25][INFO] [GPU x Batch] = 1 x 2
+[01/07 05:42:25][INFO] [UnityDataset] Found 5 sequences.
+[01/07 05:42:25][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 05:42:25][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/07 05:42:25][INFO]
+[01/07 05:42:25][INFO] [UnityDataset] Found 5 sequences.
+[01/07 05:42:25][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 05:42:25][INFO]
+[01/07 05:42:31][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 06:05:45][INFO] [Exp Name]: finetune_
+[01/07 06:05:45][INFO] [GPU x Batch] = 1 x 2
+[01/07 06:05:46][INFO] [UnityDataset] Found 5 sequences.
+[01/07 06:05:46][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 06:05:46][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/07 06:05:46][INFO]
+[01/07 06:05:46][INFO] [UnityDataset] Found 5 sequences.
+[01/07 06:05:46][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 06:05:46][INFO]
+[01/07 06:09:18][INFO] [Exp Name]: finetune_
+[01/07 06:09:18][INFO] [GPU x Batch] = 1 x 2
+[01/07 06:09:18][INFO] [UnityDataset] Found 5 sequences.
+[01/07 06:09:18][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 06:09:18][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/07 06:09:18][INFO]
+[01/07 06:09:18][INFO] [UnityDataset] Found 5 sequences.
+[01/07 06:09:18][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 06:09:18][INFO]
+[01/07 06:09:24][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 06:09:48][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[01/07 06:09:58][INFO] Start Fitting...
+[01/07 06:10:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 06:10:00][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 06:10:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 06:10:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 06:10:05][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 06:10:17][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 06:10:17][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9883 pred=+0.9900 delta(pred-gt)=+0.0017
+[01/07 06:10:17][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.01812689 -1.4423912 0.06422435] global_orient0_aa(pred)=[-0.04316897 -2.8188884 0.02261799]
+[01/07 06:10:17][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-82.62,+2.94,+1.90) pred=(-161.52,+0.62,+1.86) pred_vs_gt=(-78.89,-0.25,-2.31)
+[01/07 06:10:17][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+83.51
+[01/07 06:11:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:11:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:11:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:11:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:11:12][INFO] ✅[FIT][Epoch 0] finished! 01:13→23:14 | loss_epoch=47.5
+[01/07 06:11:12][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 06:11:22][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 06:11:22][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9986 pred=+1.0079 delta(pred-gt)=+0.0093
+[01/07 06:11:22][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.07598312 -1.661053 0.08091637] global_orient0_aa(pred)=[ 0.00927268 2.3072116 -0.01125149]
+[01/07 06:11:22][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.26,+5.65,-0.08) pred=(+132.20,+0.64,+0.18) pred_vs_gt=(-132.57,+0.20,-5.02)
+[01/07 06:11:22][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+132.63
+[01/07 06:12:16][INFO] ✅[FIT][Epoch 1] finished! 02:17→20:33 | loss_epoch=35.4
+[01/07 06:12:16][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 06:26:30][INFO] [Exp Name]: finetune_
+[01/07 06:26:30][INFO] [GPU x Batch] = 1 x 2
+[01/07 06:26:30][INFO] [UnityDataset] Found 5 sequences.
+[01/07 06:26:30][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 06:26:30][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/07 06:26:30][INFO]
+[01/07 06:26:30][INFO] [UnityDataset] Found 5 sequences.
+[01/07 06:26:30][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/07 06:26:30][INFO]
+[01/07 06:26:36][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 06:26:56][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_3/checkpoints'
+[01/07 06:27:07][INFO] Start Fitting...
+[01/07 06:27:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 06:27:09][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 06:27:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 06:27:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 06:27:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 06:27:26][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 06:27:26][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9883 pred=+0.9900 delta(pred-gt)=+0.0017
+[01/07 06:27:26][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.01812689 -1.4423912 0.06422435] global_orient0_aa(pred)=[-0.04328784 -2.821324 0.02263819]
+[01/07 06:27:26][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-82.62,+2.94,+1.90) pred=(-161.66,+0.62,+1.86) pred_vs_gt=(-79.03,-0.25,-2.31)
+[01/07 06:27:26][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+83.61
+[01/07 06:28:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:28:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:28:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:28:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:28:19][INFO] ✅[FIT][Epoch 0] finished! 01:11→22:46 | loss_epoch=47.1
+[01/07 06:28:19][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 06:28:29][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 06:28:29][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9986 pred=+1.0079 delta(pred-gt)=+0.0093
+[01/07 06:28:29][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.07598312 -1.661053 0.08091637] global_orient0_aa(pred)=[ 0.00927568 2.3087401 -0.01132225]
+[01/07 06:28:29][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.26,+5.65,-0.08) pred=(+132.28,+0.64,+0.18) pred_vs_gt=(-132.48,+0.20,-5.01)
+[01/07 06:28:29][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+132.57
+[01/07 06:43:22][INFO] [Exp Name]: finetune_
+[01/07 06:43:22][INFO] [GPU x Batch] = 1 x 2
+[01/07 06:43:22][INFO] [UnityDataset] Found 4 sequences.
+[01/07 06:43:22][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/07 06:43:22][INFO] [Train Dataset][All]: ConcatDataset size=4
+[01/07 06:43:22][INFO]
+[01/07 06:43:22][INFO] [UnityDataset] Found 4 sequences.
+[01/07 06:43:22][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/07 06:43:22][INFO]
+[01/07 06:43:28][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 06:43:51][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_4/checkpoints'
+[01/07 06:44:02][INFO] Start Fitting...
+[01/07 06:44:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 06:44:04][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 06:44:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 06:44:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 06:44:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 06:44:21][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 06:44:21][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9883 pred=+0.9900 delta(pred-gt)=+0.0017
+[01/07 06:44:21][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.01766224 -1.4602265 0.06460641] global_orient0_aa(pred)=[-0.04327518 -2.8213243 0.02263274]
+[01/07 06:44:21][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-83.64,+2.94,+1.90) pred=(-161.66,+0.62,+1.86) pred_vs_gt=(-78.01,-0.21,-2.32)
+[01/07 06:44:21][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+82.59
+[01/07 06:45:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:45:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:45:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:45:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 06:45:11][INFO] ✅[FIT][Epoch 0] finished! 01:08→21:37 | loss_epoch=20.8
+[01/07 06:45:11][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 06:45:21][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 06:45:21][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9962 pred=+0.9988 delta(pred-gt)=+0.0026
+[01/07 06:45:21][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.03770693 -1.5080367 0.03796496] global_orient0_aa(pred)=[-0.03051115 -2.3905382 0.0313379 ]
+[01/07 06:45:21][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-86.42,+2.78,+0.10) pred=(-136.97,+0.80,+1.78) pred_vs_gt=(-50.57,-1.80,-1.87)
+[01/07 06:45:21][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+52.43
+[01/07 06:46:10][INFO] ✅[FIT][Epoch 1] finished! 02:06→19:01 | loss_epoch=38.1
+[01/07 06:46:10][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 06:46:23][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 06:46:23][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9854 pred=+0.9854 delta(pred-gt)=-0.0001
+[01/07 06:46:23][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[ 0.06795976 3.1005893 -0.14478147] global_orient0_aa(pred)=[ 0.01995663 2.3589778 -0.02112819]
+[01/07 06:46:23][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+178.00,+5.39,+2.42) pred=(+135.17,+1.22,+0.47) pred_vs_gt=(-42.64,+4.10,+2.10)
+[01/07 06:46:23][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+44.86
+[01/07 06:47:07][INFO] ✅[FIT][Epoch 2] finished! 03:03→17:21 | loss_epoch=31.1
+[01/07 06:47:07][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/07 06:47:16][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 06:47:16][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9907 pred=+0.9899 delta(pred-gt)=-0.0008
+[01/07 06:47:16][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.0054173 -1.387598 0.02672811] global_orient0_aa(pred)=[-0.03627165 -2.8890948 0.02664398]
+[01/07 06:47:16][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-79.50,+1.12,+0.90) pred=(-165.54,+0.86,+1.55) pred_vs_gt=(-86.05,-0.68,-0.14)
+[01/07 06:47:16][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+87.93
+[01/07 06:48:03][INFO] ✅[FIT][Epoch 3] finished! 04:00→16:02 | loss_epoch=53.9
+[01/07 06:48:03][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[01/07 06:48:18][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 06:48:18][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9935 pred=+0.9939 delta(pred-gt)=+0.0004
+[01/07 06:48:18][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[0.12525846 2.9392116 0.04100422] global_orient0_aa(pred)=[ 0.01901468 2.640465 -0.01914267]
+[01/07 06:48:18][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+168.51,-1.10,+4.99) pred=(+151.30,+0.98,+0.58) pred_vs_gt=(-17.32,-2.91,+3.92)
+[01/07 06:48:18][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+19.84
+[01/07 06:49:01][INFO] ✅[FIT][Epoch 4] finished! 04:57→14:53 | loss_epoch=38.2
+[01/07 07:19:43][INFO] [Exp Name]: finetune_
+[01/07 07:19:43][INFO] [GPU x Batch] = 1 x 2
+[01/07 07:19:43][INFO] [UnityDataset] Found 3 sequences.
+[01/07 07:19:43][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[01/07 07:19:43][INFO] [Train Dataset][All]: ConcatDataset size=3
+[01/07 07:19:43][INFO]
+[01/07 07:19:43][INFO] [UnityDataset] Found 3 sequences.
+[01/07 07:19:43][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[01/07 07:19:43][INFO]
+[01/07 07:19:50][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 07:20:08][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_5/checkpoints'
+[01/07 07:20:19][INFO] Start Fitting...
+[01/07 07:20:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 07:20:21][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 07:20:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 07:20:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 07:20:25][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 07:20:35][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:20:35][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+1.0101 pred=+1.0077 delta(pred-gt)=-0.0024
+[01/07 07:20:35][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.00462218 -1.6701621 0.02482653] global_orient0_aa(pred)=[ 0.00486426 2.3575027 -0.00760259]
+[01/07 07:20:35][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.69,+1.09,+0.67) pred=(+135.08,+0.40,+0.07) pred_vs_gt=(-129.23,+0.67,-0.63)
+[01/07 07:20:35][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+130.78
+[01/07 07:21:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:21:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:21:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:21:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:21:23][INFO] ✅[FIT][Epoch 0] finished! 01:03→51:52 | loss_epoch=29.8
+[01/07 07:21:23][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 07:21:33][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:21:33][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9814 pred=+0.9992 delta(pred-gt)=+0.0178
+[01/07 07:21:33][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[-0.04618232 -1.7468934 0.08627693] global_orient0_aa(pred)=[ 0.01040597 2.6122596 -0.02349268]
+[01/07 07:21:33][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-100.09,+1.83,+4.57) pred=(+149.68,+1.08,+0.17) pred_vs_gt=(-110.12,+4.46,+0.03)
+[01/07 07:21:33][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+117.03
+[01/07 07:22:21][INFO] ✅[FIT][Epoch 1] finished! 02:01→48:25 | loss_epoch=22
+[01/07 07:22:21][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 07:22:34][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9881 pred=+2.3167 delta(pred-gt)=+1.3286
+[01/07 07:22:34][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[ 0.05537782 3.0384176 -0.01842257] global_orient0_aa(pred)=[ 0.01720549 2.768193 -0.02550463]
+[01/07 07:22:34][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+174.13,+0.80,+2.05) pred=(+158.62,+1.15,+0.50) pred_vs_gt=(-15.49,-0.51,+1.51)
+[01/07 07:22:34][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+17.07
+[01/07 07:23:18][INFO] ✅[FIT][Epoch 2] finished! 02:58→46:34 | loss_epoch=35.3
+[01/07 07:23:18][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/07 07:23:26][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:23:26][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9650 pred=+0.9907 delta(pred-gt)=+0.0257
+[01/07 07:23:26][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.21672568 -1.6502059 0.27961227] global_orient0_aa(pred)=[ 0.02838628 2.0229151 -0.03337292]
+[01/07 07:23:26][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.19,+17.96,+1.47) pred=(+115.93,+2.08,+0.31) pred_vs_gt=(-149.04,+2.57,-15.72)
+[01/07 07:23:26][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+147.40
+[01/07 07:24:15][INFO] ✅[FIT][Epoch 3] finished! 03:55→45:09 | loss_epoch=28
+[01/07 07:24:15][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[01/07 07:24:28][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 root_y0: gt=+1.0045 pred=+2.3078 delta(pred-gt)=+1.3033
+[01/07 07:24:28][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[0.05175653 2.9739416 0.0085205 ] global_orient0_aa(pred)=[ 0.0156543 2.6022294 -0.02999048]
+[01/07 07:24:28][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+170.42,-0.16,+2.01) pred=(+149.11,+1.40,+0.30) pred_vs_gt=(-21.31,-1.83,+1.42)
+[01/07 07:24:28][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+22.91
+[01/07 07:24:53][INFO] [Exp Name]: finetune_
+[01/07 07:24:53][INFO] [GPU x Batch] = 1 x 2
+[01/07 07:24:53][INFO] [UnityDataset] Found 3 sequences.
+[01/07 07:24:53][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[01/07 07:24:53][INFO] [Train Dataset][All]: ConcatDataset size=3
+[01/07 07:24:53][INFO]
+[01/07 07:24:53][INFO] [UnityDataset] Found 3 sequences.
+[01/07 07:24:53][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[01/07 07:24:53][INFO]
+[01/07 07:24:59][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 07:25:17][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_6/checkpoints'
+[01/07 07:25:28][INFO] Start Fitting...
+[01/07 07:25:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 07:25:29][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 07:25:31][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 07:25:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 07:25:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 07:25:44][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:25:44][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+1.0101 pred=+1.0077 delta(pred-gt)=-0.0024
+[01/07 07:25:44][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.00462218 -1.6701621 0.02482653] global_orient0_aa(pred)=[ 0.00486434 2.357581 -0.00765351]
+[01/07 07:25:44][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.69,+1.09,+0.67) pred=(+135.08,+0.40,+0.07) pred_vs_gt=(-129.22,+0.67,-0.63)
+[01/07 07:25:44][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+130.78
+[01/07 07:26:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:26:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:26:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:26:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:26:33][INFO] ✅[FIT][Epoch 0] finished! 01:05→53:20 | loss_epoch=29.8
+[01/07 07:26:33][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 07:26:43][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:26:43][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9814 pred=+0.9992 delta(pred-gt)=+0.0178
+[01/07 07:26:43][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[-0.04618232 -1.7468934 0.08627693] global_orient0_aa(pred)=[ 0.01036585 2.613009 -0.02345056]
+[01/07 07:26:43][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-100.09,+1.83,+4.57) pred=(+149.72,+1.07,+0.16) pred_vs_gt=(-110.08,+4.47,+0.03)
+[01/07 07:26:43][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+116.98
+[01/07 07:27:32][INFO] ✅[FIT][Epoch 1] finished! 02:03→49:26 | loss_epoch=22
+[01/07 07:27:32][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 07:27:45][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9881 pred=+2.3191 delta(pred-gt)=+1.3310
+[01/07 07:27:45][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[ 0.05537782 3.0384176 -0.01842257] global_orient0_aa(pred)=[ 0.01707162 2.768348 -0.02544959]
+[01/07 07:27:45][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+174.13,+0.80,+2.05) pred=(+158.63,+1.15,+0.49) pred_vs_gt=(-15.49,-0.50,+1.51)
+[01/07 07:27:45][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+17.09
+[01/07 07:28:28][INFO] ✅[FIT][Epoch 2] finished! 03:00→47:01 | loss_epoch=35.2
+[01/07 07:28:28][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/07 07:28:37][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:28:37][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9650 pred=+0.9907 delta(pred-gt)=+0.0257
+[01/07 07:28:37][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.21672568 -1.6502059 0.27961227] global_orient0_aa(pred)=[ 0.028394 2.0232513 -0.0334259]
+[01/07 07:28:37][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.19,+17.96,+1.47) pred=(+115.94,+2.08,+0.30) pred_vs_gt=(-149.02,+2.57,-15.72)
+[01/07 07:28:37][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+147.37
+[01/07 07:29:25][INFO] ✅[FIT][Epoch 3] finished! 03:57→45:27 | loss_epoch=28
+[01/07 07:29:25][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[01/07 07:29:39][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 root_y0: gt=+1.0045 pred=+2.3078 delta(pred-gt)=+1.3033
+[01/07 07:29:39][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[0.05175653 2.9739416 0.0085205 ] global_orient0_aa(pred)=[ 0.01563186 2.5991051 -0.02976354]
+[01/07 07:29:39][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+170.42,-0.16,+2.01) pred=(+148.93,+1.40,+0.30) pred_vs_gt=(-21.49,-1.82,+1.42)
+[01/07 07:29:39][INFO] [VisUnityVal] e004_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+23.08
+[01/07 07:30:23][INFO] ✅[FIT][Epoch 4] finished! 04:54→44:11 | loss_epoch=30
+[01/07 07:31:45][INFO] 🚀[FIT][Epoch 5] Data: unity Experiment: finetune_
+[01/07 07:31:54][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:31:54][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9883 pred=+0.9897 delta(pred-gt)=+0.0014
+[01/07 07:31:54][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.00328026 -1.3843807 0.02146241] global_orient0_aa(pred)=[-0.02728831 -2.9355476 0.02402241]
+[01/07 07:31:54][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-79.32,+0.86,+0.76) pred=(-168.20,+0.82,+1.15) pred_vs_gt=(-88.89,-0.39,+0.03)
+[01/07 07:31:54][INFO] [VisUnityVal] e005_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+88.41
+[01/07 07:32:43][INFO] ✅[FIT][Epoch 5] finished! 07:14→53:08 | loss_epoch=38.9
+[01/07 07:32:43][INFO] 🚀[FIT][Epoch 6] Data: unity Experiment: finetune_
+[01/07 07:32:56][INFO] [VisUnityVal] e006_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9889 pred=+2.3254 delta(pred-gt)=+1.3365
+[01/07 07:32:56][INFO] [VisUnityVal] e006_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[ 0.02824898 2.9115853 -0.10126713] global_orient0_aa(pred)=[0.0068105 2.2914934 0.00237804]
+[01/07 07:32:56][INFO] [VisUnityVal] e006_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+166.94,+4.06,+0.65) pred=(+131.29,+0.03,+0.33) pred_vs_gt=(-35.60,+3.85,+1.22)
+[01/07 07:32:56][INFO] [VisUnityVal] e006_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+32.17
+[01/07 07:33:41][INFO] ✅[FIT][Epoch 6] finished! 08:12→50:24 | loss_epoch=105
+[01/07 07:33:41][INFO] 🚀[FIT][Epoch 7] Data: unity Experiment: finetune_
+[01/07 07:33:53][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9892 pred=+2.3183 delta(pred-gt)=+1.3291
+[01/07 07:33:53][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[ 0.03620023 2.966415 -0.00844817] global_orient0_aa(pred)=[0.01044766 2.3912926 0.01862939]
+[01/07 07:33:53][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+169.98,+0.45,+1.36) pred=(+137.01,-0.60,+0.74) pred_vs_gt=(-32.96,+0.92,+0.79)
+[01/07 07:33:53][INFO] [VisUnityVal] e007_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+30.24
+[01/07 07:34:39][INFO] ✅[FIT][Epoch 7] finished! 09:10→48:11 | loss_epoch=137
+[01/07 07:34:39][INFO] 🚀[FIT][Epoch 8] Data: unity Experiment: finetune_
+[01/07 07:34:52][INFO] [VisUnityVal] e008_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9876 pred=+2.3286 delta(pred-gt)=+1.3410
+[01/07 07:34:52][INFO] [VisUnityVal] e008_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[-0.00820042 -3.131931 0.03640478] global_orient0_aa(pred)=[0.01706261 2.7117357 0.0072985 ]
+[01/07 07:34:52][INFO] [VisUnityVal] e008_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(-179.46,+1.33,+0.31) pred=(+155.37,-0.14,+0.75) pred_vs_gt=(-25.18,+1.47,-0.46)
+[01/07 07:34:52][INFO] [VisUnityVal] e008_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+22.37
+[01/07 07:36:29][INFO] [Exp Name]: finetune_
+[01/07 07:36:29][INFO] [GPU x Batch] = 1 x 2
+[01/07 07:36:29][INFO] [UnityDataset] Found 3 sequences.
+[01/07 07:36:29][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[01/07 07:36:29][INFO] [Train Dataset][All]: ConcatDataset size=3
+[01/07 07:36:29][INFO]
+[01/07 07:36:29][INFO] [UnityDataset] Found 3 sequences.
+[01/07 07:36:29][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[01/07 07:36:29][INFO]
+[01/07 07:36:36][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 07:37:08][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_7/checkpoints'
+[01/07 07:37:21][INFO] Start Fitting...
+[01/07 07:37:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 07:37:23][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 07:37:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 07:37:25][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 07:37:25][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 07:37:35][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:37:36][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+1.0101 pred=+1.0077 delta(pred-gt)=-0.0024
+[01/07 07:37:36][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.00462218 -1.6701621 0.02482653] global_orient0_aa(pred)=[ 0.00486426 2.3575027 -0.00760259]
+[01/07 07:37:36][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.69,+1.09,+0.67) pred=(+135.08,+0.40,+0.07) pred_vs_gt=(-129.23,+0.67,-0.63)
+[01/07 07:37:36][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+130.78
+[01/07 07:38:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:38:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:38:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:38:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:38:24][INFO] ✅[FIT][Epoch 0] finished! 01:02→51:19 | loss_epoch=29.8
+[01/07 07:38:24][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 07:38:33][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:38:33][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9814 pred=+0.9992 delta(pred-gt)=+0.0178
+[01/07 07:38:33][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[-0.04618232 -1.7468934 0.08627693] global_orient0_aa(pred)=[ 0.01040597 2.6122596 -0.02349268]
+[01/07 07:38:33][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-100.09,+1.83,+4.57) pred=(+149.68,+1.08,+0.17) pred_vs_gt=(-110.12,+4.46,+0.03)
+[01/07 07:38:33][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+117.03
+[01/07 07:39:23][INFO] ✅[FIT][Epoch 1] finished! 02:01→48:35 | loss_epoch=22
+[01/07 07:39:23][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 07:39:35][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9881 pred=+2.3167 delta(pred-gt)=+1.3286
+[01/07 07:39:35][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[ 0.05537782 3.0384176 -0.01842257] global_orient0_aa(pred)=[ 0.01720549 2.768193 -0.02550463]
+[01/07 07:39:35][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+174.13,+0.80,+2.05) pred=(+158.62,+1.15,+0.50) pred_vs_gt=(-15.49,-0.51,+1.51)
+[01/07 07:39:35][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+17.07
+[01/07 07:40:18][INFO] ✅[FIT][Epoch 2] finished! 02:57→46:16 | loss_epoch=35.3
+[01/07 07:40:18][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/07 07:40:27][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:40:27][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9650 pred=+0.9907 delta(pred-gt)=+0.0257
+[01/07 07:40:27][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.21672568 -1.6502059 0.27961227] global_orient0_aa(pred)=[ 0.02838628 2.0229151 -0.03337292]
+[01/07 07:40:27][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.19,+17.96,+1.47) pred=(+115.93,+2.08,+0.31) pred_vs_gt=(-149.04,+2.57,-15.72)
+[01/07 07:40:27][INFO] [VisUnityVal] e003_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+147.40
+[01/07 07:44:36][INFO] [Exp Name]: finetune_
+[01/07 07:44:36][INFO] [GPU x Batch] = 1 x 2
+[01/07 07:44:37][INFO] [UnityDataset] Found 3 sequences.
+[01/07 07:44:37][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[01/07 07:44:37][INFO] [Train Dataset][All]: ConcatDataset size=3
+[01/07 07:44:37][INFO]
+[01/07 07:44:37][INFO] [UnityDataset] Found 3 sequences.
+[01/07 07:44:37][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[01/07 07:44:37][INFO]
+[01/07 07:44:43][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/07 07:45:09][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_8/checkpoints'
+[01/07 07:45:22][INFO] Start Fitting...
+[01/07 07:45:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/07 07:45:24][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/07 07:45:26][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/07 07:45:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/07 07:45:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/07 07:45:40][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:45:40][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 root_y0: gt=+1.0101 pred=+1.0065 delta(pred-gt)=-0.0037
+[01/07 07:45:40][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[ 0.00462218 -1.6701621 0.02482653] global_orient0_aa(pred)=[-0.12235975 2.1420956 0.09738921]
+[01/07 07:45:40][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-95.69,+1.09,+0.67) pred=(+123.10,-6.76,-2.88) pred_vs_gt=(-141.66,+4.28,-7.48)
+[01/07 07:45:40][INFO] [VisUnityVal] e000_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+144.41
+[01/07 07:46:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:46:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:46:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:46:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/07 07:46:32][INFO] ✅[FIT][Epoch 0] finished! 01:09→56:26 | loss_epoch=14.5
+[01/07 07:46:32][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/07 07:46:42][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/07 07:46:42][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 root_y0: gt=+0.9814 pred=+0.9986 delta(pred-gt)=+0.0172
+[01/07 07:46:42][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_aa(gt)=[-0.04618232 -1.7468934 0.08627693] global_orient0_aa(pred)=[-0.05748235 2.5460894 0.00647037]
+[01/07 07:46:42][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 global_orient0_yxz_deg gt=(-100.09,+1.83,+4.57) pred=(+145.93,-0.99,-2.28) pred_vs_gt=(-114.04,+7.24,-1.58)
+[01/07 07:46:42][INFO] [VisUnityVal] e001_100_biboo_birthday_speech_explosion_1 yaw0_deg(pred_vs_gt)=+122.58
+[01/07 07:47:32][INFO] ✅[FIT][Epoch 1] finished! 02:09→51:47 | loss_epoch=12.5
+[01/07 07:47:32][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/07 07:47:46][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 root_y0: gt=+0.9881 pred=+2.3204 delta(pred-gt)=+1.3322
+[01/07 07:47:46][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 global_orient0_aa(gt)=[ 0.05537782 3.0384176 -0.01842257] global_orient0_aa(pred)=[0.01367732 2.7470856 0.06572023]
+[01/07 07:47:46][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 global_orient0_yxz_deg gt=(+174.13,+0.80,+2.05) pred=(+157.41,-2.53,+1.08) pred_vs_gt=(-16.70,+3.21,+1.31)
+[01/07 07:47:46][INFO] [VisUnityVal] e002_102_biboo_birthday_speech_explosion_3 yaw0_deg(pred_vs_gt)=+10.43
+[01/08 04:10:04][INFO] [Exp Name]: finetune_
+[01/08 04:10:04][INFO] [GPU x Batch] = 1 x 2
+[01/08 04:10:04][INFO] [UnityDataset] Found 13 sequences.
+[01/08 04:10:04][INFO] [Train Dataset][9/9]: name=unity, size=13, genmo.datasets.unity_dataset.UnityDataset
+[01/08 04:10:04][INFO] [Train Dataset][All]: ConcatDataset size=13
+[01/08 04:10:04][INFO]
+[01/08 04:10:04][INFO] [UnityDataset] Found 13 sequences.
+[01/08 04:10:04][INFO] [Val Dataset][7/7]: name=unity_val, size=13, genmo.datasets.unity_dataset.UnityDataset
+[01/08 04:10:04][INFO]
+[01/08 04:10:11][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 04:10:35][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[01/08 04:10:45][INFO] Start Fitting...
+[01/08 04:10:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 04:10:50][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 04:10:51][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 04:10:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 04:10:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 04:11:06][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 04:11:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9922 pred=+1.0038 delta(pred-gt)=+0.0117
+[01/08 04:11:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03738328 3.0321708 -0.06908838] global_orient0_aa(pred)=[ 0.07231037 -2.8579803 0.05438893]
+[01/08 04:11:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+173.82,+2.68,+1.27) pred=(-163.88,+2.54,-2.54) pred_vs_gt=(+22.47,-0.28,+3.80)
+[01/08 04:11:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-14.76
+[01/08 04:12:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 04:12:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 04:12:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 04:12:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 04:12:23][INFO] ✅[FIT][Epoch 0] finished! 01:36→1:18:58 | loss_epoch=25.2
+[01/08 04:12:23][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/08 04:12:42][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 04:12:42][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9919 pred=+0.9901 delta(pred-gt)=-0.0018
+[01/08 04:12:42][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[-0.03146022 -3.0983016 0.0466447 ] global_orient0_aa(pred)=[ 0.01753556 -2.3668067 0.05622979]
+[01/08 04:12:42][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-177.53,+1.70,+1.20) pred=(-135.63,+2.63,+0.22) pred_vs_gt=(+41.93,-0.89,+1.01)
+[01/08 04:12:42][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-23.27
+[01/08 04:13:51][INFO] ✅[FIT][Epoch 1] finished! 03:04→1:13:56 | loss_epoch=31.1
+[01/08 04:13:51][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/08 04:14:33][INFO] [VisUnityVal] e002_110_biboo_birthday_speech_showcasin_herself_3 root_y0: gt=+0.9952 pred=+2.3044 delta(pred-gt)=+1.3091
+[01/08 04:14:33][INFO] [VisUnityVal] e002_110_biboo_birthday_speech_showcasin_herself_3 global_orient0_aa(gt)=[-0.0009854 -0.06169528 -0.00223152] global_orient0_aa(pred)=[-0.14286354 2.540022 0.0518393 ]
+[01/08 04:14:33][INFO] [VisUnityVal] e002_110_biboo_birthday_speech_showcasin_herself_3 global_orient0_yxz_deg gt=(-3.53,-0.06,-0.13) pred=(+145.92,-3.94,-5.23) pred_vs_gt=(+149.44,-3.56,-5.33)
+[01/08 04:14:33][INFO] [VisUnityVal] e002_110_biboo_birthday_speech_showcasin_herself_3 yaw0_deg(pred_vs_gt)=-152.73
+[01/08 04:15:26][INFO] ✅[FIT][Epoch 2] finished! 04:40→1:13:14 | loss_epoch=35.7
+[01/08 04:15:26][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/08 04:15:45][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 04:15:45][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+0.9984 pred=+0.9962 delta(pred-gt)=-0.0022
+[01/08 04:15:45][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04151112 3.0344355 -0.05964002] global_orient0_aa(pred)=[ 0.04018781 -2.010865 -0.0707708 ]
+[01/08 04:15:45][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+173.94,+2.33,+1.44) pred=(-115.21,-1.84,-3.46) pred_vs_gt=(+71.04,+3.62,+5.32)
+[01/08 04:15:45][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-70.01
+[01/08 05:22:22][INFO] [Exp Name]: finetune_
+[01/08 05:22:22][INFO] [GPU x Batch] = 1 x 2
+[01/08 05:22:22][INFO] [UnityDataset] Found 2 sequences.
+[01/08 05:22:22][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 05:22:22][INFO] [Train Dataset][All]: ConcatDataset size=2
+[01/08 05:22:22][INFO]
+[01/08 05:22:22][INFO] [UnityDataset] Found 2 sequences.
+[01/08 05:22:22][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 05:22:22][INFO]
+[01/08 05:22:29][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 05:22:52][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[01/08 05:23:02][INFO] Start Fitting...
+[01/08 05:23:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 05:23:04][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 05:23:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 05:23:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 05:23:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 05:23:21][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 05:23:22][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9930 pred=+1.0044 delta(pred-gt)=+0.0113
+[01/08 05:23:22][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03662891 3.003706 -0.05900013] global_orient0_aa(pred)=[ 0.03848156 -2.8835566 0.08368719]
+[01/08 05:23:22][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+172.17,+2.34,+1.24) pred=(-165.33,+3.46,-1.08) pred_vs_gt=(+22.60,-1.44,+2.14)
+[01/08 05:23:22][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-18.54
+[01/08 05:24:05][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 05:24:05][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 05:24:05][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 05:24:05][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 05:24:05][INFO] ✅[FIT][Epoch 0] finished! 01:02→51:06 | loss_epoch=13.9
+[01/08 05:24:05][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/08 05:24:16][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 05:24:16][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9988 pred=+0.9995 delta(pred-gt)=+0.0007
+[01/08 05:24:16][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03469782 2.991596 -0.04141502] global_orient0_aa(pred)=[-0.00907645 2.2415967 -0.03054151]
+[01/08 05:24:16][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+171.45,+1.68,+1.20) pred=(+128.43,+1.08,-0.99) pred_vs_gt=(-42.96,+0.26,+2.25)
+[01/08 05:24:16][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+48.66
+[01/08 05:28:34][INFO] [Exp Name]: finetune_
+[01/08 05:28:34][INFO] [GPU x Batch] = 1 x 2
+[01/08 05:28:35][INFO] [UnityDataset] Found 2 sequences.
+[01/08 05:28:35][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 05:28:35][INFO] [Train Dataset][All]: ConcatDataset size=2
+[01/08 05:28:35][INFO]
+[01/08 05:28:35][INFO] [UnityDataset] Found 2 sequences.
+[01/08 05:28:35][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 05:28:35][INFO]
+[01/08 05:28:41][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 05:29:04][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[01/08 05:29:16][INFO] Start Fitting...
+[01/08 05:29:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 05:29:18][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 05:29:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 05:29:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 05:29:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 05:29:35][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 05:29:35][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9930 pred=+1.0044 delta(pred-gt)=+0.0113
+[01/08 05:29:35][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03662891 3.003706 -0.05900013] global_orient0_aa(pred)=[ 0.03839137 -2.8835602 0.08367579]
+[01/08 05:29:35][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+172.17,+2.34,+1.24) pred=(-165.33,+3.46,-1.08) pred_vs_gt=(+22.60,-1.43,+2.14)
+[01/08 05:29:35][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-18.54
+[01/08 05:30:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 05:30:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 05:30:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 05:30:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 05:30:19][INFO] ✅[FIT][Epoch 0] finished! 01:02→51:02 | loss_epoch=13.9
+[01/08 05:30:19][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/08 05:30:27][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 05:30:27][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9988 pred=+0.9995 delta(pred-gt)=+0.0007
+[01/08 05:30:27][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03469782 2.991596 -0.04141502] global_orient0_aa(pred)=[-0.00915245 2.2415955 -0.03067119]
+[01/08 05:30:27][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+171.45,+1.68,+1.20) pred=(+128.43,+1.09,-0.99) pred_vs_gt=(-42.96,+0.25,+2.26)
+[01/08 05:30:27][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+48.66
+[01/08 10:08:28][INFO] [Exp Name]: finetune_
+[01/08 10:08:28][INFO] [GPU x Batch] = 1 x 2
+[01/08 10:08:28][INFO] [UnityDataset] Found 5 sequences.
+[01/08 10:08:28][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/08 10:08:28][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/08 10:08:28][INFO]
+[01/08 10:08:28][INFO] [UnityDataset] Found 5 sequences.
+[01/08 10:08:28][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/08 10:08:28][INFO]
+[01/08 10:08:34][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 10:08:42][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_3/checkpoints'
+[01/08 10:08:54][INFO] Start Fitting...
+[01/08 10:08:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 10:08:56][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 10:08:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 10:08:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 10:09:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 10:09:11][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 10:09:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.7836 pred=+0.7598 delta(pred-gt)=-0.0239
+[01/08 10:09:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[-3.107824 0.04372303 0.13508186] global_orient0_aa(pred)=[ 0.02059126 -2.8832476 0.05009659]
+[01/08 10:09:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-175.05,-1.82,-178.47) pred=(-165.23,+2.06,-0.55) pred_vs_gt=(+19.78,-0.43,-177.95)
+[01/08 10:09:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+171.29
+[01/08 10:16:16][INFO] [Exp Name]: finetune_
+[01/08 10:16:16][INFO] [GPU x Batch] = 1 x 2
+[01/08 10:16:16][INFO] [UnityDataset] Found 5 sequences.
+[01/08 10:16:16][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/08 10:16:16][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/08 10:16:16][INFO]
+[01/08 10:16:16][INFO] [UnityDataset] Found 5 sequences.
+[01/08 10:16:16][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/08 10:16:16][INFO]
+[01/08 10:16:22][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 10:16:39][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_4/checkpoints'
+[01/08 10:17:01][INFO] Start Fitting...
+[01/08 10:17:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 10:17:03][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 10:17:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 10:17:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 10:17:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 10:17:28][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 10:17:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9981 delta(pred-gt)=+0.0026
+[01/08 10:17:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04296682 3.0540771 -0.04666773] global_orient0_aa(pred)=[ 0.02067284 -2.883098 0.05018289]
+[01/08 10:17:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+175.05,+1.82,+1.53) pred=(-165.23,+2.07,-0.55) pred_vs_gt=(+19.79,-0.43,+2.06)
+[01/08 10:17:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-16.26
+[01/08 10:18:55][INFO] [Exp Name]: finetune_
+[01/08 10:18:55][INFO] [GPU x Batch] = 1 x 2
+[01/08 10:18:55][INFO] [UnityDataset] Found 5 sequences.
+[01/08 10:18:55][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/08 10:18:55][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/08 10:18:55][INFO]
+[01/08 10:18:55][INFO] [UnityDataset] Found 5 sequences.
+[01/08 10:18:55][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/08 10:18:55][INFO]
+[01/08 10:19:02][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 10:19:18][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_5/checkpoints'
+[01/08 10:19:30][INFO] Start Fitting...
+[01/08 10:19:31][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 10:19:31][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 10:19:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 10:19:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 10:19:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 10:19:48][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 10:19:48][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9981 delta(pred-gt)=+0.0026
+[01/08 10:19:49][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04296682 3.0540771 -0.04666773] global_orient0_aa(pred)=[ 0.02067284 -2.883098 0.05018289]
+[01/08 10:19:49][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+175.05,+1.82,+1.53) pred=(-165.23,+2.07,-0.55) pred_vs_gt=(+19.79,-0.43,+2.06)
+[01/08 10:19:49][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-16.26
+[01/08 10:20:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 10:20:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 10:20:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 10:20:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 10:20:46][INFO] ✅[FIT][Epoch 0] finished! 01:15→1:01:56 | loss_epoch=55.8
+[01/08 10:20:46][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/08 10:20:56][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 10:20:56][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9902 pred=+1.0115 delta(pred-gt)=+0.0213
+[01/08 10:20:56][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04027311 3.0210712 -0.16103524] global_orient0_aa(pred)=[ 0.0195952 -2.1532001 0.02625812]
+[01/08 10:20:56][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+173.41,+6.17,+1.17) pred=(-123.38,+1.52,-0.22) pred_vs_gt=(+63.38,+4.46,+1.93)
+[01/08 10:20:56][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-56.11
+[01/08 10:51:14][INFO] [Exp Name]: finetune_
+[01/08 10:51:14][INFO] [GPU x Batch] = 1 x 2
+[01/08 10:51:14][INFO] [UnityDataset] Found 5 sequences.
+[01/08 10:51:14][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/08 10:51:14][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/08 10:51:14][INFO]
+[01/08 10:51:14][INFO] [UnityDataset] Found 5 sequences.
+[01/08 10:51:14][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/08 10:51:14][INFO]
+[01/08 10:51:21][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 10:51:28][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_6/checkpoints'
+[01/08 10:51:40][INFO] Start Fitting...
+[01/08 10:51:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 10:51:41][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 10:51:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 10:51:44][INFO] [LossBreakdown] body_pose=1.6208 betas=0.7734 go_c=1.3910 go_gv=7.2930 transl_vel=0.2319
+[01/08 10:51:44][INFO] [LossBreakdown] body_pose=0.1255 betas=0.1341 go_c=0.0672 go_gv=6.1465 transl_vel=0.0916
+[01/08 10:51:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 10:51:46][INFO] [LossBreakdown] body_pose=1.0020 betas=0.5506 go_c=0.2316 go_gv=0.0373 transl_vel=0.2547
+[01/08 10:51:46][INFO] [LossBreakdown] body_pose=0.4181 betas=0.2614 go_c=0.2051 go_gv=0.0128 transl_vel=0.2408
+[01/08 10:51:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 10:51:58][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 10:51:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9981 delta(pred-gt)=+0.0026
+[01/08 10:51:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04296682 3.0540771 -0.04666773] global_orient0_aa(pred)=[ 0.02067284 -2.883098 0.05018289]
+[01/08 10:51:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+175.05,+1.82,+1.53) pred=(-165.23,+2.07,-0.55) pred_vs_gt=(+19.79,-0.43,+2.06)
+[01/08 10:51:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-16.26
+[01/08 13:36:07][INFO] [Exp Name]: finetune_
+[01/08 13:36:07][INFO] [GPU x Batch] = 1 x 2
+[01/08 13:36:07][INFO] [UnityDataset] Found 5 sequences.
+[01/08 13:36:07][INFO] [Train Dataset][9/9]: name=unity, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/08 13:36:07][INFO] [Train Dataset][All]: ConcatDataset size=5
+[01/08 13:36:07][INFO]
+[01/08 13:36:07][INFO] [UnityDataset] Found 5 sequences.
+[01/08 13:36:07][INFO] [Val Dataset][7/7]: name=unity_val, size=5, genmo.datasets.unity_dataset.UnityDataset
+[01/08 13:36:07][INFO]
+[01/08 13:36:14][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 13:36:36][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_7/checkpoints'
+[01/08 13:36:46][INFO] Start Fitting...
+[01/08 13:36:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 13:36:48][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 13:36:50][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 13:36:50][INFO] [LossBreakdown] body_pose=1.6293 betas=0.8841 go_c=1.8671 go_gv=14178.5850 transl_vel=0.5528
+[01/08 13:36:51][INFO] [LossBreakdown] body_pose=5.0417 betas=4.4841 go_c=26.4494 go_gv=11394.1895 transl_vel=5.4827
+[01/08 13:36:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 13:36:52][INFO] [LossBreakdown] body_pose=1.2509 betas=0.7162 go_c=1.1011 go_gv=0.4573 transl_vel=0.5269
+[01/08 13:36:53][INFO] [LossBreakdown] body_pose=0.5688 betas=0.5065 go_c=1.0587 go_gv=0.4510 transl_vel=0.4793
+[01/08 13:36:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 13:37:06][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 13:37:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9979 delta(pred-gt)=+0.0023
+[01/08 13:37:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04296682 3.0540771 -0.04666773] global_orient0_aa(pred)=[-0.02643642 -2.9357648 0.04450835]
+[01/08 13:37:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+175.05,+1.82,+1.53) pred=(-168.21,+1.61,+1.20) pred_vs_gt=(+16.75,+0.17,+0.35)
+[01/08 13:37:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-13.29
+[01/08 13:49:04][INFO] [Exp Name]: finetune_
+[01/08 13:49:04][INFO] [GPU x Batch] = 1 x 2
+[01/08 13:49:04][INFO] [UnityDataset] Found 2 sequences.
+[01/08 13:49:04][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 13:49:04][INFO] [Train Dataset][All]: ConcatDataset size=2
+[01/08 13:49:04][INFO]
+[01/08 13:49:04][INFO] [UnityDataset] Found 2 sequences.
+[01/08 13:49:04][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 13:49:04][INFO]
+[01/08 13:49:10][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 13:49:55][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_8/checkpoints'
+[01/08 13:50:43][INFO] Start Fitting...
+[01/08 13:50:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 13:50:47][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 13:50:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 13:50:50][INFO] [LossBreakdown] body_pose=4.0921 betas=1.8877 go_c=477.0122 go_gv=27704.1738 transl_vel=0.4509
+[01/08 13:50:50][INFO] [LossBreakdown] body_pose=10.0288 betas=13.6996 go_c=131.0738 go_gv=21548.9082 transl_vel=16.3315
+[01/08 13:50:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 13:50:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 13:51:06][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 13:51:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9296 pred=+0.9122 delta(pred-gt)=-0.0174
+[01/08 13:51:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 3.0952396 -0.03775275 0.21231844] global_orient0_aa(pred)=[-0.02454188 -2.9309487 0.04634325]
+[01/08 13:51:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+172.18,+2.32,-178.76) pred=(-167.94,+1.69,+1.14) pred_vs_gt=(+4.21,-3.96,-179.35)
+[01/08 13:51:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+159.97
+[01/08 14:06:06][INFO] [Exp Name]: finetune_
+[01/08 14:06:06][INFO] [GPU x Batch] = 1 x 2
+[01/08 14:06:06][INFO] [UnityDataset] Found 2 sequences.
+[01/08 14:06:06][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 14:06:06][INFO] [Train Dataset][All]: ConcatDataset size=2
+[01/08 14:06:06][INFO]
+[01/08 14:06:06][INFO] [UnityDataset] Found 2 sequences.
+[01/08 14:06:06][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 14:06:06][INFO]
+[01/08 14:06:13][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 14:06:43][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_9/checkpoints'
+[01/08 14:06:52][INFO] Start Fitting...
+[01/08 14:06:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 14:06:53][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 14:06:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 14:06:55][INFO] [LossBreakdown] body_pose=0.9869 betas=0.8731 go_c=0.3009 go_gv=28401.8750 transl_vel=0.5147
+[01/08 14:06:55][INFO] [LossBreakdown] body_pose=11.2257 betas=10.8412 go_c=59.8282 go_gv=22406.7422 transl_vel=13.7868
+[01/08 14:06:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 14:06:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 14:07:08][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 14:07:08][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9433 pred=+0.9053 delta(pred-gt)=-0.0380
+[01/08 14:07:08][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[-3.0952399 0.03775293 0.21231854] global_orient0_aa(pred)=[-0.02454189 -2.9309492 0.04634853]
+[01/08 14:07:08][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-172.18,-2.32,-178.76) pred=(-167.94,+1.69,+1.14) pred_vs_gt=(+19.89,+0.60,-179.82)
+[01/08 14:07:08][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+175.92
+[01/08 14:18:02][INFO] [Exp Name]: finetune_
+[01/08 14:18:02][INFO] [GPU x Batch] = 1 x 2
+[01/08 14:18:02][INFO] [UnityDataset] Found 2 sequences.
+[01/08 14:18:02][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 14:18:02][INFO] [Train Dataset][All]: ConcatDataset size=2
+[01/08 14:18:02][INFO]
+[01/08 14:18:02][INFO] [UnityDataset] Found 2 sequences.
+[01/08 14:18:02][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 14:18:02][INFO]
+[01/08 14:18:09][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 14:18:29][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_10/checkpoints'
+[01/08 14:18:40][INFO] Start Fitting...
+[01/08 14:18:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 14:18:42][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 14:18:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 14:18:44][INFO] [LossBreakdown] body_pose=0.9869 betas=0.8731 go_c=0.3009 go_gv=7741.4668 transl_vel=0.5147
+[01/08 14:18:45][INFO] [LossBreakdown] body_pose=3.4799 betas=3.3583 go_c=21.0509 go_gv=5966.9849 transl_vel=3.5267
+[01/08 14:18:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 14:18:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 14:19:00][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 14:19:00][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9433 pred=+0.9053 delta(pred-gt)=-0.0380
+[01/08 14:19:00][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[-3.0952399 0.03775293 0.21231854] global_orient0_aa(pred)=[-0.02454189 -2.9309492 0.04634853]
+[01/08 14:19:00][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-172.18,-2.32,-178.76) pred=(-167.94,+1.69,+1.14) pred_vs_gt=(+19.89,+0.60,-179.82)
+[01/08 14:19:00][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+175.92
+[01/08 14:19:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 14:19:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 14:19:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 14:19:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 14:19:47][INFO] ✅[FIT][Epoch 0] finished! 01:06→54:14 | loss_epoch=670
+[01/08 14:19:47][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/08 14:19:47][INFO] [LossBreakdown] body_pose=0.9567 betas=0.8544 go_c=0.3896 go_gv=7902.5303 transl_vel=0.4568
+[01/08 14:19:48][INFO] [LossBreakdown] body_pose=2.4480 betas=2.6319 go_c=10.3859 go_gv=6778.3052 transl_vel=1.4358
+[01/08 14:19:56][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 14:19:57][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.7834 pred=+0.7548 delta(pred-gt)=-0.0286
+[01/08 14:19:57][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[-3.1055129 0.03659544 0.23006329] global_orient0_aa(pred)=[ 0.01304308 2.4179664 -0.01848861]
+[01/08 14:19:57][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-171.54,-1.66,-178.77) pred=(+138.55,+0.97,+0.25) pred_vs_gt=(-32.97,+0.54,-178.93)
+[01/08 14:19:57][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-128.63
+[01/08 14:32:52][INFO] [Exp Name]: finetune_
+[01/08 14:32:52][INFO] [GPU x Batch] = 1 x 2
+[01/08 14:32:52][INFO] [UnityDataset] Found 2 sequences.
+[01/08 14:32:52][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 14:32:52][INFO] [Train Dataset][All]: ConcatDataset size=2
+[01/08 14:32:52][INFO]
+[01/08 14:32:52][INFO] [UnityDataset] Found 2 sequences.
+[01/08 14:32:52][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 14:32:52][INFO]
+[01/08 14:32:59][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 14:33:24][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_11/checkpoints'
+[01/08 14:33:38][INFO] Start Fitting...
+[01/08 14:33:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 14:33:41][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 14:33:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 14:33:44][INFO] [LossBreakdown] body_pose=0.9869 betas=0.8731 go_c=0.3010 go_gv=1.2483 transl_vel=0.4102
+[01/08 14:33:45][INFO] [LossBreakdown] body_pose=0.2522 betas=0.1316 go_c=0.1134 go_gv=1.0279 transl_vel=0.3698
+[01/08 14:33:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 14:33:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 14:34:01][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 14:34:01][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9957 pred=+0.9848 delta(pred-gt)=-0.0109
+[01/08 14:34:01][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[-0.02574082 -2.9222999 0.05326347] global_orient0_aa(pred)=[-0.02454225 -2.931001 0.04633833]
+[01/08 14:34:01][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-167.45,+1.95,+1.22) pred=(-167.94,+1.69,+1.14) pred_vs_gt=(-0.49,+0.27,+0.03)
+[01/08 14:34:01][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+3.97
+[01/08 14:35:58][INFO] [Exp Name]: finetune_
+[01/08 14:35:58][INFO] [GPU x Batch] = 1 x 2
+[01/08 14:35:58][INFO] [UnityDataset] Found 2 sequences.
+[01/08 14:35:58][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 14:35:58][INFO] [Train Dataset][All]: ConcatDataset size=2
+[01/08 14:35:58][INFO]
+[01/08 14:35:58][INFO] [UnityDataset] Found 2 sequences.
+[01/08 14:35:58][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 14:35:58][INFO]
+[01/08 14:36:06][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 14:36:20][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_12/checkpoints'
+[01/08 14:36:32][INFO] Start Fitting...
+[01/08 14:36:34][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 14:36:34][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 14:36:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 14:36:37][INFO] [LossBreakdown] body_pose=0.8291 betas=0.3350 go_c=0.0589 go_gv=0.0141 transl_vel=0.1972
+[01/08 14:36:37][INFO] [LossBreakdown] body_pose=0.1209 betas=0.0453 go_c=0.0206 go_gv=0.0057 transl_vel=0.1815
+[01/08 14:36:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 14:36:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 14:36:50][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 14:36:50][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9957 pred=+0.9855 delta(pred-gt)=-0.0102
+[01/08 14:36:50][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[-0.02574082 -2.9222999 0.05326347] global_orient0_aa(pred)=[ 0.03869041 -2.8841364 0.08326001]
+[01/08 14:36:50][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-167.45,+1.95,+1.22) pred=(-165.36,+3.45,-1.09) pred_vs_gt=(+2.18,-0.96,+2.58)
+[01/08 14:36:50][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+1.91
+[01/08 14:37:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 14:37:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 14:37:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 14:37:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 14:37:36][INFO] ✅[FIT][Epoch 0] finished! 01:03→52:04 | loss_epoch=14.6
+[01/08 14:37:36][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/08 14:37:36][INFO] [LossBreakdown] body_pose=1.2378 betas=0.6622 go_c=0.0970 go_gv=0.0459 transl_vel=0.1807
+[01/08 14:37:36][INFO] [LossBreakdown] body_pose=0.7828 betas=0.4367 go_c=0.0819 go_gv=0.0389 transl_vel=0.1779
+[01/08 14:37:46][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 14:37:46][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9980 pred=+0.9986 delta(pred-gt)=+0.0006
+[01/08 14:37:46][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[-0.02766505 -2.933438 0.03653343] global_orient0_aa(pred)=[-0.00903099 2.241836 -0.03052489]
+[01/08 14:37:46][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-168.08,+1.30,+1.22) pred=(+128.45,+1.08,-0.99) pred_vs_gt=(-63.42,+0.66,+2.11)
+[01/08 14:37:46][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+69.15
+[01/08 14:38:30][INFO] ✅[FIT][Epoch 1] finished! 01:58→47:14 | loss_epoch=25.9
+[01/08 14:38:30][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/08 14:38:30][INFO] [LossBreakdown] body_pose=1.8561 betas=0.3655 go_c=0.0928 go_gv=0.0165 transl_vel=1.1971
+[01/08 14:38:31][INFO] [LossBreakdown] body_pose=0.6024 betas=0.2469 go_c=0.0861 go_gv=0.0060 transl_vel=1.1773
+[01/08 14:38:39][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 14:38:39][INFO] [VisUnityVal] e002_0_biboo_birthday_speech root_y0: gt=+0.9940 pred=+0.9826 delta(pred-gt)=-0.0113
+[01/08 14:38:39][INFO] [VisUnityVal] e002_0_biboo_birthday_speech global_orient0_aa(gt)=[-0.02967529 -2.9226146 0.00410439] global_orient0_aa(pred)=[ 0.14458069 -1.6616713 0.00584495]
+[01/08 14:38:39][INFO] [VisUnityVal] e002_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-167.46,+0.03,+1.17) pred=(-95.59,+5.17,-5.26) pred_vs_gt=(+71.93,-3.62,+7.38)
+[01/08 14:38:39][INFO] [VisUnityVal] e002_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-47.67
+[01/08 14:39:24][INFO] ✅[FIT][Epoch 2] finished! 02:52→44:57 | loss_epoch=25.9
+[01/08 14:39:24][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[01/08 14:39:25][INFO] [LossBreakdown] body_pose=1.3128 betas=0.7732 go_c=0.0699 go_gv=0.0302 transl_vel=0.1938
+[01/08 14:39:25][INFO] [LossBreakdown] body_pose=0.9783 betas=0.5179 go_c=0.0630 go_gv=0.0204 transl_vel=0.1960
+[01/08 14:39:33][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 14:39:33][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+1.0029 pred=+0.9911 delta(pred-gt)=-0.0118
+[01/08 14:39:33][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.07784979 -2.7070167 -0.00577145] global_orient0_aa(pred)=[ 0.03644992 -2.7865043 0.08228384]
+[01/08 14:39:33][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-155.17,+0.46,-3.19) pred=(-159.75,+3.54,-0.87) pred_vs_gt=(-4.64,-3.77,-0.82)
+[01/08 14:39:33][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+8.15
+[01/08 15:00:35][INFO] [Exp Name]: finetune_
+[01/08 15:00:35][INFO] [GPU x Batch] = 1 x 2
+[01/08 15:00:35][INFO] [UnityDataset] Found 20 sequences.
+[01/08 15:00:35][INFO] [Train Dataset][9/9]: name=unity, size=20, genmo.datasets.unity_dataset.UnityDataset
+[01/08 15:00:35][INFO] [Train Dataset][All]: ConcatDataset size=20
+[01/08 15:00:35][INFO]
+[01/08 15:00:35][INFO] [UnityDataset] Found 20 sequences.
+[01/08 15:00:35][INFO] [Val Dataset][7/7]: name=unity_val, size=20, genmo.datasets.unity_dataset.UnityDataset
+[01/08 15:00:35][INFO]
+[01/08 15:00:41][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 15:00:59][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[01/08 15:01:06][INFO] Start Fitting...
+[01/08 15:01:07][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 15:01:07][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 15:01:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 15:01:08][INFO] [LossBreakdown] body_pose=2.3099 betas=0.6608 go_c=1.3761 go_gv=0.4972 transl_vel=0.2213
+[01/08 15:01:08][INFO] [LossBreakdown] body_pose=0.1445 betas=0.0858 go_c=0.0189 go_gv=0.0116 transl_vel=0.0510
+[01/08 15:01:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 15:01:09][INFO] [LossBreakdown] body_pose=2.7166 betas=1.0093 go_c=0.1371 go_gv=0.0203 transl_vel=0.2496
+[01/08 15:01:09][INFO] [LossBreakdown] body_pose=0.9439 betas=0.2089 go_c=0.0700 go_gv=0.0143 transl_vel=0.2235
+[01/08 15:01:10][INFO] [LossBreakdown] body_pose=1.0708 betas=0.6250 go_c=0.1457 go_gv=0.0246 transl_vel=0.3218
+[01/08 15:01:10][INFO] [LossBreakdown] body_pose=0.6377 betas=0.2190 go_c=0.0995 go_gv=0.0169 transl_vel=0.1592
+[01/08 15:01:10][INFO] [LossBreakdown] body_pose=1.0293 betas=0.5956 go_c=0.1397 go_gv=0.0411 transl_vel=0.2386
+[01/08 15:01:11][INFO] [LossBreakdown] body_pose=0.4230 betas=0.3316 go_c=0.0929 go_gv=0.0195 transl_vel=0.2689
+[01/08 15:01:11][INFO] [LossBreakdown] body_pose=0.9015 betas=0.4689 go_c=0.1901 go_gv=0.0219 transl_vel=0.2070
+[01/08 15:01:11][INFO] [LossBreakdown] body_pose=0.1025 betas=0.0557 go_c=0.0347 go_gv=0.0055 transl_vel=0.1126
+[01/08 15:01:12][INFO] [LossBreakdown] body_pose=1.1721 betas=0.7369 go_c=0.1806 go_gv=0.0222 transl_vel=0.3885
+[01/08 15:01:12][INFO] [LossBreakdown] body_pose=0.1736 betas=0.0832 go_c=0.0652 go_gv=0.0087 transl_vel=0.2769
+[01/08 15:01:13][INFO] [LossBreakdown] body_pose=1.1533 betas=0.7936 go_c=0.0771 go_gv=0.0252 transl_vel=0.3946
+[01/08 15:01:13][INFO] [LossBreakdown] body_pose=0.1618 betas=0.0809 go_c=0.0382 go_gv=0.0077 transl_vel=0.2791
+[01/08 15:01:13][INFO] [LossBreakdown] body_pose=1.9356 betas=0.6991 go_c=0.0713 go_gv=0.0299 transl_vel=0.3371
+[01/08 15:01:13][INFO] [LossBreakdown] body_pose=0.3676 betas=0.2192 go_c=0.0515 go_gv=0.0168 transl_vel=0.3002
+[01/08 15:01:14][INFO] [LossBreakdown] body_pose=0.8529 betas=0.8695 go_c=0.1312 go_gv=0.0423 transl_vel=0.2882
+[01/08 15:01:14][INFO] [LossBreakdown] body_pose=0.6058 betas=0.1657 go_c=0.0547 go_gv=0.0325 transl_vel=0.2088
+[01/08 15:01:15][INFO] [LossBreakdown] body_pose=1.4969 betas=0.5997 go_c=0.2971 go_gv=0.0351 transl_vel=0.2217
+[01/08 15:01:15][INFO] [LossBreakdown] body_pose=0.4118 betas=0.2030 go_c=0.2612 go_gv=0.0152 transl_vel=0.2707
+[01/08 15:01:15][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 15:01:26][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 15:01:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9815 pred=+0.9605 delta(pred-gt)=-0.0209
+[01/08 15:01:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[-0.03086768 -2.9082391 0.06629074] global_orient0_aa(pred)=[ 0.01921018 -3.056784 0.04412569]
+[01/08 15:01:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-166.65,+2.44,+1.50) pred=(-175.17,+1.68,-0.65) pred_vs_gt=(-8.43,+1.23,+1.92)
+[01/08 15:01:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+14.06
+[01/08 15:02:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 15:02:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 15:02:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 15:02:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 15:02:57][INFO] ✅[FIT][Epoch 0] finished! 01:51→1:30:43 | loss_epoch=38.3
+[01/08 15:02:57][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/08 15:02:57][INFO] [LossBreakdown] body_pose=1.3267 betas=0.7893 go_c=0.1886 go_gv=0.0360 transl_vel=0.2503
+[01/08 15:02:57][INFO] [LossBreakdown] body_pose=0.7171 betas=0.3193 go_c=0.1675 go_gv=0.0208 transl_vel=0.1720
+[01/08 15:02:58][INFO] [LossBreakdown] body_pose=2.6334 betas=0.6415 go_c=0.3264 go_gv=0.0417 transl_vel=0.1954
+[01/08 15:02:58][INFO] [LossBreakdown] body_pose=0.4748 betas=0.2245 go_c=0.0851 go_gv=0.0187 transl_vel=0.2403
+[01/08 15:02:59][INFO] [LossBreakdown] body_pose=1.0506 betas=0.6419 go_c=0.1778 go_gv=0.0650 transl_vel=0.3638
+[01/08 15:02:59][INFO] [LossBreakdown] body_pose=0.7674 betas=0.3402 go_c=0.1103 go_gv=0.0383 transl_vel=0.4103
+[01/08 15:02:59][INFO] [LossBreakdown] body_pose=1.9224 betas=0.7286 go_c=0.0871 go_gv=0.0218 transl_vel=0.0734
+[01/08 15:02:59][INFO] [LossBreakdown] body_pose=0.1680 betas=0.1376 go_c=0.0347 go_gv=0.0068 transl_vel=0.0322
+[01/08 15:03:00][INFO] [LossBreakdown] body_pose=1.5492 betas=0.6744 go_c=0.0890 go_gv=0.0237 transl_vel=0.2087
+[01/08 15:03:00][INFO] [LossBreakdown] body_pose=0.2466 betas=0.1494 go_c=0.0435 go_gv=0.0058 transl_vel=0.1100
+[01/08 15:03:01][INFO] [LossBreakdown] body_pose=0.9864 betas=0.4790 go_c=0.0929 go_gv=0.0236 transl_vel=0.3233
+[01/08 15:03:01][INFO] [LossBreakdown] body_pose=0.1906 betas=0.0530 go_c=0.0217 go_gv=0.0100 transl_vel=0.1148
+[01/08 15:03:02][INFO] [LossBreakdown] body_pose=1.9984 betas=0.7792 go_c=0.2131 go_gv=0.0512 transl_vel=0.3440
+[01/08 15:03:02][INFO] [LossBreakdown] body_pose=0.4613 betas=0.3455 go_c=0.1698 go_gv=0.0324 transl_vel=0.3265
+[01/08 15:03:02][INFO] [LossBreakdown] body_pose=2.2150 betas=0.8340 go_c=0.1666 go_gv=0.0323 transl_vel=0.1718
+[01/08 15:03:03][INFO] [LossBreakdown] body_pose=0.5062 betas=0.3010 go_c=0.1030 go_gv=0.0206 transl_vel=0.2017
+[01/08 15:03:03][INFO] [LossBreakdown] body_pose=1.2047 betas=0.5913 go_c=0.4913 go_gv=0.1197 transl_vel=0.4824
+[01/08 15:03:03][INFO] [LossBreakdown] body_pose=0.2954 betas=0.1181 go_c=0.1664 go_gv=0.0119 transl_vel=0.4306
+[01/08 15:03:04][INFO] [LossBreakdown] body_pose=0.8140 betas=0.9277 go_c=0.1634 go_gv=0.0256 transl_vel=0.3722
+[01/08 15:03:04][INFO] [LossBreakdown] body_pose=0.2470 betas=0.2048 go_c=0.1061 go_gv=0.0158 transl_vel=0.3548
+[01/08 15:03:23][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 15:03:23][INFO] [VisUnityVal] e001_103_biboo_birthday_speech_explosion_4 root_y0: gt=+0.9897 pred=+0.9713 delta(pred-gt)=-0.0184
+[01/08 15:03:23][INFO] [VisUnityVal] e001_103_biboo_birthday_speech_explosion_4 global_orient0_aa(gt)=[-0.00844617 -0.12936251 0.03557836] global_orient0_aa(pred)=[-0.05436275 -2.2652457 -0.23814122]
+[01/08 15:03:23][INFO] [VisUnityVal] e001_103_biboo_birthday_speech_explosion_4 global_orient0_yxz_deg gt=(-7.42,-0.35,+2.06) pred=(-130.06,-10.93,-2.35) pred_vs_gt=(-122.58,-9.92,-5.75)
+[01/08 15:03:23][INFO] [VisUnityVal] e001_103_biboo_birthday_speech_explosion_4 yaw0_deg(pred_vs_gt)=+119.18
+[01/08 15:13:47][INFO] [Exp Name]: finetune_
+[01/08 15:13:47][INFO] [GPU x Batch] = 1 x 2
+[01/08 15:13:47][INFO] [UnityDataset] Found 2 sequences.
+[01/08 15:13:47][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 15:13:47][INFO] [Train Dataset][All]: ConcatDataset size=2
+[01/08 15:13:47][INFO]
+[01/08 15:13:47][INFO] [UnityDataset] Found 2 sequences.
+[01/08 15:13:47][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 15:13:47][INFO]
+[01/08 15:13:54][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 15:15:19][INFO] [Exp Name]: finetune_
+[01/08 15:15:19][INFO] [GPU x Batch] = 1 x 2
+[01/08 15:15:20][INFO] [UnityDataset] Found 2 sequences.
+[01/08 15:15:20][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 15:15:20][INFO] [Train Dataset][All]: ConcatDataset size=2
+[01/08 15:15:20][INFO]
+[01/08 15:15:20][INFO] [UnityDataset] Found 2 sequences.
+[01/08 15:15:20][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[01/08 15:15:20][INFO]
+[01/08 15:15:28][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 15:16:02][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[01/08 15:16:15][INFO] Start Fitting...
+[01/08 15:16:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 15:16:17][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 15:16:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 15:16:19][INFO] [LossBreakdown] body_pose=0.8291 betas=0.3350 go_c=0.0589 go_gv=0.0141 transl_vel=0.1972
+[01/08 15:16:20][INFO] [LossBreakdown] body_pose=0.1209 betas=0.0453 go_c=0.0206 go_gv=0.0057 transl_vel=0.1815
+[01/08 15:16:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 15:16:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 15:16:33][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 15:16:33][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9957 pred=+0.9855 delta(pred-gt)=-0.0102
+[01/08 15:16:33][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[-0.02574082 -2.9222999 0.05326347] global_orient0_aa(pred)=[ 0.03869041 -2.8841364 0.08326001]
+[01/08 15:16:33][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-167.45,+1.95,+1.22) pred=(-165.36,+3.45,-1.09) pred_vs_gt=(+2.18,-0.96,+2.58)
+[01/08 15:16:33][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+1.91
+[01/08 15:43:35][INFO] [Exp Name]: finetune_
+[01/08 15:43:35][INFO] [GPU x Batch] = 1 x 2
+[01/08 15:43:36][INFO] [UnityDataset] Found 25 sequences.
+[01/08 15:43:36][INFO] [Train Dataset][9/9]: name=unity, size=25, genmo.datasets.unity_dataset.UnityDataset
+[01/08 15:43:36][INFO] [Train Dataset][All]: ConcatDataset size=25
+[01/08 15:43:36][INFO]
+[01/08 15:43:36][INFO] [UnityDataset] Found 25 sequences.
+[01/08 15:43:36][INFO] [Val Dataset][7/7]: name=unity_val, size=25, genmo.datasets.unity_dataset.UnityDataset
+[01/08 15:43:36][INFO]
+[01/08 15:43:43][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 15:44:05][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[01/08 15:44:18][INFO] Start Fitting...
+[01/08 15:44:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 15:44:20][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 15:44:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 15:44:23][INFO] [LossBreakdown] body_pose=1.4104 betas=0.4427 go_c=0.3776 go_gv=0.0991 transl_vel=0.1995
+[01/08 15:44:23][INFO] [LossBreakdown] body_pose=0.1537 betas=0.0732 go_c=0.0165 go_gv=0.0062 transl_vel=0.1327
+[01/08 15:44:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 15:44:24][INFO] [LossBreakdown] body_pose=1.1538 betas=0.5432 go_c=0.1315 go_gv=0.0589 transl_vel=0.1730
+[01/08 15:44:24][INFO] [LossBreakdown] body_pose=0.3487 betas=0.2571 go_c=0.0905 go_gv=0.0267 transl_vel=0.1992
+[01/08 15:44:25][INFO] [LossBreakdown] body_pose=0.8455 betas=0.5991 go_c=0.1257 go_gv=0.0145 transl_vel=0.1340
+[01/08 15:44:25][INFO] [LossBreakdown] body_pose=0.4771 betas=0.2344 go_c=0.0793 go_gv=0.0165 transl_vel=0.0641
+[01/08 15:44:26][INFO] [LossBreakdown] body_pose=1.2418 betas=0.5307 go_c=0.2504 go_gv=0.0343 transl_vel=0.2066
+[01/08 15:44:26][INFO] [LossBreakdown] body_pose=0.5373 betas=0.2543 go_c=0.1768 go_gv=0.0229 transl_vel=0.2019
+[01/08 15:44:26][INFO] [LossBreakdown] body_pose=1.8683 betas=0.4638 go_c=0.4046 go_gv=0.0552 transl_vel=0.3509
+[01/08 15:44:27][INFO] [LossBreakdown] body_pose=0.1714 betas=0.0932 go_c=0.0342 go_gv=0.0058 transl_vel=0.1257
+[01/08 15:44:27][INFO] [LossBreakdown] body_pose=2.0272 betas=0.4703 go_c=0.1425 go_gv=0.0168 transl_vel=0.3722
+[01/08 15:44:27][INFO] [LossBreakdown] body_pose=0.2555 betas=0.1295 go_c=0.0336 go_gv=0.0092 transl_vel=0.2194
+[01/08 15:44:28][INFO] [LossBreakdown] body_pose=1.6725 betas=0.9567 go_c=2.2563 go_gv=0.8581 transl_vel=0.1733
+[01/08 15:44:28][INFO] [LossBreakdown] body_pose=0.1619 betas=0.0883 go_c=0.0852 go_gv=0.0096 transl_vel=0.1261
+[01/08 15:44:29][INFO] [LossBreakdown] body_pose=1.8862 betas=0.5141 go_c=0.0599 go_gv=0.0179 transl_vel=0.2425
+[01/08 15:44:29][INFO] [LossBreakdown] body_pose=0.5242 betas=0.1453 go_c=0.0500 go_gv=0.0124 transl_vel=0.2075
+[01/08 15:44:29][INFO] [LossBreakdown] body_pose=1.7759 betas=0.6438 go_c=0.1025 go_gv=0.0131 transl_vel=0.1941
+[01/08 15:44:30][INFO] [LossBreakdown] body_pose=0.8699 betas=0.3686 go_c=0.0939 go_gv=0.0147 transl_vel=0.0469
+[01/08 15:44:30][INFO] [LossBreakdown] body_pose=0.8483 betas=0.6600 go_c=0.0920 go_gv=0.0145 transl_vel=0.1034
+[01/08 15:44:30][INFO] [LossBreakdown] body_pose=0.2330 betas=0.1876 go_c=0.1178 go_gv=0.0203 transl_vel=0.1146
+[01/08 15:44:31][INFO] [LossBreakdown] body_pose=1.8775 betas=1.2187 go_c=0.2761 go_gv=0.0356 transl_vel=0.1687
+[01/08 15:44:31][INFO] [LossBreakdown] body_pose=0.3863 betas=0.2544 go_c=0.1054 go_gv=0.0208 transl_vel=0.1422
+[01/08 15:44:32][INFO] [LossBreakdown] body_pose=2.7518 betas=0.5641 go_c=0.0440 go_gv=0.0239 transl_vel=0.1462
+[01/08 15:44:32][INFO] [LossBreakdown] body_pose=0.4323 betas=0.2280 go_c=0.0365 go_gv=0.0188 transl_vel=0.1404
+[01/08 15:44:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 15:44:44][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 15:44:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9791 pred=+0.9650 delta(pred-gt)=-0.0141
+[01/08 15:44:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[-0.0270079 -2.9185286 0.02240671] global_orient0_aa(pred)=[-0.01264505 -3.0631845 0.08623575]
+[01/08 15:44:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-167.22,+0.75,+1.14) pred=(-175.56,+3.20,+0.60) pred_vs_gt=(-8.34,-2.27,+1.08)
+[01/08 15:44:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+15.42
+[01/08 15:46:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 15:46:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 15:46:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 15:46:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 15:46:35][INFO] ✅[FIT][Epoch 0] finished! 02:16→1:51:20 | loss_epoch=39.1
+[01/08 15:46:35][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/08 15:46:35][INFO] [LossBreakdown] body_pose=2.0046 betas=0.8070 go_c=0.2167 go_gv=0.0814 transl_vel=0.2620
+[01/08 15:46:35][INFO] [LossBreakdown] body_pose=0.6415 betas=0.6697 go_c=0.1854 go_gv=0.0443 transl_vel=0.2613
+[01/08 15:46:36][INFO] [LossBreakdown] body_pose=1.3542 betas=0.5071 go_c=0.1304 go_gv=0.0487 transl_vel=0.2630
+[01/08 15:46:36][INFO] [LossBreakdown] body_pose=0.1928 betas=0.0691 go_c=0.0237 go_gv=0.0090 transl_vel=0.1370
+[01/08 15:46:36][INFO] [LossBreakdown] body_pose=1.7831 betas=0.8509 go_c=0.1395 go_gv=0.0269 transl_vel=0.1853
+[01/08 15:46:36][INFO] [LossBreakdown] body_pose=0.2147 betas=0.1332 go_c=0.0423 go_gv=0.0082 transl_vel=0.1132
+[01/08 15:46:37][INFO] [LossBreakdown] body_pose=1.6782 betas=0.6305 go_c=0.0959 go_gv=0.0250 transl_vel=0.3085
+[01/08 15:46:37][INFO] [LossBreakdown] body_pose=0.2742 betas=0.1295 go_c=0.0377 go_gv=0.0134 transl_vel=0.1682
+[01/08 15:46:38][INFO] [LossBreakdown] body_pose=1.1641 betas=0.7158 go_c=0.1358 go_gv=0.0215 transl_vel=0.1949
+[01/08 15:46:38][INFO] [LossBreakdown] body_pose=0.2995 betas=0.2470 go_c=0.1138 go_gv=0.0274 transl_vel=0.2353
+[01/08 15:46:39][INFO] [LossBreakdown] body_pose=1.0379 betas=0.5108 go_c=0.2713 go_gv=0.0332 transl_vel=0.1111
+[01/08 15:46:39][INFO] [LossBreakdown] body_pose=0.5577 betas=0.2543 go_c=0.2702 go_gv=0.0105 transl_vel=0.1128
+[01/08 15:46:39][INFO] [LossBreakdown] body_pose=1.5709 betas=0.5579 go_c=0.1852 go_gv=0.0320 transl_vel=0.1787
+[01/08 15:46:40][INFO] [LossBreakdown] body_pose=0.4280 betas=0.1774 go_c=0.1124 go_gv=0.0162 transl_vel=0.1856
+[01/08 15:46:40][INFO] [LossBreakdown] body_pose=0.8884 betas=0.4979 go_c=0.0829 go_gv=0.0162 transl_vel=0.2531
+[01/08 15:46:41][INFO] [LossBreakdown] body_pose=0.2889 betas=0.1368 go_c=0.0610 go_gv=0.0106 transl_vel=0.1361
+[01/08 15:46:41][INFO] [LossBreakdown] body_pose=1.7260 betas=0.5761 go_c=0.2656 go_gv=0.0219 transl_vel=0.4510
+[01/08 15:46:41][INFO] [LossBreakdown] body_pose=0.9338 betas=0.2938 go_c=0.1771 go_gv=0.0193 transl_vel=0.3831
+[01/08 15:46:42][INFO] [LossBreakdown] body_pose=1.2809 betas=0.8732 go_c=0.1360 go_gv=0.0384 transl_vel=0.1873
+[01/08 15:46:42][INFO] [LossBreakdown] body_pose=0.4099 betas=0.4301 go_c=0.0630 go_gv=0.0143 transl_vel=0.1958
+[01/08 15:46:43][INFO] [LossBreakdown] body_pose=1.2064 betas=0.5455 go_c=0.0974 go_gv=0.0199 transl_vel=0.0905
+[01/08 15:46:43][INFO] [LossBreakdown] body_pose=0.1662 betas=0.0351 go_c=0.0171 go_gv=0.0050 transl_vel=0.0426
+[01/08 15:46:43][INFO] [LossBreakdown] body_pose=1.1871 betas=0.5391 go_c=0.1617 go_gv=0.0330 transl_vel=0.1906
+[01/08 15:46:44][INFO] [LossBreakdown] body_pose=0.2081 betas=0.1509 go_c=0.0320 go_gv=0.0099 transl_vel=0.1895
+[01/08 15:47:09][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 15:47:09][INFO] [VisUnityVal] e001_105_biboo_birthday_speech_explosion_6 root_y0: gt=+0.9908 pred=+0.9910 delta(pred-gt)=+0.0002
+[01/08 15:47:09][INFO] [VisUnityVal] e001_105_biboo_birthday_speech_explosion_6 global_orient0_aa(gt)=[-0.09144535 1.6675488 0.12195748] global_orient0_aa(pred)=[-0.04917006 -2.7455845 -0.03057923]
+[01/08 15:47:09][INFO] [VisUnityVal] e001_105_biboo_birthday_speech_explosion_6 global_orient0_yxz_deg gt=(+95.66,-7.72,+0.72) pred=(-157.36,-1.62,+1.73) pred_vs_gt=(+107.04,+0.40,-6.17)
+[01/08 15:47:09][INFO] [VisUnityVal] e001_105_biboo_birthday_speech_explosion_6 yaw0_deg(pred_vs_gt)=-117.61
+[01/08 15:48:53][INFO] ✅[FIT][Epoch 1] finished! 04:34→1:49:45 | loss_epoch=31.5
+[01/08 15:48:53][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/08 15:48:53][INFO] [LossBreakdown] body_pose=1.0772 betas=0.3259 go_c=0.7178 go_gv=0.0642 transl_vel=0.2390
+[01/08 15:48:53][INFO] [LossBreakdown] body_pose=0.8264 betas=0.3793 go_c=0.5687 go_gv=0.0503 transl_vel=0.2804
+[01/08 15:48:54][INFO] [LossBreakdown] body_pose=1.0844 betas=0.3500 go_c=0.4244 go_gv=0.0623 transl_vel=0.4536
+[01/08 15:48:54][INFO] [LossBreakdown] body_pose=0.3935 betas=0.3862 go_c=0.1504 go_gv=0.0243 transl_vel=0.3625
+[01/08 15:48:56][INFO] [LossBreakdown] body_pose=2.3411 betas=0.4738 go_c=0.5657 go_gv=0.0730 transl_vel=0.2237
+[01/08 15:48:56][INFO] [LossBreakdown] body_pose=0.8738 betas=0.3705 go_c=0.1943 go_gv=0.0343 transl_vel=0.2555
+[01/08 15:48:58][INFO] [LossBreakdown] body_pose=2.3061 betas=0.4617 go_c=0.3433 go_gv=0.0785 transl_vel=0.1407
+[01/08 15:48:59][INFO] [LossBreakdown] body_pose=0.5292 betas=0.4946 go_c=0.0945 go_gv=0.0181 transl_vel=0.0916
+[01/08 15:49:02][INFO] [LossBreakdown] body_pose=1.5345 betas=0.4026 go_c=0.2951 go_gv=0.0921 transl_vel=0.2503
+[01/08 15:49:03][INFO] [LossBreakdown] body_pose=0.3697 betas=0.4745 go_c=0.0983 go_gv=0.0246 transl_vel=0.1298
+[01/08 15:49:05][INFO] [LossBreakdown] body_pose=1.2130 betas=0.3151 go_c=0.6376 go_gv=0.0557 transl_vel=0.6078
+[01/08 15:49:05][INFO] [LossBreakdown] body_pose=0.7428 betas=0.4238 go_c=0.5327 go_gv=0.0327 transl_vel=0.8007
+[01/08 15:49:06][INFO] [LossBreakdown] body_pose=1.1557 betas=0.3861 go_c=0.4798 go_gv=0.0955 transl_vel=0.4793
+[01/08 15:49:06][INFO] [LossBreakdown] body_pose=0.3221 betas=0.3828 go_c=0.2329 go_gv=0.0270 transl_vel=0.3291
+[01/08 15:49:07][INFO] [LossBreakdown] body_pose=2.8292 betas=0.3765 go_c=0.1757 go_gv=0.0679 transl_vel=0.0767
+[01/08 15:49:07][INFO] [LossBreakdown] body_pose=0.6149 betas=0.5952 go_c=0.0688 go_gv=0.0193 transl_vel=0.1107
+[01/08 15:49:09][INFO] [LossBreakdown] body_pose=2.6862 betas=0.4472 go_c=0.2639 go_gv=0.0875 transl_vel=0.2749
+[01/08 15:49:09][INFO] [LossBreakdown] body_pose=1.3755 betas=0.5075 go_c=0.1308 go_gv=0.0654 transl_vel=0.3791
+[01/08 15:49:11][INFO] [LossBreakdown] body_pose=0.5852 betas=0.2592 go_c=0.3199 go_gv=0.0475 transl_vel=0.1754
+[01/08 15:49:11][INFO] [LossBreakdown] body_pose=0.3326 betas=0.2836 go_c=0.1161 go_gv=0.0252 transl_vel=0.1840
+[01/08 15:49:12][INFO] [LossBreakdown] body_pose=1.7400 betas=0.4860 go_c=0.6451 go_gv=0.0667 transl_vel=0.6958
+[01/08 15:49:12][INFO] [LossBreakdown] body_pose=0.5424 betas=0.3675 go_c=0.2616 go_gv=0.0327 transl_vel=0.3511
+[01/08 15:49:14][INFO] [LossBreakdown] body_pose=1.0867 betas=0.3258 go_c=0.7050 go_gv=0.0746 transl_vel=0.2732
+[01/08 15:49:14][INFO] [LossBreakdown] body_pose=0.3502 betas=0.4452 go_c=0.1196 go_gv=0.0172 transl_vel=0.1367
+[01/08 15:50:22][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 15:50:22][INFO] [VisUnityVal] e002_11_biboo_birthday_speech_talking_in_front_of_a_stuatue root_y0: gt=+0.9764 pred=+0.9662 delta(pred-gt)=-0.0102
+[01/08 15:50:22][INFO] [VisUnityVal] e002_11_biboo_birthday_speech_talking_in_front_of_a_stuatue global_orient0_aa(gt)=[0.02158001 3.0896657 0.13888471] global_orient0_aa(pred)=[ 0.00332632 -2.0512533 0.02624967]
+[01/08 15:50:22][INFO] [VisUnityVal] e002_11_biboo_birthday_speech_talking_in_front_of_a_stuatue global_orient0_yxz_deg gt=(+177.16,-5.12,+0.93) pred=(-117.53,+1.15,+0.51) pred_vs_gt=(+65.29,-6.29,+0.10)
+[01/08 15:50:22][INFO] [VisUnityVal] e002_11_biboo_birthday_speech_talking_in_front_of_a_stuatue yaw0_deg(pred_vs_gt)=-70.15
+[01/08 17:01:16][INFO] [Exp Name]: finetune_
+[01/08 17:01:16][INFO] [GPU x Batch] = 1 x 2
+[01/08 17:01:16][INFO] [UnityDataset] Found 1 sequences.
+[01/08 17:01:16][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[01/08 17:01:16][INFO] [Train Dataset][All]: ConcatDataset size=1
+[01/08 17:01:16][INFO]
+[01/08 17:01:16][INFO] [UnityDataset] Found 1 sequences.
+[01/08 17:01:16][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[01/08 17:01:16][INFO]
+[01/08 17:01:22][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 17:01:42][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[01/08 17:01:52][INFO] Start Fitting...
+[01/08 17:01:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 17:01:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/data.py:106: Total length of `DataLoader` across ranks is zero. Please make sure this was your intention.
+
+[01/08 17:01:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/data.py:106: Total length of `CombinedLoader` across ranks is zero. Please make sure this was your intention.
+
+[01/08 17:01:55][INFO] End of script.
+[01/08 17:07:25][INFO] [Exp Name]: finetune_
+[01/08 17:07:25][INFO] [GPU x Batch] = 1 x 2
+[01/08 17:07:25][INFO] [UnityDataset] Found 4 sequences.
+[01/08 17:07:25][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/08 17:07:25][INFO] [Train Dataset][All]: ConcatDataset size=4
+[01/08 17:07:25][INFO]
+[01/08 17:07:25][INFO] [UnityDataset] Found 4 sequences.
+[01/08 17:07:25][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/08 17:07:25][INFO]
+[01/08 17:07:32][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 17:07:52][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[01/08 17:08:02][INFO] Start Fitting...
+[01/08 17:08:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 17:08:04][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 17:08:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 17:08:06][INFO] [LossBreakdown] body_pose=1.6683 betas=0.7046 go_c=1.4150 go_gv=3.2022(masked) transl_vel=0.1286
+[01/08 17:08:07][INFO] [LossBreakdown] body_pose=0.1387 betas=0.0817 go_c=0.0465 go_gv=3.5283(masked) transl_vel=0.0443
+[01/08 17:08:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 17:08:08][INFO] [LossBreakdown] body_pose=1.3458 betas=0.6247 go_c=0.1120 go_gv=3.5619(masked) transl_vel=0.0685
+[01/08 17:08:08][INFO] [LossBreakdown] body_pose=0.8287 betas=0.1476 go_c=0.0636 go_gv=3.4855(masked) transl_vel=0.0539
+[01/08 17:08:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 17:08:21][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 17:08:21][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9981 delta(pred-gt)=+0.0026
+[01/08 17:08:21][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04296682 3.0540771 -0.04666773] global_orient0_aa(pred)=[ 0.02062318 -2.8829534 0.05018083]
+[01/08 17:08:21][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+175.05,+1.82,+1.53) pred=(-165.22,+2.07,-0.55) pred_vs_gt=(+19.80,-0.43,+2.05)
+[01/08 17:08:21][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-16.26
+[01/08 17:10:09][INFO] [Exp Name]: finetune_
+[01/08 17:10:09][INFO] [GPU x Batch] = 1 x 2
+[01/08 17:10:09][INFO] [UnityDataset] Found 4 sequences.
+[01/08 17:10:09][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/08 17:10:09][INFO] [Train Dataset][All]: ConcatDataset size=4
+[01/08 17:10:09][INFO]
+[01/08 17:10:09][INFO] [UnityDataset] Found 4 sequences.
+[01/08 17:10:09][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[01/08 17:10:09][INFO]
+[01/08 17:10:16][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[01/08 17:10:28][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[01/08 17:10:39][INFO] Start Fitting...
+[01/08 17:10:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[01/08 17:10:41][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[01/08 17:10:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[01/08 17:10:43][INFO] [LossBreakdown] body_pose=1.6683 betas=0.7046 go_c=1.4150 go_gv=3.2022(masked) transl_vel=0.1286
+[01/08 17:10:44][INFO] [LossBreakdown] body_pose=0.1387 betas=0.0817 go_c=0.0465 go_gv=3.5283(masked) transl_vel=0.0443
+[01/08 17:10:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[01/08 17:10:45][INFO] [LossBreakdown] body_pose=1.3458 betas=0.6247 go_c=0.1120 go_gv=3.5619(masked) transl_vel=0.0685
+[01/08 17:10:45][INFO] [LossBreakdown] body_pose=0.8287 betas=0.1476 go_c=0.0636 go_gv=3.4855(masked) transl_vel=0.0539
+[01/08 17:10:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[01/08 17:10:58][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 17:10:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9981 delta(pred-gt)=+0.0026
+[01/08 17:10:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04296682 3.0540771 -0.04666773] global_orient0_aa(pred)=[ 0.02062318 -2.8829534 0.05018083]
+[01/08 17:10:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+175.05,+1.82,+1.53) pred=(-165.22,+2.07,-0.55) pred_vs_gt=(+19.80,-0.43,+2.05)
+[01/08 17:10:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-16.26
+[01/08 17:11:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 17:11:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 17:11:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 17:11:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[01/08 17:11:48][INFO] ✅[FIT][Epoch 0] finished! 01:07→55:24 | loss_epoch=55.8
+[01/08 17:11:48][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[01/08 17:11:48][INFO] [LossBreakdown] body_pose=1.2618 betas=0.8335 go_c=0.1153 go_gv=4.3087(masked) transl_vel=0.2157
+[01/08 17:11:48][INFO] [LossBreakdown] body_pose=0.4292 betas=0.3188 go_c=0.0725 go_gv=4.1065(masked) transl_vel=0.2150
+[01/08 17:11:49][INFO] [LossBreakdown] body_pose=1.1179 betas=0.6992 go_c=0.2799 go_gv=4.0469(masked) transl_vel=0.5573
+[01/08 17:11:49][INFO] [LossBreakdown] body_pose=0.5874 betas=0.2711 go_c=0.1778 go_gv=4.1410(masked) transl_vel=0.2764
+[01/08 17:11:57][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[01/08 17:11:57][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9967 pred=+0.9955 delta(pred-gt)=-0.0012
+[01/08 17:11:57][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04501803 3.124892 -0.09182652] global_orient0_aa(pred)=[-0.02408068 -2.193168 -0.0738423 ]
+[01/08 17:11:57][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+179.19,+3.38,+1.63) pred=(-125.69,-3.56,-0.57) pred_vs_gt=(+55.26,+6.91,+2.31)
+[01/08 17:11:57][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-52.66
+[01/08 17:12:47][INFO] ✅[FIT][Epoch 1] finished! 02:07→50:55 | loss_epoch=33
+[01/08 17:12:47][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[01/08 17:12:47][INFO] [LossBreakdown] body_pose=1.9728 betas=0.5188 go_c=0.0942 go_gv=3.6038(masked) transl_vel=0.1378
+[01/08 17:12:48][INFO] [LossBreakdown] body_pose=0.1777 betas=0.0731 go_c=0.0260 go_gv=3.3465(masked) transl_vel=0.0217
+[01/08 17:12:48][INFO] [LossBreakdown] body_pose=1.2146 betas=0.6297 go_c=0.2220 go_gv=3.9859(masked) transl_vel=0.4761
+[01/08 17:12:48][INFO] [LossBreakdown] body_pose=0.1152 betas=0.0783 go_c=0.0304 go_gv=3.8710(masked) transl_vel=0.0851
+[02/23 10:42:30][INFO] [Exp Name]: finetune_
+[02/23 10:42:30][INFO] [GPU x Batch] = 1 x 2
+[02/23 10:42:30][INFO] [UnityDataset] Found 11 sequences.
+[02/23 10:42:30][INFO] [Train Dataset][9/9]: name=unity, size=11, genmo.datasets.unity_dataset.UnityDataset
+[02/23 10:42:30][INFO] [Train Dataset][All]: ConcatDataset size=11
+[02/23 10:42:30][INFO]
+[02/23 10:42:30][INFO] [UnityDataset] Found 11 sequences.
+[02/23 10:42:30][INFO] [Val Dataset][7/7]: name=unity_val, size=11, genmo.datasets.unity_dataset.UnityDataset
+[02/23 10:42:30][INFO]
+[02/23 10:42:37][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 10:43:16][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[02/23 10:43:47][INFO] Start Fitting...
+[02/23 10:43:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 10:43:49][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 10:43:51][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 10:43:51][INFO] [LossBreakdown] body_pose=1.0725 betas=0.5613 go_c=0.1525 go_gv=4.0716(masked) transl_vel=0.2225
+[02/23 10:43:52][INFO] [LossBreakdown] body_pose=0.1525 betas=0.0662 go_c=0.0511 go_gv=3.7082(masked) transl_vel=0.0425
+[02/23 10:43:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 10:43:53][INFO] [LossBreakdown] body_pose=1.3607 betas=0.6524 go_c=0.1420 go_gv=4.3658(masked) transl_vel=0.2621
+[02/23 10:43:53][INFO] [LossBreakdown] body_pose=0.6045 betas=0.2913 go_c=0.0898 go_gv=4.1250(masked) transl_vel=0.1436
+[02/23 10:43:54][INFO] [LossBreakdown] body_pose=1.0708 betas=0.6249 go_c=0.1457 go_gv=4.5939(masked) transl_vel=0.3551
+[02/23 10:43:54][INFO] [LossBreakdown] body_pose=0.6332 betas=0.2150 go_c=0.0956 go_gv=4.5918(masked) transl_vel=0.1465
+[02/23 10:43:55][INFO] [LossBreakdown] body_pose=0.9696 betas=0.4881 go_c=0.0961 go_gv=3.9965(masked) transl_vel=0.2645
+[02/23 10:43:55][INFO] [LossBreakdown] body_pose=0.4940 betas=0.2491 go_c=0.0866 go_gv=3.9221(masked) transl_vel=0.2668
+[02/23 10:43:56][INFO] [LossBreakdown] body_pose=1.0291 betas=1.0899 go_c=0.2835 go_gv=3.9039(masked) transl_vel=0.4734
+[02/23 10:43:56][INFO] [LossBreakdown] body_pose=0.0873 betas=0.0665 go_c=0.0418 go_gv=3.8970(masked) transl_vel=0.0654
+[02/23 10:43:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[02/23 10:44:10][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[02/23 10:44:10][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9998 delta(pred-gt)=+0.0042
+[02/23 10:44:10][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04647226 2.9962249 -0.04046043] global_orient0_aa(pred)=[-0.08468574 2.5034742 -0.08740971]
+[02/23 10:44:10][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+171.73,+1.67,+1.66) pred=(+143.47,+2.45,-4.69) pred_vs_gt=(-28.13,-1.70,+6.16)
+[02/23 10:44:10][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+42.68
+[02/23 10:45:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 10:45:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 10:45:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 10:45:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 10:45:44][INFO] ✅[FIT][Epoch 0] finished! 01:56→1:35:04 | loss_epoch=32.2
+[02/23 10:45:44][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 10:45:44][INFO] [LossBreakdown] body_pose=1.4431 betas=0.5064 go_c=0.2349 go_gv=4.6544(masked) transl_vel=0.1600
+[02/23 10:45:45][INFO] [LossBreakdown] body_pose=0.1886 betas=0.1088 go_c=0.0372 go_gv=4.2531(masked) transl_vel=0.0547
+[02/23 10:45:45][INFO] [LossBreakdown] body_pose=1.0088 betas=0.3672 go_c=0.1491 go_gv=4.0803(masked) transl_vel=0.3573
+[02/23 10:45:45][INFO] [LossBreakdown] body_pose=0.1125 betas=0.0432 go_c=0.0218 go_gv=3.7225(masked) transl_vel=0.0594
+[02/23 10:45:46][INFO] [LossBreakdown] body_pose=1.2532 betas=0.9458 go_c=0.1440 go_gv=4.3591(masked) transl_vel=0.4074
+[02/23 10:45:46][INFO] [LossBreakdown] body_pose=0.3605 betas=0.2481 go_c=0.2523 go_gv=3.9619(masked) transl_vel=0.2139
+[02/23 10:45:47][INFO] [LossBreakdown] body_pose=1.0284 betas=0.6614 go_c=0.1550 go_gv=4.7681(masked) transl_vel=0.3752
+[02/23 10:45:47][INFO] [LossBreakdown] body_pose=0.3493 betas=0.3410 go_c=0.0676 go_gv=4.4903(masked) transl_vel=0.1126
+[02/23 10:45:48][INFO] [LossBreakdown] body_pose=1.1016 betas=0.6361 go_c=0.3531 go_gv=3.9972(masked) transl_vel=0.9326
+[02/23 10:45:48][INFO] [LossBreakdown] body_pose=0.2258 betas=0.2775 go_c=0.2156 go_gv=4.1025(masked) transl_vel=0.2388
+[02/23 10:46:01][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[02/23 10:46:01][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9897 pred=+0.9894 delta(pred-gt)=-0.0004
+[02/23 10:46:01][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[-0.03946379 -3.0817177 0.04087318] global_orient0_aa(pred)=[ 0.00483737 -1.7396638 -0.03523872]
+[02/23 10:46:01][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-176.58,+1.47,+1.51) pred=(-99.67,-1.20,-1.33) pred_vs_gt=(+76.98,+2.84,+2.68)
+[02/23 10:46:01][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-77.65
+[02/23 11:04:09][INFO] [Exp Name]: finetune_
+[02/23 11:04:09][INFO] [GPU x Batch] = 1 x 2
+[02/23 11:04:09][INFO] [UnityDataset] Found 1 sequences.
+[02/23 11:04:09][INFO] [Train Dataset][9/9]: name=unity, size=1, genmo.datasets.unity_dataset.UnityDataset
+[02/23 11:04:09][INFO] [Train Dataset][All]: ConcatDataset size=1
+[02/23 11:04:09][INFO]
+[02/23 11:04:09][INFO] [UnityDataset] Found 1 sequences.
+[02/23 11:04:09][INFO] [Val Dataset][7/7]: name=unity_val, size=1, genmo.datasets.unity_dataset.UnityDataset
+[02/23 11:04:09][INFO]
+[02/23 11:04:16][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 11:04:46][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[02/23 11:05:02][INFO] Start Fitting...
+[02/23 11:05:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 11:05:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/data.py:106: Total length of `DataLoader` across ranks is zero. Please make sure this was your intention.
+
+[02/23 11:05:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/data.py:106: Total length of `CombinedLoader` across ranks is zero. Please make sure this was your intention.
+
+[02/23 11:05:06][INFO] End of script.
+[02/23 11:33:33][INFO] [Exp Name]: finetune_
+[02/23 11:33:33][INFO] [GPU x Batch] = 1 x 2
+[02/23 11:33:33][INFO] [UnityDataset] Found 4 sequences.
+[02/23 11:33:33][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 11:33:33][INFO] [Train Dataset][All]: ConcatDataset size=4
+[02/23 11:33:33][INFO]
+[02/23 11:33:33][INFO] [UnityDataset] Found 4 sequences.
+[02/23 11:33:33][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 11:33:33][INFO]
+[02/23 11:33:40][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 11:40:02][INFO] [Exp Name]: finetune_
+[02/23 11:40:02][INFO] [GPU x Batch] = 1 x 2
+[02/23 11:40:02][INFO] [UnityDataset] Found 4 sequences.
+[02/23 11:40:02][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 11:40:02][INFO] [Train Dataset][All]: ConcatDataset size=4
+[02/23 11:40:02][INFO]
+[02/23 11:40:02][INFO] [UnityDataset] Found 4 sequences.
+[02/23 11:40:02][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 11:40:02][INFO]
+[02/23 11:40:10][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 11:40:36][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[02/23 11:40:48][INFO] Start Fitting...
+[02/23 11:40:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 11:40:49][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 11:40:51][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 11:40:52][INFO] [LossBreakdown] body_pose=1.6681 betas=0.7046 go_c=1.4145 go_gv=0.4977(masked) transl_vel=0.1287
+[02/23 11:40:52][INFO] [LossBreakdown] body_pose=0.1412 betas=0.0744 go_c=0.0389 go_gv=0.0105(masked) transl_vel=0.0398
+[02/23 11:40:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 11:40:53][INFO] [LossBreakdown] body_pose=1.3458 betas=0.6246 go_c=0.1120 go_gv=0.0221(masked) transl_vel=0.0685
+[02/23 11:40:54][INFO] [LossBreakdown] body_pose=0.8352 betas=0.1530 go_c=0.0654 go_gv=0.0121(masked) transl_vel=0.0487
+[02/23 11:40:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[02/23 11:41:06][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[02/23 11:41:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9981 delta(pred-gt)=+0.0026
+[02/23 11:41:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04296682 3.0540771 -0.04666773] global_orient0_aa(pred)=[ 0.02067284 -2.883098 0.05018289]
+[02/23 11:41:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+175.05,+1.82,+1.53) pred=(-165.23,+2.07,-0.55) pred_vs_gt=(+19.79,-0.43,+2.06)
+[02/23 11:41:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-16.26
+[02/23 11:41:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 11:41:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 11:41:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 11:41:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 11:41:56][INFO] ✅[FIT][Epoch 0] finished! 01:08→55:51 | loss_epoch=56.4
+[02/23 11:41:56][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 11:41:57][INFO] [LossBreakdown] body_pose=1.2618 betas=0.8335 go_c=0.1153 go_gv=0.0505(masked) transl_vel=0.2157
+[02/23 11:41:57][INFO] [LossBreakdown] body_pose=0.4270 betas=0.3173 go_c=0.0687 go_gv=0.0264(masked) transl_vel=0.2087
+[02/23 11:41:57][INFO] [LossBreakdown] body_pose=1.1172 betas=0.6993 go_c=0.2801 go_gv=0.0615(masked) transl_vel=0.5573
+[02/23 11:41:58][INFO] [LossBreakdown] body_pose=0.5848 betas=0.2742 go_c=0.1912 go_gv=0.0251(masked) transl_vel=0.2742
+[02/23 11:42:06][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[02/23 11:42:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9967 pred=+0.9955 delta(pred-gt)=-0.0012
+[02/23 11:42:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04501803 3.124892 -0.09182652] global_orient0_aa(pred)=[-0.02404302 -2.1931655 -0.0739034 ]
+[02/23 11:42:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+179.19,+3.38,+1.63) pred=(-125.69,-3.57,-0.57) pred_vs_gt=(+55.26,+6.91,+2.31)
+[02/23 11:42:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-52.66
+[02/23 11:42:55][INFO] ✅[FIT][Epoch 1] finished! 02:06→50:40 | loss_epoch=33.4
+[02/23 11:42:55][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/23 11:42:55][INFO] [LossBreakdown] body_pose=1.9728 betas=0.5189 go_c=0.0942 go_gv=0.0172(masked) transl_vel=0.1377
+[02/23 11:42:55][INFO] [LossBreakdown] body_pose=0.1839 betas=0.0661 go_c=0.0242 go_gv=0.0050(masked) transl_vel=0.0209
+[02/23 11:42:56][INFO] [LossBreakdown] body_pose=1.2146 betas=0.6297 go_c=0.2220 go_gv=0.0521(masked) transl_vel=0.4761
+[02/23 11:42:56][INFO] [LossBreakdown] body_pose=0.1191 betas=0.0817 go_c=0.0309 go_gv=0.0081(masked) transl_vel=0.0781
+[02/23 11:43:08][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[02/23 11:43:08][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9935 pred=+0.9803 delta(pred-gt)=-0.0132
+[02/23 11:43:08][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[-0.01973592 -3.0925677 0.04800931] global_orient0_aa(pred)=[-0.12548979 -1.5818633 -0.09459537]
+[02/23 11:43:08][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-177.20,+1.76,+0.77) pred=(-90.88,-8.01,+1.19) pred_vs_gt=(+86.27,+9.73,-0.90)
+[02/23 11:43:08][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-85.53
+[02/23 11:45:25][INFO] [Exp Name]: finetune_
+[02/23 11:45:25][INFO] [GPU x Batch] = 1 x 2
+[02/23 11:45:25][INFO] [UnityDataset] Found 4 sequences.
+[02/23 11:45:25][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 11:45:25][INFO] [Train Dataset][All]: ConcatDataset size=4
+[02/23 11:45:25][INFO]
+[02/23 11:45:25][INFO] [UnityDataset] Found 4 sequences.
+[02/23 11:45:25][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 11:45:25][INFO]
+[02/23 11:45:32][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 11:45:57][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[02/23 11:46:09][INFO] Start Fitting...
+[02/23 11:46:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 11:46:11][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 11:46:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 11:46:13][INFO] [LossBreakdown] body_pose=1.6681 betas=0.7046 go_c=1.4145 go_gv=0.4977 transl_vel=0.1287 gogv_mask=1.00
+[02/23 11:46:13][INFO] [LossBreakdown] body_pose=0.1412 betas=0.0744 go_c=0.0389 go_gv=0.0105 transl_vel=0.0398 gogv_mask=1.00
+[02/23 11:46:15][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 11:46:15][INFO] [LossBreakdown] body_pose=1.3458 betas=0.6246 go_c=0.1120 go_gv=0.0221 transl_vel=0.0685 gogv_mask=1.00
+[02/23 11:46:16][INFO] [LossBreakdown] body_pose=0.8352 betas=0.1530 go_c=0.0654 go_gv=0.0121 transl_vel=0.0487 gogv_mask=1.00
+[02/23 11:46:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[02/23 11:46:28][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[02/23 11:46:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9981 delta(pred-gt)=+0.0026
+[02/23 11:46:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04296682 3.0540771 -0.04666773] global_orient0_aa(pred)=[ 0.02067284 -2.883098 0.05018289]
+[02/23 11:46:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+175.05,+1.82,+1.53) pred=(-165.23,+2.07,-0.55) pred_vs_gt=(+19.79,-0.43,+2.06)
+[02/23 11:46:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-16.26
+[02/23 11:47:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 11:47:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 11:47:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 11:47:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 11:47:20][INFO] ✅[FIT][Epoch 0] finished! 01:10→57:32 | loss_epoch=56.4
+[02/23 11:47:20][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 11:47:20][INFO] [LossBreakdown] body_pose=1.2618 betas=0.8335 go_c=0.1153 go_gv=0.0505 transl_vel=0.2157 gogv_mask=1.00
+[02/23 11:47:20][INFO] [LossBreakdown] body_pose=0.4270 betas=0.3173 go_c=0.0687 go_gv=0.0264 transl_vel=0.2087 gogv_mask=1.00
+[02/23 11:47:21][INFO] [LossBreakdown] body_pose=1.1172 betas=0.6993 go_c=0.2801 go_gv=0.0615 transl_vel=0.5573 gogv_mask=1.00
+[02/23 11:47:21][INFO] [LossBreakdown] body_pose=0.5848 betas=0.2742 go_c=0.1912 go_gv=0.0251 transl_vel=0.2742 gogv_mask=1.00
+[02/23 11:47:30][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[02/23 11:47:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9967 pred=+0.9955 delta(pred-gt)=-0.0012
+[02/23 11:47:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04501803 3.124892 -0.09182652] global_orient0_aa(pred)=[-0.02404302 -2.1931655 -0.0739034 ]
+[02/23 11:47:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+179.19,+3.38,+1.63) pred=(-125.69,-3.57,-0.57) pred_vs_gt=(+55.26,+6.91,+2.31)
+[02/23 11:47:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-52.66
+[02/23 11:48:20][INFO] ✅[FIT][Epoch 1] finished! 02:11→52:24 | loss_epoch=33.4
+[02/23 11:48:20][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/23 11:48:20][INFO] [LossBreakdown] body_pose=1.9728 betas=0.5189 go_c=0.0942 go_gv=0.0172 transl_vel=0.1377 gogv_mask=1.00
+[02/23 11:48:20][INFO] [LossBreakdown] body_pose=0.1839 betas=0.0661 go_c=0.0242 go_gv=0.0050 transl_vel=0.0209 gogv_mask=1.00
+[02/23 11:48:21][INFO] [LossBreakdown] body_pose=1.2146 betas=0.6297 go_c=0.2220 go_gv=0.0521 transl_vel=0.4761 gogv_mask=1.00
+[02/23 11:48:21][INFO] [LossBreakdown] body_pose=0.1191 betas=0.0817 go_c=0.0309 go_gv=0.0081 transl_vel=0.0781 gogv_mask=1.00
+[02/23 11:48:33][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[02/23 11:48:33][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9935 pred=+0.9803 delta(pred-gt)=-0.0132
+[02/23 11:48:33][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[-0.01973592 -3.0925677 0.04800931] global_orient0_aa(pred)=[-0.12548979 -1.5818633 -0.09459537]
+[02/23 11:48:33][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-177.20,+1.76,+0.77) pred=(-90.88,-8.01,+1.19) pred_vs_gt=(+86.27,+9.73,-0.90)
+[02/23 11:48:33][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-85.53
+[02/23 11:49:18][INFO] ✅[FIT][Epoch 2] finished! 03:09→49:25 | loss_epoch=30
+[02/23 11:49:18][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[02/23 11:49:19][INFO] [LossBreakdown] body_pose=1.5128 betas=1.0232 go_c=0.2511 go_gv=0.0343 transl_vel=0.1587 gogv_mask=1.00
+[02/23 11:49:19][INFO] [LossBreakdown] body_pose=0.1509 betas=0.1578 go_c=0.0821 go_gv=0.0090 transl_vel=0.0372 gogv_mask=1.00
+[02/23 11:49:19][INFO] [LossBreakdown] body_pose=0.9267 betas=0.5804 go_c=0.1038 go_gv=0.0246 transl_vel=0.2819 gogv_mask=1.00
+[02/23 11:49:19][INFO] [LossBreakdown] body_pose=0.2679 betas=0.1877 go_c=0.1070 go_gv=0.0116 transl_vel=0.1153 gogv_mask=1.00
+[02/23 11:49:28][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[02/23 11:49:28][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+0.9864 pred=+1.0136 delta(pred-gt)=+0.0272
+[02/23 11:49:28][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03733476 3.0002954 -0.3175738 ] global_orient0_aa(pred)=[ 0.03163789 -2.1958115 0.02402818]
+[02/23 11:49:28][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+172.91,+12.12,+0.67) pred=(-125.84,+1.66,-0.80) pred_vs_gt=(+61.68,+10.19,+2.79)
+[02/23 11:49:28][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-56.74
+[02/23 11:50:20][INFO] ✅[FIT][Epoch 3] finished! 04:11→48:07 | loss_epoch=36.6
+[02/23 11:50:20][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[02/23 11:50:20][INFO] [LossBreakdown] body_pose=1.0052 betas=0.5240 go_c=0.3248 go_gv=0.0217 transl_vel=0.2968 gogv_mask=1.00
+[02/23 11:50:21][INFO] [LossBreakdown] body_pose=0.3843 betas=0.2304 go_c=0.0976 go_gv=0.0104 transl_vel=0.1398 gogv_mask=1.00
+[02/23 11:50:21][INFO] [LossBreakdown] body_pose=1.5549 betas=0.6119 go_c=0.1627 go_gv=0.0308 transl_vel=2.0661 gogv_mask=1.00
+[02/23 11:50:21][INFO] [LossBreakdown] body_pose=0.3361 betas=0.2265 go_c=0.1074 go_gv=0.0160 transl_vel=0.5221 gogv_mask=1.00
+[02/23 11:50:36][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=Sizes of tensors must match except in dimension 1. Expected size 1 but got size 2 for tensor number 1 in the list.
+[02/23 11:50:36][INFO] [VisUnityVal] e004_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9933 pred=+0.9888 delta(pred-gt)=-0.0045
+[02/23 11:50:36][INFO] [VisUnityVal] e004_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.06922989 2.915525 -0.01220466] global_orient0_aa(pred)=[ 0.01451751 -1.5911481 0.09048174]
+[02/23 11:50:36][INFO] [VisUnityVal] e004_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(+167.11,+0.78,+2.63) pred=(-91.13,+3.85,+2.73) pred_vs_gt=(+101.78,-2.97,-0.78)
+[02/23 11:50:36][INFO] [VisUnityVal] e004_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-105.62
+[02/23 11:58:53][INFO] [Exp Name]: finetune_
+[02/23 11:58:53][INFO] [GPU x Batch] = 1 x 2
+[02/23 11:58:54][INFO] [UnityDataset] Found 4 sequences.
+[02/23 11:58:54][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 11:58:54][INFO] [Train Dataset][All]: ConcatDataset size=4
+[02/23 11:58:54][INFO]
+[02/23 11:58:54][INFO] [UnityDataset] Found 4 sequences.
+[02/23 11:58:54][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 11:58:54][INFO]
+[02/23 11:59:00][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 11:59:26][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_3/checkpoints'
+[02/23 11:59:36][INFO] Start Fitting...
+[02/23 11:59:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 11:59:38][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 11:59:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 11:59:39][INFO] [LossBreakdown] body_pose=1.4765 betas=0.8095 go_c=0.6506 go_gv=0.2391 transl_vel=0.1411 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=nan tv_abs_mean(tgt)=nan tv_zero_frac(pred)=nan static_conf_mean=0.63 static_conf_hi_frac=0.41
+[02/23 11:59:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 11:59:40][INFO] [LossBreakdown] body_pose=1.4157 betas=0.5889 go_c=0.0886 go_gv=0.0232 transl_vel=0.0679 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=nan tv_abs_mean(tgt)=nan tv_zero_frac(pred)=nan static_conf_mean=0.71 static_conf_hi_frac=0.61
+[02/23 11:59:51][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=shape '[1, 63, -1, 3]' is invalid for input of size 66 shapes={'static_conf_logits': (2, 120, 6), 'pred_smpl_params_global.transl': (120, 3), 'pred_smpl_params_global.global_orient': (120, 3), 'pred_smpl_params_global.body_pose': (120, 63), 'pred_smpl_params_global.betas': (120, 10), 'render_pred_params_global.transl': (120, 3), 'render_pred_params_global.global_orient': (120, 3), 'render_pred_params_global.body_pose': (120, 63), 'render_pred_params_global.betas': (120, 10)}
+[02/23 11:59:51][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9836 pred=+0.9966 delta(pred-gt)=+0.0129
+[02/23 11:59:51][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04353233 3.0365002 -0.1219962 ] global_orient0_aa(pred)=[-0.02932962 -3.0059474 -0.01442418]
+[02/23 11:59:51][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+174.19,+4.67,+1.41) pred=(-172.24,-0.62,+1.08) pred_vs_gt=(+13.62,+5.23,+0.87)
+[02/23 11:59:51][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-14.94
+[02/23 12:00:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:00:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:00:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:00:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:00:40][INFO] ✅[FIT][Epoch 0] finished! 01:03→52:00 | loss_epoch=1.67
+[02/23 12:00:40][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 12:00:40][INFO] [LossBreakdown] body_pose=0.8968 betas=0.5734 go_c=0.0672 go_gv=0.0314 transl_vel=0.1976 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=nan tv_abs_mean(tgt)=nan tv_zero_frac(pred)=nan static_conf_mean=0.64 static_conf_hi_frac=0.45
+[02/23 12:00:41][INFO] [LossBreakdown] body_pose=0.5815 betas=0.4294 go_c=0.2357 go_gv=0.0279 transl_vel=0.6054 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=nan tv_abs_mean(tgt)=nan tv_zero_frac(pred)=nan static_conf_mean=0.48 static_conf_hi_frac=0.23
+[02/23 12:00:49][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=shape '[1, 63, -1, 3]' is invalid for input of size 66 shapes={'static_conf_logits': (2, 120, 6), 'pred_smpl_params_global.transl': (120, 3), 'pred_smpl_params_global.global_orient': (120, 3), 'pred_smpl_params_global.body_pose': (120, 63), 'pred_smpl_params_global.betas': (120, 10), 'render_pred_params_global.transl': (120, 3), 'render_pred_params_global.global_orient': (120, 3), 'render_pred_params_global.body_pose': (120, 63), 'render_pred_params_global.betas': (120, 10)}
+[02/23 12:00:49][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9911 pred=+0.9981 delta(pred-gt)=+0.0070
+[02/23 12:00:49][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03549716 2.923601 -0.0567783 ] global_orient0_aa(pred)=[0.05707453 2.1862755 0.03500119]
+[02/23 12:00:49][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+167.57,+2.35,+1.14) pred=(+125.30,-0.23,+3.11) pred_vs_gt=(-42.35,+2.94,-1.37)
+[02/23 12:00:49][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+40.97
+[02/23 12:01:41][INFO] ✅[FIT][Epoch 1] finished! 02:05→50:00 | loss_epoch=1.09
+[02/23 12:01:41][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/23 12:01:42][INFO] [LossBreakdown] body_pose=1.6485 betas=0.2772 go_c=0.0459 go_gv=0.0308 transl_vel=0.0742 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=nan tv_abs_mean(tgt)=nan tv_zero_frac(pred)=nan static_conf_mean=0.66 static_conf_hi_frac=0.58
+[02/23 12:01:42][INFO] [LossBreakdown] body_pose=0.5120 betas=0.2072 go_c=0.1026 go_gv=0.0487 transl_vel=0.5541 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=nan tv_abs_mean(tgt)=nan tv_zero_frac(pred)=nan static_conf_mean=0.66 static_conf_hi_frac=0.61
+[02/23 12:01:54][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=shape '[1, 63, -1, 3]' is invalid for input of size 66 shapes={'static_conf_logits': (2, 120, 6), 'pred_smpl_params_global.transl': (120, 3), 'pred_smpl_params_global.global_orient': (120, 3), 'pred_smpl_params_global.body_pose': (120, 63), 'pred_smpl_params_global.betas': (120, 10), 'render_pred_params_global.transl': (120, 3), 'render_pred_params_global.global_orient': (120, 3), 'render_pred_params_global.body_pose': (120, 63), 'render_pred_params_global.betas': (120, 10)}
+[02/23 12:01:54][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9881 pred=+0.9904 delta(pred-gt)=+0.0023
+[02/23 12:01:54][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[0.02152805 2.9734776 0.01178094] global_orient0_aa(pred)=[-0.1009894 -1.9680827 -0.08459024]
+[02/23 12:01:54][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(+170.37,-0.38,+0.86) pred=(-112.99,-6.12,+1.82) pred_vs_gt=(+76.69,+5.82,+0.02)
+[02/23 12:01:54][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-83.30
+[02/23 12:02:39][INFO] ✅[FIT][Epoch 2] finished! 03:02→47:36 | loss_epoch=1.4
+[02/23 12:02:39][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[02/23 12:02:39][INFO] [LossBreakdown] body_pose=0.8978 betas=0.2160 go_c=0.0939 go_gv=0.0472 transl_vel=0.2621 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=nan tv_abs_mean(tgt)=nan tv_zero_frac(pred)=nan static_conf_mean=0.41 static_conf_hi_frac=0.05
+[02/23 12:02:39][INFO] [LossBreakdown] body_pose=0.3063 betas=0.1305 go_c=0.0451 go_gv=0.0279 transl_vel=0.2729 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=nan tv_abs_mean(tgt)=nan tv_zero_frac(pred)=nan static_conf_mean=0.51 static_conf_hi_frac=0.23
+[02/23 12:02:47][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=shape '[1, 63, -1, 3]' is invalid for input of size 66 shapes={'static_conf_logits': (2, 120, 6), 'pred_smpl_params_global.transl': (120, 3), 'pred_smpl_params_global.global_orient': (120, 3), 'pred_smpl_params_global.body_pose': (120, 63), 'pred_smpl_params_global.betas': (120, 10), 'render_pred_params_global.transl': (120, 3), 'render_pred_params_global.global_orient': (120, 3), 'render_pred_params_global.body_pose': (120, 63), 'render_pred_params_global.betas': (120, 10)}
+[02/23 12:02:47][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+0.9738 pred=+1.0002 delta(pred-gt)=+0.0263
+[02/23 12:02:47][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.06205852 3.0070703 -0.13520636] global_orient0_aa(pred)=[0.10226834 2.964591 0.1197371 ]
+[02/23 12:02:47][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+172.59,+5.28,+2.02) pred=(+169.92,-4.25,+4.33) pred_vs_gt=(-2.78,+9.74,-1.07)
+[02/23 12:02:47][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+0.45
+[02/23 12:03:38][INFO] ✅[FIT][Epoch 3] finished! 04:01→46:12 | loss_epoch=1.26
+[02/23 12:03:38][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[02/23 12:03:38][INFO] [LossBreakdown] body_pose=0.3193 betas=0.0961 go_c=0.0481 go_gv=0.0344 transl_vel=0.2395 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=nan tv_abs_mean(tgt)=nan tv_zero_frac(pred)=nan static_conf_mean=0.61 static_conf_hi_frac=0.43
+[02/23 12:03:38][INFO] [LossBreakdown] body_pose=0.8192 betas=0.1728 go_c=0.1654 go_gv=0.0387 transl_vel=2.0901 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=nan tv_abs_mean(tgt)=nan tv_zero_frac(pred)=nan static_conf_mean=0.55 static_conf_hi_frac=0.34
+[02/23 12:03:50][WARNING] [VisUnityVal] Global postprocess failed; using raw outputs. err=shape '[1, 63, -1, 3]' is invalid for input of size 66 shapes={'static_conf_logits': (2, 120, 6), 'pred_smpl_params_global.transl': (120, 3), 'pred_smpl_params_global.global_orient': (120, 3), 'pred_smpl_params_global.body_pose': (120, 63), 'pred_smpl_params_global.betas': (120, 10), 'render_pred_params_global.transl': (120, 3), 'render_pred_params_global.global_orient': (120, 3), 'render_pred_params_global.body_pose': (120, 63), 'render_pred_params_global.betas': (120, 10)}
+[02/23 12:03:50][INFO] [VisUnityVal] e004_101_biboo_birthday_speech_explosion_2 root_y0: gt=+1.0020 pred=+0.9945 delta(pred-gt)=-0.0075
+[02/23 12:03:50][INFO] [VisUnityVal] e004_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.02742666 3.0154672 -0.04629429] global_orient0_aa(pred)=[-0.1426638 -1.839236 -0.04631034]
+[02/23 12:03:50][INFO] [VisUnityVal] e004_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(+172.81,+1.82,+0.93) pred=(-105.77,-6.10,+4.26) pred_vs_gt=(+81.37,+8.27,-2.32)
+[02/23 12:03:50][INFO] [VisUnityVal] e004_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-84.71
+[02/23 12:04:36][INFO] ✅[FIT][Epoch 4] finished! 04:59→44:56 | loss_epoch=1.78
+[02/23 12:15:28][INFO] [Exp Name]: finetune_
+[02/23 12:15:28][INFO] [GPU x Batch] = 1 x 2
+[02/23 12:15:28][INFO] [UnityDataset] Found 4 sequences.
+[02/23 12:15:28][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 12:15:28][INFO] [Train Dataset][All]: ConcatDataset size=4
+[02/23 12:15:28][INFO]
+[02/23 12:15:28][INFO] [UnityDataset] Found 4 sequences.
+[02/23 12:15:28][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 12:15:28][INFO]
+[02/23 12:15:36][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 12:16:01][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_4/checkpoints'
+[02/23 12:16:12][INFO] Start Fitting...
+[02/23 12:16:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 12:16:14][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 12:16:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 12:16:16][INFO] [LossBreakdown] body_pose=1.5766 betas=0.8351 go_c=1.2840 go_gv=0.9389 transl_vel=3.6813 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.00 static_conf_mean=0.63 static_conf_hi_frac=0.41
+[02/23 12:16:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 12:16:17][INFO] [LossBreakdown] body_pose=1.4308 betas=1.0697 go_c=0.3230 go_gv=3.3309 transl_vel=3.9670 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0013 tv_zero_frac(pred)=0.00 static_conf_mean=0.71 static_conf_hi_frac=0.61
+[02/23 12:16:29][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9836 pred=+2.3192 delta(pred-gt)=+1.3356
+[02/23 12:16:29][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04353233 3.0365002 -0.1219962 ] global_orient0_aa(pred)=[-0.02432382 -3.0136063 0.04340264]
+[02/23 12:16:29][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+174.19,+4.67,+1.41) pred=(-172.68,+1.58,+1.03) pred_vs_gt=(+13.17,+3.03,+0.69)
+[02/23 12:16:29][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-13.62
+[02/23 12:17:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:17:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:17:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:17:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:17:21][INFO] ✅[FIT][Epoch 0] finished! 01:08→55:53 | loss_epoch=2.38
+[02/23 12:17:21][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 12:17:21][INFO] [LossBreakdown] body_pose=0.7765 betas=0.5609 go_c=0.3311 go_gv=0.5497 transl_vel=4.1842 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.67 static_conf_hi_frac=0.50
+[02/23 12:17:22][INFO] [LossBreakdown] body_pose=0.7527 betas=0.5592 go_c=1.0511 go_gv=0.9876 transl_vel=10.5087 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0013 tv_abs_mean(tgt)=0.0035 tv_zero_frac(pred)=0.00 static_conf_mean=0.52 static_conf_hi_frac=0.33
+[02/23 12:17:29][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9911 pred=+2.3256 delta(pred-gt)=+1.3345
+[02/23 12:17:29][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03549716 2.923601 -0.0567783 ] global_orient0_aa(pred)=[ 0.01339545 2.380264 -0.01157499]
+[02/23 12:17:29][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+167.57,+2.35,+1.14) pred=(+136.38,+0.70,+0.36) pred_vs_gt=(-31.16,+1.44,+1.11)
+[02/23 12:17:29][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+31.20
+[02/23 12:52:25][INFO] [Exp Name]: finetune_
+[02/23 12:52:25][INFO] [GPU x Batch] = 1 x 2
+[02/23 12:52:25][INFO] [UnityDataset] Found 4 sequences.
+[02/23 12:52:25][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 12:52:25][INFO] [Train Dataset][All]: ConcatDataset size=4
+[02/23 12:52:25][INFO]
+[02/23 12:52:25][INFO] [UnityDataset] Found 4 sequences.
+[02/23 12:52:25][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 12:52:25][INFO]
+[02/23 12:52:32][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 12:52:56][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_5/checkpoints'
+[02/23 12:53:07][INFO] Start Fitting...
+[02/23 12:53:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 12:53:08][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 12:53:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 12:53:11][INFO] [LossBreakdown] body_pose=1.5787 betas=0.8361 go_c=1.2835 go_gv=0.9397 transl_vel=3.6829 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.00 static_conf_mean=0.62 static_conf_hi_frac=0.40
+[02/23 12:53:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 12:53:12][INFO] [LossBreakdown] body_pose=1.4304 betas=1.0695 go_c=0.3231 go_gv=3.3310 transl_vel=3.9648 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0013 tv_zero_frac(pred)=0.00 static_conf_mean=0.71 static_conf_hi_frac=0.61
+[02/23 12:53:24][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9836 pred=+2.3195 delta(pred-gt)=+1.3359
+[02/23 12:53:25][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04353233 3.0365002 -0.1219962 ] global_orient0_aa(pred)=[-0.02440614 -3.0214126 0.04362149]
+[02/23 12:53:25][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+174.19,+4.67,+1.41) pred=(-173.12,+1.59,+1.02) pred_vs_gt=(+12.73,+3.02,+0.69)
+[02/23 12:53:25][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-13.15
+[02/23 12:54:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:54:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:54:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:54:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 12:54:17][INFO] ✅[FIT][Epoch 0] finished! 01:10→57:26 | loss_epoch=2.38
+[02/23 12:54:18][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 12:54:18][INFO] [LossBreakdown] body_pose=0.7812 betas=0.5562 go_c=0.3313 go_gv=0.5481 transl_vel=4.2051 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.56
+[02/23 12:54:18][INFO] [LossBreakdown] body_pose=0.7493 betas=0.5651 go_c=1.0513 go_gv=0.9852 transl_vel=10.4663 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0013 tv_abs_mean(tgt)=0.0035 tv_zero_frac(pred)=0.00 static_conf_mean=0.55 static_conf_hi_frac=0.42
+[02/23 12:54:26][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9911 pred=+2.3269 delta(pred-gt)=+1.3359
+[02/23 12:54:26][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03549716 2.923601 -0.0567783 ] global_orient0_aa(pred)=[ 0.01401115 2.400957 -0.01374081]
+[02/23 12:54:26][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+167.57,+2.35,+1.14) pred=(+137.57,+0.80,+0.36) pred_vs_gt=(-29.97,+1.35,+1.09)
+[02/23 12:54:26][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+30.09
+[02/23 12:55:23][INFO] ✅[FIT][Epoch 1] finished! 02:15→54:12 | loss_epoch=1.99
+[02/23 12:55:23][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/23 12:55:23][INFO] [LossBreakdown] body_pose=1.3414 betas=0.3831 go_c=0.3956 go_gv=0.3923 transl_vel=2.3372 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0013 tv_zero_frac(pred)=0.00 static_conf_mean=0.74 static_conf_hi_frac=0.68
+[02/23 12:55:23][INFO] [LossBreakdown] body_pose=0.7989 betas=0.4087 go_c=0.7784 go_gv=1.0955 transl_vel=9.6571 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0034 tv_zero_frac(pred)=0.00 static_conf_mean=0.71 static_conf_hi_frac=0.65
+[02/23 12:55:37][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9881 pred=+2.3069 delta(pred-gt)=+1.3189
+[02/23 12:55:37][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[0.02152805 2.9734776 0.01178094] global_orient0_aa(pred)=[-0.03877718 -2.1807377 0.02061607]
+[02/23 12:55:37][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(+170.37,-0.38,+0.86) pred=(-124.96,+0.02,+2.05) pred_vs_gt=(+64.67,-0.19,-1.23)
+[02/23 12:55:37][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-67.15
+[02/23 14:37:25][INFO] [Exp Name]: finetune_
+[02/23 14:37:25][INFO] [GPU x Batch] = 1 x 2
+[02/23 14:37:25][INFO] [UnityDataset] Found 4 sequences.
+[02/23 14:37:25][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 14:37:25][INFO] [Train Dataset][All]: ConcatDataset size=4
+[02/23 14:37:25][INFO]
+[02/23 14:37:25][INFO] [UnityDataset] Found 4 sequences.
+[02/23 14:37:25][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 14:37:25][INFO]
+[02/23 14:37:32][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 14:37:53][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[02/23 14:38:03][INFO] Start Fitting...
+[02/23 14:38:05][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 14:38:05][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 14:38:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 14:38:09][INFO] [LossBreakdown] body_pose=1.5785 betas=0.8365 go_c=1.2833 go_gv=4.6988 transl_vel=2647.6008 gogv_mask=1.00 tv_mask=0.67 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0053 tv_zero_frac(pred)=0.00 static_conf_mean=0.62 static_conf_hi_frac=0.40
+[02/23 14:38:10][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 14:38:10][INFO] [LossBreakdown] body_pose=1.4304 betas=1.0695 go_c=0.3231 go_gv=16.6549 transl_vel=3.0988 gogv_mask=1.00 tv_mask=0.67 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0011 tv_zero_frac(pred)=0.00 static_conf_mean=0.71 static_conf_hi_frac=0.61
+[02/23 14:38:21][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9832 pred=+3.0004 delta(pred-gt)=+2.0172
+[02/23 14:38:22][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04353093 3.0365 -0.1219992 ] global_orient0_aa(pred)=[-0.0242567 -2.9832227 0.04410294]
+[02/23 14:38:22][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+174.19,+4.67,+1.41) pred=(-170.93,+1.61,+1.06) pred_vs_gt=(+14.91,+3.01,+0.65)
+[02/23 14:38:22][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-15.20
+[02/23 14:39:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 14:39:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 14:39:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 14:39:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 14:39:23][INFO] ✅[FIT][Epoch 0] finished! 01:19→1:04:40 | loss_epoch=29.2
+[02/23 14:39:23][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 14:39:23][INFO] [LossBreakdown] body_pose=0.8022 betas=0.5730 go_c=0.3352 go_gv=2.7380 transl_vel=68.4894 gogv_mask=1.00 tv_mask=0.67 tv_abs_mean(pred)=0.0012 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.62 static_conf_hi_frac=0.37
+[02/23 14:39:23][INFO] [LossBreakdown] body_pose=0.7813 betas=0.5549 go_c=1.0777 go_gv=4.9519 transl_vel=9.7527 gogv_mask=1.00 tv_mask=0.67 tv_abs_mean(pred)=0.0017 tv_abs_mean(tgt)=0.0030 tv_zero_frac(pred)=0.00 static_conf_mean=0.47 static_conf_hi_frac=0.19
+[02/23 14:39:32][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9932 pred=+3.0055 delta(pred-gt)=+2.0123
+[02/23 14:39:32][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03550317 2.923601 -0.05677668] global_orient0_aa(pred)=[ 0.01445789 2.395239 -0.01973981]
+[02/23 14:39:32][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+167.57,+2.35,+1.14) pred=(+137.24,+1.05,+0.28) pred_vs_gt=(-30.29,+1.08,+1.12)
+[02/23 14:39:32][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+31.87
+[02/23 14:42:53][INFO] [Exp Name]: finetune_
+[02/23 14:42:53][INFO] [GPU x Batch] = 1 x 2
+[02/23 14:42:53][INFO] [UnityDataset] Found 4 sequences.
+[02/23 14:42:53][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 14:42:53][INFO] [Train Dataset][All]: ConcatDataset size=4
+[02/23 14:42:53][INFO]
+[02/23 14:42:53][INFO] [UnityDataset] Found 4 sequences.
+[02/23 14:42:53][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 14:42:53][INFO]
+[02/23 14:42:59][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 14:43:21][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[02/23 14:43:30][INFO] Start Fitting...
+[02/23 14:43:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 14:43:32][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 14:43:34][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 14:43:34][INFO] [LossBreakdown] body_pose=1.5785 betas=0.8365 go_c=1.2833 go_gv=0.9398 transl_vel=0.3683 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.00 static_conf_mean=0.62 static_conf_hi_frac=0.40
+[02/23 14:43:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 14:43:35][INFO] [LossBreakdown] body_pose=1.4304 betas=1.0695 go_c=0.3231 go_gv=3.3310 transl_vel=0.3964 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0013 tv_zero_frac(pred)=0.00 static_conf_mean=0.71 static_conf_hi_frac=0.61
+[02/23 14:43:46][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9836 pred=+2.3214 delta(pred-gt)=+1.3378
+[02/23 14:43:46][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04353233 3.0365002 -0.1219962 ] global_orient0_aa(pred)=[-0.02500982 -2.9897113 0.04368195]
+[02/23 14:43:46][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+174.19,+4.67,+1.41) pred=(-171.31,+1.59,+1.08) pred_vs_gt=(+14.54,+3.03,+0.64)
+[02/23 14:43:46][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-15.41
+[02/23 14:44:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 14:44:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 14:44:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 14:44:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 14:44:44][INFO] ✅[FIT][Epoch 0] finished! 01:13→59:51 | loss_epoch=1.84
+[02/23 14:44:44][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 14:44:44][INFO] [LossBreakdown] body_pose=0.7624 betas=0.4687 go_c=0.3348 go_gv=0.5440 transl_vel=0.3956 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.65 static_conf_hi_frac=0.48
+[02/23 14:44:45][INFO] [LossBreakdown] body_pose=0.7588 betas=0.4215 go_c=1.0210 go_gv=0.9824 transl_vel=0.9842 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0035 tv_zero_frac(pred)=0.00 static_conf_mean=0.51 static_conf_hi_frac=0.28
+[02/23 14:44:53][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9911 pred=+2.3202 delta(pred-gt)=+1.3291
+[02/23 14:44:53][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03549716 2.923601 -0.0567783 ] global_orient0_aa(pred)=[ 0.01314779 2.3762538 -0.0114973 ]
+[02/23 14:44:53][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+167.57,+2.35,+1.14) pred=(+136.15,+0.70,+0.35) pred_vs_gt=(-31.38,+1.44,+1.12)
+[02/23 14:44:53][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+31.09
+[02/23 15:19:44][INFO] [Exp Name]: finetune_
+[02/23 15:19:44][INFO] [GPU x Batch] = 1 x 2
+[02/23 15:19:44][INFO] [UnityDataset] Found 3 sequences.
+[02/23 15:19:44][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[02/23 15:19:44][INFO] [Train Dataset][All]: ConcatDataset size=3
+[02/23 15:19:44][INFO]
+[02/23 15:19:44][INFO] [UnityDataset] Found 3 sequences.
+[02/23 15:19:44][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[02/23 15:19:44][INFO]
+[02/23 15:19:51][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 15:20:12][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[02/23 15:20:22][INFO] Start Fitting...
+[02/23 15:20:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 15:20:24][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 15:20:25][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 15:20:26][INFO] [LossBreakdown] body_pose=1.3200 betas=0.9864 go_c=0.2346 go_gv=0.2314 transl_vel=0.1814 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0008 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.57
+[02/23 15:20:28][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 15:20:40][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9958 pred=+3.0009 delta(pred-gt)=+2.0051
+[02/23 15:20:40][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03580198 3.0379734 -0.00811263] global_orient0_aa(pred)=[-0.02453166 -2.9984052 0.04587989]
+[02/23 15:20:40][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+174.08,+0.37,+1.33) pred=(-171.81,+1.68,+1.06) pred_vs_gt=(+14.12,-1.32,+0.14)
+[02/23 15:20:40][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-10.41
+[02/23 15:21:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:21:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:21:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:21:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:21:42][INFO] ✅[FIT][Epoch 0] finished! 01:19→1:04:56 | loss_epoch=1.57
+[02/23 15:21:42][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 15:21:42][INFO] [LossBreakdown] body_pose=1.3847 betas=0.3101 go_c=0.4762 go_gv=0.9624 transl_vel=2.4462 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.69
+[02/23 15:21:50][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9814 pred=+2.9825 delta(pred-gt)=+2.0012
+[02/23 15:21:50][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03605492 3.024373 -0.01587968] global_orient0_aa(pred)=[-0.02743093 -2.4208179 0.03038438]
+[02/23 15:21:50][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+173.31,+0.68,+1.33) pred=(-138.71,+0.83,+1.61) pred_vs_gt=(+47.99,-0.12,-0.30)
+[02/23 15:21:50][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-46.27
+[02/23 15:22:55][INFO] ✅[FIT][Epoch 1] finished! 02:32→1:00:50 | loss_epoch=3.39
+[02/23 15:22:55][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/23 15:22:55][INFO] [LossBreakdown] body_pose=1.3807 betas=0.2249 go_c=1.5140 go_gv=0.9577 transl_vel=0.5478 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0025 tv_zero_frac(pred)=0.00 static_conf_mean=0.55 static_conf_hi_frac=0.22
+[02/23 15:23:07][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9888 pred=+3.1733 delta(pred-gt)=+2.1845
+[02/23 15:23:07][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.07423093 3.1321795 -0.04776685] global_orient0_aa(pred)=[-0.03824747 -2.0422115 0.01303638]
+[02/23 15:23:07][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(+179.57,+1.76,+2.71) pred=(-117.03,-0.42,+1.89) pred_vs_gt=(+63.42,+2.18,+0.84)
+[02/23 15:23:07][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-58.17
+[02/23 15:42:57][INFO] [Exp Name]: finetune_
+[02/23 15:42:57][INFO] [GPU x Batch] = 1 x 2
+[02/23 15:42:57][INFO] [UnityDataset] Found 4 sequences.
+[02/23 15:42:57][INFO] [Train Dataset][9/9]: name=unity, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 15:42:57][INFO] [Train Dataset][All]: ConcatDataset size=4
+[02/23 15:42:57][INFO]
+[02/23 15:42:57][INFO] [UnityDataset] Found 4 sequences.
+[02/23 15:42:57][INFO] [Val Dataset][7/7]: name=unity_val, size=4, genmo.datasets.unity_dataset.UnityDataset
+[02/23 15:42:57][INFO]
+[02/23 15:43:04][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 15:43:25][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[02/23 15:43:36][INFO] Start Fitting...
+[02/23 15:43:38][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 15:43:38][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 15:43:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 15:43:40][INFO] [LossBreakdown] body_pose=1.5769 betas=0.8350 go_c=1.2842 go_gv=0.9390 transl_vel=264.9609 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0053 tv_zero_frac(pred)=0.00 static_conf_mean=0.63 static_conf_hi_frac=0.41
+[02/23 15:43:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 15:43:41][INFO] [LossBreakdown] body_pose=1.4308 betas=1.0697 go_c=0.3230 go_gv=3.3309 transl_vel=0.3243 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0011 tv_zero_frac(pred)=0.00 static_conf_mean=0.71 static_conf_hi_frac=0.61
+[02/23 15:43:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9832 pred=+0.9661 delta(pred-gt)=-0.0172
+[02/23 15:43:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04353233 3.0365002 -0.1219962 ] global_orient0_aa(pred)=[-0.02583855 -2.9500375 0.04226848]
+[02/23 15:43:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+174.19,+4.67,+1.41) pred=(-169.03,+1.53,+1.15) pred_vs_gt=(+16.81,+3.10,+0.57)
+[02/23 15:43:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-17.88
+[02/23 15:45:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:45:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:45:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:45:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:45:04][INFO] ✅[FIT][Epoch 0] finished! 01:27→1:11:26 | loss_epoch=4.54
+[02/23 15:45:04][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 15:45:04][INFO] [LossBreakdown] body_pose=0.7557 betas=0.4860 go_c=0.3306 go_gv=0.5456 transl_vel=6.7738 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.50 static_conf_hi_frac=0.19
+[02/23 15:45:07][INFO] [LossBreakdown] body_pose=0.7748 betas=0.4189 go_c=1.0174 go_gv=0.9862 transl_vel=0.8658 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0013 tv_abs_mean(tgt)=0.0030 tv_zero_frac(pred)=0.00 static_conf_mean=0.40 static_conf_hi_frac=0.04
+[02/23 15:45:17][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9932 pred=+0.9676 delta(pred-gt)=-0.0256
+[02/23 15:45:17][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03549716 2.923601 -0.0567783 ] global_orient0_aa(pred)=[ 0.01337807 2.3743515 -0.01218538]
+[02/23 15:45:17][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+167.57,+2.35,+1.14) pred=(+136.05,+0.73,+0.35) pred_vs_gt=(-31.49,+1.41,+1.11)
+[02/23 15:45:17][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+31.38
+[02/23 15:56:12][INFO] [Exp Name]: finetune_
+[02/23 15:56:12][INFO] [GPU x Batch] = 1 x 2
+[02/23 15:56:12][INFO] [UnityDataset] Found 7 sequences.
+[02/23 15:56:12][INFO] [Train Dataset][9/9]: name=unity, size=7, genmo.datasets.unity_dataset.UnityDataset
+[02/23 15:56:12][INFO] [Train Dataset][All]: ConcatDataset size=7
+[02/23 15:56:12][INFO]
+[02/23 15:56:12][INFO] [UnityDataset] Found 7 sequences.
+[02/23 15:56:12][INFO] [Val Dataset][7/7]: name=unity_val, size=7, genmo.datasets.unity_dataset.UnityDataset
+[02/23 15:56:12][INFO]
+[02/23 15:56:19][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 15:56:40][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[02/23 15:56:50][INFO] Start Fitting...
+[02/23 15:56:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 15:56:52][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 15:56:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 15:56:54][INFO] [LossBreakdown] body_pose=0.8471 betas=0.7525 go_c=1.0453 go_gv=0.7465 transl_vel=6.4394 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0031 tv_zero_frac(pred)=0.00 static_conf_mean=0.66 static_conf_hi_frac=0.59
+[02/23 15:56:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 15:56:55][INFO] [LossBreakdown] body_pose=0.7835 betas=0.8316 go_c=1.2897 go_gv=0.5276 transl_vel=0.6992 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0027 tv_zero_frac(pred)=0.00 static_conf_mean=0.78 static_conf_hi_frac=0.70
+[02/23 15:56:55][INFO] [LossBreakdown] body_pose=1.3344 betas=0.7926 go_c=0.4490 go_gv=0.4045 transl_vel=2.3376 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0019 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.64
+[02/23 15:57:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9859 pred=+0.9481 delta(pred-gt)=-0.0378
+[02/23 15:57:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03447273 3.047226 -0.11250602] global_orient0_aa(pred)=[-0.02344985 -2.3500917 0.03190917]
+[02/23 15:57:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+174.76,+4.28,+1.10) pred=(-134.65,+0.92,+1.53) pred_vs_gt=(+50.57,+3.39,-0.12)
+[02/23 15:57:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-50.03
+[02/23 15:58:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:58:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:58:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:58:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 15:58:32][INFO] ✅[FIT][Epoch 0] finished! 01:41→1:22:57 | loss_epoch=1.74
+[02/23 15:58:32][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 15:58:33][INFO] [LossBreakdown] body_pose=1.1288 betas=0.8065 go_c=0.8118 go_gv=0.6377 transl_vel=2.6879 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0019 tv_abs_mean(tgt)=0.0035 tv_zero_frac(pred)=0.00 static_conf_mean=0.17 static_conf_hi_frac=0.00
+[02/23 15:58:33][INFO] [LossBreakdown] body_pose=0.8703 betas=0.8201 go_c=0.4008 go_gv=0.5681 transl_vel=1.0606 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0020 tv_abs_mean(tgt)=0.0028 tv_zero_frac(pred)=0.00 static_conf_mean=0.20 static_conf_hi_frac=0.00
+[02/23 15:58:33][INFO] [LossBreakdown] body_pose=1.4836 betas=0.5674 go_c=1.0350 go_gv=1.3728 transl_vel=0.7300 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0013 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.00 static_conf_mean=0.34 static_conf_hi_frac=0.03
+[02/23 15:58:48][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+1.0010 pred=+0.9653 delta(pred-gt)=-0.0357
+[02/23 15:58:48][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[-0.07981459 -3.097574 0.11909015] global_orient0_aa(pred)=[0.01264388 2.3902996 0.00669794]
+[02/23 15:58:48][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-177.55,+4.34,+3.04) pred=(+136.96,-0.07,+0.63) pred_vs_gt=(-45.32,+4.50,+2.23)
+[02/23 15:58:48][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=+47.17
+[02/23 16:31:27][INFO] [Exp Name]: finetune_
+[02/23 16:31:27][INFO] [GPU x Batch] = 1 x 2
+[02/23 16:31:28][INFO] [UnityDataset] Found 7 sequences.
+[02/23 16:31:28][INFO] [Train Dataset][9/9]: name=unity, size=7, genmo.datasets.unity_dataset.UnityDataset
+[02/23 16:31:28][INFO] [Train Dataset][All]: ConcatDataset size=7
+[02/23 16:31:28][INFO]
+[02/23 16:31:28][INFO] [UnityDataset] Found 7 sequences.
+[02/23 16:31:28][INFO] [Val Dataset][7/7]: name=unity_val, size=7, genmo.datasets.unity_dataset.UnityDataset
+[02/23 16:31:28][INFO]
+[02/23 16:45:07][INFO] [Exp Name]: finetune_
+[02/23 16:45:07][INFO] [GPU x Batch] = 1 x 2
+[02/23 16:45:07][INFO] [UnityDataset] Found 9 sequences.
+[02/23 16:45:07][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 16:45:07][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 16:45:07][INFO]
+[02/23 16:45:07][INFO] [UnityDataset] Found 9 sequences.
+[02/23 16:45:07][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 16:45:07][INFO]
+[02/23 16:45:13][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 16:45:46][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[02/23 16:46:12][INFO] Start Fitting...
+[02/23 16:46:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 16:46:15][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 16:46:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 16:46:22][INFO] [LossBreakdown] body_pose=0.7120 betas=0.9778 go_c=1.2894 go_gv=0.9293 transl_vel=0.7432 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 16:46:25][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 16:46:25][INFO] [LossBreakdown] body_pose=1.3988 betas=0.7286 go_c=0.4247 go_gv=0.3379 transl_vel=0.4614 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0015 tv_zero_frac(pred)=0.00 static_conf_mean=0.76 static_conf_hi_frac=0.54
+[02/23 16:46:25][INFO] [LossBreakdown] body_pose=1.0166 betas=0.6744 go_c=1.0562 go_gv=1.1169 transl_vel=0.6442 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 16:46:26][INFO] [LossBreakdown] body_pose=0.8701 betas=0.7228 go_c=0.8129 go_gv=0.9255 transl_vel=0.5427 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 16:46:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9972 pred=+0.9743 delta(pred-gt)=-0.0229
+[02/23 16:46:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03038667 3.0147462 -0.04490549] global_orient0_aa(pred)=[ 0.01432409 2.2424948 -0.01199703]
+[02/23 16:46:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+172.78,+1.77,+1.04) pred=(+128.49,+0.78,+0.35) pred_vs_gt=(-44.26,+0.89,+0.81)
+[02/23 16:46:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+45.53
+[02/23 16:48:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 16:48:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 16:48:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 16:48:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 16:48:03][INFO] ✅[FIT][Epoch 0] finished! 01:49→1:29:25 | loss_epoch=1.65
+[02/23 16:48:03][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 16:48:03][INFO] [LossBreakdown] body_pose=0.4384 betas=0.4989 go_c=0.6980 go_gv=0.4411 transl_vel=0.6734 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.48 static_conf_hi_frac=0.28
+[02/23 16:48:03][INFO] [LossBreakdown] body_pose=1.0455 betas=0.3832 go_c=0.7566 go_gv=0.7875 transl_vel=1.4174 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.00 static_conf_mean=0.55 static_conf_hi_frac=0.33
+[02/23 16:48:04][INFO] [LossBreakdown] body_pose=0.5220 betas=0.5085 go_c=0.3845 go_gv=0.4008 transl_vel=0.6110 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.33 static_conf_hi_frac=0.02
+[02/23 16:48:04][INFO] [LossBreakdown] body_pose=0.5147 betas=0.3794 go_c=1.5473 go_gv=1.3410 transl_vel=17.7164 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0013 tv_abs_mean(tgt)=0.0030 tv_zero_frac(pred)=0.00 static_conf_mean=0.31 static_conf_hi_frac=0.01
+[02/23 16:48:18][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9957 pred=+0.9597 delta(pred-gt)=-0.0360
+[02/23 16:48:18][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[-0.06166399 -3.0616815 0.07028337] global_orient0_aa(pred)=[-0.03816987 -2.0780106 0.01635037]
+[02/23 16:48:18][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-175.45,+2.53,+2.41) pred=(-119.08,-0.25,+1.96) pred_vs_gt=(+56.38,+2.81,+0.23)
+[02/23 16:48:18][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-53.75
+[02/23 16:49:18][INFO] ✅[FIT][Epoch 1] finished! 03:04→1:13:56 | loss_epoch=1.52
+[02/23 16:49:18][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/23 16:49:18][INFO] [LossBreakdown] body_pose=1.2933 betas=0.3397 go_c=0.3365 go_gv=0.3346 transl_vel=0.4953 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0015 tv_zero_frac(pred)=0.00 static_conf_mean=0.81 static_conf_hi_frac=0.67
+[02/23 16:49:19][INFO] [LossBreakdown] body_pose=0.4511 betas=0.3314 go_c=0.8895 go_gv=0.5686 transl_vel=18.0899 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0031 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.58
+[02/23 16:49:19][INFO] [LossBreakdown] body_pose=0.4561 betas=0.3125 go_c=0.8659 go_gv=0.9075 transl_vel=0.6702 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.62 static_conf_hi_frac=0.42
+[02/23 16:49:19][INFO] [LossBreakdown] body_pose=0.3526 betas=0.3095 go_c=0.6875 go_gv=0.1582 transl_vel=0.3309 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.65 static_conf_hi_frac=0.50
+[02/23 16:49:46][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 root_y0: gt=+0.9947 pred=+0.9625 delta(pred-gt)=-0.0321
+[02/23 16:49:46][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(gt)=[-0.05458534 -2.9650407 0.14522971] global_orient0_aa(pred)=[-0.03907072 -2.0148664 0.01323624]
+[02/23 16:49:46][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_yxz_deg gt=(-169.98,+5.38,+2.58) pred=(-115.46,-0.46,+1.93) pred_vs_gt=(+54.53,+5.87,-0.37)
+[02/23 16:49:46][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 yaw0_deg(pred_vs_gt)=-54.03
+[02/23 16:50:41][INFO] ✅[FIT][Epoch 2] finished! 04:27→1:09:46 | loss_epoch=1.44
+[02/23 16:50:41][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[02/23 16:50:41][INFO] [LossBreakdown] body_pose=0.3436 betas=0.1886 go_c=0.2754 go_gv=0.2694 transl_vel=0.4069 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.48 static_conf_hi_frac=0.16
+[02/23 16:50:41][INFO] [LossBreakdown] body_pose=0.8783 betas=0.2864 go_c=0.6919 go_gv=0.6089 transl_vel=1.1866 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0014 tv_zero_frac(pred)=0.00 static_conf_mean=0.77 static_conf_hi_frac=0.57
+[02/23 16:50:41][INFO] [LossBreakdown] body_pose=0.3560 betas=0.2357 go_c=0.8153 go_gv=0.8277 transl_vel=0.5527 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.59 static_conf_hi_frac=0.32
+[02/23 16:50:42][INFO] [LossBreakdown] body_pose=0.3434 betas=0.3001 go_c=0.6758 go_gv=0.3248 transl_vel=0.4634 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.58 static_conf_hi_frac=0.41
+[02/23 16:50:51][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+0.9993 pred=+0.9615 delta(pred-gt)=-0.0378
+[02/23 16:50:51][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03401112 3.0013862 -0.00415359] global_orient0_aa(pred)=[ 0.01800341 2.2240171 -0.00902304]
+[02/23 16:50:51][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+171.98,+0.25,+1.28) pred=(+127.43,+0.74,+0.56) pred_vs_gt=(-44.54,-0.59,+0.64)
+[02/23 16:50:51][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+44.05
+[02/23 16:52:03][INFO] ✅[FIT][Epoch 3] finished! 05:49→1:07:01 | loss_epoch=1.01
+[02/23 16:52:03][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[02/23 16:52:03][INFO] [LossBreakdown] body_pose=0.3518 betas=0.2053 go_c=1.1858 go_gv=0.6589 transl_vel=17.8894 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0031 tv_zero_frac(pred)=0.00 static_conf_mean=0.52 static_conf_hi_frac=0.15
+[02/23 16:52:04][INFO] [LossBreakdown] body_pose=1.1139 betas=0.2517 go_c=0.6324 go_gv=0.5986 transl_vel=0.3239 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0013 tv_zero_frac(pred)=0.00 static_conf_mean=0.74 static_conf_hi_frac=0.53
+[02/23 16:52:04][INFO] [LossBreakdown] body_pose=0.2582 betas=0.1865 go_c=0.8526 go_gv=1.0254 transl_vel=0.6697 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.61 static_conf_hi_frac=0.36
+[02/23 16:52:04][INFO] [LossBreakdown] body_pose=0.2735 betas=0.1762 go_c=0.5979 go_gv=0.1297 transl_vel=0.3209 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.00 static_conf_mean=0.57 static_conf_hi_frac=0.16
+[02/23 16:52:24][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 root_y0: gt=+0.9958 pred=+0.9468 delta(pred-gt)=-0.0490
+[02/23 16:52:24][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 global_orient0_aa(gt)=[0.14268374 0.2737292 0.0679609 ] global_orient0_aa(pred)=[ 0.0158926 2.2407336 -0.01069876]
+[02/23 16:52:24][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 global_orient0_yxz_deg gt=(+16.04,+7.53,+4.99) pred=(+128.39,+0.76,+0.44) pred_vs_gt=(+112.85,-7.75,-2.53)
+[02/23 16:52:24][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 yaw0_deg(pred_vs_gt)=-113.45
+[02/23 16:53:25][INFO] ✅[FIT][Epoch 4] finished! 07:11→1:04:47 | loss_epoch=1.1
+[02/23 16:59:13][INFO] [Exp Name]: finetune_
+[02/23 16:59:13][INFO] [GPU x Batch] = 1 x 2
+[02/23 16:59:13][INFO] [UnityDataset] Found 9 sequences.
+[02/23 16:59:13][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 16:59:13][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 16:59:13][INFO]
+[02/23 16:59:13][INFO] [UnityDataset] Found 9 sequences.
+[02/23 16:59:13][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 16:59:13][INFO]
+[02/23 16:59:20][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 16:59:44][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[02/23 16:59:55][INFO] Start Fitting...
+[02/23 16:59:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 16:59:56][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 16:59:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 16:59:59][INFO] [LossBreakdown] body_pose=0.9139 betas=0.6626 go_c=0.5414 go_gv=0.0717 transl_vel=0.2179 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 17:00:00][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 17:00:00][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7402 go_c=0.1314 go_gv=0.0420 transl_vel=0.1586 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0015 tv_zero_frac(pred)=0.01 static_conf_mean=0.76 static_conf_hi_frac=0.54
+[02/23 17:00:00][INFO] [LossBreakdown] body_pose=1.1267 betas=0.7342 go_c=0.2199 go_gv=0.0269 transl_vel=0.1672 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 17:00:00][INFO] [LossBreakdown] body_pose=0.9276 betas=0.6987 go_c=0.0835 go_gv=0.0195 transl_vel=0.1791 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 17:00:13][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9972 pred=+0.9860 delta(pred-gt)=-0.0111
+[02/23 17:00:13][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03038667 3.0147462 -0.04490549] global_orient0_aa(pred)=[0.02428909 1.9573448 0.07525613]
+[02/23 17:00:13][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+172.78,+1.77,+1.04) pred=(+112.13,-2.37,+3.02) pred_vs_gt=(-60.69,+4.36,-1.44)
+[02/23 17:00:13][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+59.48
+[02/23 17:01:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 17:01:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 17:01:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 17:01:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 17:01:20][INFO] ✅[FIT][Epoch 0] finished! 01:24→1:08:58 | loss_epoch=1.8
+[02/23 17:01:20][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 17:01:20][INFO] [LossBreakdown] body_pose=0.4430 betas=0.3901 go_c=0.1551 go_gv=0.0201 transl_vel=0.1891 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.46 static_conf_hi_frac=0.25
+[02/23 17:01:20][INFO] [LossBreakdown] body_pose=1.1920 betas=0.4918 go_c=0.0980 go_gv=0.0226 transl_vel=0.1663 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.00 static_conf_mean=0.51 static_conf_hi_frac=0.27
+[02/23 17:01:21][INFO] [LossBreakdown] body_pose=0.3939 betas=0.5380 go_c=0.0696 go_gv=0.0355 transl_vel=0.2019 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0013 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.32 static_conf_hi_frac=0.01
+[02/23 17:01:21][INFO] [LossBreakdown] body_pose=0.3755 betas=0.4500 go_c=0.3561 go_gv=0.0351 transl_vel=7.7847 gogv_mask=1.00 tv_mask=1.00 tv_abs_mean(pred)=0.0025 tv_abs_mean(tgt)=0.0030 tv_zero_frac(pred)=0.00 static_conf_mean=0.35 static_conf_hi_frac=0.02
+[02/23 17:05:03][INFO] [Exp Name]: finetune_
+[02/23 17:05:03][INFO] [GPU x Batch] = 1 x 2
+[02/23 17:05:04][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:05:04][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:05:04][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 17:05:04][INFO]
+[02/23 17:05:04][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:05:04][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:05:04][INFO]
+[02/23 17:05:11][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 17:05:34][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[02/23 17:05:44][INFO] Start Fitting...
+[02/23 17:05:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 17:05:46][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 17:05:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 17:05:48][INFO] [LossBreakdown] body_pose=0.9139 betas=0.6626 go_c=0.5414 go_gv=0.0717 transl_vel=0.2179 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.75/35.47 go_gv_deg(mean/max)=13.83/28.14 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 17:05:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 17:05:49][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7402 go_c=0.1314 go_gv=0.0420 transl_vel=0.1586 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.52/27.78 go_gv_deg(mean/max)=10.63/27.29 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0015 tv_zero_frac(pred)=0.01 static_conf_mean=0.76 static_conf_hi_frac=0.54
+[02/23 17:05:51][INFO] [LossBreakdown] body_pose=1.1267 betas=0.7342 go_c=0.2199 go_gv=0.0269 transl_vel=0.1672 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.77/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 17:05:51][INFO] [LossBreakdown] body_pose=0.9276 betas=0.6987 go_c=0.0835 go_gv=0.0195 transl_vel=0.1791 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.13/9.82 go_gv_deg(mean/max)=8.96/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 17:06:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9972 pred=+0.9860 delta(pred-gt)=-0.0111
+[02/23 17:06:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03038667 3.0147462 -0.04490549] global_orient0_aa(pred)=[0.02428909 1.9573448 0.07525613]
+[02/23 17:06:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+172.78,+1.77,+1.04) pred=(+112.13,-2.37,+3.02) pred_vs_gt=(-60.69,+4.36,-1.44)
+[02/23 17:06:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+59.48
+[02/23 17:21:40][INFO] [Exp Name]: finetune_
+[02/23 17:21:40][INFO] [GPU x Batch] = 1 x 2
+[02/23 17:21:40][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:21:40][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:21:40][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 17:21:40][INFO]
+[02/23 17:21:40][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:21:40][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:21:40][INFO]
+[02/23 17:21:48][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 17:22:17][INFO] [Exp Name]: finetune_
+[02/23 17:22:17][INFO] [GPU x Batch] = 1 x 2
+[02/23 17:22:17][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:22:17][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:22:17][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 17:22:17][INFO]
+[02/23 17:22:17][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:22:17][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:22:17][INFO]
+[02/23 17:22:23][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 17:22:31][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_3/checkpoints'
+[02/23 17:22:42][INFO] Start Fitting...
+[02/23 17:22:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 17:22:43][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 17:22:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 17:22:46][INFO] [LossBreakdown] body_pose=0.9203 betas=0.6590 go_c=0.5448 go_gv=0.0725 transl_vel=0.2262 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.81/35.75 go_gv_deg(mean/max)=13.86/28.49 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.41
+[02/23 17:22:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 17:22:47][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7464 go_c=0.1315 go_gv=0.0418 transl_vel=0.1971 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.42/28.02 go_gv_deg(mean/max)=10.52/27.78 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 17:22:47][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 17:22:48][INFO] [LossBreakdown] body_pose=0.9267 betas=0.6977 go_c=0.0836 go_gv=0.0196 transl_vel=0.1916 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/9.82 go_gv_deg(mean/max)=8.98/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 17:23:02][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9975 pred=+0.9862 delta(pred-gt)=-0.0113
+[02/23 17:23:02][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.0300002 3.0145898 -0.04516966] global_orient0_aa(pred)=[0.02919572 1.9638917 0.06257642]
+[02/23 17:23:02][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+172.77,+1.78,+1.03) pred=(+112.52,-1.74,+2.86) pred_vs_gt=(-60.30,+3.72,-1.38)
+[02/23 17:23:02][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+58.89
+[02/23 17:29:05][INFO] [Exp Name]: finetune_
+[02/23 17:29:05][INFO] [GPU x Batch] = 1 x 2
+[02/23 17:29:05][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:29:05][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:29:05][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 17:29:05][INFO]
+[02/23 17:29:05][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:29:05][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:29:05][INFO]
+[02/23 17:29:11][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 17:29:27][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_4/checkpoints'
+[02/23 17:29:38][INFO] Start Fitting...
+[02/23 17:29:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 17:29:40][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 17:29:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 17:29:42][INFO] [LossBreakdown] body_pose=0.9203 betas=0.6590 go_c=0.5448 go_gv=0.0725 transl_vel=0.2262 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.81/35.75 go_gv_deg(mean/max)=13.86/28.49 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.41
+[02/23 17:29:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 17:29:43][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7464 go_c=0.1315 go_gv=0.0418 transl_vel=0.1971 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.42/28.02 go_gv_deg(mean/max)=10.52/27.78 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 17:29:44][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 17:29:44][INFO] [LossBreakdown] body_pose=0.9267 betas=0.6977 go_c=0.0836 go_gv=0.0196 transl_vel=0.1916 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/9.82 go_gv_deg(mean/max)=8.98/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 17:29:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[02/23 17:29:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9975 pred=+0.9834 delta(pred-gt)=-0.0141
+[02/23 17:29:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.0300002 3.0145898 -0.04516966] global_orient0_aa(pred)=[-0.01938733 2.2798643 -0.05634961]
+[02/23 17:29:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+172.77,+1.78,+1.03) pred=(+130.62,+1.97,-1.88) pred_vs_gt=(-42.06,-0.55,+2.86)
+[02/23 17:29:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+48.84
+[02/23 17:31:45][INFO] [Exp Name]: finetune_
+[02/23 17:31:45][INFO] [GPU x Batch] = 1 x 2
+[02/23 17:31:45][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:31:45][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:31:45][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 17:31:45][INFO]
+[02/23 17:31:45][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:31:45][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:31:45][INFO]
+[02/23 17:31:52][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 17:32:10][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_5/checkpoints'
+[02/23 17:32:23][INFO] Start Fitting...
+[02/23 17:32:25][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 17:32:25][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 17:32:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 17:32:30][INFO] [LossBreakdown] body_pose=0.9203 betas=0.6590 go_c=0.5448 go_gv=0.0725 transl_vel=0.2262 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.81/35.75 go_gv_deg(mean/max)=13.86/28.49 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.41
+[02/23 17:32:30][INFO] [GlobalDebug] transl_w_use_pred_go=True go_w_deg(mean/max)=143.89/168.54 transl_w_loss=0.0451
+[02/23 17:32:31][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 17:32:31][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7464 go_c=0.1315 go_gv=0.0418 transl_vel=0.1971 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.42/28.02 go_gv_deg(mean/max)=10.52/27.78 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 17:32:31][INFO] [GlobalDebug] transl_w_use_pred_go=True go_w_deg(mean/max)=110.07/180.06 transl_w_loss=0.0424
+[02/23 17:32:32][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 17:32:32][INFO] [GlobalDebug] transl_w_use_pred_go=True go_w_deg(mean/max)=91.18/106.54 transl_w_loss=0.0375
+[02/23 17:32:32][INFO] [LossBreakdown] body_pose=0.9267 betas=0.6977 go_c=0.0836 go_gv=0.0196 transl_vel=0.1916 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/9.82 go_gv_deg(mean/max)=8.98/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 17:32:32][INFO] [GlobalDebug] transl_w_use_pred_go=True go_w_deg(mean/max)=75.99/91.13 transl_w_loss=0.0289
+[02/23 17:32:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/optim/lr_scheduler.py:143: UserWarning: Detected call of `lr_scheduler.step()` before `optimizer.step()`. In PyTorch 1.1.0 and later, you should call them in the opposite order: `optimizer.step()` before `lr_scheduler.step()`. Failure to do this will result in PyTorch skipping the first value of the learning rate schedule. See more details at https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate
+ warnings.warn("Detected call of `lr_scheduler.step()` before `optimizer.step()`. "
+
+[02/23 17:38:38][INFO] [Exp Name]: finetune_
+[02/23 17:38:38][INFO] [GPU x Batch] = 1 x 2
+[02/23 17:38:38][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:38:38][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:38:38][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 17:38:38][INFO]
+[02/23 17:38:38][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:38:38][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:38:38][INFO]
+[02/23 17:38:45][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 17:39:07][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_6/checkpoints'
+[02/23 17:39:18][INFO] Start Fitting...
+[02/23 17:39:21][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 17:39:21][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 17:39:23][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 17:39:23][INFO] [LossBreakdown] body_pose=0.9203 betas=0.6590 go_c=0.5448 go_gv=0.0725 transl_vel=0.2262 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.81/35.75 go_gv_deg(mean/max)=13.86/28.49 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.41
+[02/23 17:39:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 17:39:25][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7464 go_c=0.1315 go_gv=0.0418 transl_vel=0.1971 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.42/28.02 go_gv_deg(mean/max)=10.52/27.78 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 17:39:25][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 17:39:25][INFO] [LossBreakdown] body_pose=0.9267 betas=0.6977 go_c=0.0836 go_gv=0.0196 transl_vel=0.1916 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/9.82 go_gv_deg(mean/max)=8.98/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 17:45:29][INFO] [Exp Name]: finetune_
+[02/23 17:45:29][INFO] [GPU x Batch] = 1 x 2
+[02/23 17:45:29][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:45:29][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:45:29][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 17:45:29][INFO]
+[02/23 17:45:29][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:45:29][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:45:29][INFO]
+[02/23 17:45:36][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 17:45:55][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_7/checkpoints'
+[02/23 17:46:06][INFO] Start Fitting...
+[02/23 17:46:07][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 17:46:08][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 17:46:09][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 17:46:10][INFO] [LossBreakdown] body_pose=0.9151 betas=0.6620 go_c=0.5413 go_gv=0.0710 transl_vel=0.1780 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.73/35.37 go_gv_deg(mean/max)=13.78/28.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 17:46:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 17:46:11][INFO] [LossBreakdown] body_pose=2.1546 betas=0.7537 go_c=0.1314 go_gv=0.0420 transl_vel=0.2005 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.54/27.54 go_gv_deg(mean/max)=10.66/26.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 17:46:11][INFO] [LossBreakdown] body_pose=1.1252 betas=0.7384 go_c=0.2187 go_gv=0.0270 transl_vel=0.1787 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.71/20.36 go_gv_deg(mean/max)=7.89/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 17:46:12][INFO] [LossBreakdown] body_pose=0.9275 betas=0.6969 go_c=0.0838 go_gv=0.0195 transl_vel=0.1868 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/10.14 go_gv_deg(mean/max)=8.97/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 17:46:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9986 pred=+2.4090 delta(pred-gt)=+1.4104
+[02/23 17:46:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[-0.11266945 2.1451623 0.00838481] global_orient0_aa(pred)=[-0.00639139 1.9595624 0.07586171]
+[02/23 17:46:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+123.12,-2.86,-4.46) pred=(+112.27,-3.23,+1.80) pred_vs_gt=(-10.67,+5.44,-3.10)
+[02/23 17:46:26][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+11.27
+[02/23 17:52:06][INFO] [Exp Name]: finetune_
+[02/23 17:52:06][INFO] [GPU x Batch] = 1 x 2
+[02/23 17:52:06][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:52:06][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:52:06][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 17:52:06][INFO]
+[02/23 17:52:06][INFO] [UnityDataset] Found 9 sequences.
+[02/23 17:52:06][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 17:52:06][INFO]
+[02/23 17:52:13][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 17:52:26][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_8/checkpoints'
+[02/23 17:52:34][INFO] Start Fitting...
+[02/23 17:52:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 17:52:35][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 17:52:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 17:52:36][INFO] [LossBreakdown] body_pose=0.9151 betas=0.6620 go_c=0.5413 go_gv=0.0710 transl_vel=0.1780 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.73/35.37 go_gv_deg(mean/max)=13.78/28.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 17:52:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 17:52:37][INFO] [LossBreakdown] body_pose=2.1546 betas=0.7537 go_c=0.1314 go_gv=0.0420 transl_vel=0.2005 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.54/27.54 go_gv_deg(mean/max)=10.66/26.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 17:52:37][INFO] [LossBreakdown] body_pose=1.1252 betas=0.7384 go_c=0.2187 go_gv=0.0270 transl_vel=0.1787 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.71/20.36 go_gv_deg(mean/max)=7.89/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 17:52:37][INFO] [LossBreakdown] body_pose=0.9275 betas=0.6969 go_c=0.0838 go_gv=0.0195 transl_vel=0.1868 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/10.14 go_gv_deg(mean/max)=8.97/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 18:16:16][INFO] [Exp Name]: finetune_
+[02/23 18:16:16][INFO] [GPU x Batch] = 1 x 2
+[02/23 18:16:16][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:16:16][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:16:16][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 18:16:16][INFO]
+[02/23 18:16:16][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:16:16][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:16:16][INFO]
+[02/23 18:16:24][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 18:16:37][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_9/checkpoints'
+[02/23 18:16:47][INFO] Start Fitting...
+[02/23 18:16:49][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 18:16:49][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 18:16:51][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 18:16:52][INFO] [LossBreakdown] body_pose=0.9151 betas=0.6620 go_c=0.5413 go_gv=0.0710 transl_vel=0.1780 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.73/35.37 go_gv_deg(mean/max)=13.78/28.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 18:16:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 18:16:54][INFO] [LossBreakdown] body_pose=2.1546 betas=0.7537 go_c=0.1314 go_gv=0.0420 transl_vel=0.2005 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.54/27.54 go_gv_deg(mean/max)=10.66/26.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 18:16:54][INFO] [LossBreakdown] body_pose=1.1252 betas=0.7384 go_c=0.2187 go_gv=0.0270 transl_vel=0.1787 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.71/20.36 go_gv_deg(mean/max)=7.89/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 18:16:55][INFO] [LossBreakdown] body_pose=0.9275 betas=0.6969 go_c=0.0838 go_gv=0.0195 transl_vel=0.1868 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/10.14 go_gv_deg(mean/max)=8.97/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 18:17:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9959 pred=+2.3761 delta(pred-gt)=+1.3803
+[02/23 18:17:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[-0.00639139 1.9595624 0.07586171]
+[02/23 18:17:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+112.27,-3.23,+1.80) pred_vs_gt=(-8.64,+3.24,+3.90)
+[02/23 18:17:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+9.23
+[02/23 18:18:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:18:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:18:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:18:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:18:11][INFO] ✅[FIT][Epoch 0] finished! 01:23→1:08:07 | loss_epoch=2.74
+[02/23 18:18:11][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 18:18:12][INFO] [LossBreakdown] body_pose=0.5721 betas=0.4798 go_c=0.1758 go_gv=0.0260 transl_vel=0.3435 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.46/25.53 go_gv_deg(mean/max)=12.54/23.65 tv_abs_mean(pred)=0.0019 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.20 static_conf_hi_frac=0.00
+[02/23 18:18:12][INFO] [LossBreakdown] body_pose=1.3473 betas=0.5137 go_c=0.1137 go_gv=0.0295 transl_vel=0.5054 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.31/31.08 go_gv_deg(mean/max)=9.90/32.36 tv_abs_mean(pred)=0.0036 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.21 static_conf_hi_frac=0.01
+[02/23 18:18:12][INFO] [LossBreakdown] body_pose=0.6728 betas=0.7593 go_c=0.0739 go_gv=0.0450 transl_vel=0.3804 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.55/20.04 go_gv_deg(mean/max)=9.96/22.51 tv_abs_mean(pred)=0.0032 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.16 static_conf_hi_frac=0.00
+[02/23 18:18:12][INFO] [LossBreakdown] body_pose=0.7139 betas=0.6396 go_c=0.5781 go_gv=0.0300 transl_vel=0.8181 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=18.07/122.55 go_gv_deg(mean/max)=8.86/128.22 tv_abs_mean(pred)=0.0047 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.19 static_conf_hi_frac=0.00
+[02/23 18:18:23][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+1.0413 pred=+2.3780 delta(pred-gt)=+1.3367
+[02/23 18:18:23][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.00300199 -1.5857557 0.06613082] global_orient0_aa(pred)=[-0.24409217 -1.5249343 -0.01273079]
+[02/23 18:18:23][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-90.84,+2.53,+2.28) pred=(-88.46,-9.55,+8.38) pred_vs_gt=(+3.37,-5.84,-12.22)
+[02/23 18:18:23][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-1.51
+[02/23 18:22:10][INFO] [Exp Name]: finetune_
+[02/23 18:22:10][INFO] [GPU x Batch] = 1 x 2
+[02/23 18:22:10][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:22:10][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:22:10][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 18:22:10][INFO]
+[02/23 18:22:10][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:22:10][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:22:10][INFO]
+[02/23 18:22:17][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 18:22:42][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_10/checkpoints'
+[02/23 18:22:52][INFO] Start Fitting...
+[02/23 18:22:53][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 18:22:53][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 18:22:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 18:22:54][INFO] [LossBreakdown] body_pose=0.9151 betas=0.6620 go_c=0.5413 go_gv=0.0710 transl_vel=0.1780 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.73/35.37 go_gv_deg(mean/max)=13.78/28.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 18:22:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 18:22:55][INFO] [LossBreakdown] body_pose=2.1546 betas=0.7537 go_c=0.1314 go_gv=0.0420 transl_vel=0.2005 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.54/27.54 go_gv_deg(mean/max)=10.66/26.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 18:22:55][INFO] [LossBreakdown] body_pose=1.1252 betas=0.7384 go_c=0.2187 go_gv=0.0270 transl_vel=0.1787 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.71/20.36 go_gv_deg(mean/max)=7.89/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 18:22:55][INFO] [LossBreakdown] body_pose=0.9275 betas=0.6969 go_c=0.0838 go_gv=0.0195 transl_vel=0.1868 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/10.14 go_gv_deg(mean/max)=8.97/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 18:23:07][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9959 pred=+2.3761 delta(pred-gt)=+1.3803
+[02/23 18:23:07][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[-0.00639139 1.9595624 0.07586171]
+[02/23 18:23:07][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+112.27,-3.23,+1.80) pred_vs_gt=(-8.64,+3.24,+3.90)
+[02/23 18:23:07][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+9.23
+[02/23 18:24:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:24:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:24:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:24:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:24:16][INFO] ✅[FIT][Epoch 0] finished! 01:23→1:08:15 | loss_epoch=2.74
+[02/23 18:24:16][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 18:24:16][INFO] [LossBreakdown] body_pose=0.5721 betas=0.4798 go_c=0.1758 go_gv=0.0260 transl_vel=0.3435 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.46/25.53 go_gv_deg(mean/max)=12.54/23.65 tv_abs_mean(pred)=0.0019 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.20 static_conf_hi_frac=0.00
+[02/23 18:24:16][INFO] [LossBreakdown] body_pose=1.3473 betas=0.5137 go_c=0.1137 go_gv=0.0295 transl_vel=0.5054 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.31/31.08 go_gv_deg(mean/max)=9.90/32.36 tv_abs_mean(pred)=0.0036 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.21 static_conf_hi_frac=0.01
+[02/23 18:24:17][INFO] [LossBreakdown] body_pose=0.6728 betas=0.7593 go_c=0.0739 go_gv=0.0450 transl_vel=0.3804 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.55/20.04 go_gv_deg(mean/max)=9.96/22.51 tv_abs_mean(pred)=0.0032 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.16 static_conf_hi_frac=0.00
+[02/23 18:24:17][INFO] [LossBreakdown] body_pose=0.7139 betas=0.6396 go_c=0.5781 go_gv=0.0300 transl_vel=0.8181 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=18.07/122.55 go_gv_deg(mean/max)=8.86/128.22 tv_abs_mean(pred)=0.0047 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.19 static_conf_hi_frac=0.00
+[02/23 18:24:32][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+1.0413 pred=+2.3780 delta(pred-gt)=+1.3367
+[02/23 18:24:32][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.00300199 -1.5857557 0.06613082] global_orient0_aa(pred)=[-0.24409217 -1.5249343 -0.01273079]
+[02/23 18:24:32][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-90.84,+2.53,+2.28) pred=(-88.46,-9.55,+8.38) pred_vs_gt=(+3.37,-5.84,-12.22)
+[02/23 18:24:32][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-1.51
+[02/23 18:29:54][INFO] [Exp Name]: finetune_
+[02/23 18:29:54][INFO] [GPU x Batch] = 1 x 2
+[02/23 18:29:55][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:29:55][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:29:55][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 18:29:55][INFO]
+[02/23 18:29:55][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:29:55][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:29:55][INFO]
+[02/23 18:30:01][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 18:30:27][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_11/checkpoints'
+[02/23 18:30:39][INFO] Start Fitting...
+[02/23 18:30:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 18:30:41][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 18:30:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 18:30:46][INFO] [LossBreakdown] body_pose=0.9151 betas=0.6620 go_c=0.5413 go_gv=0.0710 transl_vel=0.1780 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.73/35.37 go_gv_deg(mean/max)=13.78/28.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 18:30:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 18:30:47][INFO] [LossBreakdown] body_pose=2.1546 betas=0.7537 go_c=0.1314 go_gv=0.0420 transl_vel=0.2005 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.54/27.54 go_gv_deg(mean/max)=10.66/26.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 18:30:48][INFO] [LossBreakdown] body_pose=1.1252 betas=0.7384 go_c=0.2187 go_gv=0.0270 transl_vel=0.1787 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.71/20.36 go_gv_deg(mean/max)=7.89/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 18:30:48][INFO] [LossBreakdown] body_pose=0.9275 betas=0.6969 go_c=0.0838 go_gv=0.0195 transl_vel=0.1868 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/10.14 go_gv_deg(mean/max)=8.97/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 18:31:02][INFO] [VisUnityVal] Global postprocess: minY_all -1.331->-0.000 minY_foot -1.331->-0.000
+[02/23 18:31:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9959 pred=+2.3761 delta(pred-gt)=+1.3803
+[02/23 18:31:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[-0.00639139 1.9595624 0.07586171]
+[02/23 18:31:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+112.27,-3.23,+1.80) pred_vs_gt=(-8.64,+3.24,+3.90)
+[02/23 18:31:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+9.23
+[02/23 18:37:56][INFO] [Exp Name]: finetune_
+[02/23 18:37:56][INFO] [GPU x Batch] = 1 x 2
+[02/23 18:37:56][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:37:56][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:37:56][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 18:37:56][INFO]
+[02/23 18:37:56][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:37:56][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:37:56][INFO]
+[02/23 18:38:03][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 18:38:28][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_12/checkpoints'
+[02/23 18:38:41][INFO] Start Fitting...
+[02/23 18:38:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 18:38:42][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 18:38:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 18:38:43][INFO] [LossBreakdown] body_pose=0.9151 betas=0.6620 go_c=0.5413 go_gv=0.0710 transl_vel=0.1780 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.73/35.37 go_gv_deg(mean/max)=13.78/28.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 18:38:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 18:38:44][INFO] [LossBreakdown] body_pose=2.1546 betas=0.7537 go_c=0.1314 go_gv=0.0420 transl_vel=0.2005 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.54/27.54 go_gv_deg(mean/max)=10.66/26.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 18:38:44][INFO] [LossBreakdown] body_pose=1.1252 betas=0.7384 go_c=0.2187 go_gv=0.0270 transl_vel=0.1787 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.71/20.36 go_gv_deg(mean/max)=7.89/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 18:38:44][INFO] [LossBreakdown] body_pose=0.9275 betas=0.6969 go_c=0.0838 go_gv=0.0195 transl_vel=0.1868 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/10.14 go_gv_deg(mean/max)=8.97/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 18:38:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9959 pred=+1.0516 delta(pred-gt)=+0.0557
+[02/23 18:38:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[-0.00639139 1.9595624 0.07586171]
+[02/23 18:38:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+112.27,-3.23,+1.80) pred_vs_gt=(-8.64,+3.24,+3.90)
+[02/23 18:38:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+9.23
+[02/23 18:40:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:40:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:40:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:40:02][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:40:02][INFO] ✅[FIT][Epoch 0] finished! 01:20→1:06:05 | loss_epoch=2.74
+[02/23 18:40:02][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 18:40:02][INFO] [LossBreakdown] body_pose=0.5721 betas=0.4798 go_c=0.1758 go_gv=0.0260 transl_vel=0.3435 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.46/25.53 go_gv_deg(mean/max)=12.54/23.65 tv_abs_mean(pred)=0.0019 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.20 static_conf_hi_frac=0.00
+[02/23 18:40:02][INFO] [LossBreakdown] body_pose=1.3473 betas=0.5137 go_c=0.1137 go_gv=0.0295 transl_vel=0.5054 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.31/31.08 go_gv_deg(mean/max)=9.90/32.36 tv_abs_mean(pred)=0.0036 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.21 static_conf_hi_frac=0.01
+[02/23 18:40:03][INFO] [LossBreakdown] body_pose=0.6728 betas=0.7593 go_c=0.0739 go_gv=0.0450 transl_vel=0.3804 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.55/20.04 go_gv_deg(mean/max)=9.96/22.51 tv_abs_mean(pred)=0.0032 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.16 static_conf_hi_frac=0.00
+[02/23 18:40:03][INFO] [LossBreakdown] body_pose=0.7139 betas=0.6396 go_c=0.5781 go_gv=0.0300 transl_vel=0.8181 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=18.07/122.55 go_gv_deg(mean/max)=8.86/128.22 tv_abs_mean(pred)=0.0047 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.19 static_conf_hi_frac=0.00
+[02/23 18:40:15][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+1.0413 pred=+0.9957 delta(pred-gt)=-0.0456
+[02/23 18:40:15][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.00300199 -1.5857557 0.06613082] global_orient0_aa(pred)=[-0.24409217 -1.5249343 -0.01273079]
+[02/23 18:40:15][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-90.84,+2.53,+2.28) pred=(-88.46,-9.55,+8.38) pred_vs_gt=(+3.37,-5.84,-12.22)
+[02/23 18:40:15][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-1.51
+[02/23 18:41:16][INFO] ✅[FIT][Epoch 1] finished! 02:35→1:02:08 | loss_epoch=1.62
+[02/23 18:41:16][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/23 18:41:16][INFO] [LossBreakdown] body_pose=1.9732 betas=0.3161 go_c=0.0532 go_gv=0.0247 transl_vel=0.2173 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=5.97/10.14 go_gv_deg(mean/max)=9.19/13.19 tv_abs_mean(pred)=0.0014 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.00 static_conf_mean=0.28 static_conf_hi_frac=0.01
+[02/23 18:41:17][INFO] [LossBreakdown] body_pose=0.3161 betas=0.5132 go_c=0.2202 go_gv=0.0304 transl_vel=0.5039 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.01/31.30 go_gv_deg(mean/max)=8.73/30.53 tv_abs_mean(pred)=0.0016 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.23 static_conf_hi_frac=0.00
+[02/23 18:41:17][INFO] [LossBreakdown] body_pose=0.3845 betas=0.4911 go_c=0.2111 go_gv=0.0459 transl_vel=0.4909 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.36/24.61 go_gv_deg(mean/max)=11.99/26.80 tv_abs_mean(pred)=0.0029 tv_abs_mean(tgt)=0.0029 tv_zero_frac(pred)=0.00 static_conf_mean=0.23 static_conf_hi_frac=0.00
+[02/23 18:41:17][INFO] [LossBreakdown] body_pose=0.3628 betas=0.5799 go_c=0.1924 go_gv=0.0384 transl_vel=0.2646 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.65/23.09 go_gv_deg(mean/max)=11.49/21.46 tv_abs_mean(pred)=0.0022 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.15 static_conf_hi_frac=0.00
+[02/23 18:41:46][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 root_y0: gt=+1.0023 pred=+1.0245 delta(pred-gt)=+0.0222
+[02/23 18:41:46][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(gt)=[ 0.04780671 -1.4102775 0.10208653] global_orient0_aa(pred)=[-0.04267681 -1.3940506 0.03433248]
+[02/23 18:41:46][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_yxz_deg gt=(-80.76,+5.40,+2.47) pred=(-79.91,-0.56,+2.83) pred_vs_gt=(+0.91,-1.32,-5.83)
+[02/23 18:41:46][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 yaw0_deg(pred_vs_gt)=-7.42
+[02/23 18:42:32][INFO] ✅[FIT][Epoch 2] finished! 03:51→1:00:19 | loss_epoch=1.42
+[02/23 18:42:32][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[02/23 18:42:32][INFO] [LossBreakdown] body_pose=0.2403 betas=0.4097 go_c=0.1121 go_gv=0.0443 transl_vel=0.2047 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.01/26.30 go_gv_deg(mean/max)=13.50/27.78 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.29 static_conf_hi_frac=0.00
+[02/23 18:42:32][INFO] [LossBreakdown] body_pose=0.8840 betas=0.2409 go_c=0.0873 go_gv=0.0365 transl_vel=0.2376 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.28/34.69 go_gv_deg(mean/max)=12.45/34.88 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0015 tv_zero_frac(pred)=0.00 static_conf_mean=0.48 static_conf_hi_frac=0.07
+[02/23 18:42:33][INFO] [LossBreakdown] body_pose=0.2540 betas=0.1578 go_c=0.1446 go_gv=0.0187 transl_vel=0.1970 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.70/17.98 go_gv_deg(mean/max)=5.14/15.45 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.43 static_conf_hi_frac=0.00
+[02/23 18:42:33][INFO] [LossBreakdown] body_pose=0.2605 betas=0.2469 go_c=0.1369 go_gv=0.0179 transl_vel=0.1348 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.60/18.34 go_gv_deg(mean/max)=7.13/17.61 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.35 static_conf_hi_frac=0.00
+[02/23 18:42:43][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+1.0048 pred=+1.0174 delta(pred-gt)=+0.0126
+[02/23 18:42:43][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[0.02579728 2.1115918 0.0080679 ] global_orient0_aa(pred)=[ 0.05715384 2.1228173 -0.00830148]
+[02/23 18:42:43][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.99,+0.27,+1.25) pred=(+121.69,+1.65,+2.16) pred_vs_gt=(+0.68,+0.07,-1.66)
+[02/23 18:42:43][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-4.45
+[02/23 18:50:00][INFO] [Exp Name]: finetune_
+[02/23 18:50:00][INFO] [GPU x Batch] = 1 x 2
+[02/23 18:50:00][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:50:00][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:50:00][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 18:50:00][INFO]
+[02/23 18:50:00][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:50:00][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:50:00][INFO]
+[02/23 18:50:06][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 18:50:33][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_13/checkpoints'
+[02/23 18:50:40][INFO] Start Fitting...
+[02/23 18:50:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 18:50:41][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 18:50:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 18:50:43][INFO] [LossBreakdown] body_pose=0.9151 betas=0.6620 go_c=0.5413 go_gv=0.0710 transl_vel=0.1780 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.73/35.37 go_gv_deg(mean/max)=13.78/28.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 18:50:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 18:50:44][INFO] [LossBreakdown] body_pose=2.1546 betas=0.7537 go_c=0.1314 go_gv=0.0420 transl_vel=0.2005 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.54/27.54 go_gv_deg(mean/max)=10.66/26.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 18:50:44][INFO] [LossBreakdown] body_pose=1.1252 betas=0.7384 go_c=0.2187 go_gv=0.0270 transl_vel=0.1787 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.71/20.36 go_gv_deg(mean/max)=7.89/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 18:50:44][INFO] [LossBreakdown] body_pose=0.9275 betas=0.6969 go_c=0.0838 go_gv=0.0195 transl_vel=0.1868 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/10.14 go_gv_deg(mean/max)=8.97/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 18:51:00][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9959 pred=+1.0529 delta(pred-gt)=+0.0570
+[02/23 18:51:00][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[-0.00785214 1.9580591 0.07441644]
+[02/23 18:51:00][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[-0.11266945 2.1451623 0.00838481]
+[02/23 18:51:00][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+112.18,-3.21,+1.70) pred_vs_gt=(-8.73,+3.15,+3.94)
+[02/23 18:51:00][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+9.41
+[02/23 18:52:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:52:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:52:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:52:08][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 18:52:08][INFO] ✅[FIT][Epoch 0] finished! 01:27→1:11:31 | loss_epoch=2.83
+[02/23 18:52:08][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 18:52:08][INFO] [LossBreakdown] body_pose=0.5714 betas=0.4836 go_c=0.1783 go_gv=0.0260 transl_vel=0.3446 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.52/25.53 go_gv_deg(mean/max)=12.54/23.65 tv_abs_mean(pred)=0.0019 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.20 static_conf_hi_frac=0.00
+[02/23 18:54:51][INFO] [Exp Name]: finetune_
+[02/23 18:54:51][INFO] [GPU x Batch] = 1 x 2
+[02/23 18:54:52][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:54:52][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:54:52][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 18:54:52][INFO]
+[02/23 18:54:52][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:54:52][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:54:52][INFO]
+[02/23 18:55:01][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 18:55:25][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_14/checkpoints'
+[02/23 18:55:37][INFO] Start Fitting...
+[02/23 18:55:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 18:55:39][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 18:55:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 18:55:41][INFO] [LossBreakdown] body_pose=0.9151 betas=0.6620 go_c=0.5413 go_gv=0.0710 transl_vel=0.1780 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.73/35.37 go_gv_deg(mean/max)=13.78/28.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 18:55:41][INFO] [IncamDebug] transl_c_loss=0.0419 pred_cam_abs_mean=0.887 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/23 18:55:42][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 18:55:42][INFO] [LossBreakdown] body_pose=2.1546 betas=0.7537 go_c=0.1314 go_gv=0.0420 transl_vel=0.2005 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.54/27.54 go_gv_deg(mean/max)=10.66/26.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 18:55:42][INFO] [IncamDebug] transl_c_loss=0.1123 pred_cam_abs_mean=0.972 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/23 18:55:42][INFO] [LossBreakdown] body_pose=1.1252 betas=0.7384 go_c=0.2187 go_gv=0.0270 transl_vel=0.1787 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.71/20.36 go_gv_deg(mean/max)=7.89/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 18:55:42][INFO] [IncamDebug] transl_c_loss=0.1976 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/23 18:55:43][INFO] [LossBreakdown] body_pose=0.9275 betas=0.6969 go_c=0.0838 go_gv=0.0195 transl_vel=0.1868 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/10.14 go_gv_deg(mean/max)=8.97/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 18:55:43][INFO] [IncamDebug] transl_c_loss=0.0029 pred_cam_abs_mean=0.903 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/23 18:55:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9959 pred=+1.0529 delta(pred-gt)=+0.0570
+[02/23 18:55:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[-0.00785214 1.9580591 0.07441644]
+[02/23 18:55:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[-0.11266945 2.1451623 0.00838481]
+[02/23 18:55:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+112.18,-3.21,+1.70) pred_vs_gt=(-8.73,+3.15,+3.94)
+[02/23 18:55:55][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+9.41
+[02/23 18:58:28][INFO] [Exp Name]: finetune_
+[02/23 18:58:28][INFO] [GPU x Batch] = 1 x 2
+[02/23 18:58:28][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:58:28][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:58:28][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 18:58:28][INFO]
+[02/23 18:58:28][INFO] [UnityDataset] Found 9 sequences.
+[02/23 18:58:28][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 18:58:28][INFO]
+[02/23 18:58:34][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 18:59:00][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_15/checkpoints'
+[02/23 18:59:11][INFO] Start Fitting...
+[02/23 18:59:12][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 18:59:12][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 18:59:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 18:59:14][INFO] [LossBreakdown] body_pose=0.9151 betas=0.6620 go_c=0.5413 go_gv=0.0710 transl_vel=0.1780 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.73/35.37 go_gv_deg(mean/max)=13.78/28.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 18:59:14][INFO] [IncamDebug] transl_c_loss=0.0419 pred_cam_abs_mean=0.887 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/23 18:59:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 18:59:15][INFO] [LossBreakdown] body_pose=2.1546 betas=0.7537 go_c=0.1314 go_gv=0.0420 transl_vel=0.2005 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.54/27.54 go_gv_deg(mean/max)=10.66/26.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 18:59:15][INFO] [IncamDebug] transl_c_loss=0.1123 pred_cam_abs_mean=0.972 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/23 18:59:15][INFO] [LossBreakdown] body_pose=1.1252 betas=0.7384 go_c=0.2187 go_gv=0.0270 transl_vel=0.1787 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.71/20.36 go_gv_deg(mean/max)=7.89/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 18:59:15][INFO] [IncamDebug] transl_c_loss=0.1976 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/23 18:59:15][INFO] [LossBreakdown] body_pose=0.9275 betas=0.6969 go_c=0.0838 go_gv=0.0195 transl_vel=0.1868 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/10.14 go_gv_deg(mean/max)=8.97/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 18:59:15][INFO] [IncamDebug] transl_c_loss=0.0029 pred_cam_abs_mean=0.903 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/23 18:59:24][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(323.4846496582031, 162.34628295898438, 810.02392578125, 1712.0618896484375, 0.23628446459770203)
+[02/23 18:59:24][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(314.986572265625, 158.0487823486328, 879.7130126953125, 1737.4412841796875, 0.2394775003194809)
+[02/23 18:59:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.195 max=0.385
+[02/23 18:59:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9959 pred=+1.0529 delta(pred-gt)=+0.0570
+[02/23 18:59:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[-0.00785214 1.9580591 0.07441644]
+[02/23 18:59:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[-0.11266945 2.1451623 0.00838481]
+[02/23 18:59:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+112.18,-3.21,+1.70) pred_vs_gt=(-8.73,+3.15,+3.94)
+[02/23 18:59:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+9.41
+[02/23 19:00:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 19:00:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 19:00:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 19:00:40][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 19:00:40][INFO] ✅[FIT][Epoch 0] finished! 01:28→1:12:21 | loss_epoch=2.83
+[02/23 19:00:40][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 19:00:40][INFO] [LossBreakdown] body_pose=0.5714 betas=0.4836 go_c=0.1783 go_gv=0.0260 transl_vel=0.3446 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.52/25.53 go_gv_deg(mean/max)=12.54/23.65 tv_abs_mean(pred)=0.0019 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.20 static_conf_hi_frac=0.00
+[02/23 19:00:40][INFO] [IncamDebug] transl_c_loss=0.0019 pred_cam_abs_mean=0.902 gt_pred_cam_abs_mean=0.936 valid_frac=1.00
+[02/23 19:00:40][INFO] [LossBreakdown] body_pose=1.3535 betas=0.5144 go_c=0.1131 go_gv=0.0297 transl_vel=0.5096 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.40/31.51 go_gv_deg(mean/max)=9.91/32.67 tv_abs_mean(pred)=0.0037 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.21 static_conf_hi_frac=0.00
+[02/23 19:00:40][INFO] [IncamDebug] transl_c_loss=0.0108 pred_cam_abs_mean=0.965 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/23 19:00:40][INFO] [LossBreakdown] body_pose=0.6804 betas=0.7457 go_c=0.0743 go_gv=0.0478 transl_vel=0.4180 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.27/21.92 go_gv_deg(mean/max)=10.46/23.65 tv_abs_mean(pred)=0.0034 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.16 static_conf_hi_frac=0.00
+[02/23 19:00:41][INFO] [IncamDebug] transl_c_loss=0.1396 pred_cam_abs_mean=0.907 gt_pred_cam_abs_mean=1.136 valid_frac=0.89
+[02/23 19:00:41][INFO] [LossBreakdown] body_pose=0.7086 betas=0.6356 go_c=0.5558 go_gv=0.0292 transl_vel=0.8051 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.64/122.92 go_gv_deg(mean/max)=8.53/128.36 tv_abs_mean(pred)=0.0047 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.19 static_conf_hi_frac=0.00
+[02/23 19:00:41][INFO] [IncamDebug] transl_c_loss=0.0843 pred_cam_abs_mean=0.880 gt_pred_cam_abs_mean=1.094 valid_frac=1.00
+[02/23 19:00:56][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof f0 gt=(549.347900390625, 253.296630859375, 737.7196655273438, 1149.4931640625, 0.3052249550819397) pred=(478.292236328125, 247.044189453125, 820.6392822265625, 1121.7333984375, 0.28142234683036804)
+[02/23 19:00:56][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof fl gt=(554.8383178710938, 266.72760009765625, 695.2646484375, 1172.6390380859375, 0.28505077958106995) pred=(479.3900451660156, 255.25628662109375, 747.6093139648438, 1059.07666015625, 0.26153844594955444)
+[02/23 19:00:58][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_transl_err_m mean=0.066 max=0.129
+[02/23 19:00:58][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+1.0413 pred=+0.9964 delta(pred-gt)=-0.0449
+[02/23 19:00:58][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.00300199 -1.5857557 0.06613082] global_orient0_aa(pred)=[-0.24534993 -1.5222888 -0.01460404]
+[02/23 19:00:58][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(raw_gt_world)=[-0.05838358 0.00389506 2.1818743 ]
+[02/23 19:00:58][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-90.84,+2.53,+2.28) pred=(-88.32,-9.68,+8.35) pred_vs_gt=(+3.52,-5.80,-12.35)
+[02/23 19:00:58][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-1.39
+[02/23 19:02:10][INFO] ✅[FIT][Epoch 1] finished! 02:58→1:11:25 | loss_epoch=1.68
+[02/23 19:02:10][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/23 19:02:10][INFO] [LossBreakdown] body_pose=1.9700 betas=0.3159 go_c=0.0549 go_gv=0.0247 transl_vel=0.2214 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=5.84/9.49 go_gv_deg(mean/max)=9.04/13.43 tv_abs_mean(pred)=0.0014 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.00 static_conf_mean=0.28 static_conf_hi_frac=0.01
+[02/23 19:02:10][INFO] [IncamDebug] transl_c_loss=0.0058 pred_cam_abs_mean=0.954 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/23 19:02:10][INFO] [LossBreakdown] body_pose=0.3102 betas=0.4999 go_c=0.2069 go_gv=0.0283 transl_vel=0.4910 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.31/26.30 go_gv_deg(mean/max)=8.42/25.79 tv_abs_mean(pred)=0.0016 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.23 static_conf_hi_frac=0.00
+[02/23 19:02:10][INFO] [IncamDebug] transl_c_loss=0.0177 pred_cam_abs_mean=0.913 gt_pred_cam_abs_mean=1.047 valid_frac=0.87
+[02/23 19:02:10][INFO] [LossBreakdown] body_pose=0.4275 betas=0.4491 go_c=0.2121 go_gv=0.0721 transl_vel=0.5087 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.91/30.64 go_gv_deg(mean/max)=16.78/32.98 tv_abs_mean(pred)=0.0031 tv_abs_mean(tgt)=0.0029 tv_zero_frac(pred)=0.00 static_conf_mean=0.24 static_conf_hi_frac=0.00
+[02/23 19:02:10][INFO] [IncamDebug] transl_c_loss=0.0648 pred_cam_abs_mean=0.990 gt_pred_cam_abs_mean=1.253 valid_frac=0.73
+[02/23 19:02:11][INFO] [LossBreakdown] body_pose=0.3765 betas=0.5900 go_c=0.1702 go_gv=0.0453 transl_vel=0.3135 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.51/22.80 go_gv_deg(mean/max)=11.79/21.00 tv_abs_mean(pred)=0.0025 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.15 static_conf_hi_frac=0.00
+[02/23 19:02:11][INFO] [IncamDebug] transl_c_loss=0.0527 pred_cam_abs_mean=0.976 gt_pred_cam_abs_mean=1.112 valid_frac=1.00
+[02/23 19:02:37][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof f0 gt=(497.9825744628906, 218.85873413085938, 754.4590454101562, 1513.018798828125, 0.4923076927661896) pred=(223.52304077148438, 192.02859497070312, 687.3060302734375, 1587.8349609375, 0.39462989568710327)
+[02/23 19:02:37][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof fl gt=(531.361083984375, 220.1790771484375, 725.6470947265625, 1474.6353759765625, 0.49013060331344604) pred=(200.69216918945312, 217.2867889404297, 687.1187133789062, 1476.0733642578125, 0.3415094316005707)
+[02/23 19:02:40][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_transl_err_m mean=0.113 max=0.202
+[02/23 19:02:40][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 root_y0: gt=+1.0023 pred=+1.0305 delta(pred-gt)=+0.0282
+[02/23 19:02:40][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(gt)=[ 0.04780671 -1.4102775 0.10208653] global_orient0_aa(pred)=[-0.05897016 -1.3987416 0.01304509]
+[02/23 19:02:40][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(raw_gt_world)=[-0.12592246 0.01295725 2.1298966 ]
+[02/23 19:02:40][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_yxz_deg gt=(-80.76,+5.40,+2.47) pred=(-80.21,-1.94,+2.53) pred_vs_gt=(+0.63,-1.24,-7.23)
+[02/23 19:02:40][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 yaw0_deg(pred_vs_gt)=-6.47
+[02/23 19:03:36][INFO] ✅[FIT][Epoch 2] finished! 04:24→1:09:11 | loss_epoch=1.47
+[02/23 19:03:36][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[02/23 19:03:36][INFO] [LossBreakdown] body_pose=0.3399 betas=0.3769 go_c=0.1054 go_gv=0.0885 transl_vel=0.2344 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=16.87/34.69 go_gv_deg(mean/max)=18.84/36.04 tv_abs_mean(pred)=0.0016 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.29 static_conf_hi_frac=0.00
+[02/23 19:03:39][INFO] [IncamDebug] transl_c_loss=0.0273 pred_cam_abs_mean=1.268 gt_pred_cam_abs_mean=1.320 valid_frac=0.91
+[02/23 19:03:39][INFO] [LossBreakdown] body_pose=0.9069 betas=0.2703 go_c=0.0883 go_gv=0.0389 transl_vel=0.2426 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.79/36.23 go_gv_deg(mean/max)=12.37/36.70 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0015 tv_zero_frac(pred)=0.00 static_conf_mean=0.47 static_conf_hi_frac=0.11
+[02/23 19:03:39][INFO] [IncamDebug] transl_c_loss=0.0472 pred_cam_abs_mean=1.034 gt_pred_cam_abs_mean=0.977 valid_frac=1.00
+[02/23 19:03:39][INFO] [LossBreakdown] body_pose=0.2412 betas=0.1520 go_c=0.1178 go_gv=0.0207 transl_vel=0.1982 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.87/17.05 go_gv_deg(mean/max)=5.09/14.36 tv_abs_mean(pred)=0.0013 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.44 static_conf_hi_frac=0.00
+[02/23 19:03:39][INFO] [IncamDebug] transl_c_loss=0.0082 pred_cam_abs_mean=0.950 gt_pred_cam_abs_mean=1.009 valid_frac=1.00
+[02/23 19:03:40][INFO] [LossBreakdown] body_pose=0.2561 betas=0.2679 go_c=0.1409 go_gv=0.0169 transl_vel=0.1346 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.50/18.34 go_gv_deg(mean/max)=6.88/15.86 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.35 static_conf_hi_frac=0.00
+[02/23 19:03:40][INFO] [IncamDebug] transl_c_loss=0.0090 pred_cam_abs_mean=0.949 gt_pred_cam_abs_mean=0.949 valid_frac=1.00
+[02/23 19:03:45][INFO] [VisUnityVal] e003_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(329.3016052246094, 216.16949462890625, 752.96728515625, 1738.2373046875, 0.39477503299713135) pred=(284.3705749511719, 113.29678344726562, 780.4983520507812, 1516.7451171875, 0.34586355090141296)
+[02/23 19:03:45][INFO] [VisUnityVal] e003_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(369.14556884765625, 216.22442626953125, 795.2476806640625, 1742.6558837890625, 0.2528301775455475) pred=(304.5757141113281, 129.3111572265625, 780.316162109375, 1570.73291015625, 0.3191581964492798)
+[02/23 19:03:48][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_transl_err_m mean=0.014 max=0.030
+[02/23 19:03:48][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+1.0048 pred=+1.0035 delta(pred-gt)=-0.0013
+[02/23 19:03:48][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[0.02579728 2.1115918 0.0080679 ] global_orient0_aa(pred)=[ 0.05493122 2.111311 -0.00523172]
+[02/23 19:03:48][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[-0.1246493 2.1389952 0.01005784]
+[02/23 19:03:48][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.99,+0.27,+1.25) pred=(+121.02,+1.49,+2.14) pred_vs_gt=(+0.01,+0.13,-1.51)
+[02/23 19:03:48][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-2.97
+[02/23 19:05:03][INFO] ✅[FIT][Epoch 3] finished! 05:51→1:07:24 | loss_epoch=1.12
+[02/23 19:05:03][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[02/23 19:05:03][INFO] [LossBreakdown] body_pose=0.1873 betas=0.1867 go_c=0.2510 go_gv=0.0471 transl_vel=0.4139 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.78/41.41 go_gv_deg(mean/max)=10.34/25.66 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.26 static_conf_hi_frac=0.00
+[02/23 19:05:03][INFO] [IncamDebug] transl_c_loss=0.0110 pred_cam_abs_mean=0.982 gt_pred_cam_abs_mean=1.048 valid_frac=0.78
+[02/23 19:05:03][INFO] [LossBreakdown] body_pose=1.6423 betas=0.1596 go_c=0.0711 go_gv=0.0235 transl_vel=0.1195 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=5.83/8.78 go_gv_deg(mean/max)=8.06/12.69 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0015 tv_zero_frac(pred)=0.00 static_conf_mean=0.37 static_conf_hi_frac=0.02
+[02/23 19:05:03][INFO] [IncamDebug] transl_c_loss=0.0071 pred_cam_abs_mean=0.963 gt_pred_cam_abs_mean=0.975 valid_frac=1.00
+[02/23 19:05:04][INFO] [LossBreakdown] body_pose=0.1738 betas=0.0870 go_c=0.1195 go_gv=0.0291 transl_vel=0.2041 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.33/18.69 go_gv_deg(mean/max)=9.08/16.27 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0026 tv_zero_frac(pred)=0.00 static_conf_mean=0.30 static_conf_hi_frac=0.00
+[02/23 19:05:04][INFO] [IncamDebug] transl_c_loss=0.0100 pred_cam_abs_mean=0.944 gt_pred_cam_abs_mean=1.013 valid_frac=1.00
+[02/23 19:05:04][INFO] [LossBreakdown] body_pose=0.1690 betas=0.2138 go_c=0.1482 go_gv=0.0280 transl_vel=0.0881 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.47/18.51 go_gv_deg(mean/max)=7.00/13.43 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0019 tv_zero_frac(pred)=0.00 static_conf_mean=0.20 static_conf_hi_frac=0.00
+[02/23 19:05:04][INFO] [IncamDebug] transl_c_loss=0.0295 pred_cam_abs_mean=1.122 gt_pred_cam_abs_mean=1.092 valid_frac=1.00
+[02/23 19:05:20][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 incam_proj_bbox_oof f0 gt=(568.93359375, 268.9086608886719, 791.1539916992188, 1237.05029296875, 0.32917270064353943) pred=(607.3878784179688, 281.0969543457031, 824.7445068359375, 1032.6229248046875, 0.15820029377937317)
+[02/23 19:05:20][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 incam_proj_bbox_oof fl gt=(541.4463500976562, 271.5432434082031, 756.655517578125, 1234.872802734375, 0.3378809690475464) pred=(584.858154296875, 275.6537780761719, 764.3785400390625, 1041.16162109375, 0.16008707880973816)
+[02/23 19:05:25][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 global_transl_err_m mean=0.067 max=0.125
+[02/23 19:05:25][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 root_y0: gt=+1.0514 pred=+0.9880 delta(pred-gt)=-0.0634
+[02/23 19:05:25][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 global_orient0_aa(gt)=[ 0.16767852 2.2633636 -0.10367614] global_orient0_aa(pred)=[ 0.09144579 2.4612672 -0.03438561]
+[02/23 19:05:25][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 global_orient0_aa(raw_gt_world)=[ 0.0239017 0.13060565 -0.15083215]
+[02/23 19:05:25][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 global_orient0_yxz_deg gt=(+130.33,+7.54,+4.98) pred=(+141.18,+2.76,+3.28) pred_vs_gt=(+11.07,+1.80,+4.74)
+[02/23 19:05:25][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 yaw0_deg(pred_vs_gt)=-10.20
+[02/23 19:06:24][INFO] ✅[FIT][Epoch 4] finished! 07:13→1:05:00 | loss_epoch=1.07
+[02/23 19:09:59][INFO] 🚀[FIT][Epoch 5] Data: unity Experiment: finetune_
+[02/23 19:10:10][INFO] [LossBreakdown] body_pose=0.1297 betas=0.1091 go_c=0.1581 go_gv=0.0254 transl_vel=0.1584 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.31/17.98 go_gv_deg(mean/max)=8.25/15.86 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.18 static_conf_hi_frac=0.00
+[02/23 19:10:10][INFO] [IncamDebug] transl_c_loss=0.0118 pred_cam_abs_mean=1.067 gt_pred_cam_abs_mean=1.136 valid_frac=0.85
+[02/23 19:10:10][INFO] [LossBreakdown] body_pose=0.1288 betas=0.0822 go_c=0.0648 go_gv=0.0363 transl_vel=0.1525 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.17/14.81 go_gv_deg(mean/max)=10.54/17.61 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.16 static_conf_hi_frac=0.00
+[02/23 19:10:10][INFO] [IncamDebug] transl_c_loss=0.0108 pred_cam_abs_mean=1.028 gt_pred_cam_abs_mean=1.070 valid_frac=1.00
+[02/23 19:10:11][INFO] [LossBreakdown] body_pose=0.1447 betas=0.0518 go_c=0.0720 go_gv=0.0474 transl_vel=0.2829 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.10/24.74 go_gv_deg(mean/max)=15.40/26.43 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0027 tv_zero_frac(pred)=0.00 static_conf_mean=0.21 static_conf_hi_frac=0.00
+[02/23 19:10:11][INFO] [IncamDebug] transl_c_loss=0.0055 pred_cam_abs_mean=0.945 gt_pred_cam_abs_mean=0.992 valid_frac=1.00
+[02/23 19:10:11][INFO] [LossBreakdown] body_pose=0.1696 betas=0.0931 go_c=0.1731 go_gv=0.0233 transl_vel=0.1425 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.81/18.51 go_gv_deg(mean/max)=7.34/17.24 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.26 static_conf_hi_frac=0.00
+[02/23 19:10:11][INFO] [IncamDebug] transl_c_loss=0.0109 pred_cam_abs_mean=0.955 gt_pred_cam_abs_mean=1.025 valid_frac=1.00
+[02/23 19:10:25][INFO] [VisUnityVal] e005_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof f0 gt=(558.434814453125, 263.2696533203125, 709.9166259765625, 1185.8602294921875, 0.29158198833465576) pred=(525.0533447265625, 261.77734375, 690.3836059570312, 1060.760498046875, 0.2654571831226349)
+[02/23 19:10:25][INFO] [VisUnityVal] e005_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof fl gt=(565.3956909179688, 252.05532836914062, 719.9682006835938, 1131.3701171875, 0.30798256397247314) pred=(518.8792724609375, 250.21383666992188, 678.6074829101562, 1109.66650390625, 0.28606677055358887)
+[02/23 19:10:28][INFO] [VisUnityVal] e005_101_biboo_birthday_speech_explosion_2 global_transl_err_m mean=0.123 max=0.236
+[02/23 19:10:28][INFO] [VisUnityVal] e005_101_biboo_birthday_speech_explosion_2 root_y0: gt=+1.0506 pred=+0.9964 delta(pred-gt)=-0.0542
+[02/23 19:10:28][INFO] [VisUnityVal] e005_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.01898461 -1.5685334 0.05229336] global_orient0_aa(pred)=[-0.05844741 -1.7847359 -0.04598482]
+[02/23 19:10:28][INFO] [VisUnityVal] e005_101_biboo_birthday_speech_explosion_2 global_orient0_aa(raw_gt_world)=[-0.14174439 0.01252025 2.123059 ]
+[02/23 19:10:28][INFO] [VisUnityVal] e005_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-89.86,+2.60,+1.22) pred=(-102.32,-3.62,+0.83) pred_vs_gt=(-12.48,+0.37,-6.22)
+[02/23 19:10:28][INFO] [VisUnityVal] e005_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=+7.36
+[02/23 19:12:00][INFO] ✅[FIT][Epoch 5] finished! 12:49→1:33:59 | loss_epoch=0.735
+[02/23 19:12:00][INFO] 🚀[FIT][Epoch 6] Data: unity Experiment: finetune_
+[02/23 19:12:01][INFO] [LossBreakdown] body_pose=0.1176 betas=0.0705 go_c=0.0723 go_gv=0.0365 transl_vel=0.2613 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.35/20.84 go_gv_deg(mean/max)=12.82/20.04 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0027 tv_zero_frac(pred)=0.00 static_conf_mean=0.29 static_conf_hi_frac=0.00
+[02/23 19:12:01][INFO] [IncamDebug] transl_c_loss=0.0051 pred_cam_abs_mean=0.942 gt_pred_cam_abs_mean=0.994 valid_frac=1.00
+[02/23 19:12:01][INFO] [LossBreakdown] body_pose=0.1215 betas=0.0598 go_c=0.0507 go_gv=0.0348 transl_vel=0.2918 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.92/19.21 go_gv_deg(mean/max)=10.64/18.86 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0027 tv_zero_frac(pred)=0.00 static_conf_mean=0.24 static_conf_hi_frac=0.00
+[02/23 19:12:01][INFO] [IncamDebug] transl_c_loss=0.0100 pred_cam_abs_mean=1.061 gt_pred_cam_abs_mean=1.157 valid_frac=0.74
+[02/23 19:12:02][INFO] [LossBreakdown] body_pose=0.1103 betas=0.0558 go_c=0.0629 go_gv=0.0346 transl_vel=0.1346 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.34/22.80 go_gv_deg(mean/max)=9.99/22.51 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.20 static_conf_hi_frac=0.00
+[02/23 19:12:02][INFO] [IncamDebug] transl_c_loss=0.0079 pred_cam_abs_mean=1.065 gt_pred_cam_abs_mean=1.090 valid_frac=1.00
+[02/23 19:12:02][INFO] [LossBreakdown] body_pose=0.1093 betas=0.0479 go_c=0.2636 go_gv=0.0150 transl_vel=0.0841 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.95/18.16 go_gv_deg(mean/max)=6.68/11.90 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.02 static_conf_mean=0.29 static_conf_hi_frac=0.00
+[02/23 19:12:02][INFO] [IncamDebug] transl_c_loss=0.0058 pred_cam_abs_mean=0.982 gt_pred_cam_abs_mean=1.022 valid_frac=1.00
+[02/23 19:12:27][INFO] [VisUnityVal] e006_105_biboo_birthday_speech_explosion_6 incam_proj_bbox_oof f0 gt=(499.1485290527344, 217.60928344726562, 776.417724609375, 2137.31884765625, 0.5338171124458313) pred=(547.6002197265625, 150.18568420410156, 909.763427734375, 1484.0111083984375, 0.43628445267677307)
+[02/23 19:12:27][INFO] [VisUnityVal] e006_105_biboo_birthday_speech_explosion_6 incam_proj_bbox_oof fl gt=(530.3594360351562, 229.9683837890625, 815.789306640625, 2208.416015625, 0.5288824439048767) pred=(554.02197265625, 166.2881622314453, 899.3921508789062, 1425.3974609375, 0.42859214544296265)
+[02/23 19:12:32][INFO] [VisUnityVal] e006_105_biboo_birthday_speech_explosion_6 global_transl_err_m mean=0.084 max=0.156
+[02/23 19:12:32][INFO] [VisUnityVal] e006_105_biboo_birthday_speech_explosion_6 root_y0: gt=+0.9810 pred=+0.9941 delta(pred-gt)=+0.0131
+[02/23 19:12:32][INFO] [VisUnityVal] e006_105_biboo_birthday_speech_explosion_6 global_orient0_aa(gt)=[ 0.00965729 1.5042514 -0.01759416] global_orient0_aa(pred)=[ 0.0307307 1.9132756 -0.06672973]
+[02/23 19:12:32][INFO] [VisUnityVal] e006_105_biboo_birthday_speech_explosion_6 global_orient0_aa(raw_gt_world)=[-1.1115327 -1.2904961 -1.0112354]
+[02/23 19:12:32][INFO] [VisUnityVal] e006_105_biboo_birthday_speech_explosion_6 global_orient0_yxz_deg gt=(+86.19,+0.99,-0.33) pred=(+109.64,+3.54,-0.65) pred_vs_gt=(+23.47,-0.16,-2.56)
+[02/23 19:12:32][INFO] [VisUnityVal] e006_105_biboo_birthday_speech_explosion_6 yaw0_deg(pred_vs_gt)=-23.88
+[02/23 19:13:55][INFO] ✅[FIT][Epoch 6] finished! 14:43→1:30:28 | loss_epoch=0.7
+[02/23 19:13:55][INFO] 🚀[FIT][Epoch 7] Data: unity Experiment: finetune_
+[02/23 19:13:55][INFO] [LossBreakdown] body_pose=0.0859 betas=0.0522 go_c=0.1687 go_gv=0.0367 transl_vel=0.1525 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.39/18.51 go_gv_deg(mean/max)=11.83/21.31 tv_abs_mean(pred)=0.0003 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.02 static_conf_mean=0.28 static_conf_hi_frac=0.00
+[02/23 19:13:55][INFO] [IncamDebug] transl_c_loss=0.0062 pred_cam_abs_mean=0.983 gt_pred_cam_abs_mean=0.998 valid_frac=1.00
+[02/23 19:13:55][INFO] [LossBreakdown] body_pose=0.0826 betas=0.0589 go_c=0.1306 go_gv=0.0249 transl_vel=0.1116 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.44/16.66 go_gv_deg(mean/max)=7.70/14.36 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.27 static_conf_hi_frac=0.00
+[02/23 19:13:55][INFO] [IncamDebug] transl_c_loss=0.0031 pred_cam_abs_mean=1.005 gt_pred_cam_abs_mean=1.024 valid_frac=1.00
+[02/23 19:13:56][INFO] [LossBreakdown] body_pose=0.0823 betas=0.0583 go_c=0.0814 go_gv=0.0445 transl_vel=0.1709 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.24/24.47 go_gv_deg(mean/max)=13.32/25.01 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.01 static_conf_mean=0.21 static_conf_hi_frac=0.00
+[02/23 19:13:56][INFO] [IncamDebug] transl_c_loss=0.0077 pred_cam_abs_mean=1.073 gt_pred_cam_abs_mean=1.082 valid_frac=1.00
+[02/23 19:13:56][INFO] [LossBreakdown] body_pose=0.1052 betas=0.0567 go_c=0.0614 go_gv=0.0379 transl_vel=0.2094 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.86/24.61 go_gv_deg(mean/max)=11.70/24.34 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.19 static_conf_hi_frac=0.00
+[02/23 19:13:56][INFO] [IncamDebug] transl_c_loss=0.0066 pred_cam_abs_mean=1.102 gt_pred_cam_abs_mean=1.160 valid_frac=0.80
+[02/23 19:14:19][INFO] [VisUnityVal] e007_105_biboo_birthday_speech_explosion_6 incam_proj_bbox_oof f0 gt=(291.9891357421875, 158.2655792236328, 1007.9339599609375, 4402.673828125, 0.5322906970977783) pred=(444.91583251953125, -25.974456787109375, 797.6522827148438, 2780.3017578125, 0.5297532677650452)
+[02/23 19:14:19][INFO] [VisUnityVal] e007_105_biboo_birthday_speech_explosion_6 incam_proj_bbox_oof fl gt=(501.8149719238281, 236.8048095703125, 818.722412109375, 2088.8740234375, 0.5071117281913757) pred=(560.4085083007812, 173.38119506835938, 887.2272338867188, 1527.4852294921875, 0.4409288763999939)
+[02/23 19:14:22][INFO] [VisUnityVal] e007_105_biboo_birthday_speech_explosion_6 global_transl_err_m mean=0.133 max=0.244
+[02/23 19:14:22][INFO] [VisUnityVal] e007_105_biboo_birthday_speech_explosion_6 root_y0: gt=+1.1162 pred=+0.9985 delta(pred-gt)=-0.1177
+[02/23 19:14:22][INFO] [VisUnityVal] e007_105_biboo_birthday_speech_explosion_6 global_orient0_aa(gt)=[ 0.03752447 -2.0311391 0.06943011] global_orient0_aa(pred)=[ 0.00467517 -1.8424778 -0.01905086]
+[02/23 19:14:22][INFO] [VisUnityVal] e007_105_biboo_birthday_speech_explosion_6 global_orient0_aa(raw_gt_world)=[ 1.2816364 1.0833013 -1.1528068]
+[02/23 19:14:22][INFO] [VisUnityVal] e007_105_biboo_birthday_speech_explosion_6 global_orient0_yxz_deg gt=(-116.42,+3.78,+0.22) pred=(-105.57,-0.61,-0.76) pred_vs_gt=(+10.79,+2.83,-3.50)
+[02/23 19:14:22][INFO] [VisUnityVal] e007_105_biboo_birthday_speech_explosion_6 yaw0_deg(pred_vs_gt)=-10.01
+[02/23 19:15:19][INFO] ✅[FIT][Epoch 7] finished! 16:07→1:24:40 | loss_epoch=0.647
+[02/23 19:15:19][INFO] 🚀[FIT][Epoch 8] Data: unity Experiment: finetune_
+[02/23 19:15:19][INFO] [LossBreakdown] body_pose=0.0671 betas=0.0540 go_c=0.1491 go_gv=0.0404 transl_vel=0.1666 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.88/25.66 go_gv_deg(mean/max)=13.48/26.55 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.01 static_conf_mean=0.16 static_conf_hi_frac=0.00
+[02/23 19:15:19][INFO] [IncamDebug] transl_c_loss=0.0010 pred_cam_abs_mean=0.918 gt_pred_cam_abs_mean=0.932 valid_frac=1.00
+[02/23 19:15:19][INFO] [LossBreakdown] body_pose=1.0722 betas=0.1623 go_c=0.0532 go_gv=0.0240 transl_vel=0.0971 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=5.39/13.90 go_gv_deg(mean/max)=7.05/16.86 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0014 tv_zero_frac(pred)=0.00 static_conf_mean=0.51 static_conf_hi_frac=0.33
+[02/23 19:15:19][INFO] [IncamDebug] transl_c_loss=0.0067 pred_cam_abs_mean=0.996 gt_pred_cam_abs_mean=1.049 valid_frac=1.00
+[02/23 19:15:19][INFO] [LossBreakdown] body_pose=0.0726 betas=0.0469 go_c=0.0358 go_gv=0.0256 transl_vel=0.1953 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=4.85/14.59 go_gv_deg(mean/max)=7.52/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.22 static_conf_hi_frac=0.00
+[02/23 19:15:19][INFO] [IncamDebug] transl_c_loss=0.0076 pred_cam_abs_mean=1.099 gt_pred_cam_abs_mean=1.142 valid_frac=0.91
+[02/23 19:15:20][INFO] [LossBreakdown] body_pose=0.0768 betas=0.0365 go_c=0.1537 go_gv=0.0189 transl_vel=0.0865 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.95/16.66 go_gv_deg(mean/max)=6.14/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.01 static_conf_mean=0.24 static_conf_hi_frac=0.00
+[02/23 19:15:20][INFO] [IncamDebug] transl_c_loss=0.0091 pred_cam_abs_mean=1.133 gt_pred_cam_abs_mean=1.183 valid_frac=1.00
+[02/23 19:15:41][INFO] [VisUnityVal] e008_105_biboo_birthday_speech_explosion_6 incam_proj_bbox_oof f0 gt=(519.421875, 240.859375, 902.351806640625, 2130.34033203125, 0.5030478835105896) pred=(561.9110107421875, 189.54425048828125, 910.9813842773438, 1467.9703369140625, 0.42264148592948914)
+[02/23 19:15:41][INFO] [VisUnityVal] e008_105_biboo_birthday_speech_explosion_6 incam_proj_bbox_oof fl gt=(497.9971618652344, 208.18832397460938, 782.8516235351562, 2232.325439453125, 0.5551523566246033) pred=(548.7047729492188, 177.19873046875, 917.4993896484375, 1624.821044921875, 0.4583454132080078)
+[02/23 19:15:43][INFO] [VisUnityVal] e008_105_biboo_birthday_speech_explosion_6 global_transl_err_m mean=0.069 max=0.182
+[02/23 19:15:43][INFO] [VisUnityVal] e008_105_biboo_birthday_speech_explosion_6 root_y0: gt=+1.0184 pred=+0.9843 delta(pred-gt)=-0.0341
+[02/23 19:15:43][INFO] [VisUnityVal] e008_105_biboo_birthday_speech_explosion_6 global_orient0_aa(gt)=[0.05331048 1.909154 0.00906792] global_orient0_aa(pred)=[ 0.06317088 2.0355601 -0.07008975]
+[02/23 19:15:43][INFO] [VisUnityVal] e008_105_biboo_birthday_speech_explosion_6 global_orient0_aa(raw_gt_world)=[ 1.4551488 1.0548187 -1.0416094]
+[02/23 19:15:43][INFO] [VisUnityVal] e008_105_biboo_birthday_speech_explosion_6 global_orient0_yxz_deg gt=(+109.43,+1.15,+2.39) pred=(+116.73,+4.45,+0.82) pred_vs_gt=(+7.43,-2.58,-2.59)
+[02/23 19:15:43][INFO] [VisUnityVal] e008_105_biboo_birthday_speech_explosion_6 yaw0_deg(pred_vs_gt)=-6.90
+[02/23 19:16:14][INFO] [Exp Name]: finetune_
+[02/23 19:16:14][INFO] [GPU x Batch] = 1 x 2
+[02/23 19:16:14][INFO] [UnityDataset] Found 9 sequences.
+[02/23 19:16:14][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 19:16:14][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 19:16:14][INFO]
+[02/23 19:16:14][INFO] [UnityDataset] Found 9 sequences.
+[02/23 19:16:14][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 19:16:14][INFO]
+[02/23 19:16:20][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 19:16:46][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_16/checkpoints'
+[02/23 19:16:55][INFO] Start Fitting...
+[02/23 19:16:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 19:16:56][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 19:16:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 19:16:58][INFO] [LossBreakdown] body_pose=0.9151 betas=0.6620 go_c=0.5413 go_gv=0.0710 transl_vel=0.1780 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.73/35.37 go_gv_deg(mean/max)=13.78/28.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/23 19:16:58][INFO] [IncamDebug] transl_c_loss=0.0419 pred_cam_abs_mean=0.887 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/23 19:16:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 19:16:59][INFO] [LossBreakdown] body_pose=2.1546 betas=0.7537 go_c=0.1314 go_gv=0.0420 transl_vel=0.2005 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.54/27.54 go_gv_deg(mean/max)=10.66/26.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 19:16:59][INFO] [IncamDebug] transl_c_loss=0.1123 pred_cam_abs_mean=0.972 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/23 19:17:00][INFO] [LossBreakdown] body_pose=1.1252 betas=0.7384 go_c=0.2187 go_gv=0.0270 transl_vel=0.1787 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.71/20.36 go_gv_deg(mean/max)=7.89/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 19:17:00][INFO] [IncamDebug] transl_c_loss=0.1976 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/23 19:17:00][INFO] [LossBreakdown] body_pose=0.9275 betas=0.6969 go_c=0.0838 go_gv=0.0195 transl_vel=0.1868 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/10.14 go_gv_deg(mean/max)=8.97/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 19:17:00][INFO] [IncamDebug] transl_c_loss=0.0029 pred_cam_abs_mean=0.903 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/23 19:17:09][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(323.4846496582031, 162.34628295898438, 810.02392578125, 1712.0618896484375, 0.23628446459770203)
+[02/23 19:17:09][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(314.986572265625, 158.0487823486328, 879.7130126953125, 1737.4412841796875, 0.2394775003194809)
+[02/23 19:17:09][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof_padded f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.0) pred=(323.4846496582031, 162.34628295898438, 810.02392578125, 1712.0618896484375, 0.0)
+[02/23 19:17:09][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof_padded fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.0) pred=(314.986572265625, 158.0487823486328, 879.7130126953125, 1737.4412841796875, 0.0)
+[02/23 19:17:09][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_canvas_pad l/t/r/b=0/0/0/1140 new_size=1280x1860
+[02/23 19:17:15][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.195 max=0.385
+[02/23 19:17:15][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9959 pred=+1.0529 delta(pred-gt)=+0.0570
+[02/23 19:17:15][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[-0.00785214 1.9580591 0.07441644]
+[02/23 19:17:15][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[-0.11266945 2.1451623 0.00838481]
+[02/23 19:17:15][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+112.18,-3.21,+1.70) pred_vs_gt=(-8.73,+3.15,+3.94)
+[02/23 19:17:15][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+9.41
+[02/23 19:19:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 19:19:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 19:19:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 19:19:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 19:19:59][INFO] ✅[FIT][Epoch 0] finished! 03:03→2:30:09 | loss_epoch=2.83
+[02/23 19:19:59][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 19:19:59][INFO] [LossBreakdown] body_pose=0.5714 betas=0.4836 go_c=0.1783 go_gv=0.0260 transl_vel=0.3446 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.52/25.53 go_gv_deg(mean/max)=12.54/23.65 tv_abs_mean(pred)=0.0019 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.20 static_conf_hi_frac=0.00
+[02/23 19:19:59][INFO] [IncamDebug] transl_c_loss=0.0019 pred_cam_abs_mean=0.902 gt_pred_cam_abs_mean=0.936 valid_frac=1.00
+[02/23 19:49:06][INFO] [Exp Name]: finetune_
+[02/23 19:49:06][INFO] [GPU x Batch] = 1 x 2
+[02/23 19:49:06][INFO] [UnityDataset] Found 9 sequences.
+[02/23 19:49:06][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 19:49:06][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 19:49:06][INFO]
+[02/23 19:49:06][INFO] [UnityDataset] Found 9 sequences.
+[02/23 19:49:06][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 19:49:06][INFO]
+[02/23 19:49:13][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 19:49:40][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_17/checkpoints'
+[02/23 19:49:50][INFO] Start Fitting...
+[02/23 19:49:52][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 19:49:52][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 19:49:54][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 19:49:54][INFO] [LossBreakdown] body_pose=0.9203 betas=0.6590 go_c=0.5448 go_gv=3.4784 transl_vel=0.2262 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.81/35.75 go_gv_deg(mean/max)=91.52/103.93 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.41
+[02/23 19:49:54][INFO] [IncamDebug] transl_c_loss=0.0421 pred_cam_abs_mean=0.886 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/23 19:49:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 19:49:56][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7464 go_c=0.1315 go_gv=3.6927 transl_vel=0.1971 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.42/28.02 go_gv_deg(mean/max)=97.93/107.33 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 19:49:56][INFO] [IncamDebug] transl_c_loss=0.1115 pred_cam_abs_mean=0.973 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/23 19:49:56][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=4.4246 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=92.83/99.58 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 19:49:56][INFO] [IncamDebug] transl_c_loss=0.1979 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/23 19:49:56][INFO] [LossBreakdown] body_pose=0.9267 betas=0.6977 go_c=0.0836 go_gv=4.8499 transl_vel=0.1916 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/9.82 go_gv_deg(mean/max)=97.38/102.21 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 19:49:56][INFO] [IncamDebug] transl_c_loss=0.0029 pred_cam_abs_mean=0.903 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/23 19:50:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(348.0452575683594, 141.31192016601562, 827.3465576171875, 1670.8978271484375, 0.35747459530830383)
+[02/23 19:50:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(333.13543701171875, 140.355224609375, 826.5665283203125, 1701.203125, 0.3582002818584442)
+[02/23 19:50:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof_padded f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.0) pred=(348.0452575683594, 141.31192016601562, 827.3465576171875, 1670.8978271484375, 0.0)
+[02/23 19:50:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof_padded fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.0) pred=(333.13543701171875, 140.355224609375, 826.5665283203125, 1701.203125, 0.0)
+[02/23 19:50:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_canvas_pad l/t/r/b=0/0/0/1042 new_size=1280x1762
+[02/23 19:50:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.085 max=0.106 gt_z_mean=1.140 pred_z_mean=1.104
+[02/23 19:50:13][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.021 max=0.048
+[02/23 19:50:13][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.3569 pred=+0.9998 delta(pred-gt)=+0.6429
+[02/23 19:50:13][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 1.148579 1.0835718 -1.2705066] global_orient0_aa(pred)=[0.01908925 1.9270833 0.06778753]
+[02/23 19:50:13][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.0300002 3.0145898 -0.04516966]
+[02/23 19:50:13][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(-14.13,+82.55,-102.74) pred=(+110.40,-2.19,+2.65) pred_vs_gt=(+27.66,+0.86,+94.14)
+[02/23 19:50:13][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-26.18
+[02/23 19:56:52][INFO] [Exp Name]: finetune_
+[02/23 19:56:52][INFO] [GPU x Batch] = 1 x 2
+[02/23 19:56:52][INFO] [UnityDataset] Found 9 sequences.
+[02/23 19:56:52][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 19:56:52][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 19:56:52][INFO]
+[02/23 19:56:52][INFO] [UnityDataset] Found 9 sequences.
+[02/23 19:56:52][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 19:56:52][INFO]
+[02/23 19:56:59][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 19:57:25][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_18/checkpoints'
+[02/23 19:57:32][INFO] Start Fitting...
+[02/23 19:57:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 19:57:33][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 19:57:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 19:57:36][INFO] [LossBreakdown] body_pose=0.9203 betas=0.6590 go_c=0.5448 go_gv=0.0725 transl_vel=0.2262 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.81/35.75 go_gv_deg(mean/max)=13.86/28.49 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.41
+[02/23 19:57:36][INFO] [IncamDebug] transl_c_loss=0.0421 pred_cam_abs_mean=0.886 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/23 19:57:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 19:57:37][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7464 go_c=0.1315 go_gv=0.0418 transl_vel=0.1971 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.42/28.02 go_gv_deg(mean/max)=10.52/27.78 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 19:57:37][INFO] [IncamDebug] transl_c_loss=0.1115 pred_cam_abs_mean=0.973 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/23 19:57:37][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 19:57:37][INFO] [IncamDebug] transl_c_loss=0.1979 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/23 19:57:37][INFO] [LossBreakdown] body_pose=0.9267 betas=0.6977 go_c=0.0836 go_gv=0.0196 transl_vel=0.1916 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/9.82 go_gv_deg(mean/max)=8.98/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 19:57:37][INFO] [IncamDebug] transl_c_loss=0.0029 pred_cam_abs_mean=0.903 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/23 19:57:48][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(346.3108825683594, 142.3296661376953, 845.6177978515625, 1665.8909912109375, 0.3571843206882477)
+[02/23 19:57:48][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(332.3451232910156, 141.380859375, 822.3335571289062, 1695.947509765625, 0.3587808310985565)
+[02/23 19:57:48][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof_padded f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.0) pred=(346.3108825683594, 142.3296661376953, 845.6177978515625, 1665.8909912109375, 0.0)
+[02/23 19:57:48][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof_padded fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.0) pred=(332.3451232910156, 141.380859375, 822.3335571289062, 1695.947509765625, 0.0)
+[02/23 19:57:48][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_canvas_pad l/t/r/b=0/0/0/1042 new_size=1280x1762
+[02/23 19:57:48][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.083 max=0.103 gt_z_mean=1.140 pred_z_mean=1.108
+[02/23 19:57:54][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.021 max=0.043
+[02/23 19:57:54][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9975 pred=+1.0002 delta(pred-gt)=+0.0027
+[02/23 19:57:54][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[0.02282428 1.9486594 0.06558114]
+[02/23 19:57:54][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.0300002 3.0145898 -0.04516966]
+[02/23 19:57:54][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+111.64,-2.02,+2.71) pred_vs_gt=(-9.31,+3.40,+2.39)
+[02/23 19:57:54][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+8.01
+[02/23 20:00:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:00:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:00:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:00:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:00:29][INFO] ✅[FIT][Epoch 0] finished! 02:56→2:24:28 | loss_epoch=1.83
+[02/23 20:00:29][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 20:00:30][INFO] [LossBreakdown] body_pose=0.4396 betas=0.3765 go_c=0.1585 go_gv=0.0191 transl_vel=0.1995 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.28/22.51 go_gv_deg(mean/max)=10.50/19.04 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.53 static_conf_hi_frac=0.36
+[02/23 20:00:30][INFO] [IncamDebug] transl_c_loss=0.0013 pred_cam_abs_mean=0.909 gt_pred_cam_abs_mean=0.936 valid_frac=1.00
+[02/23 20:00:30][INFO] [LossBreakdown] body_pose=1.1988 betas=0.4801 go_c=0.0968 go_gv=0.0235 transl_vel=0.1599 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.60/35.95 go_gv_deg(mean/max)=9.46/38.26 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.58 static_conf_hi_frac=0.44
+[02/23 20:00:30][INFO] [IncamDebug] transl_c_loss=0.0081 pred_cam_abs_mean=0.969 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/23 20:00:30][INFO] [LossBreakdown] body_pose=0.3774 betas=0.4978 go_c=0.0803 go_gv=0.0470 transl_vel=0.1821 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.34/22.51 go_gv_deg(mean/max)=12.57/24.20 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.40 static_conf_hi_frac=0.07
+[02/23 20:00:30][INFO] [IncamDebug] transl_c_loss=0.1115 pred_cam_abs_mean=0.930 gt_pred_cam_abs_mean=1.136 valid_frac=0.89
+[02/23 20:00:30][INFO] [LossBreakdown] body_pose=0.3566 betas=0.4171 go_c=0.3248 go_gv=0.0353 transl_vel=0.8402 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.14/116.63 go_gv_deg(mean/max)=8.78/97.49 tv_abs_mean(pred)=0.0017 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.43 static_conf_hi_frac=0.09
+[02/23 20:00:30][INFO] [IncamDebug] transl_c_loss=0.0539 pred_cam_abs_mean=0.919 gt_pred_cam_abs_mean=1.094 valid_frac=1.00
+[02/23 20:00:44][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof f0 gt=(549.347900390625, 253.296630859375, 737.7196655273438, 1149.4931640625, 0.3052249550819397) pred=(514.33935546875, 230.67835998535156, 698.5016479492188, 1156.628662109375, 0.28824383020401)
+[02/23 20:00:44][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof fl gt=(554.8383178710938, 266.72760009765625, 695.2646484375, 1172.6390380859375, 0.28505077958106995) pred=(505.1429443359375, 240.62225341796875, 677.645263671875, 1097.8228759765625, 0.274455726146698)
+[02/23 20:00:44][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof_padded f0 gt=(549.347900390625, 253.296630859375, 737.7196655273438, 1149.4931640625, 0.0) pred=(514.33935546875, 230.67835998535156, 698.5016479492188, 1156.628662109375, 0.0)
+[02/23 20:00:44][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof_padded fl gt=(554.8383178710938, 266.72760009765625, 695.2646484375, 1172.6390380859375, 0.0) pred=(505.1429443359375, 240.62225341796875, 677.645263671875, 1097.8228759765625, 0.0)
+[02/23 20:00:44][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_canvas_pad l/t/r/b=0/0/0/474 new_size=1280x1194
+[02/23 20:00:44][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_transl_err_m mean=0.084 max=0.104 gt_z_mean=1.421 pred_z_mean=1.437
+[02/23 20:00:47][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_transl_err_m mean=0.073 max=0.160
+[02/23 20:00:47][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9925 pred=+0.9878 delta(pred-gt)=-0.0047
+[02/23 20:00:47][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.00300199 -1.5857557 0.06613082] global_orient0_aa(pred)=[-0.06903021 -1.5818751 -0.04569769]
+[02/23 20:00:47][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(raw_gt_world)=[-0.05748826 -3.0477192 0.07022861]
+[02/23 20:00:47][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-90.84,+2.53,+2.28) pred=(-90.71,-4.17,+0.88) pred_vs_gt=(+0.02,+1.50,-6.69)
+[02/23 20:00:47][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-0.83
+[02/23 20:02:30][INFO] ✅[FIT][Epoch 1] finished! 04:57→1:58:48 | loss_epoch=1.26
+[02/23 20:02:30][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/23 20:02:30][INFO] [LossBreakdown] body_pose=1.8983 betas=0.3241 go_c=0.0665 go_gv=0.0134 transl_vel=0.1578 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=5.98/10.14 go_gv_deg(mean/max)=3.95/8.01 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.00 static_conf_mean=0.85 static_conf_hi_frac=0.73
+[02/23 20:02:30][INFO] [IncamDebug] transl_c_loss=0.0050 pred_cam_abs_mean=0.964 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/23 20:02:30][INFO] [LossBreakdown] body_pose=0.2910 betas=0.2544 go_c=0.2179 go_gv=0.0306 transl_vel=0.9065 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.44/166.57 go_gv_deg(mean/max)=6.84/172.40 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.66
+[02/23 20:02:30][INFO] [IncamDebug] transl_c_loss=0.0140 pred_cam_abs_mean=0.938 gt_pred_cam_abs_mean=1.047 valid_frac=0.87
+[02/23 20:02:30][INFO] [LossBreakdown] body_pose=0.2887 betas=0.2964 go_c=0.1442 go_gv=0.0609 transl_vel=0.3467 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=16.28/32.46 go_gv_deg(mean/max)=15.99/33.89 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0026 tv_zero_frac(pred)=0.00 static_conf_mean=0.81 static_conf_hi_frac=0.69
+[02/23 20:02:30][INFO] [IncamDebug] transl_c_loss=0.0312 pred_cam_abs_mean=1.062 gt_pred_cam_abs_mean=1.253 valid_frac=0.73
+[02/23 20:02:31][INFO] [LossBreakdown] body_pose=0.2462 betas=0.3329 go_c=0.1615 go_gv=0.0332 transl_vel=0.1704 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.52/21.31 go_gv_deg(mean/max)=10.22/19.04 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.62
+[02/23 20:02:31][INFO] [IncamDebug] transl_c_loss=0.0270 pred_cam_abs_mean=1.015 gt_pred_cam_abs_mean=1.112 valid_frac=1.00
+[02/23 20:02:57][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof f0 gt=(497.9825744628906, 218.85873413085938, 754.4590454101562, 1513.018798828125, 0.4923076927661896) pred=(390.26239013671875, 183.29501342773438, 681.729248046875, 1543.8746337890625, 0.39404934644699097)
+[02/23 20:02:57][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof fl gt=(531.361083984375, 220.1790771484375, 725.6470947265625, 1474.6353759765625, 0.49013060331344604) pred=(392.94744873046875, 205.8941192626953, 672.3618774414062, 1463.5728759765625, 0.355878084897995)
+[02/23 20:02:57][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof_padded f0 gt=(497.9825744628906, 218.85873413085938, 754.4590454101562, 1513.018798828125, 0.0) pred=(390.26239013671875, 183.29501342773438, 681.729248046875, 1543.8746337890625, 0.0)
+[02/23 20:02:57][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof_padded fl gt=(531.361083984375, 220.1790771484375, 725.6470947265625, 1474.6353759765625, 0.0) pred=(392.94744873046875, 205.8941192626953, 672.3618774414062, 1463.5728759765625, 0.0)
+[02/23 20:02:57][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_canvas_pad l/t/r/b=0/0/0/869 new_size=1280x1589
+[02/23 20:02:57][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_transl_err_m mean=0.134 max=0.210 gt_z_mean=1.193 pred_z_mean=1.268
+[02/23 20:02:57][ERROR] height not divisible by 2 (1280x1589)
+[02/23 20:04:46][INFO] [Exp Name]: finetune_
+[02/23 20:04:46][INFO] [GPU x Batch] = 1 x 2
+[02/23 20:04:46][INFO] [UnityDataset] Found 9 sequences.
+[02/23 20:04:46][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 20:04:46][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 20:04:46][INFO]
+[02/23 20:04:46][INFO] [UnityDataset] Found 9 sequences.
+[02/23 20:04:46][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 20:04:46][INFO]
+[02/23 20:04:52][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 20:05:17][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_19/checkpoints'
+[02/23 20:05:27][INFO] Start Fitting...
+[02/23 20:05:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 20:05:29][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 20:05:30][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 20:05:31][INFO] [LossBreakdown] body_pose=0.9203 betas=0.6590 go_c=0.5448 go_gv=0.0725 transl_vel=0.2262 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.81/35.75 go_gv_deg(mean/max)=13.86/28.49 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.41
+[02/23 20:05:31][INFO] [IncamDebug] transl_c_loss=0.0421 pred_cam_abs_mean=0.886 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/23 20:05:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 20:05:32][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7464 go_c=0.1315 go_gv=0.0418 transl_vel=0.1971 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.42/28.02 go_gv_deg(mean/max)=10.52/27.78 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 20:05:32][INFO] [IncamDebug] transl_c_loss=0.1115 pred_cam_abs_mean=0.973 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/23 20:05:32][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 20:05:32][INFO] [IncamDebug] transl_c_loss=0.1979 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/23 20:05:32][INFO] [LossBreakdown] body_pose=0.9267 betas=0.6977 go_c=0.0836 go_gv=0.0196 transl_vel=0.1916 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/9.82 go_gv_deg(mean/max)=8.98/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 20:05:32][INFO] [IncamDebug] transl_c_loss=0.0029 pred_cam_abs_mean=0.903 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/23 20:05:40][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(346.3108825683594, 142.3296661376953, 845.6177978515625, 1665.8909912109375, 0.3571843206882477)
+[02/23 20:05:40][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(332.3451232910156, 141.380859375, 822.3335571289062, 1695.947509765625, 0.3587808310985565)
+[02/23 20:05:40][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.083 max=0.103 gt_z_mean=1.140 pred_z_mean=1.108
+[02/23 20:05:43][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.021 max=0.043
+[02/23 20:05:43][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9975 pred=+1.0002 delta(pred-gt)=+0.0027
+[02/23 20:05:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[0.02282428 1.9486594 0.06558114]
+[02/23 20:05:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.0300002 3.0145898 -0.04516966]
+[02/23 20:05:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+111.64,-2.02,+2.71) pred_vs_gt=(-9.31,+3.40,+2.39)
+[02/23 20:05:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+8.01
+[02/23 20:06:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:06:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:06:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:06:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:06:48][INFO] ✅[FIT][Epoch 0] finished! 01:20→1:05:38 | loss_epoch=1.83
+[02/23 20:06:48][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 20:06:48][INFO] [LossBreakdown] body_pose=0.4396 betas=0.3765 go_c=0.1585 go_gv=0.0191 transl_vel=0.1995 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.28/22.51 go_gv_deg(mean/max)=10.50/19.04 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.53 static_conf_hi_frac=0.36
+[02/23 20:06:48][INFO] [IncamDebug] transl_c_loss=0.0013 pred_cam_abs_mean=0.909 gt_pred_cam_abs_mean=0.936 valid_frac=1.00
+[02/23 20:06:48][INFO] [LossBreakdown] body_pose=1.1988 betas=0.4801 go_c=0.0968 go_gv=0.0235 transl_vel=0.1599 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.60/35.95 go_gv_deg(mean/max)=9.46/38.26 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.58 static_conf_hi_frac=0.44
+[02/23 20:06:48][INFO] [IncamDebug] transl_c_loss=0.0081 pred_cam_abs_mean=0.969 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/23 20:06:49][INFO] [LossBreakdown] body_pose=0.3774 betas=0.4978 go_c=0.0803 go_gv=0.0470 transl_vel=0.1821 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.34/22.51 go_gv_deg(mean/max)=12.57/24.20 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.40 static_conf_hi_frac=0.07
+[02/23 20:06:49][INFO] [IncamDebug] transl_c_loss=0.1115 pred_cam_abs_mean=0.930 gt_pred_cam_abs_mean=1.136 valid_frac=0.89
+[02/23 20:06:49][INFO] [LossBreakdown] body_pose=0.3566 betas=0.4171 go_c=0.3248 go_gv=0.0353 transl_vel=0.8402 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.14/116.63 go_gv_deg(mean/max)=8.78/97.49 tv_abs_mean(pred)=0.0017 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.43 static_conf_hi_frac=0.09
+[02/23 20:06:49][INFO] [IncamDebug] transl_c_loss=0.0539 pred_cam_abs_mean=0.919 gt_pred_cam_abs_mean=1.094 valid_frac=1.00
+[02/23 20:07:01][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof f0 gt=(549.347900390625, 253.296630859375, 737.7196655273438, 1149.4931640625, 0.3052249550819397) pred=(514.33935546875, 230.67835998535156, 698.5016479492188, 1156.628662109375, 0.28824383020401)
+[02/23 20:07:01][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof fl gt=(554.8383178710938, 266.72760009765625, 695.2646484375, 1172.6390380859375, 0.28505077958106995) pred=(505.1429443359375, 240.62225341796875, 677.645263671875, 1097.8228759765625, 0.274455726146698)
+[02/23 20:07:01][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_transl_err_m mean=0.084 max=0.104 gt_z_mean=1.421 pred_z_mean=1.437
+[02/23 20:07:03][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_transl_err_m mean=0.073 max=0.160
+[02/23 20:07:03][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9925 pred=+0.9878 delta(pred-gt)=-0.0047
+[02/23 20:07:03][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.00300199 -1.5857557 0.06613082] global_orient0_aa(pred)=[-0.06903021 -1.5818751 -0.04569769]
+[02/23 20:07:03][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(raw_gt_world)=[-0.05748826 -3.0477192 0.07022861]
+[02/23 20:07:03][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-90.84,+2.53,+2.28) pred=(-90.71,-4.17,+0.88) pred_vs_gt=(+0.02,+1.50,-6.69)
+[02/23 20:07:03][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-0.83
+[02/23 20:08:00][INFO] ✅[FIT][Epoch 1] finished! 02:32→1:00:52 | loss_epoch=1.26
+[02/23 20:08:00][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/23 20:08:00][INFO] [LossBreakdown] body_pose=1.8983 betas=0.3241 go_c=0.0665 go_gv=0.0134 transl_vel=0.1578 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=5.98/10.14 go_gv_deg(mean/max)=3.95/8.01 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.00 static_conf_mean=0.85 static_conf_hi_frac=0.73
+[02/23 20:08:00][INFO] [IncamDebug] transl_c_loss=0.0050 pred_cam_abs_mean=0.964 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/23 20:08:00][INFO] [LossBreakdown] body_pose=0.2910 betas=0.2544 go_c=0.2179 go_gv=0.0306 transl_vel=0.9065 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.44/166.57 go_gv_deg(mean/max)=6.84/172.40 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.66
+[02/23 20:08:00][INFO] [IncamDebug] transl_c_loss=0.0140 pred_cam_abs_mean=0.938 gt_pred_cam_abs_mean=1.047 valid_frac=0.87
+[02/23 20:08:01][INFO] [LossBreakdown] body_pose=0.2887 betas=0.2964 go_c=0.1442 go_gv=0.0609 transl_vel=0.3467 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=16.28/32.46 go_gv_deg(mean/max)=15.99/33.89 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0026 tv_zero_frac(pred)=0.00 static_conf_mean=0.81 static_conf_hi_frac=0.69
+[02/23 20:08:01][INFO] [IncamDebug] transl_c_loss=0.0312 pred_cam_abs_mean=1.062 gt_pred_cam_abs_mean=1.253 valid_frac=0.73
+[02/23 20:08:01][INFO] [LossBreakdown] body_pose=0.2462 betas=0.3329 go_c=0.1615 go_gv=0.0332 transl_vel=0.1704 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.52/21.31 go_gv_deg(mean/max)=10.22/19.04 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.62
+[02/23 20:08:01][INFO] [IncamDebug] transl_c_loss=0.0270 pred_cam_abs_mean=1.015 gt_pred_cam_abs_mean=1.112 valid_frac=1.00
+[02/23 20:08:26][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof f0 gt=(497.9825744628906, 218.85873413085938, 754.4590454101562, 1513.018798828125, 0.4923076927661896) pred=(390.26239013671875, 183.29501342773438, 681.729248046875, 1543.8746337890625, 0.39404934644699097)
+[02/23 20:08:26][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof fl gt=(531.361083984375, 220.1790771484375, 725.6470947265625, 1474.6353759765625, 0.49013060331344604) pred=(392.94744873046875, 205.8941192626953, 672.3618774414062, 1463.5728759765625, 0.355878084897995)
+[02/23 20:08:26][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_transl_err_m mean=0.134 max=0.210 gt_z_mean=1.193 pred_z_mean=1.268
+[02/23 20:08:28][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_transl_err_m mean=0.075 max=0.194
+[02/23 20:08:28][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 root_y0: gt=+0.9911 pred=+1.0021 delta(pred-gt)=+0.0110
+[02/23 20:08:28][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(gt)=[ 0.04780671 -1.4102775 0.10208653] global_orient0_aa(pred)=[-0.08581002 -1.3626306 -0.05114591]
+[02/23 20:08:28][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(raw_gt_world)=[-0.05267622 -2.9777708 0.14565562]
+[02/23 20:08:28][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_yxz_deg gt=(-80.76,+5.40,+2.47) pred=(-78.17,-5.24,+0.76) pred_vs_gt=(+2.60,-0.02,-10.77)
+[02/23 20:08:28][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 yaw0_deg(pred_vs_gt)=-6.96
+[02/23 20:12:20][INFO] [Exp Name]: finetune_
+[02/23 20:12:20][INFO] [GPU x Batch] = 1 x 2
+[02/23 20:12:20][INFO] [UnityDataset] Found 9 sequences.
+[02/23 20:12:20][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 20:12:20][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 20:12:20][INFO]
+[02/23 20:12:20][INFO] [UnityDataset] Found 9 sequences.
+[02/23 20:12:20][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 20:12:20][INFO]
+[02/23 20:12:29][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 20:12:51][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_20/checkpoints'
+[02/23 20:13:03][INFO] Start Fitting...
+[02/23 20:13:05][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 20:13:05][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 20:13:06][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 20:13:07][INFO] [LossBreakdown] body_pose=0.9203 betas=0.6590 go_c=0.5448 go_gv=0.0725 transl_vel=0.2262 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.81/35.75 go_gv_deg(mean/max)=13.86/28.49 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.41
+[02/23 20:13:07][INFO] [IncamDebug] transl_c_loss=0.0421 pred_cam_abs_mean=0.886 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/23 20:13:07][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 20:13:07][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7464 go_c=0.1315 go_gv=0.0418 transl_vel=0.1971 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.42/28.02 go_gv_deg(mean/max)=10.52/27.78 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 20:13:07][INFO] [IncamDebug] transl_c_loss=0.1115 pred_cam_abs_mean=0.973 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/23 20:13:08][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 20:13:08][INFO] [IncamDebug] transl_c_loss=0.1979 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/23 20:13:08][INFO] [LossBreakdown] body_pose=0.9267 betas=0.6977 go_c=0.0836 go_gv=0.0196 transl_vel=0.1916 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/9.82 go_gv_deg(mean/max)=8.98/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 20:13:08][INFO] [IncamDebug] transl_c_loss=0.0029 pred_cam_abs_mean=0.903 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/23 20:13:16][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(346.3108825683594, 142.3296661376953, 845.6177978515625, 1665.8909912109375, 0.3571843206882477)
+[02/23 20:13:16][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(332.3451232910156, 141.380859375, 822.3335571289062, 1695.947509765625, 0.3587808310985565)
+[02/23 20:13:16][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.083 max=0.103 gt_z_mean=1.140 pred_z_mean=1.108
+[02/23 20:13:20][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.021 max=0.043
+[02/23 20:13:20][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9975 pred=+1.0002 delta(pred-gt)=+0.0027
+[02/23 20:13:20][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[0.02282428 1.9486594 0.06558114]
+[02/23 20:13:20][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.0300002 3.0145898 -0.04516966]
+[02/23 20:13:20][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+111.64,-2.02,+2.71) pred_vs_gt=(-9.31,+3.40,+2.39)
+[02/23 20:13:20][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+8.01
+[02/23 20:18:31][INFO] [Exp Name]: finetune_
+[02/23 20:18:31][INFO] [GPU x Batch] = 1 x 2
+[02/23 20:18:32][INFO] [UnityDataset] Found 9 sequences.
+[02/23 20:18:32][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 20:18:32][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/23 20:18:32][INFO]
+[02/23 20:18:32][INFO] [UnityDataset] Found 9 sequences.
+[02/23 20:18:32][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/23 20:18:32][INFO]
+[02/23 20:18:38][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/23 20:19:02][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_21/checkpoints'
+[02/23 20:19:12][INFO] Start Fitting...
+[02/23 20:19:14][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/23 20:19:14][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/23 20:19:18][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/23 20:19:18][INFO] [LossBreakdown] body_pose=0.9203 betas=0.6590 go_c=0.5448 go_gv=0.0725 transl_vel=0.2262 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.81/35.75 go_gv_deg(mean/max)=13.86/28.49 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.41
+[02/23 20:19:18][INFO] [IncamDebug] transl_c_loss=0.0421 pred_cam_abs_mean=0.886 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/23 20:19:19][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/23 20:19:19][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7464 go_c=0.1315 go_gv=0.0418 transl_vel=0.1971 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.42/28.02 go_gv_deg(mean/max)=10.52/27.78 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/23 20:19:19][INFO] [IncamDebug] transl_c_loss=0.1115 pred_cam_abs_mean=0.973 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/23 20:19:20][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/23 20:19:20][INFO] [IncamDebug] transl_c_loss=0.1979 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/23 20:19:20][INFO] [LossBreakdown] body_pose=0.9267 betas=0.6977 go_c=0.0836 go_gv=0.0196 transl_vel=0.1916 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/9.82 go_gv_deg(mean/max)=8.98/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/23 20:19:20][INFO] [IncamDebug] transl_c_loss=0.0029 pred_cam_abs_mean=0.903 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/23 20:19:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(346.3108825683594, 142.3296661376953, 845.6177978515625, 1665.8909912109375, 0.3571843206882477)
+[02/23 20:19:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(332.3451232910156, 141.380859375, 822.3335571289062, 1695.947509765625, 0.3587808310985565)
+[02/23 20:19:28][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.083 max=0.103 gt_z_mean=1.140 pred_z_mean=1.108
+[02/23 20:19:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.021 max=0.043
+[02/23 20:19:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9975 pred=+1.0002 delta(pred-gt)=+0.0027
+[02/23 20:19:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[0.02282428 1.9486594 0.06558114]
+[02/23 20:19:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.0300002 3.0145898 -0.04516966]
+[02/23 20:19:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+111.64,-2.02,+2.71) pred_vs_gt=(-9.31,+3.40,+2.39)
+[02/23 20:19:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+8.01
+[02/23 20:20:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:20:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:20:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:20:39][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/23 20:20:39][INFO] ✅[FIT][Epoch 0] finished! 01:26→1:10:40 | loss_epoch=1.83
+[02/23 20:20:39][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/23 20:20:39][INFO] [LossBreakdown] body_pose=0.4396 betas=0.3765 go_c=0.1585 go_gv=0.0191 transl_vel=0.1995 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.28/22.51 go_gv_deg(mean/max)=10.50/19.04 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.53 static_conf_hi_frac=0.36
+[02/23 20:20:39][INFO] [IncamDebug] transl_c_loss=0.0013 pred_cam_abs_mean=0.909 gt_pred_cam_abs_mean=0.936 valid_frac=1.00
+[02/23 20:20:40][INFO] [LossBreakdown] body_pose=1.1988 betas=0.4801 go_c=0.0968 go_gv=0.0235 transl_vel=0.1599 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.60/35.95 go_gv_deg(mean/max)=9.46/38.26 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.58 static_conf_hi_frac=0.44
+[02/23 20:20:40][INFO] [IncamDebug] transl_c_loss=0.0081 pred_cam_abs_mean=0.969 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/23 20:20:40][INFO] [LossBreakdown] body_pose=0.3774 betas=0.4978 go_c=0.0803 go_gv=0.0470 transl_vel=0.1821 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.34/22.51 go_gv_deg(mean/max)=12.57/24.20 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.40 static_conf_hi_frac=0.07
+[02/23 20:20:40][INFO] [IncamDebug] transl_c_loss=0.1115 pred_cam_abs_mean=0.930 gt_pred_cam_abs_mean=1.136 valid_frac=0.89
+[02/23 20:20:40][INFO] [LossBreakdown] body_pose=0.3566 betas=0.4171 go_c=0.3248 go_gv=0.0353 transl_vel=0.8402 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.14/116.63 go_gv_deg(mean/max)=8.78/97.49 tv_abs_mean(pred)=0.0017 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.43 static_conf_hi_frac=0.09
+[02/23 20:20:40][INFO] [IncamDebug] transl_c_loss=0.0539 pred_cam_abs_mean=0.919 gt_pred_cam_abs_mean=1.094 valid_frac=1.00
+[02/23 20:20:50][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof f0 gt=(549.347900390625, 253.296630859375, 737.7196655273438, 1149.4931640625, 0.3052249550819397) pred=(514.33935546875, 230.67835998535156, 698.5016479492188, 1156.628662109375, 0.28824383020401)
+[02/23 20:20:50][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof fl gt=(554.8383178710938, 266.72760009765625, 695.2646484375, 1172.6390380859375, 0.28505077958106995) pred=(505.1429443359375, 240.62225341796875, 677.645263671875, 1097.8228759765625, 0.274455726146698)
+[02/23 20:20:50][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_transl_err_m mean=0.084 max=0.104 gt_z_mean=1.421 pred_z_mean=1.437
+[02/23 20:20:53][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_transl_err_m mean=0.073 max=0.160
+[02/23 20:20:53][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9925 pred=+0.9878 delta(pred-gt)=-0.0047
+[02/23 20:20:53][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.00300199 -1.5857557 0.06613082] global_orient0_aa(pred)=[-0.06903021 -1.5818751 -0.04569769]
+[02/23 20:20:53][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(raw_gt_world)=[-0.05748826 -3.0477192 0.07022861]
+[02/23 20:20:53][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-90.84,+2.53,+2.28) pred=(-90.71,-4.17,+0.88) pred_vs_gt=(+0.02,+1.50,-6.69)
+[02/23 20:20:53][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-0.83
+[02/23 20:21:54][INFO] ✅[FIT][Epoch 1] finished! 02:41→1:04:25 | loss_epoch=1.26
+[02/23 20:21:54][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/23 20:21:54][INFO] [LossBreakdown] body_pose=1.8983 betas=0.3241 go_c=0.0665 go_gv=0.0134 transl_vel=0.1578 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=5.98/10.14 go_gv_deg(mean/max)=3.95/8.01 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.00 static_conf_mean=0.85 static_conf_hi_frac=0.73
+[02/23 20:21:54][INFO] [IncamDebug] transl_c_loss=0.0050 pred_cam_abs_mean=0.964 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/23 20:21:54][INFO] [LossBreakdown] body_pose=0.2910 betas=0.2544 go_c=0.2179 go_gv=0.0306 transl_vel=0.9065 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.44/166.57 go_gv_deg(mean/max)=6.84/172.40 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.66
+[02/23 20:21:54][INFO] [IncamDebug] transl_c_loss=0.0140 pred_cam_abs_mean=0.938 gt_pred_cam_abs_mean=1.047 valid_frac=0.87
+[02/23 20:21:55][INFO] [LossBreakdown] body_pose=0.2887 betas=0.2964 go_c=0.1442 go_gv=0.0609 transl_vel=0.3467 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=16.28/32.46 go_gv_deg(mean/max)=15.99/33.89 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0026 tv_zero_frac(pred)=0.00 static_conf_mean=0.81 static_conf_hi_frac=0.69
+[02/23 20:21:55][INFO] [IncamDebug] transl_c_loss=0.0312 pred_cam_abs_mean=1.062 gt_pred_cam_abs_mean=1.253 valid_frac=0.73
+[02/23 20:21:55][INFO] [LossBreakdown] body_pose=0.2462 betas=0.3329 go_c=0.1615 go_gv=0.0332 transl_vel=0.1704 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.52/21.31 go_gv_deg(mean/max)=10.22/19.04 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.62
+[02/23 20:21:55][INFO] [IncamDebug] transl_c_loss=0.0270 pred_cam_abs_mean=1.015 gt_pred_cam_abs_mean=1.112 valid_frac=1.00
+[02/23 20:22:20][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof f0 gt=(497.9825744628906, 218.85873413085938, 754.4590454101562, 1513.018798828125, 0.4923076927661896) pred=(390.26239013671875, 183.29501342773438, 681.729248046875, 1543.8746337890625, 0.39404934644699097)
+[02/23 20:22:20][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof fl gt=(531.361083984375, 220.1790771484375, 725.6470947265625, 1474.6353759765625, 0.49013060331344604) pred=(392.94744873046875, 205.8941192626953, 672.3618774414062, 1463.5728759765625, 0.355878084897995)
+[02/23 20:22:20][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_transl_err_m mean=0.134 max=0.210 gt_z_mean=1.193 pred_z_mean=1.268
+[02/23 20:22:23][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_transl_err_m mean=0.075 max=0.194
+[02/23 20:22:23][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 root_y0: gt=+0.9911 pred=+1.0021 delta(pred-gt)=+0.0110
+[02/23 20:22:23][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(gt)=[ 0.04780671 -1.4102775 0.10208653] global_orient0_aa(pred)=[-0.08581002 -1.3626306 -0.05114591]
+[02/23 20:22:23][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(raw_gt_world)=[-0.05267622 -2.9777708 0.14565562]
+[02/23 20:22:23][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_yxz_deg gt=(-80.76,+5.40,+2.47) pred=(-78.17,-5.24,+0.76) pred_vs_gt=(+2.60,-0.02,-10.77)
+[02/23 20:22:23][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 yaw0_deg(pred_vs_gt)=-6.96
+[02/23 20:23:07][INFO] ✅[FIT][Epoch 2] finished! 03:54→1:01:14 | loss_epoch=1.52
+[02/23 20:23:07][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[02/23 20:23:07][INFO] [LossBreakdown] body_pose=0.1849 betas=0.2766 go_c=0.0975 go_gv=0.0724 transl_vel=0.2140 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=15.63/33.99 go_gv_deg(mean/max)=17.87/35.56 tv_abs_mean(pred)=0.0013 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.55 static_conf_hi_frac=0.34
+[02/23 20:23:07][INFO] [IncamDebug] transl_c_loss=0.0261 pred_cam_abs_mean=1.269 gt_pred_cam_abs_mean=1.320 valid_frac=0.91
+[02/23 20:23:08][INFO] [LossBreakdown] body_pose=0.8323 betas=0.3311 go_c=0.0965 go_gv=0.0306 transl_vel=0.1464 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.27/37.17 go_gv_deg(mean/max)=10.91/36.98 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0014 tv_zero_frac(pred)=0.00 static_conf_mean=0.73 static_conf_hi_frac=0.53
+[02/23 20:23:08][INFO] [IncamDebug] transl_c_loss=0.0327 pred_cam_abs_mean=1.031 gt_pred_cam_abs_mean=0.977 valid_frac=1.00
+[02/23 20:23:08][INFO] [LossBreakdown] body_pose=0.1983 betas=0.1451 go_c=0.1150 go_gv=0.0332 transl_vel=0.2349 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=7.77/12.94 go_gv_deg(mean/max)=8.21/14.13 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.67 static_conf_hi_frac=0.60
+[02/23 20:23:08][INFO] [IncamDebug] transl_c_loss=0.0083 pred_cam_abs_mean=1.004 gt_pred_cam_abs_mean=1.009 valid_frac=1.00
+[02/23 20:23:08][INFO] [LossBreakdown] body_pose=0.2181 betas=0.1625 go_c=0.1645 go_gv=0.0199 transl_vel=0.1818 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.05/19.04 go_gv_deg(mean/max)=7.60/15.45 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0019 tv_zero_frac(pred)=0.00 static_conf_mean=0.62 static_conf_hi_frac=0.49
+[02/23 20:23:08][INFO] [IncamDebug] transl_c_loss=0.0142 pred_cam_abs_mean=0.966 gt_pred_cam_abs_mean=0.949 valid_frac=1.00
+[02/23 20:23:13][INFO] [VisUnityVal] e003_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(329.3016052246094, 216.16949462890625, 752.96728515625, 1738.2373046875, 0.39477503299713135) pred=(238.99203491210938, 126.36845397949219, 776.3062133789062, 1482.35693359375, 0.34542813897132874)
+[02/23 20:23:13][INFO] [VisUnityVal] e003_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(369.14556884765625, 216.22442626953125, 795.2476806640625, 1742.6558837890625, 0.2528301775455475) pred=(296.6162414550781, 142.52706909179688, 819.9044799804688, 1567.852783203125, 0.23352684080600739)
+[02/23 20:23:13][INFO] [VisUnityVal] e003_0_biboo_birthday_speech incam_transl_err_m mean=0.094 max=0.112 gt_z_mean=1.144 pred_z_mean=1.152
+[02/23 20:23:17][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_transl_err_m mean=0.019 max=0.040
+[02/23 20:23:17][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+0.9996 pred=+0.9879 delta(pred-gt)=-0.0117
+[02/23 20:23:17][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[0.02579728 2.1115918 0.0080679 ] global_orient0_aa(pred)=[0.06060402 2.151457 0.05912083]
+[02/23 20:23:17][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.03318783 3.0026174 -0.00475879]
+[02/23 20:23:17][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.99,+0.27,+1.25) pred=(+123.29,-1.09,+3.82) pred_vs_gt=(+2.31,+2.90,-0.16)
+[02/23 20:23:17][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-7.33
+[02/23 20:24:20][INFO] ✅[FIT][Epoch 3] finished! 05:07→58:50 | loss_epoch=0.846
+[02/23 20:24:20][INFO] 🚀[FIT][Epoch 4] Data: unity Experiment: finetune_
+[02/23 20:24:20][INFO] [LossBreakdown] body_pose=0.1986 betas=0.1144 go_c=0.3467 go_gv=0.0640 transl_vel=0.9023 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.99/74.80 go_gv_deg(mean/max)=13.92/70.67 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.57 static_conf_hi_frac=0.32
+[02/23 20:24:20][INFO] [IncamDebug] transl_c_loss=0.0101 pred_cam_abs_mean=0.968 gt_pred_cam_abs_mean=1.048 valid_frac=0.78
+[02/23 20:24:20][INFO] [LossBreakdown] body_pose=1.5346 betas=0.1945 go_c=0.0925 go_gv=0.0235 transl_vel=0.1152 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=7.42/12.94 go_gv_deg(mean/max)=8.96/13.90 tv_abs_mean(pred)=0.0003 tv_abs_mean(tgt)=0.0015 tv_zero_frac(pred)=0.00 static_conf_mean=0.73 static_conf_hi_frac=0.65
+[02/23 20:24:20][INFO] [IncamDebug] transl_c_loss=0.0063 pred_cam_abs_mean=0.966 gt_pred_cam_abs_mean=0.975 valid_frac=1.00
+[02/23 20:24:20][INFO] [LossBreakdown] body_pose=0.1591 betas=0.0721 go_c=0.1163 go_gv=0.0351 transl_vel=0.2384 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.80/17.43 go_gv_deg(mean/max)=11.53/17.43 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.61 static_conf_hi_frac=0.46
+[02/23 20:24:20][INFO] [IncamDebug] transl_c_loss=0.0046 pred_cam_abs_mean=0.977 gt_pred_cam_abs_mean=1.013 valid_frac=1.00
+[02/23 20:24:21][INFO] [LossBreakdown] body_pose=0.1667 betas=0.1290 go_c=0.1659 go_gv=0.0210 transl_vel=0.1361 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.40/18.34 go_gv_deg(mean/max)=8.53/15.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0019 tv_zero_frac(pred)=0.00 static_conf_mean=0.61 static_conf_hi_frac=0.43
+[02/23 20:24:21][INFO] [IncamDebug] transl_c_loss=0.0128 pred_cam_abs_mean=1.076 gt_pred_cam_abs_mean=1.092 valid_frac=1.00
+[02/23 20:24:37][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 incam_proj_bbox_oof f0 gt=(568.93359375, 268.9086608886719, 791.1539916992188, 1237.05029296875, 0.32917270064353943) pred=(595.958984375, 279.16021728515625, 865.5813598632812, 1044.164306640625, 0.16255442798137665)
+[02/23 20:24:37][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 incam_proj_bbox_oof fl gt=(541.4463500976562, 271.5432434082031, 756.655517578125, 1234.872802734375, 0.3378809690475464) pred=(579.2265625, 272.05352783203125, 770.5712890625, 1061.3896484375, 0.16748911142349243)
+[02/23 20:24:37][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 incam_transl_err_m mean=0.174 max=0.239 gt_z_mean=1.743 pred_z_mean=1.915
+[02/23 20:24:40][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 global_transl_err_m mean=0.068 max=0.132
+[02/23 20:24:40][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 root_y0: gt=+0.9958 pred=+0.9842 delta(pred-gt)=-0.0116
+[02/23 20:24:40][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 global_orient0_aa(gt)=[ 0.16767852 2.2633636 -0.10367614] global_orient0_aa(pred)=[0.10294407 2.5176198 0.02350744]
+[02/23 20:24:40][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 global_orient0_aa(raw_gt_world)=[0.14257036 0.26876903 0.0681826 ]
+[02/23 20:24:40][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 global_orient0_yxz_deg gt=(+130.33,+7.54,+4.98) pred=(+144.36,+0.40,+4.56) pred_vs_gt=(+14.27,+4.29,+5.73)
+[02/23 20:24:40][INFO] [VisUnityVal] e004_103_biboo_birthday_speech_explosion_4 yaw0_deg(pred_vs_gt)=-18.58
+[02/23 20:25:32][INFO] ✅[FIT][Epoch 4] finished! 06:19→56:55 | loss_epoch=0.909
+[02/26 14:23:35][INFO] [Exp Name]: finetune_
+[02/26 14:23:35][INFO] [GPU x Batch] = 1 x 2
+[02/26 14:23:35][INFO] [UnityDataset] Found 9 sequences.
+[02/26 14:23:35][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 14:23:35][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/26 14:23:35][INFO]
+[02/26 14:23:35][INFO] [UnityDataset] Found 9 sequences.
+[02/26 14:23:35][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 14:23:35][INFO]
+[02/26 14:23:43][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/26 14:24:08][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[02/26 14:24:19][INFO] Start Fitting...
+[02/26 14:24:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/26 14:24:20][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/26 14:24:22][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/26 14:24:23][INFO] [LossBreakdown] body_pose=0.7345 betas=0.7643 go_c=0.6235 go_gv=0.0695 transl_vel=0.2282 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=19.89/41.24 go_gv_deg(mean/max)=16.08/32.88 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.65 static_conf_hi_frac=0.36
+[02/26 14:24:23][INFO] [IncamDebug] transl_c_loss=0.0653 pred_cam_abs_mean=0.863 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/26 14:24:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/26 14:24:24][INFO] [LossBreakdown] body_pose=2.1123 betas=0.6598 go_c=0.1534 go_gv=0.0317 transl_vel=0.1941 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.14/20.20 go_gv_deg(mean/max)=8.61/17.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.00 static_conf_mean=0.80 static_conf_hi_frac=0.58
+[02/26 14:24:24][INFO] [IncamDebug] transl_c_loss=0.1939 pred_cam_abs_mean=0.916 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/26 14:24:24][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/26 14:24:24][INFO] [IncamDebug] transl_c_loss=0.1979 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/26 14:24:25][INFO] [LossBreakdown] body_pose=0.9147 betas=0.6379 go_c=0.1056 go_gv=0.0212 transl_vel=0.1904 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=7.38/11.34 go_gv_deg(mean/max)=8.69/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.80 static_conf_hi_frac=0.67
+[02/26 14:24:25][INFO] [IncamDebug] transl_c_loss=0.0096 pred_cam_abs_mean=0.862 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/26 14:24:34][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(356.6566467285156, 143.73374938964844, 812.1099853515625, 1705.347900390625, 0.36777937412261963)
+[02/26 14:24:34][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(341.3667297363281, 144.05609130859375, 853.4410400390625, 1738.817138671875, 0.36574745178222656)
+[02/26 14:24:34][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.090 max=0.116 gt_z_mean=1.140 pred_z_mean=1.079
+[02/26 14:24:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.026 max=0.054
+[02/26 14:24:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9975 pred=+1.0038 delta(pred-gt)=+0.0063
+[02/26 14:24:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[0.01091897 2.0095408 0.06201202]
+[02/26 14:24:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.0300002 3.0145898 -0.04516966]
+[02/26 14:24:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+115.13,-2.24,+2.04) pred_vs_gt=(-5.82,+2.94,+2.92)
+[02/26 14:24:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+6.59
+[02/26 14:25:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:25:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:25:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:25:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:25:56][INFO] ✅[FIT][Epoch 0] finished! 01:37→1:19:22 | loss_epoch=1.88
+[02/26 14:25:56][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/26 14:25:57][INFO] [LossBreakdown] body_pose=0.4243 betas=0.3046 go_c=0.1867 go_gv=0.0185 transl_vel=0.2130 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.70/24.06 go_gv_deg(mean/max)=9.78/19.38 tv_abs_mean(pred)=0.0015 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.38 static_conf_hi_frac=0.13
+[02/26 14:25:57][INFO] [IncamDebug] transl_c_loss=0.0012 pred_cam_abs_mean=0.913 gt_pred_cam_abs_mean=0.936 valid_frac=1.00
+[02/26 14:25:57][INFO] [LossBreakdown] body_pose=1.1618 betas=0.4365 go_c=0.0953 go_gv=0.0273 transl_vel=0.1948 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.21/37.07 go_gv_deg(mean/max)=10.69/39.60 tv_abs_mean(pred)=0.0020 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.49 static_conf_hi_frac=0.27
+[02/26 14:25:57][INFO] [IncamDebug] transl_c_loss=0.0107 pred_cam_abs_mean=0.968 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/26 14:25:57][INFO] [LossBreakdown] body_pose=0.4318 betas=0.4826 go_c=0.0557 go_gv=0.0262 transl_vel=0.2305 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.58/18.69 go_gv_deg(mean/max)=10.67/21.15 tv_abs_mean(pred)=0.0019 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.29 static_conf_hi_frac=0.03
+[02/26 14:25:57][INFO] [IncamDebug] transl_c_loss=0.1869 pred_cam_abs_mean=0.885 gt_pred_cam_abs_mean=1.136 valid_frac=0.89
+[02/26 14:25:58][INFO] [LossBreakdown] body_pose=0.3952 betas=0.3756 go_c=0.3307 go_gv=0.0362 transl_vel=0.9241 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.09/115.32 go_gv_deg(mean/max)=10.68/111.38 tv_abs_mean(pred)=0.0032 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.37 static_conf_hi_frac=0.03
+[02/26 14:25:58][INFO] [IncamDebug] transl_c_loss=0.0389 pred_cam_abs_mean=0.940 gt_pred_cam_abs_mean=1.094 valid_frac=1.00
+[02/26 14:26:08][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof f0 gt=(549.347900390625, 253.296630859375, 737.7196655273438, 1149.4931640625, 0.3052249550819397) pred=(507.9010925292969, 233.65536499023438, 696.8511352539062, 1160.6907958984375, 0.2872278690338135)
+[02/26 14:26:08][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof fl gt=(554.8383178710938, 266.72760009765625, 695.2646484375, 1172.6390380859375, 0.28505077958106995) pred=(505.6090393066406, 243.2876434326172, 676.9757690429688, 1100.3421630859375, 0.2721335291862488)
+[02/26 14:26:08][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_transl_err_m mean=0.084 max=0.105 gt_z_mean=1.421 pred_z_mean=1.442
+[02/26 14:26:11][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_transl_err_m mean=0.076 max=0.162
+[02/26 14:26:11][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9925 pred=+0.9849 delta(pred-gt)=-0.0076
+[02/26 14:26:11][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.00300199 -1.5857557 0.06613082] global_orient0_aa(pred)=[-0.10006028 -1.5663521 -0.08050752]
+[02/26 14:26:11][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(raw_gt_world)=[-0.05748826 -3.0477192 0.07022861]
+[02/26 14:26:11][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-90.84,+2.53,+2.28) pred=(-89.90,-6.59,+0.71) pred_vs_gt=(+0.75,+1.69,-9.10)
+[02/26 14:26:11][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-1.32
+[02/26 14:27:19][INFO] ✅[FIT][Epoch 1] finished! 02:59→1:11:56 | loss_epoch=1.44
+[02/26 14:27:19][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/26 14:27:19][INFO] [LossBreakdown] body_pose=1.9301 betas=0.2283 go_c=0.0636 go_gv=0.0201 transl_vel=0.1219 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.62/17.61 go_gv_deg(mean/max)=9.03/19.38 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.00 static_conf_mean=0.70 static_conf_hi_frac=0.58
+[02/26 14:27:19][INFO] [IncamDebug] transl_c_loss=0.0046 pred_cam_abs_mean=0.976 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/26 14:27:20][INFO] [LossBreakdown] body_pose=0.2863 betas=0.2651 go_c=0.2135 go_gv=0.0220 transl_vel=0.9153 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.54/135.71 go_gv_deg(mean/max)=7.12/142.19 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.62 static_conf_hi_frac=0.41
+[02/26 14:27:20][INFO] [IncamDebug] transl_c_loss=0.0097 pred_cam_abs_mean=0.949 gt_pred_cam_abs_mean=1.047 valid_frac=0.87
+[02/26 14:27:20][INFO] [LossBreakdown] body_pose=0.2718 betas=0.3369 go_c=0.1774 go_gv=0.0563 transl_vel=0.3241 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.46/32.57 go_gv_deg(mean/max)=16.42/33.08 tv_abs_mean(pred)=0.0012 tv_abs_mean(tgt)=0.0026 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.55
+[02/26 14:27:20][INFO] [IncamDebug] transl_c_loss=0.0404 pred_cam_abs_mean=1.045 gt_pred_cam_abs_mean=1.253 valid_frac=0.73
+[02/26 14:27:20][INFO] [LossBreakdown] body_pose=0.2480 betas=0.3175 go_c=0.1572 go_gv=0.0227 transl_vel=0.1587 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.95/19.04 go_gv_deg(mean/max)=7.94/14.81 tv_abs_mean(pred)=0.0016 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.65 static_conf_hi_frac=0.53
+[02/26 14:27:20][INFO] [IncamDebug] transl_c_loss=0.0308 pred_cam_abs_mean=0.988 gt_pred_cam_abs_mean=1.112 valid_frac=1.00
+[02/26 14:27:44][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof f0 gt=(497.9825744628906, 218.85873413085938, 754.4590454101562, 1513.018798828125, 0.4923076927661896) pred=(374.72998046875, 177.8643798828125, 681.2190551757812, 1631.7607421875, 0.4441218972206116)
+[02/26 14:27:44][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof fl gt=(531.361083984375, 220.1790771484375, 725.6470947265625, 1474.6353759765625, 0.49013060331344604) pred=(376.5287780761719, 201.9778594970703, 672.5849609375, 1549.49609375, 0.4211901128292084)
+[02/26 14:27:44][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_transl_err_m mean=0.114 max=0.162 gt_z_mean=1.193 pred_z_mean=1.235
+[02/26 14:27:48][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_transl_err_m mean=0.076 max=0.197
+[02/26 14:27:48][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 root_y0: gt=+0.9911 pred=+1.0090 delta(pred-gt)=+0.0179
+[02/26 14:27:48][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(gt)=[ 0.04780671 -1.4102775 0.10208653] global_orient0_aa(pred)=[-0.09083814 -1.3317943 -0.06571656]
+[02/26 14:27:48][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(raw_gt_world)=[-0.05267622 -2.9777708 0.14565562]
+[02/26 14:27:48][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_yxz_deg gt=(-80.76,+5.40,+2.47) pred=(-76.39,-5.95,+0.24) pred_vs_gt=(+4.32,+0.37,-11.56)
+[02/26 14:27:48][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 yaw0_deg(pred_vs_gt)=-8.04
+[02/26 14:28:53][INFO] ✅[FIT][Epoch 2] finished! 04:33→1:11:29 | loss_epoch=1.17
+[02/26 14:28:53][INFO] 🚀[FIT][Epoch 3] Data: unity Experiment: finetune_
+[02/26 14:28:53][INFO] [LossBreakdown] body_pose=0.1576 betas=0.2913 go_c=0.0967 go_gv=0.0915 transl_vel=0.1458 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=21.96/41.32 go_gv_deg(mean/max)=22.53/42.50 tv_abs_mean(pred)=0.0012 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.62 static_conf_hi_frac=0.42
+[02/26 14:28:53][INFO] [IncamDebug] transl_c_loss=0.0326 pred_cam_abs_mean=1.308 gt_pred_cam_abs_mean=1.320 valid_frac=0.91
+[02/26 14:28:53][INFO] [LossBreakdown] body_pose=0.7450 betas=0.2118 go_c=0.0792 go_gv=0.0305 transl_vel=0.1452 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.61/35.85 go_gv_deg(mean/max)=10.69/34.88 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0014 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.63
+[02/26 14:28:54][INFO] [IncamDebug] transl_c_loss=0.0753 pred_cam_abs_mean=1.080 gt_pred_cam_abs_mean=0.977 valid_frac=1.00
+[02/26 14:28:54][INFO] [LossBreakdown] body_pose=0.1748 betas=0.0730 go_c=0.1238 go_gv=0.0332 transl_vel=0.1761 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.66/17.05 go_gv_deg(mean/max)=9.76/16.47 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.67 static_conf_hi_frac=0.49
+[02/26 14:28:54][INFO] [IncamDebug] transl_c_loss=0.0082 pred_cam_abs_mean=0.935 gt_pred_cam_abs_mean=1.009 valid_frac=1.00
+[02/26 14:28:54][INFO] [LossBreakdown] body_pose=0.1766 betas=0.1166 go_c=0.1571 go_gv=0.0195 transl_vel=0.1757 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.32/19.71 go_gv_deg(mean/max)=8.51/18.34 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0019 tv_zero_frac(pred)=0.00 static_conf_mean=0.68 static_conf_hi_frac=0.53
+[02/26 14:28:54][INFO] [IncamDebug] transl_c_loss=0.0203 pred_cam_abs_mean=0.983 gt_pred_cam_abs_mean=0.949 valid_frac=1.00
+[02/26 14:29:00][INFO] [VisUnityVal] e003_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(329.3016052246094, 216.16949462890625, 752.96728515625, 1738.2373046875, 0.39477503299713135) pred=(292.630615234375, 136.5429229736328, 775.552978515625, 1575.6796875, 0.3638606667518616)
+[02/26 14:29:00][INFO] [VisUnityVal] e003_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(369.14556884765625, 216.22442626953125, 795.2476806640625, 1742.6558837890625, 0.2528301775455475) pred=(408.1445007324219, 150.30950927734375, 799.5399780273438, 1667.6234130859375, 0.24833090603351593)
+[02/26 14:29:00][INFO] [VisUnityVal] e003_0_biboo_birthday_speech incam_transl_err_m mean=0.095 max=0.107 gt_z_mean=1.144 pred_z_mean=1.085
+[02/26 14:29:04][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_transl_err_m mean=0.013 max=0.026
+[02/26 14:29:04][INFO] [VisUnityVal] e003_0_biboo_birthday_speech root_y0: gt=+0.9996 pred=+0.9883 delta(pred-gt)=-0.0113
+[02/26 14:29:04][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(gt)=[0.02579728 2.1115918 0.0080679 ] global_orient0_aa(pred)=[0.06705268 2.103836 0.02988167]
+[02/26 14:29:04][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.03318783 3.0026174 -0.00475879]
+[02/26 14:29:04][INFO] [VisUnityVal] e003_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.99,+0.27,+1.25) pred=(+120.59,+0.34,+3.45) pred_vs_gt=(-0.43,+1.85,-1.20)
+[02/26 14:29:04][INFO] [VisUnityVal] e003_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-4.12
+[02/26 14:44:46][INFO] [Exp Name]: finetune_
+[02/26 14:44:46][INFO] [GPU x Batch] = 1 x 2
+[02/26 14:44:46][INFO] [UnityDataset] Found 9 sequences.
+[02/26 14:44:46][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 14:44:46][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/26 14:44:46][INFO]
+[02/26 14:44:46][INFO] [UnityDataset] Found 9 sequences.
+[02/26 14:44:46][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 14:44:46][INFO]
+[02/26 14:44:53][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/26 14:45:18][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_1/checkpoints'
+[02/26 14:45:29][INFO] Start Fitting...
+[02/26 14:45:30][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/26 14:45:31][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/26 14:45:32][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/26 14:45:33][INFO] [LossBreakdown] body_pose=0.7345 betas=0.7643 go_c=0.6235 go_gv=0.0695 transl_vel=0.2282 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=19.89/41.24 go_gv_deg(mean/max)=16.08/32.88 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.65 static_conf_hi_frac=0.36
+[02/26 14:45:33][INFO] [IncamDebug] transl_c_loss=0.0653 pred_cam_abs_mean=0.863 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/26 14:45:34][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/26 14:45:34][INFO] [LossBreakdown] body_pose=2.1123 betas=0.6598 go_c=0.1534 go_gv=0.0317 transl_vel=0.1941 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.14/20.20 go_gv_deg(mean/max)=8.61/17.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.00 static_conf_mean=0.80 static_conf_hi_frac=0.58
+[02/26 14:45:34][INFO] [IncamDebug] transl_c_loss=0.1939 pred_cam_abs_mean=0.916 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/26 14:45:34][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/26 14:45:34][INFO] [IncamDebug] transl_c_loss=0.1979 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/26 14:45:35][INFO] [LossBreakdown] body_pose=0.8671 betas=0.6704 go_c=0.1078 go_gv=0.0212 transl_vel=0.1907 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=7.57/11.34 go_gv_deg(mean/max)=8.67/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.01 static_conf_mean=0.80 static_conf_hi_frac=0.67
+[02/26 14:45:35][INFO] [IncamDebug] transl_c_loss=0.0103 pred_cam_abs_mean=0.861 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/26 14:45:43][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(345.0684509277344, 142.83265686035156, 815.19189453125, 1705.7205810546875, 0.3650217652320862)
+[02/26 14:45:43][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(328.91705322265625, 143.19557189941406, 866.721435546875, 1742.77490234375, 0.35776486992836)
+[02/26 14:45:43][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.094 max=0.121 gt_z_mean=1.140 pred_z_mean=1.076
+[02/26 14:45:49][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.030 max=0.063
+[02/26 14:45:49][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9975 pred=+1.0048 delta(pred-gt)=+0.0073
+[02/26 14:45:49][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[0.00793431 2.0280874 0.07003364]
+[02/26 14:45:49][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.0300002 3.0145898 -0.04516966]
+[02/26 14:45:49][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+116.19,-2.65,+2.10) pred_vs_gt=(-4.74,+3.20,+3.25)
+[02/26 14:45:49][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+5.85
+[02/26 14:47:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:47:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:47:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:47:01][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:47:01][INFO] ✅[FIT][Epoch 0] finished! 01:31→1:14:52 | loss_epoch=1.87
+[02/26 14:47:01][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/26 14:47:01][INFO] [LossBreakdown] body_pose=0.3706 betas=0.2934 go_c=0.1882 go_gv=0.0178 transl_vel=0.2107 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.25/24.47 go_gv_deg(mean/max)=9.81/19.21 tv_abs_mean(pred)=0.0018 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.37 static_conf_hi_frac=0.10
+[02/26 14:47:01][INFO] [IncamDebug] transl_c_loss=0.0014 pred_cam_abs_mean=0.911 gt_pred_cam_abs_mean=0.936 valid_frac=1.00
+[02/26 14:47:01][INFO] [LossBreakdown] body_pose=1.1623 betas=0.4352 go_c=0.0966 go_gv=0.0283 transl_vel=0.2059 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.19/37.07 go_gv_deg(mean/max)=10.88/39.86 tv_abs_mean(pred)=0.0022 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.47 static_conf_hi_frac=0.25
+[02/26 14:47:01][INFO] [IncamDebug] transl_c_loss=0.0108 pred_cam_abs_mean=0.969 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/26 14:47:02][INFO] [LossBreakdown] body_pose=0.4084 betas=0.4669 go_c=0.0628 go_gv=0.0259 transl_vel=0.2351 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.92/18.16 go_gv_deg(mean/max)=10.88/21.77 tv_abs_mean(pred)=0.0020 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.29 static_conf_hi_frac=0.03
+[02/26 14:47:02][INFO] [IncamDebug] transl_c_loss=0.1853 pred_cam_abs_mean=0.884 gt_pred_cam_abs_mean=1.136 valid_frac=0.89
+[02/26 14:47:02][INFO] [LossBreakdown] body_pose=0.3840 betas=0.3780 go_c=0.3138 go_gv=0.0374 transl_vel=0.9534 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.80/133.43 go_gv_deg(mean/max)=11.03/124.70 tv_abs_mean(pred)=0.0035 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.36 static_conf_hi_frac=0.03
+[02/26 14:47:02][INFO] [IncamDebug] transl_c_loss=0.0383 pred_cam_abs_mean=0.941 gt_pred_cam_abs_mean=1.094 valid_frac=1.00
+[02/26 14:50:06][INFO] [Exp Name]: finetune_
+[02/26 14:50:06][INFO] [GPU x Batch] = 1 x 2
+[02/26 14:50:06][INFO] [UnityDataset] Found 9 sequences.
+[02/26 14:50:06][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 14:50:06][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/26 14:50:06][INFO]
+[02/26 14:50:06][INFO] [UnityDataset] Found 9 sequences.
+[02/26 14:50:06][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 14:50:06][INFO]
+[02/26 14:50:13][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/26 14:50:34][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_2/checkpoints'
+[02/26 14:50:45][INFO] Start Fitting...
+[02/26 14:50:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/26 14:50:46][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/26 14:50:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/26 14:50:49][INFO] [LossBreakdown] body_pose=0.7345 betas=0.7643 go_c=0.6235 go_gv=0.0695 transl_vel=0.2282 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=19.89/41.24 go_gv_deg(mean/max)=16.08/32.88 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.65 static_conf_hi_frac=0.36
+[02/26 14:50:49][INFO] [IncamDebug] transl_c_loss=0.0653 pred_cam_abs_mean=0.863 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/26 14:50:50][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/26 14:50:50][INFO] [LossBreakdown] body_pose=2.1123 betas=0.6598 go_c=0.1534 go_gv=0.0317 transl_vel=0.1941 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.14/20.20 go_gv_deg(mean/max)=8.61/17.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.00 static_conf_mean=0.80 static_conf_hi_frac=0.58
+[02/26 14:50:50][INFO] [IncamDebug] transl_c_loss=0.1939 pred_cam_abs_mean=0.916 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/26 14:50:50][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/26 14:50:50][INFO] [IncamDebug] transl_c_loss=0.1979 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/26 14:50:51][INFO] [LossBreakdown] body_pose=0.8671 betas=0.6704 go_c=0.1078 go_gv=0.0212 transl_vel=0.1907 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=7.57/11.34 go_gv_deg(mean/max)=8.67/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.01 static_conf_mean=0.80 static_conf_hi_frac=0.67
+[02/26 14:50:51][INFO] [IncamDebug] transl_c_loss=0.0103 pred_cam_abs_mean=0.861 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/26 14:50:59][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(345.0684509277344, 142.83265686035156, 815.19189453125, 1705.7205810546875, 0.3650217652320862)
+[02/26 14:50:59][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(328.91705322265625, 143.19557189941406, 866.721435546875, 1742.77490234375, 0.35776486992836)
+[02/26 14:50:59][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.094 max=0.121 gt_z_mean=1.140 pred_z_mean=1.076
+[02/26 14:51:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.030 max=0.063
+[02/26 14:51:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9975 pred=+1.0048 delta(pred-gt)=+0.0073
+[02/26 14:51:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[0.00793431 2.0280874 0.07003364]
+[02/26 14:51:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.0300002 3.0145898 -0.04516966]
+[02/26 14:51:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+116.19,-2.65,+2.10) pred_vs_gt=(-4.74,+3.20,+3.25)
+[02/26 14:51:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+5.85
+[02/26 14:52:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:52:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:52:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:52:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:52:17][INFO] ✅[FIT][Epoch 0] finished! 01:31→1:14:41 | loss_epoch=1.87
+[02/26 14:52:17][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/26 14:52:17][INFO] [LossBreakdown] body_pose=0.3706 betas=0.2934 go_c=0.1882 go_gv=0.0178 transl_vel=0.2107 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.25/24.47 go_gv_deg(mean/max)=9.81/19.21 tv_abs_mean(pred)=0.0018 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.37 static_conf_hi_frac=0.10
+[02/26 14:52:17][INFO] [IncamDebug] transl_c_loss=0.0014 pred_cam_abs_mean=0.911 gt_pred_cam_abs_mean=0.936 valid_frac=1.00
+[02/26 14:52:17][INFO] [LossBreakdown] body_pose=1.1623 betas=0.4352 go_c=0.0966 go_gv=0.0283 transl_vel=0.2059 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.19/37.07 go_gv_deg(mean/max)=10.88/39.86 tv_abs_mean(pred)=0.0022 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.47 static_conf_hi_frac=0.25
+[02/26 14:52:17][INFO] [IncamDebug] transl_c_loss=0.0108 pred_cam_abs_mean=0.969 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/26 14:52:18][INFO] [LossBreakdown] body_pose=0.4084 betas=0.4669 go_c=0.0628 go_gv=0.0259 transl_vel=0.2351 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.92/18.16 go_gv_deg(mean/max)=10.88/21.77 tv_abs_mean(pred)=0.0020 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.29 static_conf_hi_frac=0.03
+[02/26 14:52:18][INFO] [IncamDebug] transl_c_loss=0.1853 pred_cam_abs_mean=0.884 gt_pred_cam_abs_mean=1.136 valid_frac=0.89
+[02/26 14:52:18][INFO] [LossBreakdown] body_pose=0.3840 betas=0.3780 go_c=0.3138 go_gv=0.0374 transl_vel=0.9534 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.80/133.43 go_gv_deg(mean/max)=11.03/124.70 tv_abs_mean(pred)=0.0035 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.36 static_conf_hi_frac=0.03
+[02/26 14:52:18][INFO] [IncamDebug] transl_c_loss=0.0383 pred_cam_abs_mean=0.941 gt_pred_cam_abs_mean=1.094 valid_frac=1.00
+[02/26 14:53:52][INFO] [Exp Name]: finetune_
+[02/26 14:53:52][INFO] [GPU x Batch] = 1 x 2
+[02/26 14:53:52][INFO] [UnityDataset] Found 9 sequences.
+[02/26 14:53:52][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 14:53:52][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/26 14:53:52][INFO]
+[02/26 14:53:52][INFO] [UnityDataset] Found 9 sequences.
+[02/26 14:53:52][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 14:53:52][INFO]
+[02/26 14:54:00][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/26 14:54:21][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_3/checkpoints'
+[02/26 14:54:32][INFO] Start Fitting...
+[02/26 14:54:33][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/26 14:54:34][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/26 14:54:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/26 14:54:36][INFO] [LossBreakdown] body_pose=0.9203 betas=0.6590 go_c=0.5448 go_gv=0.0725 transl_vel=0.2262 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.81/35.75 go_gv_deg(mean/max)=13.86/28.49 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.41
+[02/26 14:54:36][INFO] [IncamDebug] transl_c_loss=0.0421 pred_cam_abs_mean=0.886 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/26 14:54:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/26 14:54:37][INFO] [LossBreakdown] body_pose=2.1555 betas=0.7464 go_c=0.1315 go_gv=0.0418 transl_vel=0.1971 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.42/28.02 go_gv_deg(mean/max)=10.52/27.78 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/26 14:54:37][INFO] [IncamDebug] transl_c_loss=0.1115 pred_cam_abs_mean=0.973 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/26 14:54:37][INFO] [LossBreakdown] body_pose=1.1250 betas=0.7374 go_c=0.2186 go_gv=0.0269 transl_vel=0.1815 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.20 go_gv_deg(mean/max)=7.87/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/26 14:54:37][INFO] [IncamDebug] transl_c_loss=0.1979 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/26 14:54:38][INFO] [LossBreakdown] body_pose=0.9267 betas=0.6977 go_c=0.0836 go_gv=0.0196 transl_vel=0.1916 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.11/9.82 go_gv_deg(mean/max)=8.98/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/26 14:54:38][INFO] [IncamDebug] transl_c_loss=0.0029 pred_cam_abs_mean=0.903 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/26 14:54:47][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(346.3108825683594, 142.3296661376953, 845.6177978515625, 1665.8909912109375, 0.3571843206882477)
+[02/26 14:54:47][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(332.3451232910156, 141.380859375, 822.3335571289062, 1695.947509765625, 0.3587808310985565)
+[02/26 14:54:47][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.083 max=0.103 gt_z_mean=1.140 pred_z_mean=1.108
+[02/26 14:54:53][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.021 max=0.043
+[02/26 14:54:53][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9975 pred=+1.0002 delta(pred-gt)=+0.0027
+[02/26 14:54:53][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[0.02282428 1.9486594 0.06558114]
+[02/26 14:54:53][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.0300002 3.0145898 -0.04516966]
+[02/26 14:54:53][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+111.64,-2.02,+2.71) pred_vs_gt=(-9.31,+3.40,+2.39)
+[02/26 14:54:53][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+8.01
+[02/26 14:56:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:56:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:56:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:56:04][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 14:56:04][INFO] ✅[FIT][Epoch 0] finished! 01:31→1:14:52 | loss_epoch=1.83
+[02/26 14:56:04][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/26 14:56:04][INFO] [LossBreakdown] body_pose=0.4396 betas=0.3765 go_c=0.1585 go_gv=0.0191 transl_vel=0.1995 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.28/22.51 go_gv_deg(mean/max)=10.50/19.04 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.53 static_conf_hi_frac=0.36
+[02/26 14:56:04][INFO] [IncamDebug] transl_c_loss=0.0013 pred_cam_abs_mean=0.909 gt_pred_cam_abs_mean=0.936 valid_frac=1.00
+[02/26 14:56:04][INFO] [LossBreakdown] body_pose=1.1988 betas=0.4801 go_c=0.0968 go_gv=0.0235 transl_vel=0.1599 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.60/35.95 go_gv_deg(mean/max)=9.46/38.26 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.58 static_conf_hi_frac=0.44
+[02/26 14:56:04][INFO] [IncamDebug] transl_c_loss=0.0081 pred_cam_abs_mean=0.969 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/26 14:56:05][INFO] [LossBreakdown] body_pose=0.3774 betas=0.4978 go_c=0.0803 go_gv=0.0470 transl_vel=0.1821 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.34/22.51 go_gv_deg(mean/max)=12.57/24.20 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.40 static_conf_hi_frac=0.07
+[02/26 14:56:05][INFO] [IncamDebug] transl_c_loss=0.1115 pred_cam_abs_mean=0.930 gt_pred_cam_abs_mean=1.136 valid_frac=0.89
+[02/26 14:56:05][INFO] [LossBreakdown] body_pose=0.3566 betas=0.4171 go_c=0.3248 go_gv=0.0353 transl_vel=0.8402 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.14/116.63 go_gv_deg(mean/max)=8.78/97.49 tv_abs_mean(pred)=0.0017 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.43 static_conf_hi_frac=0.09
+[02/26 14:56:05][INFO] [IncamDebug] transl_c_loss=0.0539 pred_cam_abs_mean=0.919 gt_pred_cam_abs_mean=1.094 valid_frac=1.00
+[02/26 14:56:15][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof f0 gt=(549.347900390625, 253.296630859375, 737.7196655273438, 1149.4931640625, 0.3052249550819397) pred=(514.33935546875, 230.67835998535156, 698.5016479492188, 1156.628662109375, 0.28824383020401)
+[02/26 14:56:15][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof fl gt=(554.8383178710938, 266.72760009765625, 695.2646484375, 1172.6390380859375, 0.28505077958106995) pred=(505.1429443359375, 240.62225341796875, 677.645263671875, 1097.8228759765625, 0.274455726146698)
+[02/26 14:56:15][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_transl_err_m mean=0.084 max=0.104 gt_z_mean=1.421 pred_z_mean=1.437
+[02/26 14:56:20][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_transl_err_m mean=0.073 max=0.160
+[02/26 14:56:20][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9925 pred=+0.9878 delta(pred-gt)=-0.0047
+[02/26 14:56:20][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.00300199 -1.5857557 0.06613082] global_orient0_aa(pred)=[-0.06903021 -1.5818751 -0.04569769]
+[02/26 14:56:20][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(raw_gt_world)=[-0.05748826 -3.0477192 0.07022861]
+[02/26 14:56:20][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-90.84,+2.53,+2.28) pred=(-90.71,-4.17,+0.88) pred_vs_gt=(+0.02,+1.50,-6.69)
+[02/26 14:56:20][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-0.83
+[02/26 14:57:31][INFO] ✅[FIT][Epoch 1] finished! 02:59→1:11:40 | loss_epoch=1.26
+[02/26 14:57:31][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/26 14:57:32][INFO] [LossBreakdown] body_pose=1.8983 betas=0.3241 go_c=0.0665 go_gv=0.0134 transl_vel=0.1578 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=5.98/10.14 go_gv_deg(mean/max)=3.95/8.01 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.00 static_conf_mean=0.85 static_conf_hi_frac=0.73
+[02/26 14:57:32][INFO] [IncamDebug] transl_c_loss=0.0050 pred_cam_abs_mean=0.964 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/26 14:57:32][INFO] [LossBreakdown] body_pose=0.2910 betas=0.2544 go_c=0.2179 go_gv=0.0306 transl_vel=0.9065 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.44/166.57 go_gv_deg(mean/max)=6.84/172.40 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.66
+[02/26 14:57:32][INFO] [IncamDebug] transl_c_loss=0.0140 pred_cam_abs_mean=0.938 gt_pred_cam_abs_mean=1.047 valid_frac=0.87
+[02/26 14:57:32][INFO] [LossBreakdown] body_pose=0.2887 betas=0.2964 go_c=0.1442 go_gv=0.0609 transl_vel=0.3467 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=16.28/32.46 go_gv_deg(mean/max)=15.99/33.89 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0026 tv_zero_frac(pred)=0.00 static_conf_mean=0.81 static_conf_hi_frac=0.69
+[02/26 14:57:32][INFO] [IncamDebug] transl_c_loss=0.0312 pred_cam_abs_mean=1.062 gt_pred_cam_abs_mean=1.253 valid_frac=0.73
+[02/26 14:57:33][INFO] [LossBreakdown] body_pose=0.2462 betas=0.3329 go_c=0.1615 go_gv=0.0332 transl_vel=0.1704 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.52/21.31 go_gv_deg(mean/max)=10.22/19.04 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.62
+[02/26 14:57:33][INFO] [IncamDebug] transl_c_loss=0.0270 pred_cam_abs_mean=1.015 gt_pred_cam_abs_mean=1.112 valid_frac=1.00
+[02/26 14:58:00][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof f0 gt=(497.9825744628906, 218.85873413085938, 754.4590454101562, 1513.018798828125, 0.4923076927661896) pred=(389.306396484375, 183.16238403320312, 680.8713989257812, 1543.7183837890625, 0.39346879720687866)
+[02/26 14:58:00][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_proj_bbox_oof fl gt=(531.361083984375, 220.1790771484375, 725.6470947265625, 1474.6353759765625, 0.49013060331344604) pred=(392.6751403808594, 205.9027862548828, 672.322021484375, 1464.998779296875, 0.3571843206882477)
+[02/26 14:58:00][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 incam_transl_err_m mean=0.134 max=0.209 gt_z_mean=1.193 pred_z_mean=1.269
+[02/26 14:58:04][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_transl_err_m mean=0.075 max=0.195
+[02/26 14:58:04][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 root_y0: gt=+0.9911 pred=+1.0025 delta(pred-gt)=+0.0114
+[02/26 14:58:04][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(gt)=[ 0.04780671 -1.4102775 0.10208653] global_orient0_aa(pred)=[-0.08510406 -1.3565564 -0.05139242]
+[02/26 14:58:04][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_aa(raw_gt_world)=[-0.05267622 -2.9777708 0.14565562]
+[02/26 14:58:04][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 global_orient0_yxz_deg gt=(-80.76,+5.40,+2.47) pred=(-77.82,-5.22,+0.71) pred_vs_gt=(+2.95,+0.03,-10.76)
+[02/26 14:58:04][INFO] [VisUnityVal] e002_107_biboo_birthday_speech_explosion_8 yaw0_deg(pred_vs_gt)=-7.40
+[02/26 15:19:44][INFO] [Exp Name]: finetune_
+[02/26 15:19:44][INFO] [GPU x Batch] = 1 x 2
+[02/26 15:19:45][INFO] [UnityDataset] Found 9 sequences.
+[02/26 15:19:45][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 15:19:45][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/26 15:19:45][INFO]
+[02/26 15:19:45][INFO] [UnityDataset] Found 9 sequences.
+[02/26 15:19:45][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 15:19:45][INFO]
+[02/26 15:19:52][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/26 15:20:14][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_4/checkpoints'
+[02/26 15:20:25][INFO] Start Fitting...
+[02/26 15:20:27][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/26 15:20:27][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/26 15:20:28][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/26 15:20:29][INFO] [LossBreakdown] body_pose=0.9152 betas=0.6618 go_c=0.5413 go_gv=0.0710 transl_vel=0.2262 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.72/35.27 go_gv_deg(mean/max)=13.78/28.02 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.40
+[02/26 15:20:29][INFO] [IncamDebug] transl_c_loss=0.0419 pred_cam_abs_mean=0.887 gt_pred_cam_abs_mean=1.015 valid_frac=1.00
+[02/26 15:20:30][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/26 15:20:30][INFO] [LossBreakdown] body_pose=2.1546 betas=0.7539 go_c=0.1314 go_gv=0.0420 transl_vel=0.1975 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.54/27.54 go_gv_deg(mean/max)=10.66/26.80 tv_abs_mean(pred)=0.0004 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.55
+[02/26 15:20:30][INFO] [IncamDebug] transl_c_loss=0.1123 pred_cam_abs_mean=0.972 gt_pred_cam_abs_mean=1.222 valid_frac=0.79
+[02/26 15:20:31][INFO] [LossBreakdown] body_pose=1.1252 betas=0.7382 go_c=0.2185 go_gv=0.0270 transl_vel=0.1825 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=11.70/20.36 go_gv_deg(mean/max)=7.92/19.38 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.69 static_conf_hi_frac=0.51
+[02/26 15:20:31][INFO] [IncamDebug] transl_c_loss=0.1975 pred_cam_abs_mean=0.899 gt_pred_cam_abs_mean=1.153 valid_frac=1.00
+[02/26 15:20:31][INFO] [LossBreakdown] body_pose=0.9273 betas=0.6970 go_c=0.0838 go_gv=0.0195 transl_vel=0.1915 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.10/10.14 go_gv_deg(mean/max)=8.97/13.67 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.00 static_conf_mean=0.79 static_conf_hi_frac=0.65
+[02/26 15:20:31][INFO] [IncamDebug] transl_c_loss=0.0029 pred_cam_abs_mean=0.903 gt_pred_cam_abs_mean=0.921 valid_frac=1.00
+[02/26 15:20:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(340.087890625, 215.88613891601562, 764.1021728515625, 1731.0291748046875, 0.256894052028656) pred=(346.440673828125, 142.63284301757812, 847.4159545898438, 1665.8863525390625, 0.3571843206882477)
+[02/26 15:20:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(393.56292724609375, 216.89535522460938, 769.66650390625, 1736.4815673828125, 0.2552975118160248) pred=(333.55120849609375, 141.51939392089844, 821.7772216796875, 1695.6507568359375, 0.35791000723838806)
+[02/26 15:20:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.082 max=0.102 gt_z_mean=1.140 pred_z_mean=1.110
+[02/26 15:20:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.021 max=0.044
+[02/26 15:20:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9975 pred=+1.0001 delta(pred-gt)=+0.0026
+[02/26 15:20:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03751069 2.1117876 -0.022111 ] global_orient0_aa(pred)=[0.02239942 1.9489677 0.066189 ]
+[02/26 15:20:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.0300002 3.0145898 -0.04516966]
+[02/26 15:20:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.03,+1.78,+1.03) pred=(+111.66,-2.05,+2.71) pred_vs_gt=(-9.29,+3.42,+2.42)
+[02/26 15:20:44][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+7.98
+[02/26 15:21:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 15:21:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 15:21:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 15:21:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 15:21:56][INFO] ✅[FIT][Epoch 0] finished! 01:30→1:13:39 | loss_epoch=1.83
+[02/26 15:21:56][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/26 15:21:56][INFO] [LossBreakdown] body_pose=0.4399 betas=0.3768 go_c=0.1595 go_gv=0.0193 transl_vel=0.1992 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.32/22.51 go_gv_deg(mean/max)=10.56/19.04 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0023 tv_zero_frac(pred)=0.00 static_conf_mean=0.53 static_conf_hi_frac=0.35
+[02/26 15:21:56][INFO] [IncamDebug] transl_c_loss=0.0013 pred_cam_abs_mean=0.909 gt_pred_cam_abs_mean=0.936 valid_frac=1.00
+[02/26 15:21:56][INFO] [LossBreakdown] body_pose=1.2032 betas=0.4807 go_c=0.0987 go_gv=0.0235 transl_vel=0.1607 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.64/35.08 go_gv_deg(mean/max)=9.43/35.95 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0018 tv_zero_frac(pred)=0.00 static_conf_mean=0.58 static_conf_hi_frac=0.43
+[02/26 15:21:56][INFO] [IncamDebug] transl_c_loss=0.0078 pred_cam_abs_mean=0.969 gt_pred_cam_abs_mean=0.985 valid_frac=1.00
+[02/26 15:21:57][INFO] [LossBreakdown] body_pose=0.3794 betas=0.4977 go_c=0.0799 go_gv=0.0466 transl_vel=0.1849 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.20/22.21 go_gv_deg(mean/max)=12.43/24.06 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.39 static_conf_hi_frac=0.07
+[02/26 15:21:57][INFO] [IncamDebug] transl_c_loss=0.1140 pred_cam_abs_mean=0.928 gt_pred_cam_abs_mean=1.136 valid_frac=0.89
+[02/26 15:21:57][INFO] [LossBreakdown] body_pose=0.3614 betas=0.4020 go_c=0.3283 go_gv=0.0345 transl_vel=0.9092 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.90/127.30 go_gv_deg(mean/max)=8.23/133.74 tv_abs_mean(pred)=0.0016 tv_abs_mean(tgt)=0.0022 tv_zero_frac(pred)=0.00 static_conf_mean=0.43 static_conf_hi_frac=0.10
+[02/26 15:21:57][INFO] [IncamDebug] transl_c_loss=0.0482 pred_cam_abs_mean=0.923 gt_pred_cam_abs_mean=1.094 valid_frac=1.00
+[02/26 15:22:08][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof f0 gt=(549.347900390625, 253.296630859375, 737.7196655273438, 1149.4931640625, 0.3052249550819397) pred=(514.8834838867188, 230.20785522460938, 698.5650634765625, 1157.0274658203125, 0.28824383020401)
+[02/26 15:22:08][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof fl gt=(554.8383178710938, 266.72760009765625, 695.2646484375, 1172.6390380859375, 0.28505077958106995) pred=(505.4332275390625, 240.08016967773438, 677.6663818359375, 1098.1451416015625, 0.274455726146698)
+[02/26 15:22:08][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 incam_transl_err_m mean=0.085 max=0.104 gt_z_mean=1.421 pred_z_mean=1.436
+[02/26 15:22:12][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_transl_err_m mean=0.073 max=0.159
+[02/26 15:22:12][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9925 pred=+0.9880 delta(pred-gt)=-0.0044
+[02/26 15:22:12][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.00300199 -1.5857557 0.06613082] global_orient0_aa(pred)=[-0.06720279 -1.584248 -0.04372025]
+[02/26 15:22:12][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_aa(raw_gt_world)=[-0.05748826 -3.0477192 0.07022861]
+[02/26 15:22:12][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-90.84,+2.53,+2.28) pred=(-90.84,-4.03,+0.89) pred_vs_gt=(-0.11,+1.49,-6.55)
+[02/26 15:22:12][INFO] [VisUnityVal] e001_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=-0.71
+[02/26 15:29:16][INFO] [Exp Name]: finetune_
+[02/26 15:29:16][INFO] [GPU x Batch] = 1 x 2
+[02/26 15:29:17][INFO] [UnityDataset] Found 9 sequences.
+[02/26 15:29:17][INFO] [Train Dataset][9/9]: name=unity, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 15:29:17][INFO] [Train Dataset][All]: ConcatDataset size=9
+[02/26 15:29:17][INFO]
+[02/26 15:29:17][INFO] [UnityDataset] Found 9 sequences.
+[02/26 15:29:17][INFO] [Val Dataset][7/7]: name=unity_val, size=9, genmo.datasets.unity_dataset.UnityDataset
+[02/26 15:29:17][INFO]
+[02/26 15:29:24][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/26 17:13:04][INFO] [Exp Name]: finetune_
+[02/26 17:13:04][INFO] [GPU x Batch] = 1 x 2
+[02/26 17:13:04][INFO] [UnityDataset] Found 3 sequences.
+[02/26 17:13:04][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[02/26 17:13:04][INFO] [Train Dataset][All]: ConcatDataset size=3
+[02/26 17:13:04][INFO]
+[02/26 17:13:04][INFO] [UnityDataset] Found 3 sequences.
+[02/26 17:13:04][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[02/26 17:13:04][INFO]
+[02/26 17:13:10][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/26 17:13:30][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_5/checkpoints'
+[02/26 17:13:40][INFO] Start Fitting...
+[02/26 17:13:41][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/26 17:13:42][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/26 17:13:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/26 17:13:44][INFO] [LossBreakdown] body_pose=1.6157 betas=0.5075 go_c=0.1272 go_gv=0.0220 transl_vel=0.1292 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.25/32.67 go_gv_deg(mean/max)=8.98/29.53 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.74 static_conf_hi_frac=0.61
+[02/26 17:13:44][INFO] [IncamDebug] transl_c_loss=0.0052 pred_cam_abs_mean=0.962 gt_pred_cam_abs_mean=0.982 valid_frac=1.00
+[02/26 17:13:45][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/26 17:13:53][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(349.9327697753906, 212.91502380371094, 774.9888916015625, 1723.708740234375, 0.26095789670944214) pred=(303.13665771484375, 186.48072814941406, 830.0010986328125, 1501.4534912109375, 0.23439767956733704)
+[02/26 17:13:53][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(332.5797424316406, 212.5543212890625, 824.0524291992188, 1723.397705078125, 0.25486209988594055) pred=(305.4437561035156, 181.58302307128906, 831.4202880859375, 1496.7310791015625, 0.23381711542606354)
+[02/26 17:13:53][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.119 max=0.137 gt_z_mean=1.141 pred_z_mean=1.225
+[02/26 17:13:53][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0096,0.0036,0.0108) norm_std=0.0075
+[02/26 17:13:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.025 max=0.040
+[02/26 17:13:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9982 pred=+0.9916 delta(pred-gt)=-0.0066
+[02/26 17:13:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.0374334 2.0976946 -0.01660619] global_orient0_aa(pred)=[ 0.03993812 2.1919029 -0.01066323]
+[02/26 17:13:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.03294645 2.990605 -0.03862115]
+[02/26 17:13:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.22,+1.57,+1.14) pred=(+125.62,+1.29,+1.43) pred_vs_gt=(+5.39,+0.38,+0.10)
+[02/26 17:13:58][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-6.64
+[02/26 17:14:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 17:14:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 17:14:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 17:14:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 17:14:56][INFO] ✅[FIT][Epoch 0] finished! 01:15→1:01:41 | loss_epoch=1.94
+[02/26 17:14:56][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/26 17:14:56][INFO] [LossBreakdown] body_pose=1.7758 betas=0.5442 go_c=0.1798 go_gv=0.0215 transl_vel=0.0901 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.60/22.07 go_gv_deg(mean/max)=10.21/17.61 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0012 tv_zero_frac(pred)=0.00 static_conf_mean=0.85 static_conf_hi_frac=0.74
+[02/26 17:14:56][INFO] [IncamDebug] transl_c_loss=0.0095 pred_cam_abs_mean=0.935 gt_pred_cam_abs_mean=0.995 valid_frac=1.00
+[02/26 17:15:02][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(367.8897399902344, 215.44203186035156, 740.2937622070312, 1724.3988037109375, 0.39666181802749634) pred=(433.6756286621094, 157.89495849609375, 775.2638549804688, 1312.4725341796875, 0.3367198705673218)
+[02/26 17:15:02][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(334.61187744140625, 214.93783569335938, 756.857421875, 1730.3028564453125, 0.2586356997489929) pred=(457.16290283203125, 179.7096710205078, 783.9954833984375, 1339.22265625, 0.21973875164985657)
+[02/26 17:15:02][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_transl_err_m mean=0.179 max=0.197 gt_z_mean=1.147 pred_z_mean=1.307
+[02/26 17:15:02][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0169,0.0142,0.0104) norm_std=0.0142
+[02/26 17:15:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_transl_err_m mean=0.027 max=0.049
+[02/26 17:15:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9982 pred=+0.9864 delta(pred-gt)=-0.0118
+[02/26 17:15:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[0.01964396 2.1108265 0.01723783] global_orient0_aa(pred)=[0.03263754 2.1805768 0.07929653]
+[02/26 17:15:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[0.03093894 2.9828596 0.00903419]
+[02/26 17:15:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.94,-0.25,+1.21) pred=(+124.93,-2.57,+3.06) pred_vs_gt=(+4.06,+2.78,+1.04)
+[02/26 17:15:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-10.35
+[02/26 17:19:17][INFO] [Exp Name]: finetune_
+[02/26 17:19:17][INFO] [GPU x Batch] = 1 x 2
+[02/26 17:19:17][INFO] [UnityDataset] Found 3 sequences.
+[02/26 17:19:17][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[02/26 17:19:17][INFO] [Train Dataset][All]: ConcatDataset size=3
+[02/26 17:19:17][INFO]
+[02/26 17:19:17][INFO] [UnityDataset] Found 3 sequences.
+[02/26 17:19:17][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[02/26 17:19:17][INFO]
+[02/26 17:19:22][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/26 17:19:43][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_6/checkpoints'
+[02/26 17:19:54][INFO] Start Fitting...
+[02/26 17:19:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/26 17:19:55][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/26 17:19:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/26 17:19:57][INFO] [LossBreakdown] body_pose=1.6157 betas=0.5075 go_c=0.1272 go_gv=0.0220 transl_vel=0.1292 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.25/32.67 go_gv_deg(mean/max)=8.98/29.53 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.01 static_conf_mean=0.74 static_conf_hi_frac=0.61
+[02/26 17:19:57][INFO] [IncamDebug] transl_c_loss=0.0052 pred_cam_abs_mean=0.962 gt_pred_cam_abs_mean=0.982 valid_frac=1.00
+[02/26 17:19:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/26 17:20:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(349.9327697753906, 212.91502380371094, 774.9888916015625, 1723.708740234375, 0.26095789670944214) pred=(306.21453857421875, 187.53842163085938, 832.1793823242188, 1510.89208984375, 0.2352685034275055)
+[02/26 17:20:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(332.5797424316406, 212.5543212890625, 824.0524291992188, 1723.397705078125, 0.25486209988594055) pred=(315.4848327636719, 179.31065368652344, 836.13134765625, 1532.68212890625, 0.23802611231803894)
+[02/26 17:20:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.103 max=0.124 gt_z_mean=1.141 pred_z_mean=1.201
+[02/26 17:20:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0091,0.0034,0.0233) norm_std=0.0115
+[02/26 17:20:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.023 max=0.042
+[02/26 17:20:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9982 pred=+0.9904 delta(pred-gt)=-0.0078
+[02/26 17:20:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.0374334 2.0976946 -0.01660619] global_orient0_aa(pred)=[ 0.03851418 2.17553 -0.00626194]
+[02/26 17:20:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.03294645 2.990605 -0.03862115]
+[02/26 17:20:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.22,+1.57,+1.14) pred=(+124.68,+1.09,+1.46) pred_vs_gt=(+4.45,+0.51,+0.25)
+[02/26 17:20:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-5.90
+[02/26 17:21:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 17:21:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 17:21:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 17:21:03][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 17:21:03][INFO] ✅[FIT][Epoch 0] finished! 01:08→55:55 | loss_epoch=1.94
+[02/26 17:21:03][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/26 17:21:03][INFO] [LossBreakdown] body_pose=1.7758 betas=0.5442 go_c=0.1799 go_gv=0.0215 transl_vel=0.0902 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=13.61/22.07 go_gv_deg(mean/max)=10.21/17.61 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0012 tv_zero_frac(pred)=0.00 static_conf_mean=0.85 static_conf_hi_frac=0.74
+[02/26 17:21:03][INFO] [IncamDebug] transl_c_loss=0.0095 pred_cam_abs_mean=0.935 gt_pred_cam_abs_mean=0.995 valid_frac=1.00
+[02/26 17:21:08][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(367.8897399902344, 215.44203186035156, 740.2937622070312, 1724.3988037109375, 0.39666181802749634) pred=(423.961181640625, 160.2303009033203, 773.5712890625, 1309.5406494140625, 0.3358490467071533)
+[02/26 17:21:08][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(334.61187744140625, 214.93783569335938, 756.857421875, 1730.3028564453125, 0.2586356997489929) pred=(455.142822265625, 183.3782958984375, 780.8526611328125, 1335.4176025390625, 0.21973875164985657)
+[02/26 17:21:08][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_transl_err_m mean=0.184 max=0.210 gt_z_mean=1.147 pred_z_mean=1.314
+[02/26 17:21:08][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0167,0.0147,0.0119) norm_std=0.0151
+[02/26 17:21:12][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_transl_err_m mean=0.027 max=0.049
+[02/26 17:21:12][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9982 pred=+0.9864 delta(pred-gt)=-0.0117
+[02/26 17:21:12][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[0.01964396 2.1108265 0.01723783] global_orient0_aa(pred)=[0.03506859 2.1814318 0.07227759]
+[02/26 17:21:12][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[0.03093894 2.9828596 0.00903419]
+[02/26 17:21:12][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.94,-0.25,+1.21) pred=(+124.98,-2.23,+3.00) pred_vs_gt=(+4.09,+2.56,+0.78)
+[02/26 17:21:12][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-10.38
+[02/26 17:22:12][INFO] ✅[FIT][Epoch 1] finished! 02:17→55:08 | loss_epoch=1.94
+[02/26 17:22:12][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/26 17:22:12][INFO] [LossBreakdown] body_pose=0.4036 betas=0.1975 go_c=0.1519 go_gv=0.0255 transl_vel=0.2741 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.36/21.92 go_gv_deg(mean/max)=11.28/18.16 tv_abs_mean(pred)=0.0023 tv_abs_mean(tgt)=0.0025 tv_zero_frac(pred)=0.00 static_conf_mean=0.55 static_conf_hi_frac=0.30
+[02/26 17:22:12][INFO] [IncamDebug] transl_c_loss=0.0019 pred_cam_abs_mean=0.903 gt_pred_cam_abs_mean=0.933 valid_frac=1.00
+[02/26 17:22:22][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof f0 gt=(546.1107788085938, 253.88790893554688, 734.773681640625, 1150.7977294921875, 0.30420899391174316) pred=(540.5365600585938, 241.10104370117188, 704.009033203125, 1175.422119140625, 0.295065313577652)
+[02/26 17:22:22][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 incam_proj_bbox_oof fl gt=(555.7108764648438, 266.3570251464844, 695.2114868164062, 1174.4456787109375, 0.28490564227104187) pred=(524.5050048828125, 248.99407958984375, 677.888671875, 1093.2509765625, 0.2732946276664734)
+[02/26 17:22:22][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 incam_transl_err_m mean=0.080 max=0.112 gt_z_mean=1.420 pred_z_mean=1.444
+[02/26 17:22:22][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 incam_delta_transl_std_m xyz=(0.0142,0.0052,0.0247) norm_std=0.0155
+[02/26 17:22:25][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_transl_err_m mean=0.075 max=0.165
+[02/26 17:22:25][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 root_y0: gt=+0.9918 pred=+0.9896 delta(pred-gt)=-0.0022
+[02/26 17:22:25][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_aa(gt)=[ 0.00325317 -1.5924343 0.06944188] global_orient0_aa(pred)=[-0.05805219 -1.5746638 -0.04741041]
+[02/26 17:22:25][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_aa(raw_gt_world)=[-0.06050703 -3.0577993 0.07391149]
+[02/26 17:22:25][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 global_orient0_yxz_deg gt=(-91.22,+2.67,+2.38) pred=(-90.27,-3.84,+0.40) pred_vs_gt=(+0.80,+2.12,-6.47)
+[02/26 17:22:25][INFO] [VisUnityVal] e002_101_biboo_birthday_speech_explosion_2 yaw0_deg(pred_vs_gt)=+0.26
+[02/26 20:33:14][INFO] [Exp Name]: finetune_
+[02/26 20:33:14][INFO] [GPU x Batch] = 1 x 2
+[02/26 20:33:14][INFO] [UnityDataset] Found 2 sequences.
+[02/26 20:33:14][INFO] [Train Dataset][9/9]: name=unity, size=2, genmo.datasets.unity_dataset.UnityDataset
+[02/26 20:33:14][INFO] [Train Dataset][All]: ConcatDataset size=2
+[02/26 20:33:14][INFO]
+[02/26 20:33:14][INFO] [UnityDataset] Found 2 sequences.
+[02/26 20:33:14][INFO] [Val Dataset][7/7]: name=unity_val, size=2, genmo.datasets.unity_dataset.UnityDataset
+[02/26 20:33:14][INFO]
+[02/26 20:33:21][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/26 20:33:42][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_7/checkpoints'
+[02/26 20:33:53][INFO] Start Fitting...
+[02/26 20:33:55][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/26 20:33:55][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/26 20:33:56][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/26 20:33:57][INFO] [LossBreakdown] body_pose=1.4249 betas=0.5467 go_c=0.1536 go_gv=0.0271 transl_vel=0.1225 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.85/33.99 go_gv_deg(mean/max)=11.16/31.72 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0015 tv_zero_frac(pred)=0.01 static_conf_mean=0.77 static_conf_hi_frac=0.61
+[02/26 20:33:57][INFO] [IncamDebug] transl_c_loss=0.0060 pred_cam_abs_mean=0.978 gt_pred_cam_abs_mean=1.003 valid_frac=1.00
+[02/26 20:33:58][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/26 20:34:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(349.9327697753906, 212.91502380371094, 774.9888916015625, 1723.708740234375, 0.26095789670944214) pred=(303.6613464355469, 179.0067901611328, 860.9672241210938, 1517.9632568359375, 0.23367197811603546)
+[02/26 20:34:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(332.5797424316406, 212.5543212890625, 824.0524291992188, 1723.397705078125, 0.25486209988594055) pred=(309.2574768066406, 171.33444213867188, 864.2628784179688, 1530.419921875, 0.2345428168773651)
+[02/26 20:34:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.107 max=0.123 gt_z_mean=1.141 pred_z_mean=1.199
+[02/26 20:34:06][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0096,0.0030,0.0190) norm_std=0.0089
+[02/26 20:34:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.024 max=0.041
+[02/26 20:34:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9982 pred=+0.9902 delta(pred-gt)=-0.0080
+[02/26 20:34:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.0374334 2.0976946 -0.01660619] global_orient0_aa(pred)=[ 0.04376853 2.1998293 -0.0290527 ]
+[02/26 20:34:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.03294645 2.990605 -0.03862115]
+[02/26 20:34:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.22,+1.57,+1.14) pred=(+126.09,+2.12,+1.20) pred_vs_gt=(+5.87,-0.23,-0.51)
+[02/26 20:34:11][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-6.53
+[02/26 20:35:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 20:35:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 20:35:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 20:35:20][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 20:35:20][INFO] ✅[FIT][Epoch 0] finished! 01:26→1:10:36 | loss_epoch=1.84
+[02/26 20:35:20][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/26 20:35:20][INFO] [LossBreakdown] body_pose=1.5330 betas=0.5600 go_c=0.1993 go_gv=0.0318 transl_vel=0.1242 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=15.29/28.25 go_gv_deg(mean/max)=11.93/27.78 tv_abs_mean(pred)=0.0020 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.00 static_conf_mean=0.60 static_conf_hi_frac=0.38
+[02/26 20:35:20][INFO] [IncamDebug] transl_c_loss=0.0069 pred_cam_abs_mean=0.946 gt_pred_cam_abs_mean=1.004 valid_frac=1.00
+[02/26 20:35:26][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(367.8897399902344, 215.44203186035156, 740.2937622070312, 1724.3988037109375, 0.39666181802749634) pred=(453.19488525390625, 227.6663360595703, 816.4274291992188, 1381.6439208984375, 0.22946298122406006)
+[02/26 20:35:26][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(334.61187744140625, 214.93783569335938, 756.857421875, 1730.3028564453125, 0.2586356997489929) pred=(457.32012939453125, 228.85504150390625, 801.1587524414062, 1416.9271240234375, 0.23367197811603546)
+[02/26 20:35:26][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_transl_err_m mean=0.213 max=0.238 gt_z_mean=1.147 pred_z_mean=1.357
+[02/26 20:35:26][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0169,0.0023,0.0177) norm_std=0.0178
+[02/26 20:35:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_transl_err_m mean=0.022 max=0.032
+[02/26 20:35:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9982 pred=+0.9878 delta(pred-gt)=-0.0104
+[02/26 20:35:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[0.01964396 2.1108265 0.01723783] global_orient0_aa(pred)=[ 0.02329979 2.1472476 -0.00230993]
+[02/26 20:35:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[0.03093894 2.9828596 0.00903419]
+[02/26 20:35:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.94,-0.25,+1.21) pred=(+123.04,+0.62,+0.91) pred_vs_gt=(+2.10,-0.70,-0.59)
+[02/26 20:35:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-6.24
+[02/26 20:36:56][INFO] ✅[FIT][Epoch 1] finished! 03:03→1:13:16 | loss_epoch=1.89
+[02/26 20:36:56][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/26 20:36:57][INFO] [LossBreakdown] body_pose=1.3963 betas=0.3926 go_c=0.1602 go_gv=0.0250 transl_vel=0.0707 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.52/20.84 go_gv_deg(mean/max)=11.41/18.34 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0013 tv_zero_frac(pred)=0.03 static_conf_mean=0.68 static_conf_hi_frac=0.52
+[02/26 20:36:57][INFO] [IncamDebug] transl_c_loss=0.0342 pred_cam_abs_mean=0.905 gt_pred_cam_abs_mean=0.994 valid_frac=1.00
+[02/26 20:37:02][INFO] [VisUnityVal] e002_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(360.3357238769531, 216.00372314453125, 804.5161743164062, 1745.4097900390625, 0.2528301775455475) pred=(442.31634521484375, 194.54034423828125, 927.4512939453125, 1381.0408935546875, 0.2285921573638916)
+[02/26 20:37:02][INFO] [VisUnityVal] e002_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(382.076904296875, 235.24044799804688, 846.8016357421875, 1868.976806640625, 0.24499273300170898) pred=(424.07574462890625, 204.76548767089844, 875.9785766601562, 1257.2626953125, 0.21451377868652344)
+[02/26 20:37:02][INFO] [VisUnityVal] e002_0_biboo_birthday_speech incam_transl_err_m mean=0.156 max=0.230 gt_z_mean=1.137 pred_z_mean=1.289
+[02/26 20:37:02][INFO] [VisUnityVal] e002_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0097,0.0118,0.0365) norm_std=0.0349
+[02/26 20:37:06][INFO] [VisUnityVal] e002_0_biboo_birthday_speech global_transl_err_m mean=0.032 max=0.068
+[02/26 20:37:06][INFO] [VisUnityVal] e002_0_biboo_birthday_speech root_y0: gt=+1.0000 pred=+0.9935 delta(pred-gt)=-0.0065
+[02/26 20:37:06][INFO] [VisUnityVal] e002_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.04288383 2.1391358 -0.00905175] global_orient0_aa(pred)=[0.03533553 2.1514502 0.00676705]
+[02/26 20:37:06][INFO] [VisUnityVal] e002_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.0432477 3.0403767 -0.03350845]
+[02/26 20:37:06][INFO] [VisUnityVal] e002_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+122.60,+1.34,+1.56) pred=(+123.29,+0.51,+1.61) pred_vs_gt=(+0.69,+0.49,+0.68)
+[02/26 20:37:06][INFO] [VisUnityVal] e002_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-5.11
+[02/26 20:49:55][INFO] [Exp Name]: finetune_
+[02/26 20:49:55][INFO] [GPU x Batch] = 1 x 2
+[02/26 20:49:55][INFO] [UnityDataset] Found 6 sequences.
+[02/26 20:49:55][INFO] [Train Dataset][9/9]: name=unity, size=6, genmo.datasets.unity_dataset.UnityDataset
+[02/26 20:49:55][INFO] [Train Dataset][All]: ConcatDataset size=6
+[02/26 20:49:55][INFO]
+[02/26 20:49:55][INFO] [UnityDataset] Found 6 sequences.
+[02/26 20:49:55][INFO] [Val Dataset][7/7]: name=unity_val, size=6, genmo.datasets.unity_dataset.UnityDataset
+[02/26 20:49:55][INFO]
+[02/26 20:50:02][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/26 20:50:21][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_8/checkpoints'
+[02/26 20:50:31][INFO] Start Fitting...
+[02/26 20:50:35][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/26 20:50:35][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/26 20:50:36][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/26 20:50:36][INFO] [LossBreakdown] body_pose=1.4092 betas=0.4445 go_c=0.0778 go_gv=0.0263 transl_vel=0.0909 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.42/37.35 go_gv_deg(mean/max)=9.89/34.88 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0014 tv_zero_frac(pred)=0.00 static_conf_mean=0.65 static_conf_hi_frac=0.49
+[02/26 20:50:36][INFO] [IncamDebug] transl_c_loss=0.0057 pred_cam_abs_mean=0.838 gt_pred_cam_abs_mean=0.836 valid_frac=1.00
+[02/26 20:50:37][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/26 20:50:37][INFO] [LossBreakdown] body_pose=0.7119 betas=0.3151 go_c=0.0743 go_gv=0.0164 transl_vel=0.1938 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.05/9.14 go_gv_deg(mean/max)=8.13/11.34 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0026 tv_zero_frac(pred)=0.00 static_conf_mean=0.55 static_conf_hi_frac=0.24
+[02/26 20:50:37][INFO] [IncamDebug] transl_c_loss=0.0045 pred_cam_abs_mean=0.801 gt_pred_cam_abs_mean=0.788 valid_frac=1.00
+[02/26 20:50:37][INFO] [LossBreakdown] body_pose=0.9881 betas=0.3693 go_c=0.2097 go_gv=0.0333 transl_vel=0.1975 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.11/25.27 go_gv_deg(mean/max)=13.37/22.65 tv_abs_mean(pred)=0.0003 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.03 static_conf_mean=0.89 static_conf_hi_frac=0.80
+[02/26 20:50:37][INFO] [IncamDebug] transl_c_loss=0.0122 pred_cam_abs_mean=0.780 gt_pred_cam_abs_mean=0.834 valid_frac=1.00
+[02/26 20:50:46][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(283.5253601074219, 214.48092651367188, 781.327880859375, 1731.6754150390625, 0.3937590718269348) pred=(293.1600036621094, 148.25318908691406, 785.3115234375, 1717.7625732421875, 0.2383164018392563)
+[02/26 20:50:46][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(373.0350341796875, 213.11033630371094, 758.5469360351562, 1729.399169921875, 0.25849056243896484) pred=(314.6191711425781, 151.03318786621094, 799.248046875, 1719.926513671875, 0.23802611231803894)
+[02/26 20:50:46][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.073 max=0.086 gt_z_mean=1.143 pred_z_mean=1.168
+[02/26 20:50:46][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0074,0.0055,0.0264) norm_std=0.0067
+[02/26 20:50:50][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.016 max=0.026
+[02/26 20:50:50][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9941 delta(pred-gt)=-0.0015
+[02/26 20:50:50][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03151054 2.1238003 -0.01273491] global_orient0_aa(pred)=[-0.02047118 2.0939586 0.04961594]
+[02/26 20:50:50][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.02841046 3.0254452 -0.0314105 ]
+[02/26 20:50:50][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.71,+1.25,+1.00) pred=(+119.99,-2.52,+0.34) pred_vs_gt=(-1.68,+1.41,+3.56)
+[02/26 20:50:50][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+1.78
+[02/26 20:51:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 20:51:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 20:51:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 20:51:59][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/26 20:51:59][INFO] ✅[FIT][Epoch 0] finished! 01:24→1:09:23 | loss_epoch=1.54
+[02/26 20:51:59][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/26 20:51:59][INFO] [LossBreakdown] body_pose=1.8596 betas=0.3601 go_c=0.0727 go_gv=0.0312 transl_vel=0.0721 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.18/17.24 go_gv_deg(mean/max)=11.25/18.86 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.00 static_conf_mean=0.61 static_conf_hi_frac=0.42
+[02/26 20:51:59][INFO] [IncamDebug] transl_c_loss=0.0040 pred_cam_abs_mean=0.831 gt_pred_cam_abs_mean=0.831 valid_frac=1.00
+[02/26 20:51:59][INFO] [LossBreakdown] body_pose=0.4968 betas=0.4004 go_c=0.0879 go_gv=0.0138 transl_vel=0.1476 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.23/15.02 go_gv_deg(mean/max)=7.70/15.24 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.50 static_conf_hi_frac=0.15
+[02/26 20:51:59][INFO] [IncamDebug] transl_c_loss=0.0087 pred_cam_abs_mean=0.775 gt_pred_cam_abs_mean=0.818 valid_frac=1.00
+[02/26 20:52:00][INFO] [LossBreakdown] body_pose=0.3678 betas=0.5538 go_c=0.0800 go_gv=0.0387 transl_vel=0.1620 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.81/23.93 go_gv_deg(mean/max)=14.06/23.65 tv_abs_mean(pred)=0.0012 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.54 static_conf_hi_frac=0.26
+[02/26 20:52:00][INFO] [IncamDebug] transl_c_loss=0.0049 pred_cam_abs_mean=0.767 gt_pred_cam_abs_mean=0.767 valid_frac=1.00
+[02/26 20:52:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(331.28607177734375, 215.93923950195312, 761.39013671875, 1728.7545166015625, 0.2576197385787964) pred=(363.4306335449219, 167.93292236328125, 849.427978515625, 1607.096435546875, 0.23193033039569855)
+[02/26 20:52:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(400.6131591796875, 211.72628784179688, 780.7462158203125, 1736.67626953125, 0.25602322816848755) pred=(377.06085205078125, 160.69300842285156, 775.3803100585938, 1611.353271484375, 0.23076923191547394)
+[02/26 20:52:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_transl_err_m mean=0.075 max=0.095 gt_z_mean=1.140 pred_z_mean=1.172
+[02/26 20:52:06][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0088,0.0069,0.0285) norm_std=0.0134
+[02/26 20:52:10][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_transl_err_m mean=0.016 max=0.032
+[02/26 20:52:10][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9983 pred=+0.9869 delta(pred-gt)=-0.0114
+[02/26 20:52:10][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03966505 2.1054645 -0.01940506] global_orient0_aa(pred)=[-2.9568891e-03 1.9146645e+00 4.0415957e-04]
+[02/26 20:52:10][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.03390508 3.0053003 -0.04316957]
+[02/26 20:52:10][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.67,+1.73,+1.18) pred=(+109.70,-0.10,-0.11) pred_vs_gt=(-10.95,-0.17,+2.22)
+[02/26 20:52:10][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+5.61
+[02/26 20:53:23][INFO] ✅[FIT][Epoch 1] finished! 02:49→1:07:52 | loss_epoch=1.29
+[02/26 20:53:23][INFO] 🚀[FIT][Epoch 2] Data: unity Experiment: finetune_
+[02/26 20:53:24][INFO] [LossBreakdown] body_pose=1.6923 betas=0.2258 go_c=0.0601 go_gv=0.0177 transl_vel=0.0937 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=7.30/14.81 go_gv_deg(mean/max)=7.79/14.81 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.01 static_conf_mean=0.80 static_conf_hi_frac=0.70
+[02/26 20:53:24][INFO] [IncamDebug] transl_c_loss=0.0024 pred_cam_abs_mean=0.817 gt_pred_cam_abs_mean=0.839 valid_frac=1.00
+[02/26 20:53:24][INFO] [LossBreakdown] body_pose=0.2792 betas=0.3198 go_c=0.0540 go_gv=0.0205 transl_vel=0.1727 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.51/16.86 go_gv_deg(mean/max)=9.13/16.86 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0024 tv_zero_frac(pred)=0.01 static_conf_mean=0.74 static_conf_hi_frac=0.65
+[02/26 20:53:24][INFO] [IncamDebug] transl_c_loss=0.0027 pred_cam_abs_mean=0.770 gt_pred_cam_abs_mean=0.771 valid_frac=1.00
+[02/26 20:53:24][INFO] [LossBreakdown] body_pose=0.2745 betas=0.2550 go_c=0.1242 go_gv=0.0173 transl_vel=0.1683 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.33/20.20 go_gv_deg(mean/max)=7.22/16.86 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.02 static_conf_mean=0.73 static_conf_hi_frac=0.67
+[02/26 20:53:24][INFO] [IncamDebug] transl_c_loss=0.0124 pred_cam_abs_mean=0.775 gt_pred_cam_abs_mean=0.835 valid_frac=1.00
+[02/26 20:53:41][INFO] [VisUnityVal] e002_103_biboo_birthday_speech_explosion_4 incam_proj_bbox_oof f0 gt=(479.8084411621094, 267.4416809082031, 747.5077514648438, 1206.365478515625, 0.34775036573410034) pred=(550.2503662109375, 270.34576416015625, 718.5698852539062, 1057.2393798828125, 0.22728592157363892)
+[02/26 20:53:41][INFO] [VisUnityVal] e002_103_biboo_birthday_speech_explosion_4 incam_proj_bbox_oof fl gt=(502.56866455078125, 267.55487060546875, 743.5947875976562, 1223.127197265625, 0.35544267296791077) pred=(559.9844970703125, 266.1633605957031, 727.9498291015625, 1065.187255859375, 0.21509432792663574)
+[02/26 20:53:41][INFO] [VisUnityVal] e002_103_biboo_birthday_speech_explosion_4 incam_transl_err_m mean=0.160 max=0.205 gt_z_mean=1.742 pred_z_mean=1.901
+[02/26 20:53:41][INFO] [VisUnityVal] e002_103_biboo_birthday_speech_explosion_4 incam_delta_transl_std_m xyz=(0.0094,0.0065,0.0288) norm_std=0.0284
+[02/26 20:53:44][INFO] [VisUnityVal] e002_103_biboo_birthday_speech_explosion_4 global_transl_err_m mean=0.077 max=0.147
+[02/26 20:53:44][INFO] [VisUnityVal] e002_103_biboo_birthday_speech_explosion_4 root_y0: gt=+0.9945 pred=+0.9957 delta(pred-gt)=+0.0012
+[02/26 20:53:44][INFO] [VisUnityVal] e002_103_biboo_birthday_speech_explosion_4 global_orient0_aa(gt)=[ 0.05555351 2.3103557 -0.04010025] global_orient0_aa(pred)=[ 0.05333097 2.2074246 -0.00424692]
+[02/26 20:53:44][INFO] [VisUnityVal] e002_103_biboo_birthday_speech_explosion_4 global_orient0_aa(raw_gt_world)=[0.04937809 0.19836597 0.02270691]
+[02/26 20:53:44][INFO] [VisUnityVal] e002_103_biboo_birthday_speech_explosion_4 global_orient0_yxz_deg gt=(+132.45,+2.68,+1.57) pred=(+126.52,+1.29,+2.12) pred_vs_gt=(-5.94,+1.34,+0.66)
+[02/26 20:53:44][INFO] [VisUnityVal] e002_103_biboo_birthday_speech_explosion_4 yaw0_deg(pred_vs_gt)=+0.12
+[02/28 11:56:16][INFO] [Exp Name]: finetune_
+[02/28 11:56:16][INFO] [GPU x Batch] = 1 x 2
+[02/28 11:56:16][INFO] [UnityDataset] Found 6 sequences.
+[02/28 11:56:16][INFO] [Train Dataset][9/9]: name=unity, size=6, genmo.datasets.unity_dataset.UnityDataset
+[02/28 11:56:16][INFO] [Train Dataset][All]: ConcatDataset size=6
+[02/28 11:56:16][INFO]
+[02/28 11:56:16][INFO] [UnityDataset] Found 6 sequences.
+[02/28 11:56:16][INFO] [Val Dataset][7/7]: name=unity_val, size=6, genmo.datasets.unity_dataset.UnityDataset
+[02/28 11:56:16][INFO]
+[02/28 11:56:24][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/28 11:56:56][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_9/checkpoints'
+[02/28 11:57:09][INFO] Start Fitting...
+[02/28 11:57:11][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/28 11:57:11][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/28 11:57:13][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/28 11:57:14][INFO] [LossBreakdown] body_pose=1.4092 betas=0.4445 go_c=0.0778 go_gv=0.0263 transl_vel=0.0909 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.42/37.35 go_gv_deg(mean/max)=9.89/34.88 tv_abs_mean(pred)=0.0007 tv_abs_mean(tgt)=0.0014 tv_zero_frac(pred)=0.00 static_conf_mean=0.65 static_conf_hi_frac=0.49
+[02/28 11:57:14][INFO] [IncamDebug] transl_c_loss=0.0057 pred_cam_abs_mean=0.838 gt_pred_cam_abs_mean=0.836 valid_frac=1.00
+[02/28 11:57:15][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/28 11:57:15][INFO] [LossBreakdown] body_pose=0.7119 betas=0.3151 go_c=0.0743 go_gv=0.0164 transl_vel=0.1938 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=6.05/9.14 go_gv_deg(mean/max)=8.13/11.34 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0026 tv_zero_frac(pred)=0.00 static_conf_mean=0.55 static_conf_hi_frac=0.24
+[02/28 11:57:15][INFO] [IncamDebug] transl_c_loss=0.0045 pred_cam_abs_mean=0.801 gt_pred_cam_abs_mean=0.788 valid_frac=1.00
+[02/28 11:57:15][INFO] [LossBreakdown] body_pose=0.9881 betas=0.3693 go_c=0.2097 go_gv=0.0333 transl_vel=0.1975 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.11/25.27 go_gv_deg(mean/max)=13.37/22.65 tv_abs_mean(pred)=0.0003 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.03 static_conf_mean=0.89 static_conf_hi_frac=0.80
+[02/28 11:57:16][INFO] [IncamDebug] transl_c_loss=0.0122 pred_cam_abs_mean=0.780 gt_pred_cam_abs_mean=0.834 valid_frac=1.00
+[02/28 11:57:25][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(283.5253601074219, 214.48092651367188, 781.327880859375, 1731.6754150390625, 0.3937590718269348) pred=(293.1600036621094, 148.25318908691406, 785.3115234375, 1717.7625732421875, 0.2383164018392563)
+[02/28 11:57:25][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(373.0350341796875, 213.11033630371094, 758.5469360351562, 1729.399169921875, 0.25849056243896484) pred=(314.6191711425781, 151.03318786621094, 799.248046875, 1719.926513671875, 0.23802611231803894)
+[02/28 11:57:25][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.073 max=0.086 gt_z_mean=1.143 pred_z_mean=1.168
+[02/28 11:57:25][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0074,0.0055,0.0264) norm_std=0.0067
+[02/28 11:57:30][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.016 max=0.026
+[02/28 11:57:30][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9941 delta(pred-gt)=-0.0015
+[02/28 11:57:30][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03151054 2.1238003 -0.01273491] global_orient0_aa(pred)=[-0.02047118 2.0939586 0.04961594]
+[02/28 11:57:30][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.02841046 3.0254452 -0.0314105 ]
+[02/28 11:57:30][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.71,+1.25,+1.00) pred=(+119.99,-2.52,+0.34) pred_vs_gt=(-1.68,+1.41,+3.56)
+[02/28 11:57:30][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+1.78
+[02/28 11:58:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 11:58:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 11:58:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 11:58:43][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 11:58:43][INFO] ✅[FIT][Epoch 0] finished! 01:33→1:16:03 | loss_epoch=1.54
+[02/28 11:58:43][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/28 11:58:43][INFO] [LossBreakdown] body_pose=1.8596 betas=0.3601 go_c=0.0727 go_gv=0.0312 transl_vel=0.0721 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.18/17.24 go_gv_deg(mean/max)=11.25/18.86 tv_abs_mean(pred)=0.0010 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.00 static_conf_mean=0.61 static_conf_hi_frac=0.42
+[02/28 11:58:43][INFO] [IncamDebug] transl_c_loss=0.0040 pred_cam_abs_mean=0.831 gt_pred_cam_abs_mean=0.831 valid_frac=1.00
+[02/28 11:58:43][INFO] [LossBreakdown] body_pose=0.4968 betas=0.4004 go_c=0.0879 go_gv=0.0138 transl_vel=0.1476 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.23/15.02 go_gv_deg(mean/max)=7.70/15.24 tv_abs_mean(pred)=0.0011 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.50 static_conf_hi_frac=0.15
+[02/28 11:58:43][INFO] [IncamDebug] transl_c_loss=0.0087 pred_cam_abs_mean=0.775 gt_pred_cam_abs_mean=0.818 valid_frac=1.00
+[02/28 11:58:43][INFO] [LossBreakdown] body_pose=0.3678 betas=0.5538 go_c=0.0800 go_gv=0.0387 transl_vel=0.1620 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.81/23.93 go_gv_deg(mean/max)=14.06/23.65 tv_abs_mean(pred)=0.0012 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.54 static_conf_hi_frac=0.26
+[02/28 11:58:44][INFO] [IncamDebug] transl_c_loss=0.0049 pred_cam_abs_mean=0.767 gt_pred_cam_abs_mean=0.767 valid_frac=1.00
+[02/28 11:58:51][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(331.28607177734375, 215.93923950195312, 761.39013671875, 1728.7545166015625, 0.2576197385787964) pred=(363.4306335449219, 167.93292236328125, 849.427978515625, 1607.096435546875, 0.23193033039569855)
+[02/28 11:58:51][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(400.6131591796875, 211.72628784179688, 780.7462158203125, 1736.67626953125, 0.25602322816848755) pred=(377.06085205078125, 160.69300842285156, 775.3803100585938, 1611.353271484375, 0.23076923191547394)
+[02/28 11:58:51][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_transl_err_m mean=0.075 max=0.095 gt_z_mean=1.140 pred_z_mean=1.172
+[02/28 11:58:51][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0088,0.0069,0.0285) norm_std=0.0134
+[02/28 11:58:55][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_transl_err_m mean=0.016 max=0.032
+[02/28 11:58:55][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9983 pred=+0.9869 delta(pred-gt)=-0.0114
+[02/28 11:58:55][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03966505 2.1054645 -0.01940506] global_orient0_aa(pred)=[-2.9568891e-03 1.9146645e+00 4.0415957e-04]
+[02/28 11:58:55][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.03390508 3.0053003 -0.04316957]
+[02/28 11:58:55][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.67,+1.73,+1.18) pred=(+109.70,-0.10,-0.11) pred_vs_gt=(-10.95,-0.17,+2.22)
+[02/28 11:58:55][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+5.61
+[02/28 12:03:38][INFO] [Exp Name]: finetune_
+[02/28 12:03:38][INFO] [GPU x Batch] = 1 x 2
+[02/28 12:03:38][INFO] [UnityDataset] Found 6 sequences.
+[02/28 12:03:38][INFO] [Train Dataset][9/9]: name=unity, size=6, genmo.datasets.unity_dataset.UnityDataset
+[02/28 12:03:38][INFO] [Train Dataset][All]: ConcatDataset size=6
+[02/28 12:03:38][INFO]
+[02/28 12:03:38][INFO] [UnityDataset] Found 6 sequences.
+[02/28 12:03:38][INFO] [Val Dataset][7/7]: name=unity_val, size=6, genmo.datasets.unity_dataset.UnityDataset
+[02/28 12:03:38][INFO]
+[02/28 12:03:45][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/28 12:04:15][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_10/checkpoints'
+[02/28 12:04:25][INFO] Start Fitting...
+[02/28 12:04:27][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/28 12:04:27][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/28 12:04:29][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/28 12:04:30][INFO] [LossBreakdown] body_pose=1.3740 betas=0.5189 go_c=0.0843 go_gv=0.0311 transl_vel=0.0947 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=10.13/43.56 go_gv_deg(mean/max)=11.37/42.00 tv_abs_mean(pred)=0.0008 tv_abs_mean(tgt)=0.0014 tv_zero_frac(pred)=0.00 static_conf_mean=0.65 static_conf_hi_frac=0.48
+[02/28 12:04:30][INFO] [IncamDebug] transl_c_loss=0.0046 pred_cam_abs_mean=0.806 gt_pred_cam_abs_mean=0.836 valid_frac=1.00
+[02/28 12:04:30][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/28 12:04:31][INFO] [LossBreakdown] body_pose=0.6754 betas=0.3495 go_c=0.0957 go_gv=0.0225 transl_vel=0.1987 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=8.63/15.02 go_gv_deg(mean/max)=10.22/15.24 tv_abs_mean(pred)=0.0013 tv_abs_mean(tgt)=0.0026 tv_zero_frac(pred)=0.00 static_conf_mean=0.51 static_conf_hi_frac=0.25
+[02/28 12:04:31][INFO] [IncamDebug] transl_c_loss=0.0058 pred_cam_abs_mean=0.759 gt_pred_cam_abs_mean=0.788 valid_frac=1.00
+[02/28 12:04:31][INFO] [LossBreakdown] body_pose=0.9881 betas=0.3693 go_c=0.2097 go_gv=0.0333 transl_vel=0.1975 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=17.11/25.27 go_gv_deg(mean/max)=13.37/22.65 tv_abs_mean(pred)=0.0003 tv_abs_mean(tgt)=0.0021 tv_zero_frac(pred)=0.03 static_conf_mean=0.89 static_conf_hi_frac=0.80
+[02/28 12:04:31][INFO] [IncamDebug] transl_c_loss=0.0122 pred_cam_abs_mean=0.780 gt_pred_cam_abs_mean=0.834 valid_frac=1.00
+[02/28 12:04:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(283.5253601074219, 214.48092651367188, 781.327880859375, 1731.6754150390625, 0.3937590718269348) pred=(315.7223815917969, 140.22323608398438, 793.8090209960938, 1793.8271484375, 0.24586357176303864)
+[02/28 12:04:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(373.0350341796875, 213.11033630371094, 758.5469360351562, 1729.399169921875, 0.25849056243896484) pred=(355.1200256347656, 142.59463500976562, 790.6710205078125, 1799.80908203125, 0.24528300762176514)
+[02/28 12:04:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.073 max=0.108 gt_z_mean=1.143 pred_z_mean=1.116
+[02/28 12:04:39][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0069,0.0059,0.0254) norm_std=0.0135
+[02/28 12:04:45][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.013 max=0.024
+[02/28 12:04:45][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9956 pred=+0.9912 delta(pred-gt)=-0.0044
+[02/28 12:04:45][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03151054 2.1238003 -0.01273491] global_orient0_aa(pred)=[-1.3032896e-02 2.0765851e+00 -1.2483114e-03]
+[02/28 12:04:45][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.02841046 3.0254452 -0.0314105 ]
+[02/28 12:04:45][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+121.71,+1.25,+1.00) pred=(+118.98,-0.26,-0.56) pred_vs_gt=(-2.72,-0.54,+2.11)
+[02/28 12:04:45][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+2.65
+[02/28 12:05:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:05:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:05:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:05:47][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:05:47][INFO] ✅[FIT][Epoch 0] finished! 01:20→1:06:03 | loss_epoch=1.54
+[02/28 12:05:47][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/28 12:05:47][INFO] [LossBreakdown] body_pose=1.9015 betas=0.3542 go_c=0.0520 go_gv=0.0241 transl_vel=0.0753 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=7.56/17.24 go_gv_deg(mean/max)=9.84/18.34 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0016 tv_zero_frac(pred)=0.01 static_conf_mean=0.73 static_conf_hi_frac=0.58
+[02/28 12:05:47][INFO] [IncamDebug] transl_c_loss=0.0070 pred_cam_abs_mean=0.767 gt_pred_cam_abs_mean=0.831 valid_frac=1.00
+[02/28 12:05:49][INFO] [LossBreakdown] body_pose=0.4799 betas=0.4406 go_c=0.0906 go_gv=0.0147 transl_vel=0.1507 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.55/15.86 go_gv_deg(mean/max)=7.77/15.66 tv_abs_mean(pred)=0.0009 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.52 static_conf_hi_frac=0.16
+[02/28 12:05:49][INFO] [IncamDebug] transl_c_loss=0.0074 pred_cam_abs_mean=0.782 gt_pred_cam_abs_mean=0.818 valid_frac=1.00
+[02/28 12:05:49][INFO] [LossBreakdown] body_pose=0.5853 betas=0.4739 go_c=0.1279 go_gv=0.0510 transl_vel=0.1721 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=16.07/30.42 go_gv_deg(mean/max)=16.44/28.37 tv_abs_mean(pred)=0.0005 tv_abs_mean(tgt)=0.0020 tv_zero_frac(pred)=0.00 static_conf_mean=0.62 static_conf_hi_frac=0.32
+[02/28 12:05:49][INFO] [IncamDebug] transl_c_loss=0.0055 pred_cam_abs_mean=0.747 gt_pred_cam_abs_mean=0.767 valid_frac=1.00
+[02/28 12:05:54][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(331.28607177734375, 215.93923950195312, 761.39013671875, 1728.7545166015625, 0.2576197385787964) pred=(329.10845947265625, 144.07081604003906, 760.650634765625, 1612.6663818359375, 0.23570391535758972)
+[02/28 12:05:54][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(400.6131591796875, 211.72628784179688, 780.7462158203125, 1736.67626953125, 0.25602322816848755) pred=(342.8603210449219, 135.09548950195312, 784.7772216796875, 1609.4716796875, 0.23352684080600739)
+[02/28 12:05:54][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_transl_err_m mean=0.088 max=0.122 gt_z_mean=1.140 pred_z_mean=1.101
+[02/28 12:05:54][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0071,0.0072,0.0305) norm_std=0.0162
+[02/28 12:05:58][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_transl_err_m mean=0.019 max=0.035
+[02/28 12:05:58][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9983 pred=+0.9868 delta(pred-gt)=-0.0115
+[02/28 12:05:58][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.03966505 2.1054645 -0.01940506] global_orient0_aa(pred)=[-0.00727363 2.1854496 -0.00758927]
+[02/28 12:05:58][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.03390508 3.0053003 -0.04316957]
+[02/28 12:05:58][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.67,+1.73,+1.18) pred=(+125.22,+0.16,-0.46) pred_vs_gt=(+4.56,-0.61,+2.18)
+[02/28 12:05:58][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=-6.64
+[02/28 12:42:01][INFO] [Exp Name]: finetune_
+[02/28 12:42:01][INFO] [GPU x Batch] = 1 x 2
+[02/28 12:42:02][INFO] [UnityDataset] Found 3 sequences.
+[02/28 12:42:02][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[02/28 12:42:02][INFO] [Train Dataset][All]: ConcatDataset size=3
+[02/28 12:42:02][INFO]
+[02/28 12:42:02][INFO] [UnityDataset] Found 3 sequences.
+[02/28 12:42:02][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[02/28 12:42:02][INFO]
+[02/28 12:42:08][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/28 12:42:31][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_11/checkpoints'
+[02/28 12:42:42][INFO] Start Fitting...
+[02/28 12:42:44][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/28 12:42:44][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/28 12:42:46][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/28 12:42:46][INFO] [LossBreakdown] body_pose=1.4416 betas=0.4510 go_c=0.0884 go_gv=0.0251 transl_vel=0.1456 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.01/42.83 go_gv_deg(mean/max)=10.08/40.04 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.00 static_conf_mean=0.65 static_conf_hi_frac=0.46
+[02/28 12:42:47][INFO] [IncamDebug] transl_c_loss=0.0048 pred_cam_abs_mean=0.828 gt_pred_cam_abs_mean=0.854 valid_frac=1.00
+[02/28 12:42:48][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/28 12:42:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(349.9327697753906, 212.91502380371094, 774.9888916015625, 1723.708740234375, 0.26095789670944214) pred=(483.2897033691406, 185.81089782714844, 768.710205078125, 1683.3951416015625, 0.24760521948337555)
+[02/28 12:42:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(332.5797424316406, 212.5543212890625, 824.0524291992188, 1723.397705078125, 0.25486209988594055) pred=(479.870849609375, 113.93203735351562, 816.0200805664062, 1809.1297607421875, 0.25979679822921753)
+[02/28 12:42:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.039 max=0.081 gt_z_mean=1.141 pred_z_mean=1.123
+[02/28 12:42:56][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0113,0.0054,0.0273) norm_std=0.0138
+[02/28 12:43:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.022 max=0.035
+[02/28 12:43:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9982 pred=+0.9883 delta(pred-gt)=-0.0099
+[02/28 12:43:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.0374334 2.0976946 -0.01660619] global_orient0_aa(pred)=[-0.0261304 1.8297232 -0.00515052]
+[02/28 12:43:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.03294645 2.990605 -0.03862115]
+[02/28 12:43:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.22,+1.57,+1.14) pred=(+104.85,-0.59,-1.18) pred_vs_gt=(-15.38,-0.93,+3.03)
+[02/28 12:43:03][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+11.87
+[02/28 12:43:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:43:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:43:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:43:57][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:43:57][INFO] ✅[FIT][Epoch 0] finished! 01:13→1:00:16 | loss_epoch=1.71
+[02/28 12:43:57][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/28 12:43:57][INFO] [LossBreakdown] body_pose=1.8069 betas=0.4865 go_c=0.0792 go_gv=0.0337 transl_vel=0.0845 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=14.19/19.88 go_gv_deg(mean/max)=14.69/19.54 tv_abs_mean(pred)=0.0003 tv_abs_mean(tgt)=0.0012 tv_zero_frac(pred)=0.00 static_conf_mean=0.91 static_conf_hi_frac=0.84
+[02/28 12:43:57][INFO] [IncamDebug] transl_c_loss=0.0060 pred_cam_abs_mean=0.785 gt_pred_cam_abs_mean=0.815 valid_frac=1.00
+[02/28 12:44:02][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(367.8897399902344, 215.44203186035156, 740.2937622070312, 1724.3988037109375, 0.39666181802749634) pred=(478.94940185546875, 187.37586975097656, 912.3441162109375, 1578.9793701171875, 0.24049346148967743)
+[02/28 12:44:02][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(334.61187744140625, 214.93783569335938, 756.857421875, 1730.3028564453125, 0.2586356997489929) pred=(478.1136474609375, 183.8382568359375, 837.2825317382812, 1579.8131103515625, 0.23904208838939667)
+[02/28 12:44:02][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_transl_err_m mean=0.051 max=0.082 gt_z_mean=1.147 pred_z_mean=1.160
+[02/28 12:44:02][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0115,0.0038,0.0387) norm_std=0.0105
+[02/28 12:44:08][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_transl_err_m mean=0.025 max=0.041
+[02/28 12:44:08][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9982 pred=+0.9862 delta(pred-gt)=-0.0120
+[02/28 12:44:08][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[0.01964396 2.1108265 0.01723783] global_orient0_aa(pred)=[-0.042272 1.8178447 0.03007535]
+[02/28 12:44:08][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[0.03093894 2.9828596 0.00903419]
+[02/28 12:44:08][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.94,-0.25,+1.21) pred=(+104.19,-2.47,-0.74) pred_vs_gt=(-16.81,-0.53,+2.91)
+[02/28 12:44:08][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+12.13
+[02/28 12:49:32][INFO] [Exp Name]: finetune_
+[02/28 12:49:32][INFO] [GPU x Batch] = 1 x 2
+[02/28 12:49:32][INFO] [UnityDataset] Found 3 sequences.
+[02/28 12:49:32][INFO] [Train Dataset][9/9]: name=unity, size=3, genmo.datasets.unity_dataset.UnityDataset
+[02/28 12:49:32][INFO] [Train Dataset][All]: ConcatDataset size=3
+[02/28 12:49:32][INFO]
+[02/28 12:49:32][INFO] [UnityDataset] Found 3 sequences.
+[02/28 12:49:32][INFO] [Val Dataset][7/7]: name=unity_val, size=3, genmo.datasets.unity_dataset.UnityDataset
+[02/28 12:49:32][INFO]
+[02/28 12:49:39][INFO] [PL-Trainer] Loading ckpt: ./s050000.ckpt
+[02/28 12:50:05][INFO] [Simple Ckpt Saver]: Save to `outputs/unity/finetune_/version_0/checkpoints'
+[02/28 12:50:14][INFO] Start Fitting...
+[02/28 12:50:15][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/utilities/model_summary/model_summary.py:242: Precision 16-mixed is not supported by the model summary. Estimated model size in MB will not be accurate. Using 32 bits instead.
+
+[02/28 12:50:15][INFO] 🚀[FIT][Epoch 0] Data: unity Experiment: finetune_
+[02/28 12:50:16][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/nn/modules/conv.py:306: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return F.conv1d(input, weight, bias, self.stride,
+
+[02/28 12:50:17][INFO] [LossBreakdown] body_pose=1.4168 betas=0.4625 go_c=0.0874 go_gv=0.0300 transl_vel=0.1418 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=9.78/44.05 go_gv_deg(mean/max)=11.09/42.41 tv_abs_mean(pred)=0.0006 tv_abs_mean(tgt)=0.0017 tv_zero_frac(pred)=0.00 static_conf_mean=0.65 static_conf_hi_frac=0.46
+[02/28 12:50:17][INFO] [IncamDebug] transl_c_loss=0.0044 pred_cam_abs_mean=0.830 gt_pred_cam_abs_mean=0.854 valid_frac=1.00
+[02/28 12:50:17][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/torch/autograd/graph.py:744: UserWarning: Plan failed with a cudnnException: CUDNN_BACKEND_EXECUTION_PLAN_DESCRIPTOR: cudnnFinalize Descriptor Failed cudnn_status: CUDNN_STATUS_NOT_SUPPORTED (Triggered internally at ../aten/src/ATen/native/cudnn/Conv_v8.cpp:919.)
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
+
+[02/28 12:50:27][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(349.9327697753906, 212.91502380371094, 774.9888916015625, 1723.708740234375, 0.26095789670944214) pred=(477.20745849609375, 178.5178680419922, 769.7216796875, 1674.762451171875, 0.2448475956916809)
+[02/28 12:50:27][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(332.5797424316406, 212.5543212890625, 824.0524291992188, 1723.397705078125, 0.25486209988594055) pred=(471.34783935546875, 124.71530151367188, 815.3561401367188, 1803.4664306640625, 0.25732946395874023)
+[02/28 12:50:27][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_transl_err_m mean=0.047 max=0.089 gt_z_mean=1.141 pred_z_mean=1.121
+[02/28 12:50:27][INFO] [VisUnityVal] e000_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0115,0.0053,0.0282) norm_std=0.0140
+[02/28 12:50:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_transl_err_m mean=0.021 max=0.035
+[02/28 12:50:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech root_y0: gt=+0.9982 pred=+0.9885 delta(pred-gt)=-0.0097
+[02/28 12:50:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(gt)=[ 0.0374334 2.0976946 -0.01660619] global_orient0_aa(pred)=[-0.03011364 1.8607007 0.00665575]
+[02/28 12:50:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[ 0.03294645 2.990605 -0.03862115]
+[02/28 12:50:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.22,+1.57,+1.14) pred=(+106.63,-1.15,-1.00) pred_vs_gt=(-13.60,-0.48,+3.43)
+[02/28 12:50:32][INFO] [VisUnityVal] e000_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+10.32
+[02/28 12:51:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pa_mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:51:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/mpjpe', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:51:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/pve', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:51:24][WARNING] /root/miniconda3/envs/gvhmr/lib/python3.10/site-packages/pytorch_lightning/trainer/connectors/logger_connector/result.py:433: It is recommended to use `self.log('val_metric_Unity/accel', ..., sync_dist=True)` when logging on epoch level in distributed setting to accumulate the metric across devices.
+
+[02/28 12:51:24][INFO] ✅[FIT][Epoch 0] finished! 01:09→57:01 | loss_epoch=1.72
+[02/28 12:51:24][INFO] 🚀[FIT][Epoch 1] Data: unity Experiment: finetune_
+[02/28 12:51:24][INFO] [LossBreakdown] body_pose=1.8170 betas=0.6284 go_c=0.0743 go_gv=0.0288 transl_vel=0.0889 gogv_mask=1.00 tv_mask=1.00 go_c_deg(mean/max)=12.98/18.69 go_gv_deg(mean/max)=13.35/17.98 tv_abs_mean(pred)=0.0003 tv_abs_mean(tgt)=0.0012 tv_zero_frac(pred)=0.00 static_conf_mean=0.92 static_conf_hi_frac=0.86
+[02/28 12:51:24][INFO] [IncamDebug] transl_c_loss=0.0077 pred_cam_abs_mean=0.786 gt_pred_cam_abs_mean=0.815 valid_frac=1.00
+[02/28 12:51:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof f0 gt=(367.8897399902344, 215.44203186035156, 740.2937622070312, 1724.3988037109375, 0.39666181802749634) pred=(427.47857666015625, 145.2262725830078, 746.046630859375, 1544.64306640625, 0.3625544309616089)
+[02/28 12:51:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_proj_bbox_oof fl gt=(334.61187744140625, 214.93783569335938, 756.857421875, 1730.3028564453125, 0.2586356997489929) pred=(464.0299377441406, 163.49838256835938, 749.2650146484375, 1569.6798095703125, 0.2358490526676178)
+[02/28 12:51:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_transl_err_m mean=0.082 max=0.129 gt_z_mean=1.147 pred_z_mean=1.133
+[02/28 12:51:30][INFO] [VisUnityVal] e001_0_biboo_birthday_speech incam_delta_transl_std_m xyz=(0.0139,0.0110,0.0406) norm_std=0.0202
+[02/28 12:51:34][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_transl_err_m mean=0.026 max=0.039
+[02/28 12:51:34][INFO] [VisUnityVal] e001_0_biboo_birthday_speech root_y0: gt=+0.9982 pred=+0.9845 delta(pred-gt)=-0.0137
+[02/28 12:51:34][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(gt)=[0.01964396 2.1108265 0.01723783] global_orient0_aa(pred)=[ 0.00380696 1.7861054 -0.02808194]
+[02/28 12:51:34][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_aa(raw_gt_world)=[0.03093894 2.9828596 0.00903419]
+[02/28 12:51:34][INFO] [VisUnityVal] e001_0_biboo_birthday_speech global_orient0_yxz_deg gt=(+120.94,-0.25,+1.21) pred=(+102.33,+1.21,-0.73) pred_vs_gt=(-18.59,-2.42,-0.26)
+[02/28 12:51:34][INFO] [VisUnityVal] e001_0_biboo_birthday_speech yaw0_deg(pred_vs_gt)=+14.40
diff --git a/train.sh b/train.sh
new file mode 100644
index 0000000000000000000000000000000000000000..a94b40b132587aa8c24338c5569308026afbe95d
--- /dev/null
+++ b/train.sh
@@ -0,0 +1 @@
+python scripts/train.py --config-name finetune_unity ckpt_path=./s050000.ckpt
diff --git a/verify_rotation_convention.py b/verify_rotation_convention.py
new file mode 100644
index 0000000000000000000000000000000000000000..1e21d847b74218098bd6d040753213da091e2b28
--- /dev/null
+++ b/verify_rotation_convention.py
@@ -0,0 +1,71 @@
+#!/usr/bin/env python3
+"""
+Verify that our rotation conversion matches GENMO's expected convention.
+GENMO expects: global_orient_c = R_w2c @ R_pel_w (in CV coordinates)
+"""
+import numpy as np
+from scipy.spatial.transform import Rotation as R
+
+def test_rotation_order():
+ """Test that R_w2c @ R_pel_w gives correct incam rotation."""
+
+ # Setup coordinate conversion
+ C = np.diag([1.0, -1.0, 1.0]) # Unity -> CV (flip Y axis)
+
+ # Example: Camera facing +Z in Unity, Pelvis rotated 90° around Y
+ cam_quat_unity = np.array([0.0, 0.0, 0.0, 1.0]) # Identity
+ pel_quat_unity = np.array([0.0, 0.707, 0.0, 0.707]) # 90° around Y
+
+ # OLD METHOD (wrong order - convert to CV after computing relative rotation)
+ R_cam_w_unity_old = R.from_quat(cam_quat_unity).as_matrix()
+ R_pel_w_unity_old = R.from_quat(pel_quat_unity).as_matrix()
+ R_rel_unity_old = R_cam_w_unity_old.T @ R_pel_w_unity_old
+ R_cv_old = C @ R_rel_unity_old @ C
+
+ # NEW METHOD (correct - convert to CV FIRST, then compute relative rotation)
+ R_cam_w_unity_new = R.from_quat(cam_quat_unity).as_matrix()
+ R_pel_w_unity_new = R.from_quat(pel_quat_unity).as_matrix()
+
+ R_cam_w_cv = C @ R_cam_w_unity_new @ C # Camera-to-world in CV
+ R_w2c_cv = R_cam_w_cv.T # World-to-camera in CV
+ R_pel_w_cv = C @ R_pel_w_unity_new @ C # Pelvis-to-world in CV
+
+ R_pel_c_cv = R_w2c_cv @ R_pel_w_cv # GENMO's expected formula
+
+ # Compare
+ euler_old = R.from_matrix(R_cv_old).as_euler('YXZ', degrees=True)
+ euler_new = R.from_matrix(R_pel_c_cv).as_euler('YXZ', degrees=True)
+
+ print("=== Rotation Order Verification ===")
+ print(f"OLD (Unity rel then CV): yaw={euler_old[0]:7.2f}, pitch={euler_old[1]:7.2f}, roll={euler_old[2]:7.2f}")
+ print(f"NEW (CV first then rel): yaw={euler_new[0]:7.2f}, pitch={euler_new[1]:7.2f}, roll={euler_new[2]:7.2f}")
+ print(f"Difference: yaw={euler_new[0]-euler_old[0]:7.2f}, pitch={euler_new[1]-euler_old[1]:7.2f}, roll={euler_new[2]-euler_old[2]:7.2f}")
+ print()
+
+ # Verify matrix multiplication order
+ print("Matrix multiplication order check:")
+ print(f" R_w2c @ R_pel_w (GENMO convention):")
+ print(f" - Takes pelvis in world coords")
+ print(f" - Rotates it into camera coords")
+ print(f" - Result: global_orient_c")
+ print()
+
+ # Test with actual rotation
+ test_vec_w = np.array([1.0, 0.0, 0.0]) # +X in world
+ test_vec_c = R_w2c_cv @ (R_pel_w_cv @ test_vec_w) # Apply pelvis rotation, then camera transform
+ test_vec_direct = R_pel_c_cv @ test_vec_w # Direct multiplication
+
+ print(f"Vector transformation test:")
+ print(f" Original vector (world): {test_vec_w}")
+ print(f" Via two steps: {test_vec_c}")
+ print(f" Via R_w2c @ R_pel_w: {test_vec_direct}")
+ print(f" Match: {np.allclose(test_vec_c, test_vec_direct)}")
+ print()
+
+ print("Expected results after reprocessing:")
+ print(" - In-camera rotation errors should be <10° (all axes)")
+ print(" - No 168° roll offset")
+ print(" - Model predictions should closely match GT")
+
+if __name__ == "__main__":
+ test_rotation_order()
diff --git a/videos/VRM_JG6Z7WA.txt b/videos/VRM_JG6Z7WA.txt
new file mode 100644
index 0000000000000000000000000000000000000000..25c65801ad44dfc0590c1cac379d227e537e5bb7
--- /dev/null
+++ b/videos/VRM_JG6Z7WA.txt
@@ -0,0 +1,2 @@
+00:00:11 - 00:01:45
+
diff --git a/videos/download.sh b/videos/download.sh
new file mode 100644
index 0000000000000000000000000000000000000000..3da4f9e479e69e70f425993e2348dddd673a8844
--- /dev/null
+++ b/videos/download.sh
@@ -0,0 +1,4 @@
+yt-dlp \
+ -f "(bv*[vcodec^=av01]/bv*[vcodec^=vp9]/bv*[vcodec^=avc1])+ba/best" \
+ --merge-output-format mp4 \
+ "https://www.youtube.com/watch?v=VRM_JG6Z7WA"
diff --git a/videos/process_video.sh b/videos/process_video.sh
new file mode 100644
index 0000000000000000000000000000000000000000..57e6562c9439efc9eb7f964e62412987f3d8ee7b
--- /dev/null
+++ b/videos/process_video.sh
@@ -0,0 +1,7 @@
+# Clip with high quality settings using system ffmpeg (has AV1 decoder + libx264).
+FFMPEG_BIN="/usr/bin/ffmpeg"
+if [ ! -x "$FFMPEG_BIN" ]; then
+ FFMPEG_BIN="ffmpeg"
+fi
+
+"$FFMPEG_BIN" -y -ss 00:00:16 -to 00:00:22 -i videos/video_101_biboo_birthday_speech_explosion_2.mp4 -c:v libx264 -b:v 5M -c:a copy videos/test_10.mp4
\ No newline at end of file