diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/REPORT.md b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/REPORT.md new file mode 100644 index 0000000000000000000000000000000000000000..7f2af02a5f64a7ae2fdb82b2a93c6eaac45fb57d --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/REPORT.md @@ -0,0 +1,20 @@ +# QwenOFT mean-trained checkpoints under profile simulation + +Four final step-5000 H1 checkpoints; two rounds, one evaluation per physical GPU2/3,100 episodes each (400 total). + +The training latency was fixed mean; this evaluation samples the complete archived RTX3090 temporal hidden-regime profile. Simulator FPS, seeds, horizon, limits and model/profile identities are in evaluation-plan.json. Standard deviations below use ddof=0. Returns have task-specific scales. Startup checks are separate and excluded. + +| Task | Episodes | Return mean +/- SD | Length mean +/- SD | Success | Invalid | +|---|---:|---:|---:|---:|---:| +| flappy | 100 | 384.824005 +/- 116.787774 | 3119.31 +/- 939.87 | not provided by task | 0 | +| deadly_corridor | 100 | 1620.798776 +/- 913.624278 | 148.53 +/- 49.46 | not provided by task | 0 | +| ant | 100 | 1453.844064 +/- 693.727520 | 803.85 +/- 328.81 | not provided by task | 0 | +| intercept | 100 | 3.544349 +/- 7.071923 | 60.00 +/- 0.00 | 9/100 | 0 | + +No success metric is invented for Flappy/Deadly/Ant. Intercept reports the native accumulated success flag. No policy-quality acceptance gate is claimed. + +Compatibility repairs: portable robot_type copied from each actual training manifest (weights unchanged); official ViZDoom1.2.4 VizdoomCorridor-v0 uses the same deadly_corridor WAD as SF, preserves render contract and semantic seven-button ordering; public action space is equivalent MultiBinary7. Existing native render/button/history tests passed. Full eval source/patch and original profile assets are archived. + +Flappy/Deadly seeds1000000..1000099; Ant42..141; Intercept4242424242..4242424341. Latency seed271828. Flappy10/10Hz, Deadly35/8.75Hz, Ant10/10Hz, Intercept20/20Hz. Max raw frames3600/3600/1000/60; capacities1. MIKASA H1 holds last chunk action; no prefix, no DAgger. Ant keeps its training prompt label1 while execution latency is sampled. + +Raw JSONL logs are losslessly gzip-compressed for distribution; original uncompressed records remain on the experiment host. Empty observation_attempts files are retained; admission/drop evidence is in steps/actions. Per-task CSV and full400 episode CSV are provided. diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/all_episodes.csv b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/all_episodes.csv new file mode 100644 index 0000000000000000000000000000000000000000..be5e27101cb97f6a2dd0d85a0399d3fb8a5ea839 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/all_episodes.csv @@ -0,0 +1,401 @@ +task,episode_id,seed,return_env,length,mean_latency_ms,success +flappy,0,1000000,444.6000052243471,3600,76.02271694866694, +flappy,1,1000001,444.6000052243471,3600,76.14445348705047, +flappy,2,1000002,444.6000052243471,3600,75.83047266244563, +flappy,3,1000003,444.6000052243471,3600,76.04121221698036, +flappy,4,1000004,444.6000052243471,3600,75.7789115791707, +flappy,5,1000005,228.2000027000904,1861,76.22757676162651, +flappy,6,1000006,444.6000052243471,3600,75.98373978309758, +flappy,7,1000007,444.6000052243471,3600,75.85552109823348, +flappy,8,1000008,444.6000052243471,3600,75.9782303085917, +flappy,9,1000009,444.6000052243471,3600,75.94667987356688, +flappy,10,1000010,444.6000052243471,3600,75.66396359484234, +flappy,11,1000011,444.6000052243471,3600,75.7794525026407, +flappy,12,1000012,444.6000052243471,3600,75.90110110734818, +flappy,13,1000013,444.6000052243471,3600,76.01870178237883, +flappy,14,1000014,444.6000052243471,3600,75.75567207010911, +flappy,15,1000015,444.6000052243471,3600,75.83026036637241, +flappy,16,1000016,444.6000052243471,3600,75.74502908171665, +flappy,17,1000017,444.6000052243471,3600,75.84316844302293, +flappy,18,1000018,444.6000052243471,3600,75.85876738771161, +flappy,19,1000019,265.50000313669443,2162,75.89492798135642, +flappy,20,1000020,444.6000052243471,3600,75.90859756288593, +flappy,21,1000021,444.6000052243471,3600,75.93474621914784, +flappy,22,1000022,444.6000052243471,3600,75.77022360156529, +flappy,23,1000023,444.6000052243471,3600,75.8506098974935, +flappy,24,1000024,444.6000052243471,3600,75.80511776716725, +flappy,25,1000025,116.00000138580799,955,76.07937915327228, +flappy,26,1000026,444.6000052243471,3600,75.77409482659607, +flappy,27,1000027,444.6000052243471,3600,75.82354466933252, +flappy,28,1000028,444.6000052243471,3600,75.92578714415393, +flappy,29,1000029,444.6000052243471,3600,75.77326038618416, +flappy,30,1000030,256.0000030249357,2085,75.8461606092662, +flappy,31,1000031,444.6000052243471,3600,75.87053786258159, +flappy,32,1000032,444.6000052243471,3600,75.90930861144982, +flappy,33,1000033,444.6000052243471,3600,75.80530422686525, +flappy,34,1000034,444.6000052243471,3600,76.05997569829616, +flappy,35,1000035,444.6000052243471,3600,75.67579907153437, +flappy,36,1000036,444.6000052243471,3600,76.07561842170198, +flappy,37,1000037,444.6000052243471,3600,75.87459102177027, +flappy,38,1000038,55.60000067949295,468,75.8887188983619, +flappy,39,1000039,444.6000052243471,3600,75.86536772802552, +flappy,40,1000040,432.900005094707,3512,76.00355652525975, +flappy,41,1000041,274.8000032454729,2237,75.7658282850597, +flappy,42,1000042,264.90000312775373,2156,75.95264956954799, +flappy,43,1000043,265.4000031352043,2161,75.82748305801191, +flappy,44,1000044,444.6000052243471,3600,75.9295822845668, +flappy,45,1000045,143.90000171214342,1180,75.94310218110371, +flappy,46,1000046,444.6000052243471,3600,75.69568531179425, +flappy,47,1000047,93.1000011190772,771,76.0527875505066, +flappy,48,1000048,56.10000068694353,473,76.20882901957174, +flappy,49,1000049,265.2000031322241,2159,76.05401077635972, +flappy,50,1000050,444.6000052243471,3600,75.89333271844873, +flappy,51,1000051,444.6000052243471,3600,75.89090159365671, +flappy,52,1000052,398.80000469088554,3234,75.91218218803246, +flappy,53,1000053,444.6000052243471,3600,75.86400590251726, +flappy,54,1000054,270.2000031918287,2200,76.01590238337654, +flappy,55,1000055,69.70000084489584,582,75.68422480575155, +flappy,56,1000056,444.6000052243471,3600,75.87884524455251, +flappy,57,1000057,444.6000052243471,3600,75.96981187494319, +flappy,58,1000058,444.6000052243471,3600,76.03772455115222, +flappy,59,1000059,437.90000515431166,3553,76.04088529786887, +flappy,60,1000060,348.9000041112304,2834,75.79422825165413, +flappy,61,1000061,444.6000052243471,3600,75.87728099437057, +flappy,62,1000062,78.9000009521842,656,75.96352981662133, +flappy,63,1000063,444.6000052243471,3600,75.80269270184165, +flappy,64,1000064,444.6000052243471,3600,75.88518180564401, +flappy,65,1000065,444.6000052243471,3600,75.87533034544981, +flappy,66,1000066,444.6000052243471,3600,75.94241138050401, +flappy,67,1000067,444.6000052243471,3600,75.95312277771471, +flappy,68,1000068,444.6000052243471,3600,75.8998829764233, +flappy,69,1000069,444.6000052243471,3600,75.98564617573034, +flappy,70,1000070,444.6000052243471,3600,75.68328575087021, +flappy,71,1000071,135.0000016093254,1109,75.99546963217229, +flappy,72,1000072,444.6000052243471,3600,75.9923106611263, +flappy,73,1000073,444.6000052243471,3600,75.80422251719546, +flappy,74,1000074,444.6000052243471,3600,75.95469853250815, +flappy,75,1000075,444.6000052243471,3600,75.74551875442629, +flappy,76,1000076,444.6000052243471,3600,75.93301571087362, +flappy,77,1000077,444.6000052243471,3600,75.98384926019328, +flappy,78,1000078,444.6000052243471,3600,75.85055115368883, +flappy,79,1000079,444.6000052243471,3600,75.97142616222317, +flappy,80,1000080,444.6000052243471,3600,75.97039764106849, +flappy,81,1000081,444.6000052243471,3600,75.74469321422862, +flappy,82,1000082,116.20000138878822,957,76.0366526049804, +flappy,83,1000083,444.6000052243471,3600,75.95924386190674, +flappy,84,1000084,444.6000052243471,3600,76.0310580385874, +flappy,85,1000085,36.60000045597553,314,75.8634823847272, +flappy,86,1000086,260.90000308305025,2125,75.91783880059099, +flappy,87,1000087,444.6000052243471,3600,76.16752514785735, +flappy,88,1000088,444.6000052243471,3600,75.83331254385584, +flappy,89,1000089,444.6000052243471,3600,76.2113387300584, +flappy,90,1000090,444.6000052243471,3600,75.8334932097261, +flappy,91,1000091,225.20000265538692,1831,75.75618859671614, +flappy,92,1000092,444.6000052243471,3600,75.79683788505955, +flappy,93,1000093,180.9000021442771,1478,75.96465307644473, +flappy,94,1000094,305.2000035941601,2478,75.86565754734926, +flappy,95,1000095,444.6000052243471,3600,75.74523644464854, +flappy,96,1000096,444.6000052243471,3600,75.94353591524424, +flappy,97,1000097,444.6000052243471,3600,75.81927739599219, +flappy,98,1000098,444.6000052243471,3600,75.96229410618645, +flappy,99,1000099,444.6000052243471,3600,75.94512877548694, +deadly_corridor,0,1000000,337.47547912597656,72,71.90727374040254, +deadly_corridor,1,1000001,819.0284423828125,143,73.84762082340946, +deadly_corridor,2,1000002,2284.857650756836,182,72.79171012339609, +deadly_corridor,3,1000003,2276.2068634033203,189,76.345275285376, +deadly_corridor,4,1000004,805.2153015136719,150,73.86282581373551, +deadly_corridor,5,1000005,621.8231658935547,115,74.31105893586228, +deadly_corridor,6,1000006,2276.414749145508,176,74.2226331369995, +deadly_corridor,7,1000007,2284.310989379883,176,72.9072057957754, +deadly_corridor,8,1000008,81.07798767089844,49,73.20258272646697, +deadly_corridor,9,1000009,317.2351837158203,75,72.54774919154028, +deadly_corridor,10,1000010,2282.7608489990234,176,72.78293151689127, +deadly_corridor,11,1000011,88.11907958984375,45,72.60486105128022, +deadly_corridor,12,1000012,2281.468536376953,176,72.29193331603048, +deadly_corridor,13,1000013,2276.6868591308594,178,72.73330265771509, +deadly_corridor,14,1000014,2276.1705932617188,178,73.30067987408609, +deadly_corridor,15,1000015,2282.6631622314453,177,72.49405489224537, +deadly_corridor,16,1000016,2280.300033569336,172,72.80884970803692, +deadly_corridor,17,1000017,2280.4182891845703,182,73.03539182090206, +deadly_corridor,18,1000018,2281.2594451904297,177,72.50972089313564, +deadly_corridor,19,1000019,479.8523712158203,99,72.58046231642126, +deadly_corridor,20,1000020,2279.7379455566406,181,72.47468246266928, +deadly_corridor,21,1000021,2284.9097442626953,197,83.18983231769475, +deadly_corridor,22,1000022,2286.2730407714844,172,72.84281562147524, +deadly_corridor,23,1000023,244.51919555664062,74,76.23461799191558, +deadly_corridor,24,1000024,2279.957275390625,195,72.94927214021655, +deadly_corridor,25,1000025,2283.952178955078,179,73.18968843008061, +deadly_corridor,26,1000026,2276.701370239258,178,72.87702909462648, +deadly_corridor,27,1000027,2277.142562866211,190,72.45412386128042, +deadly_corridor,28,1000028,2279.025634765625,177,74.11102172804317, +deadly_corridor,29,1000029,2285.7152099609375,177,71.63189230597281, +deadly_corridor,30,1000030,53.374298095703125,44,72.51418721312025, +deadly_corridor,31,1000031,2279.8080444335938,183,72.72702656843174, +deadly_corridor,32,1000032,2282.307357788086,178,74.33584751930213, +deadly_corridor,33,1000033,2282.834014892578,192,73.95005063555192, +deadly_corridor,34,1000034,2284.200241088867,188,76.29368894499888, +deadly_corridor,35,1000035,2287.2159118652344,179,72.81890806090988, +deadly_corridor,36,1000036,2284.693832397461,183,76.28284599973325, +deadly_corridor,37,1000037,2283.2066650390625,178,72.1797344044525, +deadly_corridor,38,1000038,2281.032196044922,178,73.74343783824916, +deadly_corridor,39,1000039,2282.960678100586,190,73.24816830891406, +deadly_corridor,40,1000040,2287.094253540039,185,72.35711232966574, +deadly_corridor,41,1000041,2279.3030853271484,179,72.42125368367608, +deadly_corridor,42,1000042,440.0892791748047,104,73.92064892672727, +deadly_corridor,43,1000043,2280.8592529296875,177,72.36020918178356, +deadly_corridor,44,1000044,2283.4308471679688,189,75.93658060557208, +deadly_corridor,45,1000045,2282.324264526367,181,73.54224681770178, +deadly_corridor,46,1000046,326.0184631347656,74,73.1983876441008, +deadly_corridor,47,1000047,2279.086135864258,182,73.00958120503027, +deadly_corridor,48,1000048,2280.3804626464844,179,73.17268244992928, +deadly_corridor,49,1000049,2276.215301513672,189,75.47590644230628, +deadly_corridor,50,1000050,2278.132034301758,182,74.50495464842548, +deadly_corridor,51,1000051,2285.6056518554688,181,73.41699294418743, +deadly_corridor,52,1000052,2287.240921020508,173,73.22110809114655, +deadly_corridor,53,1000053,310.81517028808594,73,74.09003681120738, +deadly_corridor,54,1000054,2276.6219787597656,175,72.98605010243534, +deadly_corridor,55,1000055,2276.2769470214844,194,75.17704077845171, +deadly_corridor,56,1000056,2278.861602783203,178,72.97353037051572, +deadly_corridor,57,1000057,2279.728561401367,181,73.96913002154926, +deadly_corridor,58,1000058,2280.544464111328,176,73.02432805290651, +deadly_corridor,59,1000059,487.829833984375,108,78.89398217393664, +deadly_corridor,60,1000060,567.0655517578125,113,72.64874721482185, +deadly_corridor,61,1000061,2278.210220336914,177,72.96068484971086, +deadly_corridor,62,1000062,2281.436721801758,186,75.46710866924751, +deadly_corridor,63,1000063,382.2119903564453,89,81.21157315209366, +deadly_corridor,64,1000064,246.2946014404297,70,73.9736408486285, +deadly_corridor,65,1000065,285.21240234375,76,73.13661133681993, +deadly_corridor,66,1000066,310.6737365722656,75,73.40468658737086, +deadly_corridor,67,1000067,346.1162872314453,75,72.1929723632303, +deadly_corridor,68,1000068,804.7056121826172,150,73.76397959753224, +deadly_corridor,69,1000069,2285.6442108154297,184,75.13255757158333, +deadly_corridor,70,1000070,730.5995788574219,132,73.25446825350764, +deadly_corridor,71,1000071,86.91796875,47,76.28335745963689, +deadly_corridor,72,1000072,60.30122375488281,44,76.83513093208644, +deadly_corridor,73,1000073,768.6264343261719,141,77.27057350071598, +deadly_corridor,74,1000074,2280.1071166992188,172,74.16699734355548, +deadly_corridor,75,1000075,860.9334106445312,151,73.15118478347584, +deadly_corridor,76,1000076,722.9459228515625,143,75.5655785931314, +deadly_corridor,77,1000077,2276.8687438964844,182,72.95102474014934, +deadly_corridor,78,1000078,368.3357238769531,79,71.51096709276341, +deadly_corridor,79,1000079,-76.45918273925781,17,72.24888432102617, +deadly_corridor,80,1000080,2281.5543823242188,183,73.32589540463356, +deadly_corridor,81,1000081,2281.6688842773438,171,73.10600900440717, +deadly_corridor,82,1000082,2277.5223083496094,178,73.55648700566698, +deadly_corridor,83,1000083,42.30937194824219,41,73.52700344736942, +deadly_corridor,84,1000084,2285.8980407714844,176,71.98655161011203, +deadly_corridor,85,1000085,68.90191650390625,45,72.84773487604696, +deadly_corridor,86,1000086,2286.2190551757812,171,72.82303966497733, +deadly_corridor,87,1000087,281.1173553466797,76,72.26983276661764, +deadly_corridor,88,1000088,2283.1607971191406,175,73.49638264342678, +deadly_corridor,89,1000089,2277.888946533203,177,73.44736473371472, +deadly_corridor,90,1000090,429.36326599121094,93,71.86172378947977, +deadly_corridor,91,1000091,252.0751953125,70,72.26459581736903, +deadly_corridor,92,1000092,2278.306442260742,192,80.97328482778371, +deadly_corridor,93,1000093,2285.236801147461,175,74.02717585214627, +deadly_corridor,94,1000094,857.2727355957031,152,85.59110000526613, +deadly_corridor,95,1000095,2275.9288024902344,199,73.62958803645523, +deadly_corridor,96,1000096,2286.8704833984375,179,72.31519682456816, +deadly_corridor,97,1000097,2278.048355102539,181,73.50330330803081, +deadly_corridor,98,1000098,2277.4480743408203,178,76.78472725777, +deadly_corridor,99,1000099,2276.9671478271484,178,77.9678189026336, +ant,0,42,1846.1103431567394,1000,89.89614608291177, +ant,1,43,2415.720790707953,1000,90.00308114332259, +ant,2,44,457.34421085068755,177,89.83716885697598, +ant,3,45,1421.7952163289683,1000,89.87909631338808, +ant,4,46,2037.7234409469488,937,89.82685347370092, +ant,5,47,2330.630175869275,1000,90.47193606091501, +ant,6,48,1161.643572255748,429,89.84194070141322, +ant,7,49,2351.1524624990343,1000,89.92640891799017, +ant,8,50,513.2964809479813,210,89.88895656571908, +ant,9,51,1126.8652528911032,660,89.91361550654544, +ant,10,52,1693.436933192597,1000,89.84960962337662, +ant,11,53,948.3780972955639,1000,89.94678527711802, +ant,12,54,2322.052445211472,1000,90.11873818885832, +ant,13,55,960.4026770814776,1000,90.93377411320307, +ant,14,56,1464.564005196777,1000,89.80893705661644, +ant,15,57,1110.548792782156,1000,89.99466844889166, +ant,16,58,2246.207900740156,1000,90.1624262080728, +ant,17,59,85.64836938561511,60,89.87005518664785, +ant,18,60,340.54799067574436,143,89.94240076131771, +ant,19,61,2457.088748930458,1000,89.92054036086635, +ant,20,62,2166.2512677098603,1000,89.95452553058773, +ant,21,63,2357.957592244385,1000,89.86780458600198, +ant,22,64,1654.8780938737275,871,90.08433827425095, +ant,23,65,1499.367100151414,1000,89.89663615668341, +ant,24,66,2297.4032619179525,1000,90.09818426014289, +ant,25,67,1253.360764666355,543,89.9390124443734, +ant,26,68,1221.270312709775,1000,89.84986177450952, +ant,27,69,2389.2476464763376,1000,89.95772586857817, +ant,28,70,1682.5290233886233,707,89.76145439054764, +ant,29,71,2474.676425615127,1000,89.82093759631324, +ant,30,72,382.9231146443659,256,90.69916524888657, +ant,31,73,1837.8126619276347,1000,90.03642087221974, +ant,32,74,227.19436616673684,101,89.8553742761573, +ant,33,75,1700.6312067622644,1000,89.80097198453268, +ant,34,76,960.9452812639541,372,89.8420903148968, +ant,35,77,2290.6720141359438,1000,89.91770573449698, +ant,36,78,328.5729178056416,162,90.01187187392946, +ant,37,79,1180.073938772476,1000,89.81938304804656, +ant,38,80,817.4190215442345,363,89.85140773938038, +ant,39,81,1651.2255208727013,1000,91.17610023451576, +ant,40,82,1428.174672693164,1000,89.8551155619885, +ant,41,83,1627.3838925098842,1000,90.55986754698809, +ant,42,84,1079.756369746183,680,90.17098553312343, +ant,43,85,2173.9447393037276,1000,89.84319301261918, +ant,44,86,409.90633829945847,160,89.66802828269809, +ant,45,87,2467.2636019929073,1000,89.90844708827387, +ant,46,88,657.4084558813478,248,89.86487149424892, +ant,47,89,974.7436031610902,1000,89.76305094278182, +ant,48,90,1510.5184342975385,1000,90.24355118464125, +ant,49,91,602.2339441184535,260,89.7103209703719, +ant,50,92,760.9784375126189,316,89.8206829517188, +ant,51,93,1941.172113330597,1000,90.30785204408768, +ant,52,94,624.3590446196446,281,89.94582387208622, +ant,53,95,2163.4347041279893,1000,89.84848132390947, +ant,54,96,1126.9957963444238,1000,89.84637728060243, +ant,55,97,1405.131632695366,1000,90.18855922596491, +ant,56,98,1206.2916757636292,1000,89.78065539051504, +ant,57,99,2392.7980761515178,1000,89.76388668266138, +ant,58,100,964.0216541467705,1000,89.82618651237911, +ant,59,101,2252.192880003706,1000,89.8197082349776, +ant,60,102,2471.9158497657563,1000,89.96642568195992, +ant,61,103,1902.8491241623092,1000,89.87542708971246, +ant,62,104,1435.6661382989703,1000,90.28644124851098, +ant,63,105,1668.3237703695809,1000,89.86433221097877, +ant,64,106,1813.291243529155,1000,89.85118001877315, +ant,65,107,446.72353548541076,189,89.8309544306309, +ant,66,108,130.84194814079504,74,89.82416773165995, +ant,67,109,2315.857153770824,1000,90.25959750757508, +ant,68,110,288.3915792961347,116,90.09271984792927, +ant,69,111,894.0228631227924,1000,89.89663691508213, +ant,70,112,2030.322535823717,1000,89.84028619017428, +ant,71,113,507.9449555916754,215,90.57267432538549, +ant,72,114,2377.7373967468293,1000,89.84919425782105, +ant,73,115,897.3077114027096,1000,89.90431472264346, +ant,74,116,1454.612590266188,1000,91.19515970740413, +ant,75,117,2292.457960175467,1000,89.8333901030839, +ant,76,118,1424.378337790017,1000,89.88029014661089, +ant,77,119,1441.1111023164538,1000,89.79844243631413, +ant,78,120,1265.4771503717611,1000,89.86009503143968, +ant,79,121,1662.8808067819505,1000,90.5661722205243, +ant,80,122,2508.917122342891,1000,89.87403626041336, +ant,81,123,1655.3510139158748,1000,90.05039760075688, +ant,82,124,1387.3843721247736,821,90.10235730111886, +ant,83,125,646.4356689469432,271,89.7559653760994, +ant,84,126,2172.801064037805,1000,89.84602989356796, +ant,85,127,165.9213897970373,72,89.66756877688618, +ant,86,128,1063.1483912161111,1000,89.77409215132576, +ant,87,129,1000.135342286622,1000,89.81900933661238, +ant,88,130,1977.2359176146426,1000,89.77653862908736, +ant,89,131,1937.1674235355138,1000,90.12773943823525, +ant,90,132,1344.7729257831547,1000,90.39620143170467, +ant,91,133,786.3379828975102,441,89.82893206036925, +ant,92,134,1391.060299752017,1000,89.86676880070257, +ant,93,135,503.300235688713,250,89.94819176115624, +ant,94,136,2446.7482357041768,1000,89.84848658183878, +ant,95,137,1171.9102336514923,1000,89.92785850220504, +ant,96,138,2356.7311711183065,1000,90.42933754946152, +ant,97,139,2356.12199478712,1000,89.82049779117614, +ant,98,140,1389.2987977192308,1000,90.82209581044775, +ant,99,141,967.2335383727841,1000,89.80016695371027, +intercept,0,4242424242,0.7267571190313902,60,99.89614420497905,0.0 +intercept,1,4242424243,2.9096362272975966,60,99.91707940536706,0.0 +intercept,2,4242424244,3.2060351513209753,60,97.78809018716221,0.0 +intercept,3,4242424245,0.7574528902187012,60,98.54894447730877,0.0 +intercept,4,4242424246,0.6827895979695313,60,99.11745353519741,0.0 +intercept,5,4242424247,29.923812823486514,60,99.04064156549293,1.0 +intercept,6,4242424248,0.7661087726592086,60,98.25342313549518,0.0 +intercept,7,4242424249,0.8284444468154106,60,97.98431264506286,0.0 +intercept,8,4242424250,0.9707721562881488,60,99.10671115977826,0.0 +intercept,9,4242424251,1.0944434545235708,60,99.10142489904808,0.0 +intercept,10,4242424252,0.7526731102407211,60,98.24263629181895,0.0 +intercept,11,4242424253,1.0327306617691647,60,99.90610126116793,0.0 +intercept,12,4242424254,24.08529434411321,60,99.04964452767656,1.0 +intercept,13,4242424255,0.8258126199943945,60,99.06333184347895,0.0 +intercept,14,4242424256,0.6465023508935701,60,99.96017435988418,0.0 +intercept,15,4242424257,1.2059930491086561,60,99.07293754243183,0.0 +intercept,16,4242424258,0.8975468523567542,60,99.10975490804557,0.0 +intercept,17,4242424259,0.638558203499997,60,99.9008234011206,0.0 +intercept,18,4242424260,2.473904824233614,60,99.1007534285042,0.0 +intercept,19,4242424261,0.8594156300532632,60,98.2804424689215,0.0 +intercept,20,4242424262,0.7127419076277874,60,97.42140552034121,0.0 +intercept,21,4242424263,1.1195833964738995,60,99.05356389575846,0.0 +intercept,22,4242424264,1.4589147588121705,60,99.10476263429966,0.0 +intercept,23,4242424265,22.348254217096837,60,98.14243140713285,1.0 +intercept,24,4242424266,27.43761277961312,60,99.03063235183511,1.0 +intercept,25,4242424267,0.797963114338927,60,98.95851806063928,0.0 +intercept,26,4242424268,0.6415987604705151,60,99.1202532952496,0.0 +intercept,27,4242424269,1.502438226743834,60,99.8369766656745,0.0 +intercept,28,4242424270,1.277322537265718,60,99.13638822823135,0.0 +intercept,29,4242424271,0.6413188653605175,60,99.8165233572777,0.0 +intercept,30,4242424272,26.015227647672873,60,99.93359984997578,1.0 +intercept,31,4242424273,0.7568511647114065,60,98.29798580223347,0.0 +intercept,32,4242424274,0.7758818510046694,60,95.9119617819155,0.0 +intercept,33,4242424275,0.743274000211386,60,99.16579733811342,0.0 +intercept,34,4242424276,0.9812663898337632,60,99.96913332715677,0.0 +intercept,35,4242424277,0.7364500367548317,60,98.4144170848438,0.0 +intercept,36,4242424278,0.7676261149172205,60,99.87913624991887,0.0 +intercept,37,4242424279,2.6105462690466084,60,99.0124647390605,0.0 +intercept,38,4242424280,0.8922563010128215,60,99.49582641131909,0.0 +intercept,39,4242424281,0.7909053032053635,60,99.95776157301488,0.0 +intercept,40,4242424282,27.747763212013524,60,99.89232705853966,1.0 +intercept,41,4242424283,2.830903574009426,60,99.11182141335861,0.0 +intercept,42,4242424284,3.749473527306691,60,99.89401411987875,0.0 +intercept,43,4242424285,3.2371535471174866,60,98.58941395009701,0.0 +intercept,44,4242424286,1.141169616690604,60,98.95023432158384,0.0 +intercept,45,4242424287,1.2504711685760412,60,99.8783128676535,0.0 +intercept,46,4242424288,1.1401455145678483,60,99.09364640302553,0.0 +intercept,47,4242424289,1.1743367564631626,60,98.24703755640672,0.0 +intercept,48,4242424290,0.6911400489043444,60,98.98847807253395,0.0 +intercept,49,4242424291,0.966755291854497,60,98.31297463384391,0.0 +intercept,50,4242424292,3.7725237559643574,60,98.1307167401627,0.0 +intercept,51,4242424293,0.7292428385990206,60,99.08520847604322,0.0 +intercept,52,4242424294,2.733719722367823,60,99.8949988335446,0.0 +intercept,53,4242424295,2.7277548569836654,60,99.16542541107671,0.0 +intercept,54,4242424296,0.8013565168366767,60,98.2801475641182,0.0 +intercept,55,4242424297,0.9918300381395966,60,98.74883429246843,0.0 +intercept,56,4242424298,3.8384227409260347,60,98.30485570834159,0.0 +intercept,57,4242424299,2.525593837024644,60,99.08211861473346,0.0 +intercept,58,4242424300,1.1939986812940333,60,99.95562586586023,0.0 +intercept,59,4242424301,1.1946645161951892,60,99.11887142756973,0.0 +intercept,60,4242424302,0.6632764584392135,60,99.92733404817194,0.0 +intercept,61,4242424303,0.7345126099826302,60,99.61093201950378,0.0 +intercept,62,4242424304,1.1547945403144695,60,98.88382479344455,0.0 +intercept,63,4242424305,1.0395031699445099,60,99.10455669644611,0.0 +intercept,64,4242424306,2.7713681719324086,60,99.99607387713222,0.0 +intercept,65,4242424307,3.8083399715833366,60,99.90908449191997,0.0 +intercept,66,4242424308,3.1245881704380736,60,99.86388390473627,0.0 +intercept,67,4242424309,0.9936205917911138,60,99.06530098425861,0.0 +intercept,68,4242424310,0.6479002644773573,60,97.31661851374615,0.0 +intercept,69,4242424311,1.09404552471824,60,99.06482130667098,0.0 +intercept,70,4242424312,0.725047086874838,60,99.39714496924636,0.0 +intercept,71,4242424313,2.085218493710272,60,99.91926924929075,0.0 +intercept,72,4242424314,25.112157980707707,60,99.89495984140663,1.0 +intercept,73,4242424315,0.7960666966973804,60,99.91315720008677,0.0 +intercept,74,4242424316,1.8899870013119653,60,99.8635479883608,0.0 +intercept,75,4242424317,24.77215793245705,60,99.08978442272605,1.0 +intercept,76,4242424318,0.763190906640375,60,99.1181927131107,0.0 +intercept,77,4242424319,0.8356004936795216,60,98.87472332915829,0.0 +intercept,78,4242424320,24.543561146681895,60,98.85109478338812,1.0 +intercept,79,4242424321,0.7962639288743958,60,96.64267992061197,0.0 +intercept,80,4242424322,0.6807828926102957,60,98.70551839611774,0.0 +intercept,81,4242424323,1.1704122956143692,60,99.1162675413269,0.0 +intercept,82,4242424324,0.8024117537715938,60,99.10984056283594,0.0 +intercept,83,4242424325,1.0154686415335163,60,99.90653765962175,0.0 +intercept,84,4242424326,0.6267238368745893,60,99.14313411902761,0.0 +intercept,85,4242424327,1.1180786813492887,60,99.87378109642233,0.0 +intercept,86,4242424328,1.0531825890648179,60,99.90759275984404,0.0 +intercept,87,4242424329,0.7319892354425974,60,99.07934787032669,0.0 +intercept,88,4242424330,1.1460731038823724,60,98.32584786308246,0.0 +intercept,89,4242424331,1.145515855285339,60,99.853580446333,0.0 +intercept,90,4242424332,3.1438898412743583,60,98.23875463665809,0.0 +intercept,91,4242424333,1.1678254807484336,60,99.94300368083988,0.0 +intercept,92,4242424334,1.1468605129048228,60,98.28856657896678,0.0 +intercept,93,4242424335,2.816772125195712,60,99.10609232867152,0.0 +intercept,94,4242424336,1.1577836629003286,60,99.93574594730708,0.0 +intercept,95,4242424337,1.0533778404060286,60,99.92123883389186,0.0 +intercept,96,4242424338,0.8533297177054919,60,99.06174133027585,0.0 +intercept,97,4242424339,0.7617567333800253,60,99.95316521359209,0.0 +intercept,98,4242424340,0.9895546428160742,60,99.11385213912092,0.0 +intercept,99,4242424341,0.770722996792756,60,99.44169788411487,0.0 diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.csv b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.csv new file mode 100644 index 0000000000000000000000000000000000000000..1b381df6789eea28ea56149c4b780cf0cade01c0 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.csv @@ -0,0 +1,5 @@ +task,episodes,return_mean,return_sd,length_mean,length_sd,success_count,success_rate,invalid_actions,dropped_actions +flappy,100,384.8240045265853,116.78777394316903,3119.31,939.8693174585497,,,0,64 +deadly_corridor,100,1620.7987757873534,913.6242782186637,148.53,49.455930888013825,,,0,0 +ant,100,1453.844063807972,693.7275200567642,803.85,328.8088312378486,,,0,0 +intercept,100,3.5443485127069287,7.07192296411853,60.0,0.0,9,0.09,0,10 diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.json b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.json new file mode 100644 index 0000000000000000000000000000000000000000..8486070c582599f0cb0c336c70e6569823f4990f --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.json @@ -0,0 +1,205 @@ +{ + "condition": "profile-latency", + "executor_mode": "simulated", + "latency_method": "temporal/profile_sample", + "episodes_per_checkpoint": 100, + "total_episodes": 400, + "checkpoints_metadata_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "results": { + "flappy": { + "n_episodes": 100, + "mean_return": 384.8240045265853, + "std_return": 116.78777394316903, + "min_return": 36.60000045597553, + "max_return": 444.6000052243471, + "mean_length": 3119.31, + "std_length": 939.8693174585497, + "min_length": 314.0, + "max_length": 3600.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "flappy", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "flappy", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T122201421825Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "flappy", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42", + "profile_sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 64, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy.yaml", + "execution_audit": { + "issued_action_records": 311075, + "applied_action_records": 310911, + "dropped_action_records": 64, + "nonnoop_issued_records": 30817, + "finite_action_values": true, + "latency_sample_count": 311075, + "latency_mean_ms": 75.89784633675906, + "latency_std_ms": 3.799946378622932, + "latency_p95_ms": 81.3960393048375, + "latency_p99_ms": 87.23844517488543 + } + }, + "deadly_corridor": { + "n_episodes": 100, + "mean_return": 1620.7987757873534, + "std_return": 913.6242782186637, + "min_return": -76.45918273925781, + "max_return": 2287.240921020508, + "mean_length": 148.53, + "std_length": 49.455930888013825, + "min_length": 17.0, + "max_length": 199.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "doom_deadly_corridor", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "deadly_corridor", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T171446047509Z", + "profile_ref": null, + "env_fps": 35.0, + "obs_fps": 8.75, + "frame_ms": 28.571428571428573, + "latency_type": "profile_sample", + "task": "deadly_corridor", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42", + "profile_sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor.yaml", + "execution_audit": { + "issued_action_records": 3753, + "applied_action_records": 3673, + "dropped_action_records": 0, + "nonnoop_issued_records": 3753, + "finite_action_values": true, + "latency_sample_count": 3753, + "latency_mean_ms": 74.01999621872471, + "latency_std_ms": 5.5537519652567635, + "latency_p95_ms": 89.54825982614612, + "latency_p99_ms": 95.97310052501227 + } + }, + "ant": { + "n_episodes": 100, + "mean_return": 1453.844063807972, + "std_return": 693.7275200567642, + "min_return": 85.64836938561511, + "max_return": 2508.917122342891, + "mean_length": 803.85, + "std_length": 328.8088312378486, + "min_length": 60.0, + "max_length": 1000.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "LatencyBench/AntContinuous-v0", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "ant", + "instance_id": "instance_859cf1e47bca6046", + "source_run_id": "20260911T033037730561Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "ant", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42", + "profile_sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant.yaml", + "execution_audit": { + "issued_action_records": 79573, + "applied_action_records": 79465, + "dropped_action_records": 0, + "nonnoop_issued_records": 79573, + "finite_action_values": true, + "latency_sample_count": 79573, + "latency_mean_ms": 90.00919554158884, + "latency_std_ms": 2.514492574433973, + "latency_p95_ms": 91.11971585797141, + "latency_p99_ms": 102.67108120995428 + } + }, + "intercept": { + "n_episodes": 100, + "mean_return": 3.5443485127069287, + "std_return": 7.07192296411853, + "min_return": 0.6267238368745893, + "max_return": 29.923812823486514, + "mean_length": 60.0, + "std_length": 0.0, + "min_length": 60.0, + "max_length": 60.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "mikasa_intercept_grab_fast", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "mikasa_intercept_grab_fast", + "instance_id": "instance_3a0d42681a03715c", + "source_run_id": "20260909T044501695676Z", + "profile_ref": null, + "env_fps": 20.0, + "obs_fps": 20.0, + "frame_ms": 50.0, + "latency_type": "profile_sample", + "task": "intercept", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0", + "profile_sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 10, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept.yaml", + "success_count": 9, + "success_rate": 0.09, + "execution_audit": { + "issued_action_records": 2974, + "applied_action_records": 2864, + "dropped_action_records": 10, + "nonnoop_issued_records": 2974, + "finite_action_values": true, + "latency_sample_count": 2974, + "latency_mean_ms": 99.11060319379854, + "latency_std_ms": 4.301543980874005, + "latency_p95_ms": 100.2889407458356, + "latency_p99_ms": 100.64616770379737 + } + } + }, + "quality_acceptance": "not inferred; observed statistics only" +} diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/episodes.csv b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/episodes.csv new file mode 100644 index 0000000000000000000000000000000000000000..0d65dda504fe99b22c9f6a65c0026aea21f8daf5 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/episodes.csv @@ -0,0 +1,101 @@ +episode_id,seed,return_env,length,mean_latency_ms,invalid_actions,dropped_actions +0,42,1846.1103431567394,1000,89.89614608291177,0,0 +1,43,2415.720790707953,1000,90.00308114332259,0,0 +2,44,457.34421085068755,177,89.83716885697598,0,0 +3,45,1421.7952163289683,1000,89.87909631338808,0,0 +4,46,2037.7234409469488,937,89.82685347370092,0,0 +5,47,2330.630175869275,1000,90.47193606091501,0,0 +6,48,1161.643572255748,429,89.84194070141322,0,0 +7,49,2351.1524624990343,1000,89.92640891799017,0,0 +8,50,513.2964809479813,210,89.88895656571908,0,0 +9,51,1126.8652528911032,660,89.91361550654544,0,0 +10,52,1693.436933192597,1000,89.84960962337662,0,0 +11,53,948.3780972955639,1000,89.94678527711802,0,0 +12,54,2322.052445211472,1000,90.11873818885832,0,0 +13,55,960.4026770814776,1000,90.93377411320307,0,0 +14,56,1464.564005196777,1000,89.80893705661644,0,0 +15,57,1110.548792782156,1000,89.99466844889166,0,0 +16,58,2246.207900740156,1000,90.1624262080728,0,0 +17,59,85.64836938561511,60,89.87005518664785,0,0 +18,60,340.54799067574436,143,89.94240076131771,0,0 +19,61,2457.088748930458,1000,89.92054036086635,0,0 +20,62,2166.2512677098603,1000,89.95452553058773,0,0 +21,63,2357.957592244385,1000,89.86780458600198,0,0 +22,64,1654.8780938737275,871,90.08433827425095,0,0 +23,65,1499.367100151414,1000,89.89663615668341,0,0 +24,66,2297.4032619179525,1000,90.09818426014289,0,0 +25,67,1253.360764666355,543,89.9390124443734,0,0 +26,68,1221.270312709775,1000,89.84986177450952,0,0 +27,69,2389.2476464763376,1000,89.95772586857817,0,0 +28,70,1682.5290233886233,707,89.76145439054764,0,0 +29,71,2474.676425615127,1000,89.82093759631324,0,0 +30,72,382.9231146443659,256,90.69916524888657,0,0 +31,73,1837.8126619276347,1000,90.03642087221974,0,0 +32,74,227.19436616673684,101,89.8553742761573,0,0 +33,75,1700.6312067622644,1000,89.80097198453268,0,0 +34,76,960.9452812639541,372,89.8420903148968,0,0 +35,77,2290.6720141359438,1000,89.91770573449698,0,0 +36,78,328.5729178056416,162,90.01187187392946,0,0 +37,79,1180.073938772476,1000,89.81938304804656,0,0 +38,80,817.4190215442345,363,89.85140773938038,0,0 +39,81,1651.2255208727013,1000,91.17610023451576,0,0 +40,82,1428.174672693164,1000,89.8551155619885,0,0 +41,83,1627.3838925098842,1000,90.55986754698809,0,0 +42,84,1079.756369746183,680,90.17098553312343,0,0 +43,85,2173.9447393037276,1000,89.84319301261918,0,0 +44,86,409.90633829945847,160,89.66802828269809,0,0 +45,87,2467.2636019929073,1000,89.90844708827387,0,0 +46,88,657.4084558813478,248,89.86487149424892,0,0 +47,89,974.7436031610902,1000,89.76305094278182,0,0 +48,90,1510.5184342975385,1000,90.24355118464125,0,0 +49,91,602.2339441184535,260,89.7103209703719,0,0 +50,92,760.9784375126189,316,89.8206829517188,0,0 +51,93,1941.172113330597,1000,90.30785204408768,0,0 +52,94,624.3590446196446,281,89.94582387208622,0,0 +53,95,2163.4347041279893,1000,89.84848132390947,0,0 +54,96,1126.9957963444238,1000,89.84637728060243,0,0 +55,97,1405.131632695366,1000,90.18855922596491,0,0 +56,98,1206.2916757636292,1000,89.78065539051504,0,0 +57,99,2392.7980761515178,1000,89.76388668266138,0,0 +58,100,964.0216541467705,1000,89.82618651237911,0,0 +59,101,2252.192880003706,1000,89.8197082349776,0,0 +60,102,2471.9158497657563,1000,89.96642568195992,0,0 +61,103,1902.8491241623092,1000,89.87542708971246,0,0 +62,104,1435.6661382989703,1000,90.28644124851098,0,0 +63,105,1668.3237703695809,1000,89.86433221097877,0,0 +64,106,1813.291243529155,1000,89.85118001877315,0,0 +65,107,446.72353548541076,189,89.8309544306309,0,0 +66,108,130.84194814079504,74,89.82416773165995,0,0 +67,109,2315.857153770824,1000,90.25959750757508,0,0 +68,110,288.3915792961347,116,90.09271984792927,0,0 +69,111,894.0228631227924,1000,89.89663691508213,0,0 +70,112,2030.322535823717,1000,89.84028619017428,0,0 +71,113,507.9449555916754,215,90.57267432538549,0,0 +72,114,2377.7373967468293,1000,89.84919425782105,0,0 +73,115,897.3077114027096,1000,89.90431472264346,0,0 +74,116,1454.612590266188,1000,91.19515970740413,0,0 +75,117,2292.457960175467,1000,89.8333901030839,0,0 +76,118,1424.378337790017,1000,89.88029014661089,0,0 +77,119,1441.1111023164538,1000,89.79844243631413,0,0 +78,120,1265.4771503717611,1000,89.86009503143968,0,0 +79,121,1662.8808067819505,1000,90.5661722205243,0,0 +80,122,2508.917122342891,1000,89.87403626041336,0,0 +81,123,1655.3510139158748,1000,90.05039760075688,0,0 +82,124,1387.3843721247736,821,90.10235730111886,0,0 +83,125,646.4356689469432,271,89.7559653760994,0,0 +84,126,2172.801064037805,1000,89.84602989356796,0,0 +85,127,165.9213897970373,72,89.66756877688618,0,0 +86,128,1063.1483912161111,1000,89.77409215132576,0,0 +87,129,1000.135342286622,1000,89.81900933661238,0,0 +88,130,1977.2359176146426,1000,89.77653862908736,0,0 +89,131,1937.1674235355138,1000,90.12773943823525,0,0 +90,132,1344.7729257831547,1000,90.39620143170467,0,0 +91,133,786.3379828975102,441,89.82893206036925,0,0 +92,134,1391.060299752017,1000,89.86676880070257,0,0 +93,135,503.300235688713,250,89.94819176115624,0,0 +94,136,2446.7482357041768,1000,89.84848658183878,0,0 +95,137,1171.9102336514923,1000,89.92785850220504,0,0 +96,138,2356.7311711183065,1000,90.42933754946152,0,0 +97,139,2356.12199478712,1000,89.82049779117614,0,0 +98,140,1389.2987977192308,1000,90.82209581044775,0,0 +99,141,967.2335383727841,1000,89.80016695371027,0,0 diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/eval_config.yaml b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/eval_config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1d959452e8ecd71b7df4170e6561f0be78f2b079 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/eval_config.yaml @@ -0,0 +1,203 @@ +experiment: + name: ant-mean5000-profile-simulation-100ep + seed: 42 +backend: + type: sample_factory + algo: APPO + device: cuda + train_dir: /mnt/checkpoints/latency-sensitive-bench/small_models/ant + restart_behavior: overwrite + run_mode: eval +executor: + mode: simulated + simulated_worker_capacity: 1 + simulated_inference_pool: true + inference_devices: + - cuda:0 + inference_batch_size: 16 +env: + action_space: + dtype: float32 + high: + - 1.0 + - 1.0 + - 1.0 + - 1.0 + - 1.0 + - 1.0 + - 1.0 + - 1.0 + labels: + - back_right_hip_torque + - back_right_ankle_torque + - front_left_hip_torque + - front_left_ankle_torque + - front_right_hip_torque + - front_right_ankle_torque + - back_left_hip_torque + - back_left_ankle_torque + low: + - -1.0 + - -1.0 + - -1.0 + - -1.0 + - -1.0 + - -1.0 + - -1.0 + - -1.0 + type: box + base_prompt: Make the Ant move forward as fast as possible without falling. Predict + eight continuous torques in [-1, 1] ordered as back right hip, back right ankle, + front left hip, front left ankle, front right hip, front right ankle, back left + hip, and back left ankle. + env_fps: 10.0 + env_id: LatencyBench/AntContinuous-v0 + frame_stack: 1 + make_kwargs: + base_env_id: Ant-v4 + base_make_kwargs: + exclude_current_positions_from_observation: true + use_contact_forces: false + render_mode: rgb_array + noop_action: + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + obs_fps: 10.0 + registration_imports: + - latency_bench.envs.gymnasium_ant + state_space: + labels: + - torso_z + - torso_quaternion_w + - torso_quaternion_x + - torso_quaternion_y + - torso_quaternion_z + - front_left_hip_angle + - front_left_ankle_angle + - front_right_hip_angle + - front_right_ankle_angle + - back_left_hip_angle + - back_left_ankle_angle + - back_right_hip_angle + - back_right_ankle_angle + - torso_x_velocity + - torso_y_velocity + - torso_z_velocity + - torso_angular_velocity_x + - torso_angular_velocity_y + - torso_angular_velocity_z + - front_left_hip_angular_velocity + - front_left_ankle_angular_velocity + - front_right_hip_angular_velocity + - front_right_ankle_angular_velocity + - back_left_hip_angular_velocity + - back_left_ankle_angular_velocity + - back_right_hip_angular_velocity + - back_right_ankle_angular_velocity + task_name: ant_rgb_state + name: gymnasium + obs_resize: + - 224 + - 224 +latency: + method: temporal + profile_path: /home/ubuntu/lzj/mean-profiling/reference/checkpoint-configs/latency-aware/ant/small-policy/sample-factory-qwenoft-rtx3090-v1/profile/profile.json + profile_worker_slot: 0 + seed: 271828 + add_latency_info: false +scheduler: + hold_policy: hold + ordering_policy: issue_order_fifo +policy: + type: starvla + checkpoint_path: /home/ubuntu/lzj/mean-profiling/ant/vla-publication/checkpoints/model.pt + model_config_path: /home/ubuntu/lzj/mean-profiling/ant/vla-publication/config.full.yaml + task_manifest_path: /home/ubuntu/lzj/mean-profiling/ant/vla-publication/manifest.json + device: cuda:0 + latency_prompt_map_path: /home/ubuntu/lzj/mean-profiling/ant/vla-publication/latency_prompt_map.json + latency_prompt_key: 1 + prompt_mode: raw + unnorm_key: new_embodiment + backbone_path: /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct + worker_python_executable: /home/ubuntu/lzj/conda/envs/qwenoft/bin/python +training: + train_for_env_steps: 10000000 + num_workers: 8 + num_envs_per_worker: 8 + worker_num_splits: 2 + num_policies: 1 + batch_size: 1024 + rollout: 64 + recurrence: 1 + num_epochs: 2 + num_batches_per_epoch: 4 + num_batches_to_accumulate: 2 + policy_workers_per_policy: 1 + max_policy_lag: 10000 + learning_rate: 0.00295 + lr_schedule: linear_decay + lr_schedule_kl_threshold: 0.008 + gamma: 0.99 + gae_lambda: 0.95 + ppo_clip_ratio: 0.2 + ppo_clip_value: 1.0 + value_loss_coeff: 1.3 + max_grad_norm: 3.5 + exploration_loss: entropy + exploration_loss_coeff: 0.0 + kl_loss_coeff: 0.1 + reward_scale: 1.0 + reward_clip: 1000.0 + async_rl: false + serial_mode: false + batched_sampling: false + with_vtrace: false + use_rnn: false + encoder_mlp_layers: + - 64 + - 64 + nonlinearity: tanh + adaptive_stddev: false + policy_initialization: torch_default + initial_stddev: 1.0 + actor_critic_share_weights: true + shuffle_minibatches: false + value_bootstrap: true + normalize_input: true + normalize_returns: true + decorrelate_experience_max_seconds: 10 + decorrelate_envs_on_one_worker: true + set_workers_cpu_affinity: true + force_envs_single_thread: true + save_every_sec: 600 + keep_checkpoints: 3 + save_best_every_sec: 60 + save_best_after: 100000 +evaluation: + eval_interval_steps: null + eval_episodes: 100 + eval_parallel_envs: 16 + eval_max_steps: 1000 + eval_deterministic: true + eval_latency_values: null + eval_raw_reward: true + eval_suites: + fixed: [] + normal: [] + uniform: [] +logging: + output_dir: /home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant + video: + enabled: false + save_step_records: true + save_action_records: true + save_latency_records: true + wandb_project: null + wandb_group: null + wandb_job_type: null diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py new file mode 100644 index 0000000000000000000000000000000000000000..d5edbe901e40e8e9cbb2e0763281fdfcd77dbfcc --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py @@ -0,0 +1,702 @@ +from __future__ import annotations + +import time +from collections.abc import Callable, Mapping, Sequence +from dataclasses import dataclass, field +from pathlib import Path + +import numpy as np + +from latency_bench.core.clock import EnvClock +from latency_bench.core.decision_action_history import DecisionActionHistory +from latency_bench.core.timing import StageProfiler, profiler_scope +from latency_bench.core.types import ActionEvent, EpisodeMetrics, LatencyRecord, Observation, StepRecord +from latency_bench.envs.atari import TRUE_EPISODE_END_INFO_KEY +from latency_bench.envs.base import EnvAdapter +from latency_bench.executors._simulated_timeline import ( + SimulatedResultTimeline, + SimulatedWorkerCapacity, + build_simulated_action_event, +) +from latency_bench.executors.base import BatchedExecutor +from latency_bench.executors.env_step_backend import EnvStepBackend, env_action_space +from latency_bench.latency.sample import LatencySample +from latency_bench.latency.samplers import LatencySampler +from latency_bench.logging.metrics import ( + compute_episode_metrics, + compute_episode_metrics_from_aggregates, + episode_raw_fact_metadata, + latency_type_from_source, + profile_metadata_from_source, +) +from latency_bench.logging.records import build_step_record +from latency_bench.logging.trajectory_logger import TrajectoryLogger +from latency_bench.policy.action_prefix import with_action_prefix +from latency_bench.policy.base import PolicyRunner +from latency_bench.scheduler.action_queue import ActionScheduler +from latency_bench.scheduler.decision import DecisionScheduler +from latency_bench.utils.io import write_json +from latency_bench.utils.stats import series_stats + + +@dataclass +class _EpisodeBuffers: + step_records: list[StepRecord] | None = None + action_events: list[ActionEvent] | None = None + latency_records: list[LatencyRecord] | None = None + latency_values_ms: list[float] = field(default_factory=list) + episode_return_env: float = 0.0 + survival_steps: int = 0 + game_score: float | None = None + return_raw: float | None = None + num_actions: int = 0 + num_dropped_actions: int = 0 + num_invalid_actions: int = 0 + submitted_observation_frames: int = 0 + dropped_observation_count: int = 0 + soft_reset_count: int = 0 + final_lives: int | None = None + final_is_true_episode_end: bool | None = None + task_metrics: dict | None = None + task_metric_moments: dict | None = None + + def record_step(self, *, reward: float, info: dict) -> None: + self.episode_return_env += float(reward) + self.survival_steps += 1 + if "invalid_action" in info and info["invalid_action"]: + self.num_invalid_actions += 1 + if "soft_reset" in info and info["soft_reset"]: + self.soft_reset_count += 1 + if "lives" in info: + self.final_lives = info["lives"] + if TRUE_EPISODE_END_INFO_KEY in info: + self.final_is_true_episode_end = info[TRUE_EPISODE_END_INFO_KEY] + if "game_score" in info: + self.game_score = float(info["game_score"]) + if "score" in info: + self.game_score = float(info["score"]) + if "task_metrics" in info: + self.task_metrics = info["task_metrics"] + if "task_metric_moments" in info: + self.task_metric_moments = info["task_metric_moments"] + self._update_return_raw(info) + extra_stats = info["episode_extra_stats"] if "episode_extra_stats" in info else None + if isinstance(extra_stats, dict): + self._update_return_raw(extra_stats) + + def _update_return_raw(self, stats: dict) -> None: + for key in ("return_raw", "raw_return", "episodic_raw_return", "episode/raw_return"): + if key in stats and stats[key] is not None: + self.return_raw = float(stats[key]) + + +@dataclass +class _SlotState: + slot_id: int + env: EnvAdapter + latency_source: LatencySampler + action_scheduler: ActionScheduler + result_timeline: SimulatedResultTimeline + active: bool = False + episode_id: int | None = None + episode_seed: int | None = None + env_step: int = 0 + recent_drop_count: int = 0 + decision_action_history: DecisionActionHistory | None = None + decision_admitted: bool = False + decision_issued_action: object = None + buffers: _EpisodeBuffers = field(default_factory=_EpisodeBuffers) + worker_capacity: SimulatedWorkerCapacity = field( + default_factory=lambda: SimulatedWorkerCapacity(capacity=None, busy_until_by_worker={}) + ) + + +@dataclass +class _PendingPolicyObservation: + slot: _SlotState + observation: Observation + obs_id: int + latency_sample: LatencySample + worker_slot: int + + +class BatchedSimulatedLatencyExecutor(BatchedExecutor): + """Run multiple simulated episodes concurrently with independent slot state. + + The main process owns policy inference, latency scheduling, episode accounting, + and logging. Env stepping can be serial in-process or delegated to worker + subprocesses through env_backend. + """ + + def __init__( + self, + *, + env_backend: EnvStepBackend, + policy: PolicyRunner, + decision_scheduler: DecisionScheduler, + latency_sources: Sequence[LatencySampler], + action_schedulers: Sequence[ActionScheduler], + clock: EnvClock, + logger: TrajectoryLogger | None = None, + episode_latency_source_factory: Callable[[int], LatencySampler] | None = None, + simulated_worker_capacity: int | None = None, + profile_pipeline: bool = False, + inference_pool=None, + action_prefix=None, + action_history_decisions: int | None = None, + ): + slot_count = env_backend.num_slots + self.env_backend = env_backend + self.envs = list(env_backend.slot_handles) + self.policy = policy + self.decision_scheduler = decision_scheduler + self.clock = clock + self.logger = logger + self.profile_pipeline = bool(profile_pipeline) + self.inference_pool = inference_pool + self.action_prefix = action_prefix + self._pipeline_profile_rows: list[dict[str, float]] = [] + self.simulated_worker_capacity = simulated_worker_capacity + self._collect_step_records = bool(logger is not None and logger.save_step_records) + self._collect_action_records = bool(logger is not None and logger.save_action_records) + self._collect_latency_records = bool(logger is not None and logger.save_latency_records) + self.episode_latency_source_factory = episode_latency_source_factory + self.slots = [ + _SlotState( + slot_id=slot_id, + env=self.envs[slot_id], + latency_source=latency_sources[slot_id], + action_scheduler=action_schedulers[slot_id], + result_timeline=SimulatedResultTimeline( + ordering_policy=action_schedulers[slot_id].ordering_policy + ), + decision_action_history=( + DecisionActionHistory( + env_action_space(self.envs[slot_id]), num_envs=1, decisions=action_history_decisions + ) if action_history_decisions is not None else None + ), + buffers=self._new_episode_buffers(), + worker_capacity=SimulatedWorkerCapacity( + capacity=simulated_worker_capacity, + busy_until_by_worker={}, + ), + ) + for slot_id in range(slot_count) + ] + self._next_obs_id = 0 + self._next_action_id = 0 + self.started_episodes = 0 + self.completed_episodes = 0 + self._completed_metrics: dict[int, EpisodeMetrics] = {} + self._completed_buffers: dict[int, _EpisodeBuffers] = {} + self._episode_log_order: list[int] = [] + self._next_episode_log_index = 0 + + @property + def num_slots(self) -> int: + return len(self.slots) + + def close(self) -> None: + if self.inference_pool is not None: + self.inference_pool.close() + self.env_backend.close() + + def run_episodes( + self, + *, + episode_ids: Sequence[int], + seeds: Sequence[int | None], + eval_max_steps: int = 10000, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + ) -> list[EpisodeMetrics]: + if eval_max_steps < 0: + raise ValueError("eval_max_steps must be non-negative") + episode_ids = [int(episode_id) for episode_id in episode_ids] + if len(seeds) != len(episode_ids): + raise ValueError("seeds length must match episode_ids length") + + self._reset_run_state(episode_ids) + if not episode_ids: + return [] + + next_episode_index = 0 + initial_slots = min(self.num_slots, len(episode_ids)) + for slot in self.slots[:initial_slots]: + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + + while self.completed_episodes < len(episode_ids): + active_slots = self._active_slots() + if eval_max_steps == 0: + for slot in active_slots: + self._complete_slot(slot, on_episode_complete=on_episode_complete) + if next_episode_index < len(episode_ids): + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + continue + + observations = [] + observation_slots: list[_SlotState] = [] + step_capacity_info: dict[int, dict[str, int | bool | None]] = {} + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + slot.worker_capacity.release_ready(slot.env_step) + self._deliver_arrived_results(slot, raw_frame=slot.env_step) + observation_submitted = False + observation_dropped = False + if self.decision_scheduler.should_observe(slot.env_step, current_time_ms): + prefix_request_pending = ( + self.action_prefix is not None + and self.action_prefix["mode"] != "none" + and slot.result_timeline.pending_observation_count > 0 + ) + if slot.worker_capacity.can_submit() and not prefix_request_pending: + observation_slots.append(slot) + observation_submitted = True + else: + slot.buffers.dropped_observation_count += 1 + observation_dropped = True + slot.recent_drop_count += 1 + if self.simulated_worker_capacity is not None: + step_capacity_info[slot.slot_id] = { + "observation_submitted": observation_submitted, + "observation_dropped": observation_dropped, + } + if slot.decision_action_history is not None and slot.env_step % self.clock.obs_stride_raw_frames == 0: + slot.decision_admitted = observation_submitted + slot.decision_issued_action = slot.action_scheduler.noop_action.value + + observe_ms = 0.0 + if observation_slots: + observe_start = time.perf_counter() + observations_by_slot = self.env_backend.observe_slots([slot.slot_id for slot in observation_slots]) + observe_ms = (time.perf_counter() - observe_start) * 1000.0 + pending_observations = [ + self._sample_policy_observation( + slot, + self._policy_observation( + slot, + observations_by_slot[slot.slot_id], + transport=( + slot.decision_action_history.observation()[0] + if slot.decision_action_history is not None else None + ), + ), + ) + for slot in observation_slots + ] + observations = [pending.observation for pending in pending_observations] + + profile_row = None + if observations: + profiler = StageProfiler(enabled=self.profile_pipeline) + with profiler_scope(profiler): + policy_outputs = ( + self.inference_pool.predict_batch(observations) + if self.inference_pool is not None + else self.policy.predict_batch(observations) + ) + if len(policy_outputs) != len(observations): + raise RuntimeError("policy.predict_batch returned the wrong number of outputs") + if self.profile_pipeline: + profile_row = { + "active_slots": float(len(active_slots)), + "batch_size": float(len(observations)), + "observe_slots_ms": observe_ms, + **{key: float(value) for key, value in profiler.timings.items()}, + } + for pending, policy_output in zip(pending_observations, policy_outputs): + if pending.slot.decision_action_history is not None: + pending.slot.decision_issued_action = policy_output.action.value + self._enqueue_policy_output( + pending.slot, + pending.observation, + policy_output, + obs_id=pending.obs_id, + latency_sample=pending.latency_sample, + worker_slot=pending.worker_slot, + ) + + actions_by_slot = {} + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + self._deliver_arrived_results(slot, raw_frame=slot.env_step) + active_action = slot.action_scheduler.update(slot.env_step, current_time_ms) + actions_by_slot[slot.slot_id] = active_action + + env_step_start = time.perf_counter() + step_responses = self.env_backend.step_slots(actions_by_slot) + if profile_row is not None: + profile_row["env_step_ms"] = (time.perf_counter() - env_step_start) * 1000.0 + self._pipeline_profile_rows.append(profile_row) + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + active_action = actions_by_slot[slot.slot_id] + if ( + slot.decision_action_history is not None + and (slot.env_step + 1) % self.clock.obs_stride_raw_frames == 0 + ): + slot.decision_action_history.append( + [0], [slot.decision_admitted], + [slot.decision_issued_action], [active_action.value], + ) + result = step_responses[slot.slot_id].result + soft_reset = bool(result.info.get("soft_reset")) if isinstance(result.info, dict) else False + episode_done = bool(result.done or result.truncated) and not soft_reset + if slot.buffers.step_records is not None: + record = build_step_record( + episode_id=int(slot.episode_id), + env_step=slot.env_step, + scheduled_time_ms=current_time_ms, + active_action=active_action, + reward=result.reward, + done=episode_done, + info=result.info, + active_event=slot.action_scheduler.latest_applied_event, + frame_ms=self.clock.frame_ms, + latency_type=latency_type_from_source(slot.latency_source), + ) + slot.buffers.step_records.append(record) + slot.buffers.record_step(reward=float(result.reward), info=record.info) + else: + slot.buffers.record_step(reward=float(result.reward), info=result.info) + if self.simulated_worker_capacity is not None and slot.buffers.step_records is not None: + slot.buffers.step_records[-1].info.update( + { + **step_capacity_info[slot.slot_id], + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + ) + if soft_reset: + slot.buffers.num_dropped_actions += slot.action_scheduler.dropped_events_count + slot.action_scheduler.reset() + slot.result_timeline.reset() + self._reset_policy_state(slot.slot_id) + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + slot.recent_drop_count = 0 + + slot.env_step += 1 + if episode_done or slot.env_step >= eval_max_steps: + self._complete_slot(slot, on_episode_complete=on_episode_complete) + if next_episode_index < len(episode_ids): + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + + self._write_pipeline_profile_summary() + return self._ordered_metrics(episode_ids) + + def _policy_observation( + self, + slot: _SlotState, + observation: Observation, + transport: np.ndarray | None = None, + ) -> Observation: + observation = with_action_prefix(observation, slot.action_scheduler, self.action_prefix) + metadata = dict(observation.metadata) + metadata["slot_id"] = slot.slot_id + metadata["episode_id"] = int(slot.episode_id) + metadata["action_noise_seed"] = slot.episode_seed + data = observation.data + if transport is not None: + data = {**data, "transport": transport} if isinstance(data, Mapping) else {"obs": data, "transport": transport} + return Observation( + data=data, + env_step=observation.env_step, + sim_time_ms=observation.sim_time_ms, + metadata=metadata, + ) + + def _sample_policy_observation( + self, + slot: _SlotState, + observation: Observation, + ) -> _PendingPolicyObservation: + obs_id = self._next_obs_id + self._next_obs_id += 1 + raw_frame = int(slot.env_step) + current_time_ms = self.clock.step_to_time_ms(raw_frame) + worker_slot = slot.worker_capacity.assign_worker() + latency_context = { + "observation": observation, + "obs_id": obs_id, + "env_step": raw_frame, + "raw_frame": raw_frame, + "sim_time_ms": current_time_ms, + "episode_id": slot.episode_id, + "slot_id": slot.slot_id, + "worker_slot": worker_slot, + "recent_drop_count": slot.recent_drop_count, + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + latency_sample = slot.latency_source.sample(latency_context) + metadata = dict(observation.metadata) + metadata["obs_id"] = obs_id + policy_observation = Observation( + data=observation.data, + env_step=observation.env_step, + sim_time_ms=observation.sim_time_ms, + metadata=metadata, + ) + slot.worker_capacity.submit( + worker_slot, raw_frame + latency_sample.worker_service_raw_frames + ) + return _PendingPolicyObservation( + slot=slot, + obs_id=obs_id, + latency_sample=latency_sample, + worker_slot=worker_slot, + observation=policy_observation, + ) + + def _reset_run_state(self, episode_ids: Sequence[int]) -> None: + self.started_episodes = 0 + self.completed_episodes = 0 + self._pipeline_profile_rows.clear() + self._completed_metrics.clear() + self._completed_buffers.clear() + self._episode_log_order = [int(episode_id) for episode_id in episode_ids] + self._next_episode_log_index = 0 + for slot in self.slots: + slot.active = False + slot.episode_id = None + slot.episode_seed = None + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + + def _active_slots(self) -> list[_SlotState]: + return [slot for slot in self.slots if slot.active] + + def _deliver_arrived_results(self, slot: _SlotState, *, raw_frame: int | None) -> None: + released, dropped = slot.result_timeline.release_arrived(raw_frame) + slot.buffers.num_dropped_actions += len(dropped) + for event in released: + slot.action_scheduler.enqueue(event) + + def _start_slot(self, slot: _SlotState, *, episode_id: int, seed: int | None) -> None: + if self.episode_latency_source_factory is not None: + slot.latency_source = self.episode_latency_source_factory(episode_id) + slot.active = True + slot.episode_id = int(episode_id) + slot.episode_seed = None if seed is None else int(seed) + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + self._reset_policy_state(slot.slot_id) + self.env_backend.reset_slot(slot.slot_id, episode_id=episode_id, seed=seed) + self.started_episodes += 1 + + def _reset_policy_state(self, slot_id: int) -> None: + if self.inference_pool is not None: + self.inference_pool.reset_state(slot_id) + else: + self.policy.reset_state(slot_id=slot_id) + + def _complete_slot( + self, + slot: _SlotState, + *, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + ) -> None: + if not slot.active or slot.episode_id is None: + return + episode_id = int(slot.episode_id) + self._deliver_arrived_results(slot, raw_frame=None) + slot.buffers.num_dropped_actions += slot.action_scheduler.dropped_events_count + metrics = self._compute_episode_metrics( + episode_id=episode_id, + buffers=slot.buffers, + metadata=episode_raw_fact_metadata( + mode="simulated", + episode_seed=slot.episode_seed, + env_fps=self.clock.env_fps, + obs_fps=self.clock.obs_fps, + frame_ms=self.clock.frame_ms, + latency_type=latency_type_from_source(slot.latency_source), + latency_source=slot.latency_source, + ) + | slot.action_scheduler.chunk_metrics() + | ( + { + "submitted_observation_frames": slot.buffers.submitted_observation_frames, + "dropped_observation_count": slot.buffers.dropped_observation_count, + "simulated_worker_capacity": self.simulated_worker_capacity, + "inference_worker_count": self.simulated_worker_capacity, + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + if self.simulated_worker_capacity is not None + else {} + ), + ) + self._completed_metrics[episode_id] = metrics + self._completed_buffers[episode_id] = slot.buffers + self.completed_episodes += 1 + slot.active = False + slot.episode_id = None + slot.episode_seed = None + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + self._flush_completed_in_episode_order() + if on_episode_complete is not None: + on_episode_complete(metrics) + + def _enqueue_policy_output( + self, + slot: _SlotState, + observation, + policy_output, + *, + obs_id: int, + latency_sample: LatencySample, + worker_slot: int, + ) -> None: + raw_frame = int(slot.env_step) + latency_ms = latency_sample.latency_ms + ready_raw_frame = raw_frame + latency_sample.action_ready_raw_frames + ready_time_ms = self.clock.step_to_time_ms(ready_raw_frame) + latency_type = latency_type_from_source(slot.latency_source) + profile_metadata = profile_metadata_from_source(slot.latency_source) + slot_metadata = { + "episode_id": int(slot.episode_id), + "slot_id": int(slot.slot_id), + "worker_id": int(worker_slot), + } + latency_record, event = build_simulated_action_event( + action_id=self._next_action_id, + obs_id=obs_id, + policy_output=policy_output, + raw_frame=raw_frame, + ready_raw_frame=ready_raw_frame, + ready_time_ms=ready_time_ms, + latency_sample=latency_sample, + frame_ms=self.clock.frame_ms, + latency_type=latency_type, + profile_metadata=profile_metadata, + latency_record_metadata=slot_metadata, + extra_event_metadata=slot_metadata, + ) + self._next_action_id += 1 + slot.result_timeline.submit(obs_id=obs_id, ready_raw_frame=ready_raw_frame, event=event) + slot.buffers.submitted_observation_frames += 1 + slot.recent_drop_count = 0 + slot.buffers.num_actions += 1 + if slot.buffers.action_events is not None: + slot.buffers.action_events.append(event) + slot.buffers.latency_values_ms.append(latency_ms) + if slot.buffers.latency_records is not None: + slot.buffers.latency_records.append(latency_record) + + def _flush_completed_in_episode_order(self) -> None: + if self.logger is None: + return + while self._next_episode_log_index < len(self._episode_log_order): + episode_id = self._episode_log_order[self._next_episode_log_index] + if episode_id not in self._completed_metrics: + break + buffers = self._completed_buffers[episode_id] + metrics = self._completed_metrics[episode_id] + if buffers.step_records is not None: + for record in buffers.step_records: + self.logger.log_step(record) + if buffers.action_events is not None: + for event in buffers.action_events: + self.logger.log_action_event(event) + if buffers.latency_records is not None: + for latency_record in buffers.latency_records: + self.logger.log_latency(latency_record) + self.logger.log_episode_metrics(metrics) + self._next_episode_log_index += 1 + + def _ordered_metrics(self, episode_ids: Sequence[int]) -> list[EpisodeMetrics]: + return [self._completed_metrics[int(episode_id)] for episode_id in episode_ids] + + def _new_episode_buffers(self) -> _EpisodeBuffers: + return _EpisodeBuffers( + step_records=[] if self._collect_step_records else None, + action_events=[] if self._collect_action_records else None, + latency_records=[] if self._collect_latency_records else None, + ) + + def _write_pipeline_profile_summary(self) -> None: + if not self.profile_pipeline or self.logger is None or not self._pipeline_profile_rows: + return + keys = sorted({key for row in self._pipeline_profile_rows for key in row}) + summary = { + "num_profiled_batches": len(self._pipeline_profile_rows), + **{ + key: series_stats([float(row[key]) for row in self._pipeline_profile_rows if key in row]) + for key in keys + }, + } + write_json(Path(self.logger.output_dir) / "simulated_pipeline_summary.json", summary) + + def _compute_episode_metrics( + self, + *, + episode_id: int, + buffers: _EpisodeBuffers, + metadata: dict, + ) -> EpisodeMetrics: + if buffers.task_metrics is not None: + metadata["task_metrics"] = buffers.task_metrics + if buffers.task_metric_moments is not None: + metadata["task_metric_moments"] = buffers.task_metric_moments + if buffers.final_lives is not None: + metadata["final_lives"] = buffers.final_lives + if buffers.final_is_true_episode_end is not None: + metadata["final_is_true_episode_end"] = buffers.final_is_true_episode_end + metadata["soft_reset_count"] = buffers.soft_reset_count + if buffers.step_records is not None and buffers.action_events is not None: + return compute_episode_metrics( + episode_id=episode_id, + step_records=buffers.step_records, + action_events=buffers.action_events, + latency_values_ms=buffers.latency_values_ms, + metadata=metadata, + frame_ms=self.clock.frame_ms, + ) + return compute_episode_metrics_from_aggregates( + episode_id=episode_id, + episode_return_env=buffers.episode_return_env, + survival_steps=buffers.survival_steps, + return_raw=buffers.return_raw, + game_score=buffers.game_score, + latency_values_ms=buffers.latency_values_ms, + num_actions=buffers.num_actions, + num_dropped_actions=buffers.num_dropped_actions, + num_invalid_actions=buffers.num_invalid_actions, + metadata=metadata, + ) diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch new file mode 100644 index 0000000000000000000000000000000000000000..bdeca64d184242eedaa57463d2e0a242f43e19ba --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch @@ -0,0 +1,99 @@ +diff --git a/latency_bench/envs/deadly_corridor.py b/latency_bench/envs/deadly_corridor.py +index 4dcaa48c..dc4d1186 100644 +--- a/latency_bench/envs/deadly_corridor.py ++++ b/latency_bench/envs/deadly_corridor.py +@@ -5,7 +5,7 @@ from collections import deque + from typing import Any + + import numpy as np +-from gymnasium.spaces import Box, Tuple ++from gymnasium.spaces import Box, MultiBinary, Tuple + + from latency_bench.core.types import Action, Observation, StepResult + from latency_bench.envs.base import EnvAdapter +@@ -346,6 +346,7 @@ class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + export_env_raw_rgb_frames: bool = True, + ): + import gymnasium as gym ++ import vizdoom + import vizdoom.gymnasium_wrapper # noqa: F401 (registers the env ids) + + env_cfg = config["env"] +@@ -360,30 +361,27 @@ class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + ) + if key in env_cfg + } +- attempts = [ +- ("VizdoomDeadlyCorridor-MultiBinary-v1", {}), +- ("VizdoomDeadlyCorridor-MultiBinary-v0", {}), +- ("VizdoomDeadlyCorridor-v1", {"max_buttons_pressed": 0}), +- ("VizdoomDeadlyCorridor-v0", {"max_buttons_pressed": 0}), +- ] +- last_exc: Exception | None = None +- self.gym_env = None +- for env_id, kwargs in attempts: +- try: +- # frame_skip=1: the latency_bench scheduler advances obs_stride raw +- # frames per decision and holds the action between observations. +- self.gym_env = gym.make( +- env_id, render_mode="rgb_array", frame_skip=1, **render_options, **kwargs +- ) +- self.env_id = env_id +- break +- except (gym.error.NameNotFound, gym.error.VersionNotFound, gym.error.NamespaceNotFound) as exc: +- last_exc = exc +- if self.gym_env is None: +- raise RuntimeError(f"Failed to create Deadly Corridor MultiBinary env: {last_exc}") ++ # ViZDoom registers deadly_corridor.cfg under this official Gym ID. ++ self.env_id = "VizdoomCorridor-v0" ++ self.gym_env = gym.make( ++ self.env_id, render_mode="rgb_array", frame_skip=1, max_buttons_pressed=0, ++ ) ++ game = self.gym_env.unwrapped.game ++ game.close() ++ for key, value in render_options.items(): ++ if key == "screen_resolution": ++ value = getattr(vizdoom.ScreenResolution, value) ++ getattr(game, f"set_{key}")(value) ++ game.init() ++ self.gym_env.unwrapped.observation_space.spaces["screen"] = Box( ++ 0, 255, ++ shape=(game.get_screen_height(), game.get_screen_width(), game.get_screen_channels()), ++ dtype=np.uint8, ++ ) + + self._runtime_button_order = _deadly_runtime_button_names(self.gym_env) + self._num_buttons = len(self._runtime_button_order) ++ self.gym_env.unwrapped.action_space = MultiBinary(self._num_buttons) + self.noop_action = noop_action or Action( + value=[0] * self._num_buttons, name="NOOP", is_noop=True + ) +diff --git a/tests/integration/test_deadly_render_contract.py b/tests/integration/test_deadly_render_contract.py +index 535db22a..09894b1b 100644 +--- a/tests/integration/test_deadly_render_contract.py ++++ b/tests/integration/test_deadly_render_contract.py +@@ -5,11 +5,13 @@ import json + import numpy as np + import pytest + +-pytest.importorskip("vizdoom", minversion="1.3.0") ++pytest.importorskip("vizdoom", minversion="1.2.4") + pytest.importorskip("sample_factory") + + from latency_bench.envs.deadly_corridor import DeadlyCorridorEnvAdapter, DeadlyCorridorVlaEnvAdapter + from latency_bench.envs.raw_rgb import ENV_RAW_RGB_FRAME_STACK_INFO_KEY ++from latency_bench.core.types import Action ++from gymnasium.spaces import MultiBinary + from scripts.tasks.decision_history.eval_vla_hist8 import evaluation_config + + +@@ -33,6 +35,9 @@ def test_hist8_deadly_vla_uses_the_teacher_resolution_and_hud(tmp_path): + # The health/ammo panel is stable across the two engine reset paths; + # the animated face and enemies can differ with their RNG streams. + np.testing.assert_array_equal(teacher_frame[-20:, :64], student_frame[-20:, :64]) ++ assert isinstance(student.gym_env.action_space, MultiBinary) ++ step = student.step(Action(value=[1, 0, 0, 0, 0, 0, 1], name="forward_attack")) ++ assert np.isfinite(step.reward) + finally: + teacher.close() + student.close() diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py new file mode 100644 index 0000000000000000000000000000000000000000..1016be20dca9e949c752192edb407b9a84e30e35 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py @@ -0,0 +1,455 @@ +from __future__ import annotations + +import copy +from collections import deque +from typing import Any + +import numpy as np +from gymnasium.spaces import Box, MultiBinary, Tuple + +from latency_bench.core.types import Action, Observation, StepResult +from latency_bench.envs.base import EnvAdapter +from latency_bench.utils.array import looks_chw +from latency_bench.envs.raw_rgb import RawRgbFrameStackBuffer + + +def _noop_action_from_space(space) -> Any: + n = getattr(space, "n", None) + if n is not None: + return 0 + if isinstance(space, Tuple): + return tuple(_noop_action_from_space(subspace) for subspace in space.spaces) + if isinstance(space, Box): + import numpy as np + + return np.zeros(space.shape, dtype=space.dtype) + raise TypeError(f"Unsupported action space for Deadly Corridor no-op action: {space}") + + +def _coerce_noop_action_for_space(value: Any, space) -> Any: + if isinstance(space, Tuple): + if isinstance(value, (list, tuple)): + if len(value) != len(space.spaces): + raise ValueError( + f"Deadly Corridor no-op action length {len(value)} does not match action space {space}" + ) + return tuple( + _coerce_noop_action_for_space(item, subspace) + for item, subspace in zip(value, space.spaces) + ) + if value == 0: + return _noop_action_from_space(space) + return value + + +def _spec_with_reward_scaling(spec: Any, disable_reward_scaling: bool) -> Any: + if not disable_reward_scaling: + return spec + spec_to_use = copy.copy(spec) + spec_to_use.reward_scaling = 1.0 + return spec_to_use + + +def _synchronous_eval_fps_from_config(config: dict[str, Any], default: int = 35) -> int: + env_cfg = config.get("env", {}) + try: + fps = int(float(env_cfg.get("env_fps", default))) + except (TypeError, ValueError) as exc: + raise ValueError("env_fps must be positive") from exc + if fps <= 0: + raise ValueError("env_fps must be positive") + return fps + + +def _build_sample_factory_eval_cfg(config: dict[str, Any]) -> Any: + from training.deadly_corridor_sf import integration + from training.common.utils import maybe_set_cli_override + + integration.register_deadly_corridor_components() + base_cfg = integration.SAMPLE_FACTORY_CONFIG_PARSER.parse_eval( + integration.build_cli_args_from_config(config) + ) + eval_fps = _synchronous_eval_fps_from_config(config) + cfg = copy.deepcopy(base_cfg) + if _requires_sample_factory_checkpoint_config(config): + from sample_factory.cfg.arguments import load_from_checkpoint + + cfg = load_from_checkpoint(cfg) + + for key in ( + "seed", + "res_w", + "res_h", + "wide_aspect_ratio", + ): + if hasattr(base_cfg, key): + maybe_set_cli_override(cfg, key, getattr(base_cfg, key)) + maybe_set_cli_override(cfg, "frame_stack", 1) + explicit_max_episode_steps = int(getattr(base_cfg, "max_episode_steps", 0) or 0) + if explicit_max_episode_steps > 0: + maybe_set_cli_override(cfg, "max_episode_steps", explicit_max_episode_steps) + else: + eval_max_steps = int(getattr(base_cfg, "eval_max_steps", 0) or 0) + if eval_max_steps > 0: + maybe_set_cli_override(cfg, "max_episode_steps", eval_max_steps) + + maybe_set_cli_override(cfg, "mode", "eval") + maybe_set_cli_override(cfg, "latency_type", "zero") + maybe_set_cli_override(cfg, "fixed_latency_ms", 0.0) + maybe_set_cli_override(cfg, "env_frameskip", 1) + maybe_set_cli_override(cfg, "eval_env_frameskip", 1) + maybe_set_cli_override(cfg, "num_envs", 1) + maybe_set_cli_override(cfg, "no_render", True) + maybe_set_cli_override(cfg, "save_video", False) + maybe_set_cli_override(cfg, "fps", eval_fps) + maybe_set_cli_override(cfg, "eval_deterministic", bool(getattr(base_cfg, "eval_deterministic", True))) + maybe_set_cli_override(cfg, "disable_reward_scaling", bool(getattr(base_cfg, "eval_raw_reward", False))) + return cfg + + +def _requires_sample_factory_checkpoint_config(config: dict[str, Any]) -> bool: + policy_type = str(config.get("policy", {}).get("type", "")).strip().lower() + return policy_type == "deadly_corridor_sf" + + +def _seed_initialized_vizdoom_game(env: Any, seed: int) -> bool: + unwrapped = getattr(env, "unwrapped", env) + game = getattr(unwrapped, "game", None) + if game is None: + return False + unwrapped.seed(int(seed)) + game.set_seed(int(unwrapped.curr_seed)) + return True + + +class DeadlyCorridorEnvAdapter(EnvAdapter): + """Latency-bench adapter for ViZDoom Deadly Corridor using the SF Doom env stack.""" + OBSERVATION_TYPE = "vizdoom_frame_v1" + + def __init__( + self, + *, + config: dict[str, Any], + noop_action: Action | None = None, + export_env_raw_rgb_frames: bool = False, + ): + env_cfg = config["env"] + env_id = str(env_cfg.get("env_id", "doom_deadly_corridor")) + env_fps = float(env_cfg.get("env_fps", 35)) + self.frame_stack = int(env_cfg.get("frame_stack", 1) or 1) + + from sample_factory.utils.attr_dict import AttrDict + from sf_examples.vizdoom.doom.doom_utils import DOOM_ENVS, make_doom_env_from_spec + + cfg = _build_sample_factory_eval_cfg(config) + spec = next((item for item in DOOM_ENVS if item.name == str(env_id)), None) + if spec is None: + raise ValueError(f"Unknown ViZDoom env spec: {env_id}") + spec_to_use = _spec_with_reward_scaling( + spec, + disable_reward_scaling=bool(getattr(cfg, "disable_reward_scaling", False)), + ) + self.gym_env = make_doom_env_from_spec( + spec_to_use, + str(env_id), + cfg, + AttrDict(worker_index=0, vector_index=0, env_id=0), + render_mode=None, + ) + self.cfg = cfg + self.env_id = env_id + self.env_fps = float(env_fps) + action_space = self.gym_env.action_space + noop_value = _noop_action_from_space(action_space) + if noop_action is None: + self.noop_action = Action(value=noop_value, name=str(noop_value), is_noop=True) + else: + coerced_noop_value = _coerce_noop_action_for_space(noop_action.value, action_space) + self.noop_action = Action( + value=coerced_noop_value, + name=str(coerced_noop_value), + is_noop=True, + is_oneshot=noop_action.is_oneshot, + ) + self.env_step = 0 + self.export_env_raw_rgb_frames = bool(export_env_raw_rgb_frames) + self._last_info: dict[str, Any] = {} + self._last_frame: Any = None + self._observed_frames: deque[np.ndarray] = deque(maxlen=self.frame_stack) + self._raw_rgb_frames = RawRgbFrameStackBuffer(num_frames=self.frame_stack) + + def reset(self, seed: int | None = None) -> Observation: + self.env_step = 0 + self._observed_frames.clear() + if seed is not None: + if _seed_initialized_vizdoom_game(self.gym_env, int(seed)): + obs, info = self.gym_env.reset() + else: + try: + obs, info = self.gym_env.reset(seed=seed) + except TypeError: + obs, info = self.gym_env.reset() + else: + obs, info = self.gym_env.reset() + self._last_frame = obs + self._last_info = dict(info or {}) + self._reset_frame_stack(obs) + if self.export_env_raw_rgb_frames: + self._reset_raw_rgb_frame_stack() + return self._make_observation(info=self._last_info) + + def step(self, action: Action) -> StepResult: + gym_action = action.value + obs, reward, terminated, truncated, info = self.gym_env.step(gym_action) + self.env_step += 1 + self._last_frame = obs + self._last_info = dict(info or {}) + self._append_frame(obs) + if self.export_env_raw_rgb_frames and not bool(terminated or truncated): + self._append_raw_rgb_frame() + observation = self._make_observation(info=self._last_info) + step_info = dict(self._last_info) + step_info.update( + { + "env_step": self.env_step, + "sim_time_ms": self.env_step * self.frame_ms, + "applied_action": gym_action, + "applied_action_name": action.name, + "observation": "vizdoom_frame_v1", + } + ) + return StepResult( + observation=observation, + reward=float(reward), + done=bool(terminated), + truncated=bool(truncated), + info=step_info, + ) + + def observe(self) -> Observation: + if self._last_frame is None: + raise RuntimeError("DeadlyCorridorEnvAdapter has no current observation; call reset() first") + metadata = self._metadata(self._last_info) + return Observation( + data=self._policy_frame_stack(), + env_step=self.env_step, + sim_time_ms=self.env_step * self.frame_ms, + metadata=metadata, + ) + + def render_game_frame(self) -> np.ndarray: + return np.transpose(self.gym_env.unwrapped.game.get_state().screen_buffer, (1, 2, 0)) + + def close(self) -> None: + self.gym_env.close() + + def _reset_frame_stack(self, frame: Any) -> None: + self._observed_frames.clear() + self._append_frame(frame) + + def _append_frame(self, frame: Any) -> None: + self._observed_frames.append(_single_frame_data(frame)) + + def _policy_frame_stack(self) -> np.ndarray: + frames = list(self._observed_frames) + if not frames: + raise RuntimeError("Deadly Corridor observe() has no current frame; call reset() first") + if len(frames) < self.frame_stack: + frames = [frames[0]] * (self.frame_stack - len(frames)) + frames + frames = [np.asarray(frame, dtype=np.uint8) for frame in frames[-self.frame_stack :]] + if self.frame_stack == 1: + return frames[-1] + axis = 0 if looks_chw(frames[0]) else -1 + return np.concatenate(frames, axis=axis) + + +def _single_frame_data(frame: Any) -> np.ndarray: + value = frame.get("obs") if isinstance(frame, dict) else frame + arr = np.asarray(value, dtype=np.uint8) + if arr.ndim == 2: + return arr[..., None] + if arr.ndim != 3: + raise ValueError(f"Expected Deadly Corridor image frame with 2 or 3 dims, got {arr.shape!r}") + return arr + + +# Fixed semantic button order the StarVLA multibinary head is trained against. +# Mirrors starVLA.training.rl_games.eval_core._semantic_to_runtime_multibinary. +DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER = ( + "MOVE_FORWARD", + "MOVE_BACKWARD", + "MOVE_LEFT", + "MOVE_RIGHT", + "TURN_LEFT", + "TURN_RIGHT", + "ATTACK", +) + + +def _deadly_runtime_button_names(gym_env: Any) -> list[str]: + """Return the live ViZDoom action-button order (ports eval_core helper). + + The MultiBinary action vector is indexed by the game's available-button + order, which is not guaranteed to equal the semantic order the head emits. + """ + + def _button_name(button: Any) -> str: + name = getattr(button, "name", None) + if name is not None: + return str(name) + text = str(button) + return text.split(".")[-1] if "." in text else text + + for candidate in (gym_env, getattr(gym_env, "unwrapped", None)): + if candidate is None: + continue + for attr_name in ("game", "_game"): + game = getattr(candidate, attr_name, None) + if game is None: + continue + getter = getattr(game, "get_available_buttons", None) + if getter is None: + continue + names = [_button_name(button) for button in getter()] + if names: + return names + return list(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + + +def _semantic_to_runtime_multibinary(semantic_values: list[int], runtime_order: list[str]) -> list[int]: + semantic_map = { + name: int(semantic_values[idx]) + for idx, name in enumerate(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + if idx < len(semantic_values) + } + return [semantic_map.get(name, 0) for name in runtime_order] + + +class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + """Deadly Corridor adapter for StarVLA eval, matching eval_core's env. + + Unlike :class:`DeadlyCorridorEnvAdapter` (sample_factory, factorised action + tuple), this uses the gymnasium ``VizdoomDeadlyCorridor-MultiBinary`` env so + the model's multibinary head can fire arbitrary button subsets, exactly like + ``starVLA.training.rl_games.eval_core``. Native ``frame_skip=1`` is used so + latency_bench's observation-cadence scheduler owns the obs_stride stepping + (see ObservationCadenceDecisionScheduler); setting a native skip would + double-count it. + """ + + OBSERVATION_TYPE = "vizdoom_frame_v1" + + def __init__( + self, + *, + config: dict[str, Any], + noop_action: Action | None = None, + export_env_raw_rgb_frames: bool = True, + ): + import gymnasium as gym + import vizdoom + import vizdoom.gymnasium_wrapper # noqa: F401 (registers the env ids) + + env_cfg = config["env"] + self.env_fps = float(env_cfg.get("env_fps", 35)) + self.frame_stack = int(env_cfg.get("frame_stack", 1) or 1) + # The raw teacher view is part of the policy's observation contract. + render_options = { + key: env_cfg[key] + for key in ( + "screen_resolution", "render_hud", "render_crosshair", + "render_weapon", "render_decals", "render_particles", + ) + if key in env_cfg + } + # ViZDoom registers deadly_corridor.cfg under this official Gym ID. + self.env_id = "VizdoomCorridor-v0" + self.gym_env = gym.make( + self.env_id, render_mode="rgb_array", frame_skip=1, max_buttons_pressed=0, + ) + game = self.gym_env.unwrapped.game + game.close() + for key, value in render_options.items(): + if key == "screen_resolution": + value = getattr(vizdoom.ScreenResolution, value) + getattr(game, f"set_{key}")(value) + game.init() + self.gym_env.unwrapped.observation_space.spaces["screen"] = Box( + 0, 255, + shape=(game.get_screen_height(), game.get_screen_width(), game.get_screen_channels()), + dtype=np.uint8, + ) + + self._runtime_button_order = _deadly_runtime_button_names(self.gym_env) + self._num_buttons = len(self._runtime_button_order) + self.gym_env.unwrapped.action_space = MultiBinary(self._num_buttons) + self.noop_action = noop_action or Action( + value=[0] * self._num_buttons, name="NOOP", is_noop=True + ) + self.export_env_raw_rgb_frames = bool(export_env_raw_rgb_frames) + self.env_step = 0 + self._last_info: dict[str, Any] = {} + self._last_frame: Any = None + self._raw_rgb_frames = RawRgbFrameStackBuffer(num_frames=self.frame_stack) + + def reset(self, seed: int | None = None) -> Observation: + self.env_step = 0 + try: + obs, info = self.gym_env.reset(seed=seed) + except TypeError: + obs, info = self.gym_env.reset() + self._last_frame = obs + self._last_info = dict(info or {}) + if self.export_env_raw_rgb_frames: + self._reset_raw_rgb_frame_stack() + return self._make_observation(info=self._last_info) + + def step(self, action: Action) -> StepResult: + # action.value is a 7-dim multibinary vector in semantic order; re-order + # to the live game's button layout before stepping the MultiBinary env. + semantic = [int(v) for v in np.asarray(action.value).reshape(-1).tolist()] + expected = len(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + if len(semantic) != expected: + raise ValueError( + "DeadlyCorridorVlaEnvAdapter expects a " + f"{expected}-dim multibinary action in semantic order, got " + f"{len(semantic)} values ({action.value!r}). This usually means the " + "policy decoded a non-multibinary layout; ensure the deadly head is " + "action_layout=multibinary_7 and reached the multibinary decode path." + ) + runtime_buttons = _semantic_to_runtime_multibinary(semantic, self._runtime_button_order) + gym_action = np.asarray(runtime_buttons, dtype=np.int8) + obs, reward, terminated, truncated, info = self.gym_env.step(gym_action) + self.env_step += 1 + self._last_frame = obs + self._last_info = dict(info or {}) + if self.export_env_raw_rgb_frames and not bool(terminated or truncated): + self._append_raw_rgb_frame() + observation = self._make_observation(info=self._last_info) + step_info = dict(self._last_info) + step_info.update( + { + "env_step": self.env_step, + "sim_time_ms": self.env_step * self.frame_ms, + "applied_action": runtime_buttons, + "applied_action_name": action.name, + "observation": self.OBSERVATION_TYPE, + } + ) + return StepResult( + observation=observation, + reward=float(reward), + done=bool(terminated), + truncated=bool(truncated), + info=step_info, + ) + + def observe(self) -> Observation: + return self._make_observation(info=self._last_info) + + def render_game_frame(self) -> np.ndarray: + frame = self.gym_env.render() + return np.asarray(frame, dtype=np.uint8) + + def close(self) -> None: + self.gym_env.close() diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py new file mode 100644 index 0000000000000000000000000000000000000000..13f64ef81329e0b3c9296a066e17196a7e6c1d56 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py @@ -0,0 +1,60 @@ +"""Causal action history sampled at completed decision boundaries.""" + +from __future__ import annotations + +import numpy as np +from gymnasium.spaces import Discrete, MultiBinary, Tuple + + +class DecisionActionHistory: + """Encode admission, admitted command, and last applied action for each decision.""" + + def __init__(self, action_space, *, num_envs: int, decisions: int): + self._multibinary = isinstance(action_space, MultiBinary) + if isinstance(action_space, Discrete): + self.action_sizes = (action_space.n,) + elif isinstance(action_space, Tuple) and all(isinstance(space, Discrete) for space in action_space.spaces): + self.action_sizes = tuple(space.n for space in action_space.spaces) + elif self._multibinary and action_space.shape == (7,): + self.action_sizes = (3, 3, 3, 2) + else: + raise NotImplementedError(f"Decision action history does not support {action_space!r}") + self.decisions = decisions + self.action_dim = sum(size - 1 for size in self.action_sizes) + self.step_dim = 1 + 2 * self.action_dim + self.data = np.zeros((num_envs, decisions, self.step_dim), dtype=np.float32) + self._basis = tuple(np.eye(size, dtype=np.float32)[:, 1:] for size in self.action_sizes) + + @property + def observation_dim(self) -> int: + return self.decisions * self.step_dim + + def reset(self, indices=None) -> None: + if indices is None: + self.data.fill(0) + else: + self.data[indices] = 0 + + def append(self, indices, admitted, issued_actions, applied_actions) -> None: + admitted = np.asarray(admitted, dtype=np.float32).reshape(-1) + issued = self._encode(issued_actions) * admitted[:, None] + applied = self._encode(applied_actions) + rows = self.data[indices].copy() + rows[:, :-1] = rows[:, 1:] + rows[:, -1, 0] = admitted + rows[:, -1, 1 : 1 + self.action_dim] = issued + rows[:, -1, 1 + self.action_dim :] = applied + self.data[indices] = rows + + def observation(self) -> np.ndarray: + return self.data.reshape(self.data.shape[0], self.observation_dim).copy() + + def _encode(self, actions) -> np.ndarray: + if self._multibinary: + # The VLA button order is move, strafe, turn, attack; teacher history + # encodes turn, move, strafe, attack. Keep both opposing bits if issued. + return np.asarray(actions, dtype=np.float32).reshape(-1, 7)[:, [4, 5, 0, 1, 2, 3, 6]] + values = np.asarray(actions, dtype=np.int64).reshape(-1, len(self.action_sizes)) + return np.concatenate( + [basis[values[:, index]] for index, basis in enumerate(self._basis)], axis=1 + ) diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py new file mode 100644 index 0000000000000000000000000000000000000000..fca10fa388e20c9df4abb2f235ac4ea284a3144c --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py @@ -0,0 +1,216 @@ +"""Single evaluation driver: run one config's episodes and attach metadata. + +This is the core ``run_from_config`` and its episode-side helpers. Sweep/suite +orchestration lives in :mod:`latency_bench.eval.sweeps`; the CLI in +:mod:`latency_bench.run`. +""" +from __future__ import annotations + +from collections.abc import Callable, Sequence +from pathlib import Path +from typing import Any + +import yaml + +from training.common.utils import seed_everything +from latency_bench.core.types import EpisodeMetrics, ExecutorMode +from latency_bench.eval.config import ( + _episode_seed, + _eval_episodes, + _eval_max_steps, + _evaluation_seed, + resolve_evaluation_config, +) +from latency_bench.eval.reporting import _write_non_sweep_summary +from latency_bench.envs.base import EnvAdapter +from latency_bench.executors.base import BatchedExecutor +from latency_bench.executors.factory import build_executor +from latency_bench.executors.realtime_warmup import plot_realtime_eval_latency +from latency_bench.latency.config import latency_type_from_config +from latency_bench.logging.action_trace_replay import record_videos_from_action_trace +from latency_bench.logging.video import select_episode_return_stratified + + +def run_from_config( + config: dict[str, Any], + extra_metadata: dict[str, Any] | None = None, + *, + write_summary: bool = True, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + episode_ids: Sequence[int] | None = None, + policy: Any | None = None, + env: EnvAdapter | None = None, + env_backend: Any | None = None, + inference_devices: list[str] | None = None, +) -> list[EpisodeMetrics]: + eval_max_steps = _eval_max_steps(config) + resolve_evaluation_config(config) + if ( + policy is None + and env is None + and env_backend is None + and config["policy"]["type"] == "starvla" + ): + from latency_bench.policy.starvla import prepare_starvla_checkpoint_input_config + + prepare_starvla_checkpoint_input_config(config) + + experiment_cfg = config["experiment"] + policy_cfg = config["policy"] + logging_cfg = config["logging"] + seed = _evaluation_seed(config) + configured_num_episodes = _eval_episodes(config) + selected_episode_ids = list(range(configured_num_episodes)) if episode_ids is None else list(episode_ids) + seed_everything(seed) + + executor_kwargs = {} + if policy is not None: + executor_kwargs["policy"] = policy + if env is not None: + executor_kwargs["env"] = env + if env_backend is not None: + executor_kwargs["env_backend"] = env_backend + if inference_devices is not None: + executor_kwargs["inference_devices"] = inference_devices + executor = build_executor(config, **executor_kwargs) + metrics = [] + warmup_metadata_by_episode: dict[int, dict[str, Any]] = {} + try: + output_dir = Path(logging_cfg["output_dir"]) + output_dir.mkdir(parents=True, exist_ok=True) + (output_dir / "resolved_config.yaml").write_text( + yaml.safe_dump(config, sort_keys=False), encoding="utf-8" + ) + if isinstance(executor, BatchedExecutor): + warmup_metadata = executor.run_warmup() + run_episodes_kwargs: dict[str, Any] = { + "episode_ids": selected_episode_ids, + "seeds": [_episode_seed(config, episode_id) for episode_id in selected_episode_ids], + "eval_max_steps": eval_max_steps, + } + if on_episode_complete is not None: + run_episodes_kwargs["on_episode_complete"] = on_episode_complete + metrics = list(executor.run_episodes(**run_episodes_kwargs)) + warmup_metadata_by_episode.update( + (episode_id, warmup_metadata) for episode_id in selected_episode_ids + ) + else: + warmup_metadata = executor.run_warmup() + for episode_id in selected_episode_ids: + warmup_metadata_by_episode[episode_id] = warmup_metadata + episode_metrics = executor.run_episode( + episode_id=episode_id, + seed=_episode_seed(config, episode_id), + eval_max_steps=eval_max_steps, + ) + metrics.append(episode_metrics) + if on_episode_complete is not None: + on_episode_complete(episode_metrics) + metrics.sort(key=lambda item: int(item.episode_id)) + for episode_metrics in metrics: + for key, value in _evaluation_raw_fact_metadata(config, int(episode_metrics.episode_id)).items(): + if episode_metrics.metadata.get(key) is None: + episode_metrics.metadata[key] = value + episode_metrics.metadata.update(warmup_metadata_by_episode[int(episode_metrics.episode_id)]) + if "measurement" in config: + episode_metrics.metadata["measurement"] = config["measurement"] + episode_metrics.metadata["config_name"] = experiment_cfg.get("name") + episode_metrics.metadata["run_name"] = experiment_cfg.get("name") + if "checkpoint_path" in policy_cfg: + episode_metrics.metadata["checkpoint_path"] = policy_cfg["checkpoint_path"] + if "profile_path" in config["latency"]: + episode_metrics.metadata["source_profile_path"] = config["latency"]["profile_path"] + if "checkpoint_kind" in policy_cfg: + episode_metrics.metadata["checkpoint_kind"] = str(policy_cfg["checkpoint_kind"]) + episode_metrics.metadata["output_dir"] = str(logging_cfg["output_dir"]) + if "action_prefix" in policy_cfg: + episode_metrics.metadata["action_prefix"] = policy_cfg["action_prefix"] + if extra_metadata: + episode_metrics.metadata.update(extra_metadata) + if executor.logger is not None: + executor.logger.flush() + _record_realtime_eval_latency_plot(config, executor) + if write_summary: + _write_non_sweep_summary(config, metrics) + _record_stratified_replay_videos(config, metrics, seed=seed) + finally: + executor.close() + return metrics + + +def _record_realtime_eval_latency_plot(config: dict[str, Any], executor: Any) -> None: + if ExecutorMode(config["executor"]["mode"]) != ExecutorMode.REALTIME: + return + if not config["logging"]["save_latency_records"]: + return + + latency_values = list(executor.logger.latency_ms_values) + plot_realtime_eval_latency( + latency_values, + Path(config["logging"]["output_dir"]) / "eval_latency_trace.png", + ) + + +def _record_stratified_replay_videos( + config: dict[str, Any], + metrics: list[EpisodeMetrics], + *, + seed: int, +) -> None: + if "video" not in config["logging"]: + return + video_cfg = config["logging"]["video"] + if not video_cfg["enabled"]: + return + if not config["logging"]["save_step_records"]: + # Replay reads steps.jsonl, which is only written when save_step_records is on. + # Without it (e.g. factor-sweep evals) skip video instead of crashing on a missing file. + return + if ExecutorMode(config["executor"]["mode"]) == ExecutorMode.REALTIME: + return + selections = select_episode_return_stratified( + metrics, + num_bins=video_cfg["num_bins"], + seed=seed, + ) + record_videos_from_action_trace(config, selections=selections, metrics=metrics) + + +def _evaluation_raw_fact_metadata(config: dict[str, Any], episode_id: int) -> dict[str, Any]: + env_cfg = config.get("env", {}) + policy_cfg = config.get("policy", {}) + latency_cfg = config.get("latency", {}) + executor_cfg = config.get("executor", {}) + env_fps = float(env_cfg["env_fps"]) if "env_fps" in env_cfg else None + obs_fps = float(env_cfg["obs_fps"]) if "obs_fps" in env_cfg else None + frame_ms = None if env_fps is None or env_fps <= 0 else 1000.0 / env_fps + executor_mode = str(executor_cfg.get("mode", "")).strip().lower() + latency_type = latency_type_from_config(latency_cfg) + if executor_mode == "paused": + latency_type = "zero" + elif executor_mode == "realtime": + latency_type = "measured" + return { + "mode": executor_cfg.get("mode"), + "episode_seed": _episode_seed(config, episode_id), + "policy_id": _metadata_id(policy_cfg, "policy_id", "id", "type"), + "env_id": _metadata_id(env_cfg, "env_id", "id", "name"), + "model_id": latency_cfg.get("model_id"), + "gpu_class": latency_cfg.get("gpu_class"), + "workload_id": latency_cfg.get("workload_id"), + "instance_id": latency_cfg.get("instance_id"), + "source_run_id": latency_cfg.get("source_run_id"), + "profile_ref": latency_cfg.get("profile_ref"), + "env_fps": env_fps, + "obs_fps": obs_fps, + "frame_ms": frame_ms, + "latency_type": latency_type, + } + + +def _metadata_id(config: dict[str, Any], *keys: str) -> str | None: + for key in keys: + value = config.get(key) + if value is not None: + return str(value) + return None diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py new file mode 100644 index 0000000000000000000000000000000000000000..19206ec9dfd77ce730fbaee77fe932956dac0b4d --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py @@ -0,0 +1,341 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path +from typing import Any + +import gymnasium as gym +import numpy as np +import torch +import yaml + + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT)) +sys.path.insert(0, str(ROOT / "third_party" / "MIKASA-Robo")) + +from latency_bench.core.types import Action, Observation # noqa: E402 +from latency_bench.executors.gpu_batched_env_step_backend import ( # noqa: E402 + GpuBatchedEnvStepBackendBase, + SlotStepOutcome, +) +from mikasa_robo_suite.seed_reset import ( # noqa: E402 + reset_seeded_slot as _reset_seeded_slot, + reset_seeded_slots as _reset_seeded_slots, +) + + +ENV_ID = "InterceptGrabFast-VLA-v0" +INSTRUCTION = "Intercept the rolling ball and grasp it to stop it." +START_SEED = 4242424242 +MIKASA_IMAGE_VIEWS_INFO_KEY = "mikasa_image_views" +MIKASA_STATE_INFO_KEY = "mikasa_proprio" + + +def _scalar(value: Any) -> Any: + if torch.is_tensor(value): + return value.detach().reshape(-1)[0].cpu().item() + return np.asarray(value).reshape(-1)[0].item() + + +def _make_raw_env( + obs_mode: str, + num_envs: int = 1, + simulator_device: str = "gpu", +): + import mikasa_robo_suite.vla.memory_envs # noqa: F401 + + return gym.make( + ENV_ID, + num_envs=num_envs, + obs_mode=obs_mode, + control_mode="pd_ee_delta_pose", + render_mode="all", + sim_backend=simulator_device, + render_backend=simulator_device, + reward_mode="normalized_dense", + ) + + +def _make_ppo_env(num_envs: int = 1, simulator_device: str = "gpu"): + from baselines.ppo.ppo_memtasks import FlattenRGBDObservationWrapper + from mani_skill.vector.wrappers.gymnasium import ManiSkillVectorEnv + from mikasa_robo_suite.vla.dataset_collectors.get_mikasa_robo_datasets import ( + env_info, + ) + + env = _make_raw_env( + "state", + num_envs=num_envs, + simulator_device=simulator_device, + ) + wrappers, _ = env_info(ENV_ID) + for wrapper, kwargs in wrappers: + env = wrapper(env, **kwargs) + env = FlattenRGBDObservationWrapper(env, rgb=False, depth=False, state=True) + return ManiSkillVectorEnv( + env, + num_envs, + ignore_terminations=True, + record_metrics=True, + ) + + +def _make_vla_env(num_envs: int = 1, simulator_device: str = "gpu"): + from mikasa_robo_suite.vla.utils.apply_wrappers import apply_mikasa_vla_wrappers + + return apply_mikasa_vla_wrappers( + _make_raw_env( + "rgb", + num_envs=num_envs, + simulator_device=simulator_device, + ), + include_overlays=False, + ) + + +class _PpoPolicy: + def __init__(self, env, checkpoint: Path): + from baselines.ppo.ppo_memtasks import AgentStateOnly + + self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") + self.agent = AgentStateOnly(env).to(self.device) + self.agent.load_state_dict(torch.load(checkpoint, map_location=self.device)) + self.agent.eval() + + def forward(self, observation): + with torch.no_grad(): + return self.agent.get_action( + {key: value.to(self.device) for key, value in observation.items()}, + deterministic=True, + ) + + +class MikasaEnvStepBackend(GpuBatchedEnvStepBackendBase): + """Own the native MIKASA simulator and its 7D action contract.""" + + backend_name = "mikasa_gpu_batched" + + def __init__(self, *, config: dict[str, Any], num_slots: int, env=None): + noop_action = Action( + value=np.asarray(config["env"]["noop_action"], dtype=np.float32), + name="noop", + is_noop=True, + ) + super().__init__( + config=config, + noop_action=noop_action, + num_slots=num_slots, + action_space=gym.spaces.Box(-1.0, 1.0, shape=(7,), dtype=np.float32), + ) + self.env = ( + _make_vla_env( + num_envs=num_slots, + simulator_device=config["env"]["simulator_device"], + ) + if env is None + else env + ) + self._episode_seeds = [0] * num_slots + self._success = np.zeros(num_slots, dtype=np.bool_) + self._observation, _ = self.env.reset(seed=self._episode_seeds) + + def _reset_slot_observation(self, slot_id: int, *, seed: int | None) -> Observation: + if seed is not None: + self._episode_seeds[slot_id] = int(seed) + self._observation, _ = _reset_seeded_slot( + self.env, + slot_id=slot_id, + seed=self._episode_seeds[slot_id], + ) + self._env_steps[slot_id] = 0 + self._success[slot_id] = False + return self._observation_for_slot(slot_id) + + def _observe_slot_observations( + self, + slot_ids: list[int], + ) -> dict[int, Observation]: + return {slot_id: self._observation_for_slot(slot_id) for slot_id in slot_ids} + + def _step_cores( + self, + slot_ids: list[int], + *, + actions: np.ndarray, + active_mask: np.ndarray, + ) -> Any: + del slot_ids, active_mask + tensor_actions = torch.as_tensor( + actions, + dtype=torch.float32, + device=self.env.unwrapped.device, + ) + self._observation, reward, terminated, truncated, info = self.env.step( + tensor_actions + ) + return reward, terminated, truncated, info + + def _slot_step_outcome(self, state: Any, slot_id: int) -> SlotStepOutcome: + reward, terminated, truncated, info = state + success = bool(_slot_value(info["success"], slot_id)) + self._success[slot_id] |= success + return SlotStepOutcome( + reward=float(_slot_value(reward, slot_id)), + done=bool(_slot_value(terminated, slot_id)), + truncated=bool(_slot_value(truncated, slot_id)), + info={ + "success": success, + "task_metrics": {"success": float(self._success[slot_id])}, + }, + ) + + def _observation_for_slot(self, slot_id: int) -> Observation: + rgb = self._observation["rgb"] + if torch.is_tensor(rgb): + rgb = rgb.detach().cpu().numpy() + rgb = np.asarray(rgb) + views = np.stack( + [ + np.asarray(rgb[slot_id, :, :, :3], dtype=np.uint8), + np.asarray(rgb[slot_id, :, :, 3:6], dtype=np.uint8), + ] + ) + metadata = { + MIKASA_IMAGE_VIEWS_INFO_KEY: views, + MIKASA_STATE_INFO_KEY: self._observation["proprio"][slot_id].detach().cpu().numpy(), + "slot_id": slot_id, + } + if "action_prefix_state_key" in self.config["env"]: + metadata["action_prefix_state_key"] = self.config["env"]["action_prefix_state_key"] + if "returned_action_context" in self.config["env"]: + context = self.config["env"]["returned_action_context"] + metadata["returned_action_context"] = { + **context, + "order": np.asarray(context["order"]), + "low": np.asarray(context["low"], dtype=np.float32), + "high": np.asarray(context["high"], dtype=np.float32), + } + return Observation( + data=None, + env_step=int(self._env_steps[slot_id]), + sim_time_ms=float(self._env_steps[slot_id]) * self._frame_ms, + metadata=metadata, + ) + + def close(self) -> None: + if not self.closed: + self.env.close() + super().close() + + +def _slot_value(value: Any, slot_id: int) -> Any: + if torch.is_tensor(value): + return value.detach().reshape(-1)[slot_id].cpu().item() + return np.asarray(value).reshape(-1)[slot_id].item() + + +def _evaluate(args: argparse.Namespace) -> dict[str, Any]: + env = _make_ppo_env() + policy = _PpoPolicy(env, args.checkpoint) + seeds = [] + successes = [] + returns = [] + lengths = [] + try: + for episode_index in range(args.episodes): + seed = START_SEED + episode_index + observation, _ = env.reset(seed=seed) + success_once = False + episode_return = 0.0 + for step in range(60): + action = policy.forward(observation) + observation, reward, terminated, truncated, info = env.step(action) + success_once = success_once or bool(_scalar(info["success"])) + episode_return += float(_scalar(reward)) + if bool(_scalar(terminated)) or bool(_scalar(truncated)): + break + seeds.append(seed) + successes.append(success_once) + returns.append(episode_return) + lengths.append(step + 1) + finally: + env.close() + summary = { + "seeds": seeds, + "successes": successes, + "success_rate": float(np.mean(successes)), + "returns": returns, + "lengths": lengths, + } + (args.output_dir / "summary.json").write_text( + json.dumps(summary, indent=2) + "\n", encoding="utf-8" + ) + return summary + + +def _latency_eval(argv: list[str]) -> None: + from latency_bench.core.config import load_config + from latency_bench.eval.config import apply_runtime_overrides + from latency_bench.eval.driver import run_from_config + + parser = argparse.ArgumentParser() + parser.add_argument("--eval-config", type=Path, required=True) + parser.add_argument("--checkpoint-path", type=Path) + parser.add_argument("--model-config-path", type=Path) + parser.add_argument("--task-contract-path", type=Path) + parser.add_argument("--run-name") + parser.add_argument("--output-dir", type=Path) + parser.add_argument("--latency-method", choices=("zero", "temporal")) + parser.add_argument("--profile-path", type=Path) + parser.add_argument("--latency-seed", type=int) + args = parser.parse_args(argv) + config = load_config(args.eval_config) + apply_runtime_overrides( + config, + checkpoint_path=args.checkpoint_path, + model_config_path=args.model_config_path, + task_contract_path=args.task_contract_path, + run_name=args.run_name, + output_dir=args.output_dir, + latency_method=args.latency_method, + latency_profile_path=args.profile_path, + latency_seed=args.latency_seed, + ) + output_dir = Path(config["logging"]["output_dir"]) + output_dir.mkdir(parents=True, exist_ok=True) + (output_dir / "eval_config.yaml").write_text( + yaml.safe_dump(config, sort_keys=False), encoding="utf-8" + ) + backend = MikasaEnvStepBackend( + config=config, + num_slots=int(config["evaluation"]["eval_parallel_envs"]), + ) + run_from_config( + config, + env_backend=backend, + inference_devices=config["executor"]["inference_devices"], + ) + + +def main() -> None: + if sys.argv[1:2] == ["latency-eval"]: + _latency_eval(sys.argv[2:]) + return + + parser = argparse.ArgumentParser() + parser.add_argument("--policy", choices=("ppo",), required=True) + parser.add_argument("--checkpoint", type=Path) + parser.add_argument("--episodes", type=int, default=50) + parser.add_argument("--output-dir", type=Path, required=True) + args = parser.parse_args() + args.output_dir.mkdir(parents=True, exist_ok=True) + + _evaluate(args) + + +if __name__ == "__main__": + main() diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py new file mode 100644 index 0000000000000000000000000000000000000000..7b9cab3b3fad77d6f6a3b7919ff4a8124c98d011 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py @@ -0,0 +1,1207 @@ +from __future__ import annotations + +import json +import sys +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any + +import numpy as np +import numpy.typing as npt +from PIL import Image + +from latency_bench.core.actions import ActionResolver +from latency_bench.core.clock import EnvClock +from latency_bench.core.timing import current_profiler +from latency_bench.core.types import Action, Observation, PolicyOutput +from latency_bench.data.ghost_trail import GhostTrailConfig, build_flappy_ghost_trail_window +from latency_bench.data.state_normalization import min_max_normalize_state +from latency_bench.envs.raw_rgb import ENV_RAW_RGB_FRAME_STACK_INFO_KEY +from latency_bench.envs.gymnasium_task import ( + gymnasium_action_space_contract, + gymnasium_task_contract, +) +from latency_bench.policy.base import PolicyRunner +from latency_bench.policy.starvla_prompts import load_latency_prompt_map, resolve_starvla_prompt + +from latency_bench.utils.paths import REPO_ROOT + + +STARVLA_ROOT = REPO_ROOT / "third_party" / "starVLA" +STATEFUL_STARVLA_MODEL_IDS: tuple[str, ...] = ( + "pi0", + "pi-0", + "pi05", + "pi-0.5", + "gr00t", + "qwenpi", + "qwenpi_v3", + "qwengr00t", +) +STATELESS_STARVLA_MODEL_IDS: tuple[str, ...] = ( + "openvla", + "qwenoft", +) + +DEMON_ATTACK_ACTION_LABELS: tuple[str, ...] = ( + "NOOP", + "FIRE", + "RIGHT", + "LEFT", + "RIGHTFIRE", + "LEFTFIRE", +) +DEADLY_CORRIDOR_TURN_LABELS: tuple[str, ...] = ( + "TURN_NOOP", + "TURN_LEFT", + "TURN_RIGHT", +) +DEADLY_CORRIDOR_MOVE_LABELS: tuple[str, ...] = ( + "MOVE_NOOP", + "MOVE_FORWARD", + "MOVE_BACKWARD", +) +DEADLY_CORRIDOR_STRAFE_LABELS: tuple[str, ...] = ( + "STRAFE_NOOP", + "MOVE_LEFT", + "MOVE_RIGHT", +) +DEADLY_CORRIDOR_ATTACK_LABELS: tuple[str, ...] = ( + "ATTACK_NOOP", + "ATTACK", +) + + +class StarVlaPolicyRunner(PolicyRunner): + """Translate observations and model outputs using the task action contract.""" + + def __init__( + self, + *, + wrapper: Any, + checkpoint_path: str, + device: str, + unnorm_key: str | None, + env_name: str, + action_resolver: ActionResolver, + action_refs: Sequence[Any], + latency_prompt_map: dict[str, Any] | None = None, + base_prompt: str | None = None, + latency_prompt_key: int | str | None = None, + prompt_mode: str | None = None, + obs_resize: tuple[int, int] | None = None, + image_transform_config: Mapping[str, Any] | None = None, + observation_stride_raw_frames: int, + model_cfg: Mapping[str, Any] | None = None, + state_normalization: Mapping[str, Any] | None = None, + state_source: str | None = None, + image_views_info_key: str | None = None, + action_output_type: str | None = None, + ) -> None: + self._wrapper = wrapper + self._obs_resize = tuple(obs_resize) if obs_resize else None + self._checkpoint_path = checkpoint_path + self._device = device + self._unnorm_key = unnorm_key + self._env_name = env_name + self._action_by_raw_id = { + raw_action_id: action_resolver.resolve(action_ref) + for raw_action_id, action_ref in enumerate(action_refs) + } + self._latency_prompt_map = latency_prompt_map + self._base_prompt = base_prompt + self._latency_prompt_key = latency_prompt_key + self._prompt_mode = str(prompt_mode or "default").strip().lower() + self._image_transform_config = dict(image_transform_config or {"image_transform": "raw_rgb"}) + self._image_transform = str( + self._image_transform_config.get("image_transform", "raw_rgb") or "raw_rgb" + ).strip().lower() + model_cfg = ( + _normalized_model_cfg_from_wrapper(wrapper) + if model_cfg is None + else _normalized_model_cfg(model_cfg) + ) + self._include_state = _include_state_from_model_cfg(model_cfg) + self._state_dim = _state_dim_from_model_cfg(model_cfg) if self._include_state else None + self._state_normalization = dict(state_normalization or {}) + self._state_source = state_source + self._image_views_info_key = image_views_info_key + self._action_output_type = action_output_type + vla_data = (model_cfg.get("datasets", {}) or {}).get("vla_data", {}) or {} + self._pack_image_sequence = ( + bool(vla_data["pack_image_sequence"]) + if "pack_image_sequence" in vla_data + else False + ) + self._image_sequence_length = ( + int(vla_data["image_sequence_length"]) + if self._pack_image_sequence + else 1 + ) + self._observation_stride_raw_frames = int(observation_stride_raw_frames) + self._image_sequence_raw_span = ( + 1 + + (self._image_sequence_length - 1) + * self._observation_stride_raw_frames + ) + self._num_obs_frames = int(vla_data.get("num_obs_frames", 1) or 1) + self._image_mode = str(vla_data.get("image_mode", "single")) + self._stitch_grid = tuple(vla_data.get("stitch_grid", [2, 2])) + framework_cfg = model_cfg["framework"] + kv_cfg = framework_cfg["kv_memory"] if "kv_memory" in framework_cfg else {} + self._kv_memory_enabled = bool(kv_cfg["enabled"]) if "enabled" in kv_cfg else False + + def reset_state(self, slot_id: int | None = None) -> None: + # Clears the model's per-slot KV memory at episode boundaries (req3). + # No-op unless the framework maintains KV memory. + reset = getattr(self._wrapper, "reset_memory", None) + if callable(reset): + reset(slot_id) + + def predict(self, observation: Observation) -> PolicyOutput: + return self.predict_batch([observation])[0] + + def predict_batch(self, observations: Sequence[Observation]) -> list[PolicyOutput]: + profiler = current_profiler() + with profiler.time("policy_build_example_ms"): + examples = [self._build_example(observation) for observation in observations] + with profiler.time("policy_wrapper_predict_action_ms"): + prediction = self._wrapper.predict_action( + examples=examples, unnorm_key=self._unnorm_key, profiler=profiler + ) + with profiler.time("policy_decode_ms"): + outputs = [ + self._decode_prediction( + prediction=prediction, + index=index, + observation=observation, + example=example, + ) + for index, (observation, example) in enumerate(zip(observations, examples)) + ] + return outputs + + def _decode_prediction( + self, + *, + prediction: dict[str, Any], + index: int, + observation: Observation, + example: dict[str, Any], + ) -> PolicyOutput: + actions = np.asarray(prediction["actions"]) + raw_action_scores = ( + np.asarray(prediction["raw_action_scores"]) + if "raw_action_scores" in prediction + else None + ) + return self._policy_output( + observation=observation, + example=example, + action_payload=actions[index, 0], + action_output_type=( + prediction["action_output_type"] + if self._action_output_type is None + else self._action_output_type + ), + raw_action_scores=None if raw_action_scores is None else raw_action_scores[index, 0], + ) + + def _build_example(self, observation: Observation) -> dict[str, Any]: + frame_source = observation.metadata[ + ENV_RAW_RGB_FRAME_STACK_INFO_KEY + if self._image_views_info_key is None + else self._image_views_info_key + ] + frames = observation_data_to_hwc_uint8_frames(frame_source) # oldest .. newest + transformed = self._transformed_frame(frames=frames, observation=observation) + + if self._image_views_info_key is not None: + pass + elif self._pack_image_sequence: + if transformed is not None: + raise ValueError( + "WanOFT packed image sequences require image_transform=raw_rgb" + ) + if len(frames) < self._image_sequence_raw_span: + raise ValueError( + "WanOFT packed image sequence requires " + f"{self._image_sequence_raw_span} raw frames for " + f"{self._image_sequence_length} decision observations at stride " + f"{self._observation_stride_raw_frames}, got {len(frames)}" + ) + frames = frames[ + -self._image_sequence_raw_span + :: self._observation_stride_raw_frames + ] + elif transformed is not None: + frames = [transformed] + elif self._image_mode == "single" or self._kv_memory_enabled: + frames = frames[-1:] + else: + # Select the temporal observation window to match training (_pack_sample). + raw_span = 1 + (self._num_obs_frames - 1) * self._observation_stride_raw_frames + frames = frames[-raw_span :: self._observation_stride_raw_frames] + + prompt = resolve_starvla_prompt( + env_name=self._env_name, + observation_metadata=observation.metadata, + latency_prompt_map=self._latency_prompt_map, + base_prompt=self._base_prompt, + latency_prompt_key=self._latency_prompt_key, + prompt_mode=self._prompt_mode, + ) + + if self._image_mode == "stitch": + if transformed is not None: + raise ValueError("image_transform is not compatible with image_mode=stitch") + # Tile the window into one image; matches _pack_sample's stitch branch + # (raw frames passed to stitch_frames, which resizes each cell to 224). + images = [_get_stitch_frames()(frames, grid=self._stitch_grid, size=(224, 224))] + else: + if self._obs_resize is not None: + height, width = self._obs_resize + # match training preprocessing exactly: gr00t LeRobotSingleDataset._pack_sample + # does `Image.fromarray(image).resize((224, 224))` (PIL default resample = BICUBIC). + frames = [ + np.asarray(Image.fromarray(frame).resize((width, height)), dtype=np.uint8) + for frame in frames + ] + images = [Image.fromarray(frame) for frame in frames] + + example = { + "image": images, + "lang": prompt, + } + if self._kv_memory_enabled: + example["slot_id"] = observation.metadata["slot_id"] + elif "slot_id" in observation.metadata: + example["slot_id"] = observation.metadata["slot_id"] + if self._include_state: + if self._state_source == "transport": + state = np.asarray(observation.data["transport"], dtype=np.float32) + example["state"] = state.reshape(1, self._state_dim) + elif self._state_normalization: + state = np.asarray( + observation.metadata["gymnasium_state"], dtype=np.float32 + ) + state_min = np.asarray(self._state_normalization["min"], dtype=np.float32) + state_max = np.asarray(self._state_normalization["max"], dtype=np.float32) + state = min_max_normalize_state(state, state_min, state_max) + example["state"] = state.reshape(1, self._state_dim) + else: + example["state"] = np.zeros((1, self._state_dim), dtype=np.float32) + return example + + def _transformed_frame( + self, + *, + frames: Sequence[npt.NDArray[np.uint8]], + observation: Observation, + ) -> npt.NDArray[np.uint8] | None: + if self._image_transform in {"", "none", "raw", "raw_rgb"}: + return None + if self._image_transform not in {"flappy_ghost_trail", "demon_attack_ghost_trail"}: + raise ValueError(f"Unsupported StarVLA image_transform={self._image_transform!r}") + if self._image_transform == "flappy_ghost_trail" and self._env_name != "flappy": + raise ValueError("image_transform=flappy_ghost_trail is only supported for env_name=flappy") + if self._image_transform == "demon_attack_ghost_trail" and self._env_name != "demon_attack": + raise ValueError("image_transform=demon_attack_ghost_trail is only supported for env_name=demon_attack") + + config = GhostTrailConfig( + image_transform=self._image_transform, + history_frames=int(self._image_transform_config.get("history_frames", 5)), + gamma=float(self._image_transform_config.get("gamma", 1.3)), + min_alpha=int(self._image_transform_config.get("min_alpha", 35)), + ground_fraction=float(self._image_transform_config.get("ground_fraction", 0.22)), + scroll_px_per_step=float(self._image_transform_config.get("scroll_px_per_step", 4.0)), + ) + if self._image_transform == "demon_attack_ghost_trail": + # env_step counts raw ALE frames (buffer updated 4× per decision step). + # frames[-0:] == frames, so env_step=0 falls back to the full reset-fill buffer. + valid_count = min(len(frames), int(observation.env_step)) + else: + max_frames = max(1, int(config.history_frames) + 1) + valid_count = min(len(frames), max(1, int(observation.env_step) + 1), max_frames) + window = [np.asarray(frame, dtype=np.uint8) for frame in frames[-valid_count:]] + + if self._image_transform == "demon_attack_ghost_trail": + from latency_bench.data.ghost_trail_demon import build_demon_attack_ghost_trail_window + steps_arg = list(range(len(window))) + return build_demon_attack_ghost_trail_window(window, steps_arg, config=config) + + current_step = int(observation.env_step) + start_step = current_step - valid_count + 1 + steps = list(range(start_step, current_step + 1)) + return build_flappy_ghost_trail_window(window, steps, config=config) + + def _policy_output( + self, + *, + observation: Observation, + example: dict[str, Any], + action_payload: npt.NDArray[Any], + action_output_type: str, + raw_action_scores: npt.NDArray[Any] | None, + ) -> PolicyOutput: + payload = np.asarray(action_payload) + action, action_metadata = action_from_starvla_payload( + payload=payload, + env_name=self._env_name, + action_by_raw_id=self._action_by_raw_id, + action_output_type=action_output_type, + ) + metadata = { + "policy_type": "starvla", + "prompt_source": "latency_prompt_map" if self._latency_prompt_map is not None else "base", + "checkpoint_path": self._checkpoint_path, + "unnorm_key": self._unnorm_key, + "device": self._device, + "input_frame_count": len(example["image"]), + "image_transform": self._image_transform, + "action_output_type": action_output_type, + "action_payload": to_jsonable_action_payload(payload), + "kv_memory_enabled": self._kv_memory_enabled, + **action_metadata, + } + if self._pack_image_sequence: + metadata["image_sequence_length"] = self._image_sequence_length + metadata["input_frame_raw_stride"] = self._observation_stride_raw_frames + metadata["input_frame_raw_span"] = self._image_sequence_raw_span + if "slot_id" in example: + metadata["slot_id"] = example["slot_id"] + if raw_action_scores is not None: + metadata["raw_action_scores"] = [ + float(item) for item in np.asarray(raw_action_scores, dtype=np.float32).tolist() + ] + if "latency_raw_frames" in observation.metadata: + metadata["latency_raw_frames"] = observation.metadata["latency_raw_frames"] + if "latency_ms" in observation.metadata: + metadata["latency_ms"] = observation.metadata["latency_ms"] + if self._latency_prompt_key is not None: + metadata["latency_prompt_key"] = self._latency_prompt_key + return PolicyOutput( + action=action, + raw_output=metadata["action_payload"], + metadata=metadata, + ) + + +def observation_data_to_hwc_uint8_frames(data: Any) -> list[npt.NDArray[np.uint8]]: + frame = _extract_observation_array(data) + if frame.ndim == 4 and frame.shape[-1] == 3: + return [_as_uint8_image(item) for item in frame] + if frame.ndim == 4 and frame.shape[1] == 3: + return [_as_uint8_image(np.transpose(item, (1, 2, 0))) for item in frame] + if frame.ndim == 3 and frame.shape[-1] == 3: + return [_as_uint8_image(frame)] + if frame.ndim == 3 and frame.shape[0] == 3: + return [_as_uint8_image(np.transpose(frame, (1, 2, 0)))] + if ( + frame.ndim == 3 + and frame.shape[0] % 3 == 0 + and frame.shape[0] < frame.shape[1] + and frame.shape[0] < frame.shape[2] + ): + return [ + _as_uint8_image(np.transpose(frame[start : start + 3, :, :], (1, 2, 0))) + for start in range(0, frame.shape[0], 3) + ] + if frame.ndim == 3 and frame.shape[-1] % 3 == 0: + return [ + _as_uint8_image(frame[:, :, start : start + 3]) + for start in range(0, frame.shape[-1], 3) + ] + if frame.ndim == 3 and frame.shape[0] % 3 == 0: + return [ + _as_uint8_image(np.transpose(frame[start : start + 3, :, :], (1, 2, 0))) + for start in range(0, frame.shape[0], 3) + ] + return [_as_uint8_image(frame)] + + +def decode_starvla_action( + *, + vector: npt.NDArray[Any], + env_name: str, + action_by_raw_id: Mapping[int, Action], + action_layout: str | None = None, +) -> tuple[Action, dict[str, Any]]: + deadly_layout = None + if str(env_name) == "deadly_corridor": + action_dim = int(np.asarray(vector).shape[-1]) + deadly_layouts = { + 7: "deadly_corridor_semantic_7", + 11: "deadly_corridor_factorized_11", + 54: "deadly_corridor_joint_54", + } + if action_dim not in deadly_layouts: + raise ValueError( + "Deadly Corridor StarVLA action vector expected 7, 11, or 54 " + f"values, got {action_dim}" + ) + deadly_layout = deadly_layouts[action_dim] + asterix_layout = None + if str(env_name) == "asterix": + action_dim = int(np.asarray(vector).shape[-1]) + if action_layout is not None: + asterix_layout = str(action_layout).strip().lower() + else: + asterix_layout = "factorized_6" if action_dim < 9 else "discrete_9" + + decode_rl_games_actions, _, _ = _load_rl_games_action_decode() + prediction = decode_rl_games_actions( + normalized_actions=np.asarray(vector), + env_name=str(env_name), + deadly_action_layout=(deadly_layout.removeprefix("deadly_corridor_") if deadly_layout is not None else None), + asterix_action_layout=asterix_layout, + ) + action, metadata = action_from_starvla_payload( + payload=np.asarray(prediction["actions"]), + env_name=env_name, + action_by_raw_id=action_by_raw_id, + action_output_type=prediction["action_output_type"], + ) + if deadly_layout is not None: + metadata["action_layout"] = deadly_layout + if deadly_layout == "deadly_corridor_joint_54": + turn, move, strafe, attack = action.value + metadata["raw_action_id"] = turn * 18 + move * 6 + strafe * 2 + attack + elif deadly_layout == "deadly_corridor_semantic_7": + semantic_actions = ( + [0, 1, 0, 0], + [0, 2, 0, 0], + [0, 0, 1, 0], + [0, 0, 2, 0], + [1, 0, 0, 0], + [2, 0, 0, 0], + [0, 0, 0, 1], + ) + metadata["raw_action_id"] = semantic_actions.index(action.value) + if asterix_layout is not None: + metadata["action_layout"] = asterix_layout + return action, metadata + + +# Fixed semantic button order the StarVLA multibinary head is trained against. +# Mirrors starVLA.training.rl_games.eval_core._semantic_to_runtime_multibinary; +# the env adapter re-orders this to the live ViZDoom button layout. +DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER = ( + "MOVE_FORWARD", + "MOVE_BACKWARD", + "MOVE_LEFT", + "MOVE_RIGHT", + "TURN_LEFT", + "TURN_RIGHT", + "ATTACK", +) + + +def action_from_starvla_payload( + *, + payload: npt.NDArray[Any], + env_name: str, + action_by_raw_id: Mapping[int, Action], + action_output_type: str = "", +) -> tuple[Action, dict[str, Any]]: + if str(action_output_type) == "rl_games_continuous": + values = [float(item) for item in np.asarray(payload).reshape(-1).tolist()] + return Action( + value=values, + name="continuous_torque", + is_noop=all(value == 0.0 for value in values), + is_oneshot=False, + ), {"continuous_action": values} + if str(env_name) == "demon_attack": + return demon_attack_action_from_id(int(np.asarray(payload).reshape(-1)[0])) + if str(env_name) == "deadly_corridor": + # Multibinary heads emit an already-thresholded 7-dim button vector in + # fixed semantic order; the env adapter re-orders it to the live ViZDoom + # button layout. Keep it as-is rather than reinterpreting it as a + # [turn, move, strafe, attack] categorical tuple. + if str(action_output_type) == "rl_games_deadly_corridor_multibinary": + buttons = [int(item) for item in np.asarray(payload).reshape(-1).tolist()] + active = [ + DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER[idx] + for idx, pressed in enumerate(buttons) + if idx < len(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) and pressed + ] + action_name = "+".join(active) if active else "NOOP" + return Action( + value=buttons, + name=action_name, + is_noop=not any(buttons), + is_oneshot=False, + ), { + "decoded_multibinary_buttons": buttons, + "action_label": action_name, + "action_layout": "deadly_corridor_multibinary_7", + } + return deadly_corridor_action_from_tuple( + action_value=[int(item) for item in np.asarray(payload).reshape(-1).tolist()], + metadata={"action_layout": "deadly_corridor_tuple"}, + ) + raw_action_id = int(np.asarray(payload).reshape(-1)[0]) + return action_by_raw_id[raw_action_id], {"raw_action_id": raw_action_id} + + +def to_jsonable_action_payload(payload: npt.NDArray[Any]) -> Any: + value = np.asarray(payload).tolist() + if isinstance(value, list) and len(value) == 1: + return value[0] + return value + + +def demon_attack_action_from_id(action_id: int) -> tuple[Action, dict[str, Any]]: + action = Action( + value=action_id, + name=DEMON_ATTACK_ACTION_LABELS[action_id], + is_noop=action_id == 0, + is_oneshot=False, + ) + return action, {"raw_action_id": action_id, "action_label": action.name} + + +def deadly_corridor_action_from_tuple( + *, + action_value: list[int], + metadata: dict[str, Any], +) -> tuple[Action, dict[str, Any]]: + turn, move, strafe, attack = action_value + action_value = [turn, move, strafe, attack] + turn_label = DEADLY_CORRIDOR_TURN_LABELS[turn] + move_label = DEADLY_CORRIDOR_MOVE_LABELS[move] + strafe_label = DEADLY_CORRIDOR_STRAFE_LABELS[strafe] + attack_label = DEADLY_CORRIDOR_ATTACK_LABELS[attack] + active_labels = [ + label + for label in (turn_label, move_label, strafe_label, attack_label) + if not label.endswith("_NOOP") + ] + action_name = "+".join(active_labels) if active_labels else "NOOP" + return Action( + value=action_value, + name=action_name, + is_noop=action_value == [0, 0, 0, 0], + is_oneshot=False, + ), { + "decoded_action_tuple": action_value, + "turn_label": turn_label, + "move_label": move_label, + "strafe_label": strafe_label, + "attack_label": attack_label, + "action_label": action_name, + **metadata, + } + + +def _extract_observation_array(data: Any) -> npt.NDArray[Any]: + if isinstance(data, Mapping): + return np.asarray(data["observation"]) + return np.asarray(data) + + +def _as_uint8_image(frame: npt.NDArray[Any]) -> npt.NDArray[np.uint8]: + return np.ascontiguousarray(frame, dtype=np.uint8) + + +def _normalized_model_cfg(model_cfg: Mapping[str, Any]) -> dict[str, Any]: + _ensure_starvla_path() + from omegaconf import OmegaConf + from starVLA.model.framework.share_tools import apply_config_compat + + cfg = OmegaConf.create(model_cfg) + apply_config_compat(cfg) + _apply_model_family_include_state_compat(cfg) + return OmegaConf.to_container(cfg, resolve=True) + + +def _normalized_model_cfg_from_wrapper(wrapper: Any) -> dict[str, Any]: + return _normalized_model_cfg(wrapper._model_cfg) + + +def _load_starvla_model_config(path: str | Path) -> dict[str, Any]: + from omegaconf import OmegaConf + + return _normalized_model_cfg(OmegaConf.load(path)) + + +def _apply_model_family_include_state_compat(cfg: Any) -> None: + from omegaconf import OmegaConf + + if OmegaConf.select(cfg, "datasets.vla_data.include_state") is not None: + return + + model_ids = ( + _normalized_optional_config_string(cfg, ("model",)), + _normalized_optional_config_string(cfg, ("rl_games", "model_alias")), + _normalized_optional_config_string(cfg, ("framework", "name")), + ) + if any(model_id in STATEFUL_STARVLA_MODEL_IDS for model_id in model_ids if model_id is not None): + OmegaConf.update(cfg, "datasets.vla_data.include_state", True, force_add=True) + return + if any(model_id in STATELESS_STARVLA_MODEL_IDS for model_id in model_ids if model_id is not None): + OmegaConf.update(cfg, "datasets.vla_data.include_state", False, force_add=True) + + +def _normalized_optional_config_string(cfg: Any, path: tuple[str, ...]) -> str | None: + from omegaconf import OmegaConf + + value = OmegaConf.select(cfg, ".".join(path)) + if value is None: + return None + return str(value).strip().lower() + + +def _state_dim_from_model_cfg(model_cfg: dict[str, Any]) -> int: + return model_cfg["framework"]["action_model"]["state_dim"] + + +def _include_state_from_model_cfg(model_cfg: dict[str, Any]) -> bool: + return model_cfg["datasets"]["vla_data"]["include_state"] + + +_STITCH_FRAMES = None + + +def _get_stitch_frames(): + """Lazily import starVLA's stitch_frames (starVLA path is added at runtime).""" + global _STITCH_FRAMES + if _STITCH_FRAMES is None: + _ensure_starvla_path() + from starVLA.training.trainer_utils.trainer_tools import stitch_frames + + _STITCH_FRAMES = stitch_frames + return _STITCH_FRAMES + + +def _ensure_starvla_path() -> None: + starvla_root = str(STARVLA_ROOT) + if starvla_root not in sys.path: + sys.path.insert(0, starvla_root) + + +def _observation_stride_raw_frames(config: Mapping[str, Any]) -> int: + env_cfg = config["env"] + return EnvClock( + env_fps=float(env_cfg["env_fps"]), + obs_fps=float(env_cfg["obs_fps"]), + ).obs_stride_raw_frames + + +def apply_starvla_model_input_config( + config: dict[str, Any], + *, + model_cfg: Mapping[str, Any], + image_transform: str = "raw_rgb", +) -> None: + """Match latency_bench's raw frame stack to a saved StarVLA input contract.""" + vla_data = model_cfg["datasets"]["vla_data"] + pack_image_sequence = ( + bool(vla_data["pack_image_sequence"]) + if "pack_image_sequence" in vla_data + else False + ) + normalized_transform = str(image_transform).strip().lower() + raw_image_transform = normalized_transform in {"", "none", "raw", "raw_rgb"} + if pack_image_sequence: + if not raw_image_transform: + raise ValueError( + "WanOFT packed image sequences require image_transform=raw_rgb" + ) + input_frame_count = int(vla_data["image_sequence_length"]) + else: + if not raw_image_transform: + return + framework_cfg = model_cfg["framework"] + kv_cfg = framework_cfg["kv_memory"] if "kv_memory" in framework_cfg else {} + kv_memory_enabled = bool(kv_cfg["enabled"]) if "enabled" in kv_cfg else False + if kv_memory_enabled: + return + image_mode = str(vla_data["image_mode"]) if "image_mode" in vla_data else "single" + if image_mode == "single": + return + input_frame_count = int(vla_data["num_obs_frames"]) + + observation_stride = _observation_stride_raw_frames(config) + required_raw_frames = 1 + (input_frame_count - 1) * observation_stride + config["env"]["frame_stack"] = max( + int(config["env"]["frame_stack"]), + required_raw_frames, + ) + + +def prepare_starvla_checkpoint_input_config(config: dict[str, Any]) -> None: + """Apply the saved checkpoint input contract before env construction.""" + if config["policy"]["type"] != "starvla": + return + + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + if config["env"]["name"] == "gymnasium": + contract = json.loads( + Path(policy_cfg["task_contract_path"]).read_text(encoding="utf-8") + ) + config["env"]["state_space"] = {"labels": contract["state_labels"]} + if contract["robot_type"] in ("latency_balance_profile_h8", "latency_balance_profile_h16"): + config["env"]["name"] = "balance_profile" + config["env"]["action_context_horizon"] = contract["action_horizon"] + config["env"]["frame_stack"] = 1 + return + if "model_config_path" in policy_cfg: + model_cfg = _load_starvla_model_config(policy_cfg["model_config_path"]) + else: + _ensure_starvla_path() + from starVLA.model.framework.share_tools import read_mode_config + + saved_model_cfg, _norm_stats = read_mode_config(policy_cfg["checkpoint_path"]) + model_cfg = _normalized_model_cfg(saved_model_cfg) + if config["env"]["name"] == "gymnasium": + image_size = model_cfg["rl_games"]["env_eval"]["image_size"] + config["env"]["obs_resize"] = [image_size, image_size] + image_transform_cfg = ( + policy_cfg["image_transform_config"] + if "image_transform_config" in policy_cfg + else {} + ) + image_transform = ( + image_transform_cfg["image_transform"] + if "image_transform" in image_transform_cfg + else "raw_rgb" + ) + apply_starvla_model_input_config( + config, + model_cfg=model_cfg, + image_transform=image_transform, + ) + + +def _load_policy_wrapper_class() -> Any: + _ensure_starvla_path() + from deployment.model_server.policy_wrapper import PolicyServerWrapper + + return PolicyServerWrapper + + +def _profiler_stage(profiler: Any, name: str) -> Any: + from contextlib import nullcontext + + return profiler.time(name) if profiler is not None else nullcontext() + + +def _load_rl_games_action_decode() -> tuple[Any, Any, Any]: + _ensure_starvla_path() + from deployment.model_server.rl_games_action_decode import ( + decode_rl_games_actions, + resolve_asterix_action_decode_spec, + resolve_deadly_action_decode_spec, + ) + + return decode_rl_games_actions, resolve_deadly_action_decode_spec, resolve_asterix_action_decode_spec + + +class LiveStarVlaWrapper: + """In-process stand-in for ``PolicyServerWrapper`` over a *live* framework. + + During training the trainer already holds the model in memory + (``accelerator.unwrap_model(self.model)`` — the same object eval_core calls). + This wrapper exposes only the rl_games-mode surface ``StarVlaPolicyRunner`` + uses — ``predict_action`` (framework forward + rl_games decode), + ``reset_memory`` passthrough, and the ``_model_cfg`` attribute — so no + checkpoint reload is needed. The disk-backed ``PolicyNormProcessor`` is never + built because rl_games decoding ignores un-normalization stats. + """ + + def __init__( + self, + *, + framework: Any, + model_cfg: dict[str, Any], + env_name: str, + rl_games_action_env_dim: int | None = None, + gymnasium_action_space_type: str = "discrete", + action_layout: str | None = None, + multibinary_threshold: float | None = None, + ) -> None: + self._framework = framework + self._model_cfg = model_cfg + self._rl_games_env_name = str(env_name) + self._rl_games_action_env_dim = rl_games_action_env_dim + self._gymnasium_action_space_type = gymnasium_action_space_type + ( + self._decode_rl_games_actions, + resolve_deadly_action_decode_spec, + resolve_asterix_action_decode_spec, + ) = _load_rl_games_action_decode() + self._action_layout = action_layout + self._multibinary_threshold = multibinary_threshold + if self._rl_games_env_name == "deadly_corridor": + self._action_layout, self._multibinary_threshold = resolve_deadly_action_decode_spec( + model_cfg, + action_layout=action_layout, + multibinary_threshold=multibinary_threshold, + ) + elif self._rl_games_env_name == "asterix": + self._action_layout = resolve_asterix_action_decode_spec( + model_cfg, + action_layout=action_layout, + ) + + def reset_memory(self, slot_id: int | None = None) -> None: + reset = getattr(self._framework, "reset_memory", None) + if callable(reset): + reset(slot_id) + + def predict_action( + self, + examples: list[dict[str, Any]], + unnorm_key: str | None = None, + **kwargs: Any, + ) -> dict[str, Any]: + # unnorm_key is unused in rl_games mode; kept for interface parity. + del unnorm_key + profiler = kwargs["profiler"] if "profiler" in kwargs else None + out = self._framework.predict_action(examples=examples, **kwargs) + normalized = np.asarray(out["normalized_actions"]) # (B, T, D) + decode_kwargs: dict[str, Any] = {} + if self._rl_games_env_name == "gymnasium": + decode_kwargs["action_env_dim"] = self._rl_games_action_env_dim + if self._gymnasium_action_space_type == "box": + decode_kwargs["gymnasium_action_space_type"] = "box" + with _profiler_stage(profiler, "starvla_wrapper_rl_games_decode_ms"): + return self._decode_rl_games_actions( + normalized_actions=normalized, + env_name=self._rl_games_env_name, + deadly_action_layout=( + self._action_layout + if self._rl_games_env_name == "deadly_corridor" + else None + ), + deadly_multibinary_threshold=( + self._multibinary_threshold + if self._rl_games_env_name == "deadly_corridor" + else None + ), + asterix_action_layout=( + self._action_layout + if self._rl_games_env_name == "asterix" + else None + ), + **decode_kwargs, + ) + + +_LEGACY_GYMNASIUM_TASK_NAMES = { + "ant_rgb_state": "ant", + "half_cheetah_rgb_state": "half_cheetah", + "hopper_rgb_state": "hopper", + "humanoid_rgb_state": "humanoid", + "inverted_pendulum_rgb_state": "inverted_pendulum", + "swimmer_rgb_state": "swimmer", + "walker2d_rgb_state": "walker2d", +} + + +def _canonical_gymnasium_contract_namespace( + contract: Mapping[str, Any], +) -> dict[str, Any]: + canonical = dict(contract) + task_name = canonical["task_name"] + if task_name in _LEGACY_GYMNASIUM_TASK_NAMES: + canonical["task_name"] = _LEGACY_GYMNASIUM_TASK_NAMES[task_name] + if canonical["env_id"] == "LatencyBench/HopperRgbState-v0": + canonical["env_id"] = "LatencyBench/Hopper-v0" + canonical["registration_imports"] = [ + "latency_bench.envs.gymnasium_hopper" + if module == "latency_bench.envs.gymnasium_hopper_rgb_state" + else module + for module in canonical["registration_imports"] + ] + return canonical + + +def _validate_gymnasium_starvla_contract( + *, + env_cfg: Mapping[str, Any], + policy_cfg: Mapping[str, Any], + model_cfg: Mapping[str, Any], + manifest: Mapping[str, Any], +) -> None: + eval_contract = gymnasium_task_contract(env_cfg) + manifest_task = manifest.get("gymnasium_task") + expected = policy_cfg.get( + "gymnasium_training_task_contract", manifest_task or eval_contract + ) + comparable_eval_contract = {**eval_contract, "make_kwargs": expected["make_kwargs"]} + if _canonical_gymnasium_contract_namespace( + comparable_eval_contract + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA training contract or dataset manifest" + ) + if manifest.get("integration_name", "gymnasium") != "gymnasium": + raise ValueError("StarVLA task manifest is not a Gymnasium handoff") + if manifest_task is not None: + if _canonical_gymnasium_contract_namespace( + manifest_task + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA dataset manifest" + ) + model_contract = model_cfg["datasets"]["vla_data"].get("gymnasium_task_contract") + if model_contract is not None: + if _canonical_gymnasium_contract_namespace( + model_contract + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA model config" + ) + action_space = gymnasium_action_space_contract(env_cfg) + action_layout = str(policy_cfg.get("action_layout", "") or "").strip().lower() + is_asterix_factorized = ( + str(env_cfg.get("task_name", "")) == "asterix" + and action_layout in {"factorized_6", "factorized6", "asterix_factorized_6", "asterix_factorized6"} + ) + if not is_asterix_factorized and manifest["active_action_dim"] != len(action_space["labels"]): + raise ValueError( + "StarVLA dataset active_action_dim does not match its Gymnasium action catalog" + ) + if ( + model_cfg["framework"]["action_model"]["action_env_dim"] + != manifest["active_action_dim"] + ): + raise ValueError( + "StarVLA model action_env_dim does not match the dataset manifest" + ) + model_uses_state = bool(model_cfg["datasets"]["vla_data"]["include_state"]) + manifest_has_state_metadata = ( + "uses_state" in manifest or "state_labels" in manifest + ) + manifest_uses_state = bool(manifest.get("uses_state", model_uses_state)) + if manifest_has_state_metadata: + if policy_cfg.get("state_source") != "transport" and manifest_uses_state != ("state_space" in expected): + raise ValueError( + "StarVLA dataset uses_state does not match the Gymnasium state space" + ) + if manifest_uses_state != model_uses_state: + raise ValueError( + "StarVLA dataset uses_state does not match the model include_state" + ) + if manifest_has_state_metadata and manifest_uses_state: + state_labels = manifest["state_labels"] + expected_state_labels = expected["state_space"]["labels"] if policy_cfg.get("state_source") != "transport" else state_labels + if state_labels != expected_state_labels: + raise ValueError( + "StarVLA dataset state_labels do not match the Gymnasium state space" + ) + if manifest["state_dim"] != len(state_labels): + raise ValueError( + "StarVLA dataset state_dim does not match its state_labels" + ) + if ( + model_cfg["framework"]["action_model"]["state_dim"] + != manifest["state_dim"] + ): + raise ValueError( + "StarVLA model state_dim does not match the dataset manifest" + ) + if not manifest["state_normalization"]: + raise ValueError( + "StarVLA state-enabled dataset manifest is missing state_normalization" + ) + + +def _starvla_runner_kwargs( + config: dict[str, Any], + action_resolver: ActionResolver, + model_cfg: Mapping[str, Any] | None, + *, + base_prompt: str | None, +) -> dict[str, Any]: + """Resolve task and input settings shared by checkpoint and resident models.""" + env_cfg = config["env"] + policy_cfg = config["policy"] + if env_cfg["name"] == "gymnasium": + task_manifest = json.loads( + Path(policy_cfg["task_manifest_path"]).read_text(encoding="utf-8") + ) + _validate_gymnasium_starvla_contract( + env_cfg=env_cfg, + policy_cfg=policy_cfg, + model_cfg=model_cfg, + manifest=task_manifest, + ) + semantic_env_name = env_cfg["task_name"] + action_refs = env_cfg.get("action_order", []) + base_prompt = env_cfg["base_prompt"] + state_normalization = task_manifest.get("state_normalization") + else: + semantic_env_name = env_cfg["name"] + action_refs = policy_cfg.get("actions", action_resolver.default_action_refs()) + state_normalization = policy_cfg["state_normalization"] if "state_normalization" in policy_cfg else None + return dict( + unnorm_key=policy_cfg.get("unnorm_key"), + env_name=semantic_env_name, + action_resolver=action_resolver, + action_refs=action_refs, + latency_prompt_map=( + load_latency_prompt_map(policy_cfg["latency_prompt_map_path"]) + if "latency_prompt_map_path" in policy_cfg + else None + ), + base_prompt=base_prompt, + latency_prompt_key=policy_cfg.get("latency_prompt_key"), + prompt_mode=policy_cfg.get("prompt_mode"), + obs_resize=tuple(env_cfg["obs_resize"]) if env_cfg.get("obs_resize") else None, + image_transform_config=policy_cfg.get("image_transform_config"), + observation_stride_raw_frames=_observation_stride_raw_frames(config), + model_cfg=model_cfg, + state_normalization=state_normalization, + state_source=policy_cfg["state_source"] if "state_source" in policy_cfg else None, + ) + + +def build_starvla_policy( + config: dict[str, Any], + action_resolver: ActionResolver, +) -> PolicyRunner: + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + _ensure_starvla_path() + from latency_bench.policy.starvla_tasks import build_task_starvla_policy + + return build_task_starvla_policy(config) + env_cfg = config["env"] + integration_env_name = env_cfg["name"] + model_cfg = ( + _load_starvla_model_config(policy_cfg["model_config_path"]) + if integration_env_name == "gymnasium" or "model_config_path" in policy_cfg + else None + ) + runner_kwargs = _starvla_runner_kwargs( + config, action_resolver, model_cfg, base_prompt=env_cfg.get("base_prompt") + ) + wrapper_cls = _load_policy_wrapper_class() + wrapper_kwargs: dict[str, Any] = dict( + ckpt_path=policy_cfg["checkpoint_path"], + device=policy_cfg["device"], + use_bf16=True, + unnorm_key=runner_kwargs["unnorm_key"], + action_output_mode=( + policy_cfg["action_output_mode"] + if "action_output_mode" in policy_cfg + else "rl_games" + ), + rl_games_env_name=integration_env_name, + rl_games_action_layout=( + policy_cfg["action_layout"] if "action_layout" in policy_cfg else None + ), + rl_games_multibinary_threshold=( + policy_cfg["multibinary_threshold"] + if "multibinary_threshold" in policy_cfg + else None + ), + ) + if "backbone_path" in policy_cfg: + wrapper_kwargs["backbone_path"] = policy_cfg["backbone_path"] + if integration_env_name == "gymnasium": + action_space = gymnasium_action_space_contract(env_cfg) + wrapper_kwargs["rl_games_action_env_dim"] = len(action_space["labels"]) + if action_space["type"] == "box": + wrapper_kwargs["rl_games_gymnasium_action_space_type"] = "box" + wrapper_kwargs["rl_games_env_name"] = integration_env_name + wrapper = wrapper_cls(**wrapper_kwargs) + return StarVlaPolicyRunner( + wrapper=wrapper, + checkpoint_path=policy_cfg["checkpoint_path"], + device=policy_cfg["device"], + **runner_kwargs, + image_views_info_key=( + policy_cfg["image_views_info_key"] + if "image_views_info_key" in policy_cfg + else None + ), + action_output_type=( + policy_cfg["action_output_type"] + if "action_output_type" in policy_cfg + else None + ), + ) + + +def build_live_starvla_policy( + *, + framework: Any, + model_cfg: dict[str, Any], + config: dict[str, Any], + action_resolver: ActionResolver | None = None, +) -> PolicyRunner: + """Build a StarVLA policy around a *live* in-memory framework (no reload). + + Mirrors ``build_starvla_policy`` but swaps the ckpt-loading + ``PolicyServerWrapper`` for :class:`LiveStarVlaWrapper`, so the trainer's + resident model is evaluated directly. ``model_cfg`` is the in-memory model + config (e.g. ``read_mode_config`` output) the wrapper would otherwise read + from disk. + """ + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + from latency_bench.policy.starvla_tasks import TaskStarVlaPolicyRunner + + contract = json.loads( + Path(policy_cfg["task_contract_path"]).read_text(encoding="utf-8") + ) + return TaskStarVlaPolicyRunner( + framework, + policy_config=policy_cfg, + model_config=model_cfg, + contract=contract, + ) + + env_cfg = config["env"] + integration_env_name = env_cfg["name"] + normalized_model_cfg = ( + _normalized_model_cfg(model_cfg) + if integration_env_name == "gymnasium" + else None + ) + # Resident evaluation historically takes non-Gymnasium prompts from the map. + runner_kwargs = _starvla_runner_kwargs( + config, action_resolver, normalized_model_cfg, base_prompt=None + ) + wrapper_kwargs: dict[str, Any] = dict( + framework=framework, + model_cfg=model_cfg, + env_name=integration_env_name, + action_layout=policy_cfg["action_layout"] if "action_layout" in policy_cfg else None, + multibinary_threshold=( + policy_cfg["multibinary_threshold"] + if "multibinary_threshold" in policy_cfg + else None + ), + ) + if integration_env_name == "gymnasium": + action_space = gymnasium_action_space_contract(env_cfg) + wrapper_kwargs["rl_games_action_env_dim"] = len(action_space["labels"]) + if action_space["type"] == "box": + wrapper_kwargs["gymnasium_action_space_type"] = "box" + wrapper_kwargs["env_name"] = integration_env_name + wrapper = LiveStarVlaWrapper(**wrapper_kwargs) + return StarVlaPolicyRunner( + wrapper=wrapper, + checkpoint_path=policy_cfg.get("checkpoint_path", ""), + device=policy_cfg.get("device", "cuda"), + **runner_kwargs, + ) + + +__all__ = [ + "LiveStarVlaWrapper", + "StarVlaPolicyRunner", + "apply_starvla_model_input_config", + "build_live_starvla_policy", + "build_starvla_policy", + "decode_starvla_action", + "observation_data_to_hwc_uint8_frames", + "prepare_starvla_checkpoint_input_config", +] diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py new file mode 100644 index 0000000000000000000000000000000000000000..4a5cc03813f1bc524a30a7b6b7292a4f4fca85d4 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py @@ -0,0 +1,92 @@ +"""StarVLA inference using the task's training observation/action contract.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import numpy as np +from PIL import Image + +from latency_bench.core.types import Action, Observation, PolicyOutput +from latency_bench.data.starvla_tasks import denormalize, normalize +from latency_bench.policy.base import PolicyRunner + + +class TaskStarVlaPolicyRunner(PolicyRunner): + """Map task RGB/state into a StarVLA model and decode its action chunk.""" + + def __init__(self, framework, *, policy_config: dict, model_config: dict, contract: dict): + self.framework = framework + self.policy_config = policy_config + self.model_config = model_config + self.contract = contract + + def _example(self, observation: Observation) -> dict: + cfg = self.policy_config + state = normalize( + observation.metadata[cfg["state_info_key"]], + self.contract["normalization"]["state"], + ).reshape(1, self.contract["state_dim"]) + data_cfg = self.model_config["datasets"]["vla_data"] + height, width = data_cfg["obs_image_size"] + images = [ + Image.fromarray(frame).resize((width, height)) + for frame in observation.metadata[cfg["image_views_info_key"]] + ] + if data_cfg["image_mode"] == "stitch_views": + from starVLA.training.trainer_utils.trainer_tools import stitch_frames + + # MIKASA's two simultaneous views form one Wan observation, not a video. + images = [stitch_frames(images, grid=data_cfg["stitch_grid"], size=(width, height))] + example = {"image": images, "state": state, "lang": self.contract["prompt"]} + if "action_prefix" in observation.metadata: + example["action_prefix"] = normalize( + observation.metadata["action_prefix"], + self.contract["normalization"]["action"], + ) + example["action_prefix_mask"] = observation.metadata["action_prefix_mask"] + return example + + def predict(self, observation: Observation) -> PolicyOutput: + return self.predict_batch([observation])[0] + + def predict_batch(self, observations: list[Observation]) -> list[PolicyOutput]: + prediction = self.framework.predict_action( + examples=[self._example(observation) for observation in observations] + ) + actions = denormalize( + prediction["normalized_actions"], self.contract["normalization"]["action"] + ) + # Prefix heads were excluded from the loss; retain the frozen controller plan. + for chunk, observation in zip(actions, observations): + if "action_prefix" in observation.metadata: + mask = observation.metadata["action_prefix_mask"] + chunk[mask] = observation.metadata["action_prefix"][mask] + return [ + PolicyOutput( + action=Action(value=chunk[0].tolist(), name="task_command"), + action_chunk=chunk, + raw_output=chunk.tolist(), + metadata={"policy_type": "starvla", "task": self.contract["task"]}, + ) + for chunk in actions + ] + + +def build_task_starvla_policy(config: dict) -> TaskStarVlaPolicyRunner: + # StarVLA and torch are optional in the simulator process; workers own them. + import torch + from starVLA.model.framework.base_framework import baseframework + from starVLA.model.framework.share_tools import read_mode_config + + cfg = config["policy"] + model_config, _ = read_mode_config(cfg["checkpoint_path"]) + framework = baseframework.from_pretrained( + cfg["checkpoint_path"], backbone_path=cfg["backbone_path"] + ) + framework = framework.to(device=cfg["device"], dtype=torch.bfloat16).eval() + contract = json.loads(Path(cfg["task_contract_path"]).read_text(encoding="utf-8")) + return TaskStarVlaPolicyRunner( + framework, policy_config=cfg, model_config=model_config, contract=contract + ) diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json new file mode 100644 index 0000000000000000000000000000000000000000..371542eee194a26ff888c211317abe89050b38e0 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json @@ -0,0 +1,105 @@ +{ + "condition": "profile-latency", + "executor_mode": "simulated", + "latency_method": "temporal", + "profile_source": "originalRTX3090immutableprofiles", + "episodes_per_checkpoint": 100, + "total_episodes": 400, + "rounds": [ + [ + "flappy", + "deadly_corridor" + ], + [ + "ant", + "intercept" + ] + ], + "physical_gpu_assignments": { + "flappy": 2, + "deadly_corridor": 3, + "ant": 2, + "intercept": 3 + }, + "single_gpu_per_job": true, + "round2_requires_both_round1_complete": true, + "latency_seed": 271828, + "tasks": { + "flappy": { + "gpu": 2, + "seed_start": 1000000, + "seed_end": 1000099, + "env_fps": 10, + "obs_fps": 10, + "max_raw_steps": 3600, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 75.87417450998383, + "profile": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/flappy/instance_a5037b165aa0cedc/profile.json", + "sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "deadly_corridor": { + "gpu": 3, + "seed_start": 1000000, + "seed_end": 1000099, + "env_fps": 35, + "obs_fps": 8.75, + "max_raw_steps": 3600, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 73.69250777493353, + "profile": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/deadly_corridor/instance_a5037b165aa0cedc/profile.json", + "sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "ant": { + "gpu": 2, + "seed_start": 42, + "seed_end": 141, + "env_fps": 10, + "obs_fps": 10, + "max_raw_steps": 1000, + "parallel_envs": 16, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 90.56460638563993, + "profile": "/home/ubuntu/lzj/mean-profiling/reference/checkpoint-configs/latency-aware/ant/small-policy/sample-factory-qwenoft-rtx3090-v1/profile/profile.json", + "sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "intercept": { + "gpu": 3, + "seed_start": 4242424242, + "seed_end": 4242424341, + "env_fps": 20, + "obs_fps": 20, + "max_raw_steps": 60, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0", + "profile": { + "mean_ms": 99.05021289731565, + "profile": "/home/ubuntu/lzj/profiles/intercept-published/profiles/qwenoft/1x-rtx3090/mikasa_intercept_grab_fast/instance_3a0d42681a03715c/profile.json", + "sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + } + } +} diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/execution_audit.json b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/execution_audit.json new file mode 100644 index 0000000000000000000000000000000000000000..bcf844ccb409b1f5998a6a0d11499365d7a7bc33 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/execution_audit.json @@ -0,0 +1,12 @@ +{ + "issued_action_records": 79573, + "applied_action_records": 79465, + "dropped_action_records": 0, + "nonnoop_issued_records": 79573, + "finite_action_values": true, + "latency_sample_count": 79573, + "latency_mean_ms": 90.00919554158884, + "latency_std_ms": 2.514492574433973, + "latency_p95_ms": 91.11971585797141, + "latency_p99_ms": 102.67108120995428 +} diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/files.sha256.json b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/files.sha256.json new file mode 100644 index 0000000000000000000000000000000000000000..935b89c932c196fce18a60a16aea4d30a36a09a2 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/files.sha256.json @@ -0,0 +1,130 @@ +{ + "REPORT.md": { + "bytes": 2163, + "sha256": "b2fd63b1cde2a415b2d77daf0184ce5ec3b6ada1aa199c21941fb04031f77db6" + }, + "all_episodes.csv": { + "bytes": 25198, + "sha256": "bd41f35a464250ed6f9bc16e466aa9f1b55a48ac29072bb0ec3559a24129edac" + }, + "comparison.csv": { + "bytes": 438, + "sha256": "a0c7cf191395dd851bd6222bdf61427f15782fea2db4416ddc072a5f5dc8a861" + }, + "comparison.json": { + "bytes": 7826, + "sha256": "0d5dd042439466aee84cd0d96c31a27a951e57965a7e468cb73ec07883f1f751" + }, + "episodes.csv": { + "bytes": 5314, + "sha256": "50f01becf2fc32fc3051b84314ae9c6494bf356b8cee8bf9ad4108b5bd9b9188" + }, + "eval_config.yaml": { + "bytes": 5217, + "sha256": "ecbfbe6a7642142e6ec5941eb8a9545d2d49d0d54bba649be7e4faf1824ce36e" + }, + "evaluation-code/batched_simulated.py": { + "bytes": 31282, + "sha256": "b901f966d911feab7962a32f21095cb90f7880121811f2b4eab2193afe1381db" + }, + "evaluation-code/deadly-compatibility.patch": { + "bytes": 4570, + "sha256": "623676cc4542b1eab6c9395b163b369ddc605353c1de02d17d8f713167ee07fa" + }, + "evaluation-code/deadly_corridor.py": { + "bytes": 17902, + "sha256": "47f7bc65cba9853e66d79ed2a28f844bd2a094f1285458be166045f2db1690dc" + }, + "evaluation-code/decision_action_history.py": { + "bytes": 2746, + "sha256": "14a9d223e775745b6c402dbce9e2a50a1c3f7b5b9fe528150ef8689126fe97cf" + }, + "evaluation-code/eval_driver.py": { + "bytes": 9133, + "sha256": "330030270fbb695bc5f14037ef7349650bd20c53c881c1159ee55ea066408d9e" + }, + "evaluation-code/mikasa_evaluate.py": { + "bytes": 11466, + "sha256": "6cf9ffee25fcfd6f3255c520fc544c48ff2c8f8912e5369c2410a709820c4ffd" + }, + "evaluation-code/starvla.py": { + "bytes": 48378, + "sha256": "6d9988f3a28d39e46c2f6e80da85edebc42cafa629a2b9f75000414324c1065a" + }, + "evaluation-code/starvla_tasks.py": { + "bytes": 4029, + "sha256": "3fc74169d1554d9dc3358ed85e450cca75eb69bc1fff85284c1054a605633a52" + }, + "evaluation-plan.json": { + "bytes": 4698, + "sha256": "b758a5fb72dcdef49d025e2fd168d024ebd8b18b2b00125145b3cde38b16a318" + }, + "execution_audit.json": { + "bytes": 361, + "sha256": "6e5794d7a4456ff19b474e542e5ac03e569e26680ea7cde76b4544fa97394f1c" + }, + "profile/latency_burst_model.json": { + "bytes": 26859, + "sha256": "3d41d58e5b4a51390f1984f69066b72831ef0b95b113577f37a4f0dfde65214f" + }, + "profile/latency_distribution.json": { + "bytes": 25092, + "sha256": "58b144f743a58073234a29f53355aa19b41851cae47686904c38847def37bb94" + }, + "profile/profile.json": { + "bytes": 2148, + "sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50" + }, + "provenance.json": { + "bytes": 3751, + "sha256": "990ab3236a0247b51f73e61555fd30bf3d408c8724713d28c7d9a64a55e0c55b" + }, + "queue_eval_latency_profile_sample.json": { + "bytes": 3953, + "sha256": "32ff1a934595711c9e5bfe9cc4d033bcf9778187e83c837affb3418d0f2dd853" + }, + "raw-records/actions.jsonl.gz": { + "bytes": 12313765, + "sha256": "c1354df8561fbd24c78158553cb997343065983d33497883dac41e2177a421cf" + }, + "raw-records/e2e_latencies.jsonl.gz": { + "bytes": 1639178, + "sha256": "c43d7986e0bd8cfa33d1fb30dd48832c1c48df9a557aa5d97d3b11bb7be4986a" + }, + "raw-records/episode_metrics.jsonl.gz": { + "bytes": 7869, + "sha256": "1c791b0600f8690f8d3f36f2656417e4cab6a368173a6f08adb5f23ad71028f9" + }, + "raw-records/infer_latencies.jsonl.gz": { + "bytes": 1162241, + "sha256": "eaa49aa3fbce55063a9c3037d18a6b9ff3bd327981b4a4f35799a99c434189df" + }, + "raw-records/latencies.jsonl.gz": { + "bytes": 1162235, + "sha256": "0c09425581b2ad7d7786b9489ce88bfc64eecd577fec18aa66280c27e08b8f18" + }, + "raw-records/observation_attempts.jsonl.gz": { + "bytes": 47, + "sha256": "b1a7d5db5a150efea2d3bb76abaa4d5328c50ae0918a01e1e891509155994759" + }, + "raw-records/queue_eval_results.jsonl.gz": { + "bytes": 1548, + "sha256": "163339eb50d2c2964191177ea5918ce97b681f08c2efb4694c9279cddb37854e" + }, + "raw-records/steps.jsonl.gz": { + "bytes": 16329224, + "sha256": "0ffffd6f5f47d90c8a79ffbafddbca7c98c15fe48b10563a3a5fd7d5bc5d7db3" + }, + "resolved_config.yaml": { + "bytes": 5285, + "sha256": "1c10ee3785bbf188e4a406e7e6cc934fdf637cf00b006a1792b24410b95d233a" + }, + "statistics.json": { + "bytes": 1222, + "sha256": "694b29fddd346892c7559f611fcba1949fd377dc14ea751c52e28e89809d0ff9" + }, + "stdout.log": { + "bytes": 27591, + "sha256": "f5c6dff29aacd92ac3a83c5d6df326b98145332af60278c69a3a75a34a541b77" + } +} diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json new file mode 100644 index 0000000000000000000000000000000000000000..6fbff2bd7612b75cad4e703f678de37d90692e20 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json @@ -0,0 +1,1336 @@ +{ + "burst_dwell_distribution": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.0, + 21.15, + 21.3, + 21.45, + 21.6, + 21.75, + 21.9, + 22.05, + 22.2, + 22.35, + 22.5, + 22.65, + 22.8, + 22.95, + 23.1, + 23.25, + 23.4, + 23.55, + 23.7, + 23.85, + 24.0, + 25.0, + 26.0, + 27.000000000000004, + 28.000000000000004, + 29.0, + 29.999999999999996, + 31.0, + 32.0, + 33.0, + 34.0, + 35.0, + 36.0, + 37.0, + 38.0, + 39.0, + 40.00000000000001, + 40.99999999999999, + 42.0, + 43.0, + 44.0, + 44.949999999999996, + 45.9, + 46.85000000000001, + 47.800000000000004, + 48.75, + 49.7, + 50.64999999999999, + 51.599999999999994, + 52.55, + 53.5, + 54.449999999999996, + 55.400000000000006, + 56.349999999999994, + 57.300000000000004, + 58.25, + 59.2, + 60.150000000000006, + 61.10000000000001, + 62.05, + 63.0, + 65.64999999999999, + 68.29999999999998, + 70.94999999999999, + 73.60000000000001, + 76.25, + 78.89999999999999, + 81.55000000000001, + 84.20000000000002, + 86.85000000000001, + 89.5, + 92.15000000000003, + 94.79999999999998, + 97.44999999999997, + 100.10000000000001, + 102.75, + 105.39999999999999, + 108.04999999999998, + 110.70000000000002, + 113.35000000000001, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0, + 116.0 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "burst_dwell_lengths": [ + 21, + 24, + 44, + 63, + 116 + ], + "burst_merge_gap_records": 30, + "burst_rank_processes": [ + { + "draw_count": 268, + "dwell_length_spearman_rho": 0.6, + "level_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76014185945193, + 111.76312052955231, + 111.76609919965267, + 111.76907786975305, + 111.77205653985342, + 111.77503520995378, + 111.77801388005416, + 111.78099255015454, + 111.7839712202549, + 111.78694989035527, + 111.78992856045565, + 111.79290723055601, + 111.79588590065639, + 111.79886457075675, + 111.80184324085712, + 111.8048219109575, + 111.80780058105786, + 111.81077925115824, + 111.81375792125861, + 111.81673659135897, + 111.81971526145935, + 111.88049919903278, + 111.94128313660622, + 112.00206707417965, + 112.06285101175308, + 112.12363494932652, + 112.18441888689995, + 112.24520282447338, + 112.30598676204681, + 112.36677069962025, + 112.42755463719368, + 112.48833857476711, + 112.54912251234055, + 112.60990644991398, + 112.67069038748741, + 112.73147432506084, + 112.79225826263428, + 112.85304220020771, + 112.91382613778114, + 112.97461007535458, + 113.03539401292801, + 113.24575516482194, + 113.45611631671588, + 113.66647746860981, + 113.87683862050375, + 114.08719977239768, + 114.29756092429162, + 114.50792207618555, + 114.71828322807949, + 114.9286443799734, + 115.13900553186735, + 115.34936668376127, + 115.55972783565521, + 115.77008898754914, + 115.98045013944308, + 116.19081129133701, + 116.40117244323095, + 116.61153359512488, + 116.82189474701882, + 117.03225589891275, + 117.24261705080669, + 117.25980174541473, + 117.2769864400228, + 117.29417113463084, + 117.31135582923889, + 117.32854052384695, + 117.345725218455, + 117.36290991306305, + 117.3800946076711, + 117.39727930227916, + 117.4144639968872, + 117.43164869149525, + 117.44883338610332, + 117.46601808071136, + 117.48320277531941, + 117.50038746992746, + 117.51757216453552, + 117.53475685914357, + 117.55194155375162, + 117.56912624835968, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773, + 117.58631094296773 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "level_residual_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + -6.34056695302327, + -6.34056695302327, + -6.34056695302327, + -6.34056695302327, + -6.34056695302327, + -6.34056695302327, + -6.34056695302327, + -6.34056695302327, + -6.34056695302327, + -6.34056695302327, + -6.34056695302327, + -6.34056695302327, + -6.332754473686213, + -6.12963001092275, + -5.934426395098373, + -5.799949272473652, + -5.680410925547278, + -5.623615436553957, + -5.049422933657966, + -3.070867107311886, + -1.7738834404945374, + -1.764313852787018, + -1.7532525277137756, + -1.740157015323639, + -1.6828855347633362, + -1.5814380860328672, + -1.4429068446159363, + -1.2771808218955993, + -1.16335510969162, + -1.0770060324668886, + -0.977082860469818, + -0.8721587061882018, + -0.7349506437778474, + -0.590055936574936, + -0.5727308547496796, + -0.5720452892780304, + -0.5142639031012877, + -0.45617564717929043, + -0.4475114683310238, + -0.4395893094937051, + -0.4373559707403203, + -0.4331851969162628, + -0.42087719579538, + -0.40994842906793366, + -0.40276329855124315, + -0.394058796763421, + -0.3824843714634611, + -0.3737770799795834, + -0.36897951642672694, + -0.36016353984674165, + -0.34732915023962757, + -0.33729381263256214, + -0.3293111131588657, + -0.30667495767275965, + -0.2762810901800839, + -0.26342058698336757, + -0.25701974431674157, + -0.25447803537051356, + -0.2528551677862851, + -0.2522101406256405, + -0.25169265786807216, + -0.2242364039023741, + -0.19644585728645325, + -0.18723676562309266, + -0.17905045092105895, + -0.17870542625586466, + -0.1779036621252743, + -0.1751835922400204, + -0.16621431390444927, + -0.14028289834658797, + -0.10150121072928503, + -0.038446786999703955, + 0.00013522148132323088, + 0.005345754623413106, + 0.012373673717179623, + 0.02121897876262297, + 0.05395132501919632, + 0.1042008348305973, + 0.12717157800991863, + 0.1357006212075504, + 0.15964065512020884, + 0.1892584224541934, + 0.19996284385521934, + 0.20616408765315977, + 0.20962504406769852, + 0.21272857169309786, + 0.24325262506802403, + 0.27453822116056625, + 0.2974418594439802, + 0.3181665986776349, + 0.3221864451964669, + 0.32762383421261637, + 0.3390149017174992, + 0.37237897018591165, + 0.465384041269618, + 0.5267693853378296, + 0.52842857837677, + 0.5617947413523979, + 0.6383976815144189, + 0.6814763828118594, + 0.6910308452447208, + 0.7474897046883862, + 0.8383451219399723, + 0.9931568259000759, + 1.1818277404705675, + 4.4926970267295445, + 8.953849923610676, + 10.534285777807236, + 11.428836621840787, + 11.76198321501414, + 12.021903166770938, + 12.031900087992355, + 12.031900087992355, + 12.031900087992355, + 12.031900087992355, + 12.031900087992355, + 12.031900087992355, + 12.031900087992355, + 12.031900087992355, + 12.031900087992355, + 12.031900087992355, + 12.031900087992355, + 12.031900087992355 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "severity": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.125, + 0.12589285714285714, + 0.12678571428571428, + 0.12767857142857142, + 0.12857142857142856, + 0.1294642857142857, + 0.13035714285714287, + 0.13125, + 0.13214285714285715, + 0.13303571428571428, + 0.13392857142857142, + 0.13482142857142856, + 0.1357142857142857, + 0.13660714285714284, + 0.13749999999999998, + 0.13839285714285715, + 0.1392857142857143, + 0.14017857142857143, + 0.14107142857142857, + 0.1419642857142857, + 0.14285714285714285, + 0.14523809523809522, + 0.14761904761904762, + 0.15, + 0.1523809523809524, + 0.15476190476190477, + 0.15714285714285714, + 0.1595238095238095, + 0.16190476190476188, + 0.16428571428571428, + 0.16666666666666666, + 0.16904761904761903, + 0.17142857142857143, + 0.1738095238095238, + 0.17619047619047618, + 0.17857142857142855, + 0.18095238095238095, + 0.1833333333333333, + 0.1857142857142857, + 0.1880952380952381, + 0.19047619047619047, + 0.19129720853858784, + 0.1921182266009852, + 0.19293924466338258, + 0.19376026272577995, + 0.19458128078817732, + 0.1954022988505747, + 0.19622331691297207, + 0.19704433497536944, + 0.1978653530377668, + 0.19868637110016418, + 0.19950738916256155, + 0.20032840722495895, + 0.20114942528735633, + 0.2019704433497537, + 0.20279146141215107, + 0.20361247947454844, + 0.2044334975369458, + 0.20525451559934318, + 0.20607553366174056, + 0.20689655172413793, + 0.2079153605015674, + 0.20893416927899686, + 0.20995297805642632, + 0.2109717868338558, + 0.21199059561128525, + 0.2130094043887147, + 0.2140282131661442, + 0.21504702194357367, + 0.21606583072100313, + 0.2170846394984326, + 0.2181034482758621, + 0.21912225705329152, + 0.220141065830721, + 0.22115987460815045, + 0.22217868338557994, + 0.22319749216300938, + 0.22421630094043887, + 0.22523510971786834, + 0.2262539184952978, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727, + 0.22727272727272727 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "spike_count": 52 + } + ], + "burst_slot_rank_templates": [ + { + "dwell_length": 116, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 44, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 63, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 21, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 24, + "slot_to_rank": { + "0": 0 + } + } + ], + "model_type": "hidden_regime", + "pre_worker_time_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 0.2572758197784424, + 0.25763754005432127, + 0.25947027969360353, + 0.2603460645675659, + 0.2612747564315796, + 0.2622022571563721, + 0.26243807792663576, + 0.2626762580871582, + 0.2632839403152466, + 0.26363272762298584, + 0.2637669315338135, + 0.26404299354553223, + 0.2643587684631348, + 0.2657140350341797, + 0.26658174514770505, + 0.26738574981689456, + 0.2681525468826294, + 0.26886218070983886, + 0.26952486515045165, + 0.27005916118621826, + 0.2706614589691162, + 0.2714514255523682, + 0.27218435287475584, + 0.27285959720611574, + 0.27350028038024904, + 0.2743678092956543, + 0.2750845909118652, + 0.2760720920562744, + 0.2770882177352905, + 0.27842949867248534, + 0.27990328788757324, + 0.28201286792755126, + 0.28431395053863523, + 0.28664437294006345, + 0.28922285556793215, + 0.29165172576904297, + 0.29378581047058105, + 0.2955480146408081, + 0.2971444034576416, + 0.29853535175323487, + 0.2998380756378174, + 0.3010571956634521, + 0.3020807981491089, + 0.30305158138275146, + 0.3037783861160278, + 0.3045779085159302, + 0.3050941705703735, + 0.3056695747375488, + 0.30616676330566406, + 0.3065678071975708, + 0.30704905033111574, + 0.3073734760284424, + 0.3076727342605591, + 0.30795797348022463, + 0.308273024559021, + 0.3085658121109009, + 0.308845591545105, + 0.3091064405441284, + 0.30933346748352053, + 0.30958592891693115, + 0.3098057985305786, + 0.3100399971008301, + 0.3102893924713135, + 0.31050017833709714, + 0.3107223749160767, + 0.31094244956970213, + 0.3111721992492676, + 0.31139523029327393, + 0.31161237239837647, + 0.31180827140808104, + 0.3120582962036133, + 0.31228928565979003, + 0.31247793197631835, + 0.31274282455444335, + 0.3129736089706421, + 0.3132426595687866, + 0.3134938716888428, + 0.31373076915740966, + 0.31396267890930174, + 0.31422226905822753, + 0.31449563026428223, + 0.3147676229476929, + 0.3149920606613159, + 0.3153657913208008, + 0.31567907333374023, + 0.3159314775466919, + 0.31631767749786377, + 0.31666929721832277, + 0.3170040273666382, + 0.3174827384948731, + 0.31792099952697755, + 0.31836187839508057, + 0.31882681846618655, + 0.31948143005371094, + 0.3199909162521362, + 0.3205808925628662, + 0.3213667631149292, + 0.3224719858169556, + 0.3236453628540039, + 0.3248334980010986, + 0.32641579151153566, + 0.3288518667221069, + 0.3317467498779297, + 0.3364091396331787, + 0.34346031188964843, + 0.35033455848693845, + 0.3815898180007932, + 0.6752684497833252, + 1.2073634576797472, + 1.278598713874817, + 1.302369451522827, + 1.3093395385742188, + 1.3129376239776611, + 1.3175906381607052, + 1.3359725713729873, + 1.3516274499893188, + 1.3746464385986326, + 1.3949562454223634, + 1.4172659616470336, + 1.4342637987136844, + 1.4602041645050134, + 1.4875035186767591, + 1.49049711227417 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "regime_step_counts": { + "burst": 268, + "calm": 7708 + }, + "regime_transition_counts": { + "burst": { + "burst": 263, + "calm": 5 + }, + "calm": { + "burst": 5, + "calm": 7698 + } + }, + "reset_scope": "session", + "schema_version": 12, + "spike_median_multiplier": 1.25, + "spike_threshold_ms_by_worker_slot": { + "0": 111.23815685510635 + }, + "worker_count": 1 +} diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json new file mode 100644 index 0000000000000000000000000000000000000000..b49f795ef4b9da706834f8ca618e43aa577a312a --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json @@ -0,0 +1,1027 @@ +{ + "schema_version": 3, + "worker_slots": { + "0": { + "all": { + "count": 7976, + "observation_to_action_latency_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 87.98952293395996, + 88.00625395126343, + 88.19498282814025, + 88.31266105270386, + 88.42687071514129, + 88.47465879440307, + 88.50693488693237, + 88.53092688083649, + 88.54779446983338, + 88.55876398563385, + 88.56488679409027, + 88.57280330657959, + 88.59264374256134, + 88.67249797344208, + 88.724377617836, + 88.79207693576812, + 88.84999904632568, + 88.89409097194671, + 88.92342544078826, + 88.96823967933655, + 88.99946705818176, + 89.02508285045624, + 89.04901226520538, + 89.07655789375305, + 89.10013774871827, + 89.13031135559082, + 89.15370976924896, + 89.17447654724121, + 89.19607995033265, + 89.21466581344605, + 89.23746184825897, + 89.25717034339905, + 89.2771969127655, + 89.29832275390625, + 89.31721762657166, + 89.33869683265686, + 89.35448241233826, + 89.37088777542114, + 89.387730717659, + 89.40454774379731, + 89.42069890975952, + 89.43555467128753, + 89.45022724151612, + 89.46291656017303, + 89.47820586681365, + 89.49225774765014, + 89.50589845180511, + 89.52021051883698, + 89.53634333610535, + 89.5502944946289, + 89.56533065795898, + 89.57548904418945, + 89.58862001419067, + 89.59958215236664, + 89.6106701040268, + 89.6255473279953, + 89.63833825588226, + 89.65307829856873, + 89.66765783309937, + 89.68132509231567, + 89.69334455490112, + 89.70676946640015, + 89.72026408195495, + 89.73473812103272, + 89.74919620037079, + 89.76378743171692, + 89.77904286384583, + 89.79202803611756, + 89.80719444274902, + 89.82152740478516, + 89.83681576728821, + 89.85403666496276, + 89.87273532867431, + 89.88893761634827, + 89.91062943458557, + 89.92647359848023, + 89.94606273174286, + 89.96231405735016, + 89.97694365501404, + 89.99286030292511, + 90.0110937833786, + 90.02730028629303, + 90.04637926101685, + 90.06258551597595, + 90.07847352981567, + 90.09652791976929, + 90.11794292926788, + 90.13743216991425, + 90.15522152900697, + 90.18199282169343, + 90.20373028278351, + 90.2253544807434, + 90.25468337535858, + 90.28829836368561, + 90.32118947029113, + 90.35554928779602, + 90.3950603723526, + 90.44660758495331, + 90.51204998970032, + 90.56736985683442, + 90.64274927139282, + 90.74715027809142, + 90.87436978340149, + 91.05810060977936, + 91.26494856357574, + 91.56088218212128, + 92.13188321590424, + 93.51960658550263, + 110.3865959215164, + 111.56080113887786, + 112.18723976135253, + 112.27797777271272, + 112.3118753566742, + 112.38298761463165, + 112.55841102600098, + 112.6685480260849, + 112.81127831363678, + 113.11438784122467, + 113.30635335445405, + 113.64114878845216, + 114.79401259803862, + 130.04163708992016, + 130.29119229316711 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "spearman_rho": 0.999288082060808, + "worker_service_time_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 87.72087597846985, + 87.72505484809875, + 87.91075664138793, + 88.00112848186492, + 88.12835392475128, + 88.16585436820984, + 88.20216335296631, + 88.22228655815124, + 88.23629946613312, + 88.25117263317108, + 88.26306705760956, + 88.27515149116516, + 88.29007994651795, + 88.37461434841155, + 88.42467656612396, + 88.48697556972503, + 88.54574456214905, + 88.59064786434173, + 88.6260237455368, + 88.66484241485595, + 88.69925922870635, + 88.72065505981445, + 88.74581295013428, + 88.77460636615753, + 88.80257711410522, + 88.82836333751679, + 88.85065879821778, + 88.86961028575897, + 88.89002092838287, + 88.91101721286773, + 88.93179827690125, + 88.95344092845917, + 88.97252882957459, + 88.99404806137085, + 89.01403583049775, + 89.03331364154816, + 89.05015802383423, + 89.06947086334229, + 89.08563427448273, + 89.10351252555847, + 89.11768238544464, + 89.13144860267639, + 89.14576484203339, + 89.1602001285553, + 89.1763215970993, + 89.19063215255737, + 89.20418736934661, + 89.2187579345703, + 89.23084268569946, + 89.2456736755371, + 89.26038558959961, + 89.27289342880249, + 89.28322927474976, + 89.29450441837311, + 89.30719062805176, + 89.32086008548737, + 89.33541519641877, + 89.35248424530029, + 89.36700273036956, + 89.37777100563049, + 89.39082574367524, + 89.40557551383972, + 89.41941950321197, + 89.43248313426972, + 89.4482385969162, + 89.46083876609802, + 89.47424108982086, + 89.48949390888214, + 89.50253468036652, + 89.51752857208253, + 89.53237959384919, + 89.5501193523407, + 89.56644603252411, + 89.58779298782349, + 89.6055972623825, + 89.61994940757751, + 89.63657371997833, + 89.65447835445404, + 89.6701370716095, + 89.6889020872116, + 89.70479391098023, + 89.72396631240845, + 89.74089723587036, + 89.75628419399261, + 89.775436668396, + 89.79404658794402, + 89.81355702877045, + 89.832233710289, + 89.8513820886612, + 89.87316411018372, + 89.8954009437561, + 89.91825423240661, + 89.95083716869354, + 89.98068819999695, + 90.01347800254821, + 90.04942324638367, + 90.09265229701995, + 90.14444270133973, + 90.20294900894164, + 90.25911507606506, + 90.33128764629365, + 90.43725185394287, + 90.5703069114685, + 90.7431587934494, + 90.93199528217316, + 91.24273622989654, + 91.76115915775299, + 92.93062980651855, + 109.06819209098813, + 110.25691298007965, + 110.92635715007782, + 110.98662494850159, + 111.02695600032807, + 111.1557109861374, + 111.28695208072662, + 111.41097618579865, + 111.5848327255249, + 111.8585945930481, + 112.06217238235473, + 112.4594908771515, + 113.57045895004362, + 129.3671735408784, + 129.61821103096008 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + } + }, + "steady": { + "count": 7708, + "observation_to_action_latency_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 87.98952293395996, + 88.00474726018906, + 88.19305597496033, + 88.31169596481324, + 88.42505750274658, + 88.47302298545837, + 88.50544290447235, + 88.52511432647705, + 88.54508119392395, + 88.55763299179077, + 88.56368862724304, + 88.57163988304139, + 88.58391930580139, + 88.6713106393814, + 88.71694125175476, + 88.78470235347748, + 88.84136619567872, + 88.88592077732086, + 88.91770453929901, + 88.95671053886413, + 88.99107503414155, + 89.01632058620453, + 89.04142089366913, + 89.06427590847015, + 89.09112273216248, + 89.1197328710556, + 89.14090068340302, + 89.16389953613282, + 89.18336018562317, + 89.20432123661041, + 89.22178001880646, + 89.24455571174622, + 89.26285143375397, + 89.28278621196746, + 89.30259282112121, + 89.32047615528107, + 89.34048044681549, + 89.35656083106994, + 89.37347135066986, + 89.3890373134613, + 89.40507507801055, + 89.4206326007843, + 89.43527732372284, + 89.44985229969025, + 89.4615450334549, + 89.47571161746978, + 89.49060957431793, + 89.50358631610871, + 89.51721787452698, + 89.53188828468323, + 89.54624356746673, + 89.55934813022614, + 89.57238875865936, + 89.58399452209473, + 89.5950669336319, + 89.60552267551422, + 89.61853682994843, + 89.63070479393005, + 89.64368647098541, + 89.66024139881134, + 89.67300935745239, + 89.68450391292572, + 89.69663897037506, + 89.70963977336883, + 89.72322317123412, + 89.73775261878967, + 89.75136733055115, + 89.76614410877228, + 89.78024483203887, + 89.79339708805084, + 89.807336602211, + 89.82134807109833, + 89.8357172679901, + 89.85296149253845, + 89.8696624326706, + 89.8862680053711, + 89.9060546875, + 89.92375180721282, + 89.939760055542, + 89.95744044303893, + 89.97084111213684, + 89.98641595840454, + 90.00528963565826, + 90.02034323692322, + 90.03935966014862, + 90.05414157390595, + 90.07118356227875, + 90.08569293022155, + 90.10546464443206, + 90.1239938879013, + 90.14333739280701, + 90.16480300426483, + 90.18858228206635, + 90.20850865364075, + 90.23085282325745, + 90.26068212985993, + 90.29339134693146, + 90.32349193572998, + 90.35763807296753, + 90.39734611034393, + 90.44722507953644, + 90.51184470653534, + 90.56559889316559, + 90.64097036838531, + 90.73639919757844, + 90.85822832584381, + 91.0151937007904, + 91.20243098735808, + 91.45049385070801, + 91.88321811676025, + 92.989627699852, + 93.15146862792969, + 93.30333351707459, + 93.47491034889221, + 93.5772506389618, + 93.86473531723023, + 95.11389953041056, + 108.71966855239876, + 110.97717627239227, + 111.38316594314577, + 111.66655566310887, + 112.25149490270606, + 112.35095000267029 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "spearman_rho": 0.9992113218119626, + "worker_service_time_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 87.72087597846985, + 87.72467852516175, + 87.91030064630509, + 88.00015591812134, + 88.12767535018921, + 88.16078618812561, + 88.20027036190032, + 88.2218482208252, + 88.2316044626236, + 88.2494578113556, + 88.25887827968597, + 88.27183927154542, + 88.28763780593872, + 88.37063961029052, + 88.41736738204956, + 88.48253869056701, + 88.53538405895233, + 88.58251465797424, + 88.61692441940308, + 88.6526103067398, + 88.69219470977784, + 88.71518459320069, + 88.73541054725646, + 88.76436851501465, + 88.7882429409027, + 88.8153453207016, + 88.84055385589599, + 88.86013361930847, + 88.88090627670289, + 88.89949244976043, + 88.9193493270874, + 88.9378014087677, + 88.95896531105042, + 88.97797864437103, + 88.99964028835296, + 89.01831042766571, + 89.03623116016388, + 89.05389294624328, + 89.07155320167541, + 89.08649010658264, + 89.10401025772094, + 89.11759095191955, + 89.13092276096344, + 89.14472054004669, + 89.15695644378663, + 89.17383971691132, + 89.18826713562012, + 89.2024789762497, + 89.21522489070892, + 89.22818738937377, + 89.2410564661026, + 89.25574686527253, + 89.26698334217072, + 89.27876037597656, + 89.28845839977265, + 89.30029655456543, + 89.31337325572967, + 89.32795318126678, + 89.34285112380981, + 89.35806794166565, + 89.37013675689697, + 89.38430845737457, + 89.39552543163299, + 89.40971125602722, + 89.42204663753509, + 89.43564129829407, + 89.44992272853851, + 89.4627941942215, + 89.47512364387512, + 89.4904280757904, + 89.50267264842986, + 89.51706793308259, + 89.53111134052277, + 89.54835547924041, + 89.56463825702667, + 89.58519184112549, + 89.60270733833313, + 89.61832924365997, + 89.63192135810851, + 89.64982438087463, + 89.66651585578919, + 89.68272776603699, + 89.70003828048706, + 89.71646264076233, + 89.73325060844421, + 89.74732863903046, + 89.76466166973114, + 89.7842432308197, + 89.80248938560486, + 89.82193244457245, + 89.83790985107422, + 89.85878474712372, + 89.87975364208222, + 89.90125885009766, + 89.92823070526123, + 89.95628376960754, + 89.98322794437408, + 90.01770035743714, + 90.05196633815765, + 90.0945546579361, + 90.14471434116363, + 90.20198478698731, + 90.25720650196075, + 90.327011551857, + 90.41905149459839, + 90.55038662433624, + 90.70465857982636, + 90.87903607845307, + 91.12213365077973, + 91.56424148082733, + 92.54860488414765, + 92.65843725967407, + 92.76501456165313, + 92.8867836036682, + 93.04233510875703, + 93.24323705673218, + 94.55598453712442, + 107.64949651336671, + 109.727002286911, + 110.21494954109193, + 110.4879216728211, + 111.00118846521367, + 111.11597681045532 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + } + } + } + } +} diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/profile.json b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/profile.json new file mode 100644 index 0000000000000000000000000000000000000000..bf3078569d7a19926a7fce8cbaf4b168a532139a --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/profile.json @@ -0,0 +1,66 @@ +{ + "burst_model_path": "latency_burst_model.json", + "distribution_path": "latency_distribution.json", + "env_fps": 10, + "frame_ms": 100.0, + "gpu_class": "1x-rtx3090", + "instance_id": "instance_859cf1e47bca6046", + "latency_kind": "observation_to_action_latency", + "latency_method": "temporal", + "model_id": "qwenoft", + "n_admitted_observations": 7976, + "n_capacity_drops": 294, + "n_observation_attempts": 8270, + "per_slot_summary": { + "0": { + "admitted_count": 7976, + "mean_observation_to_action_latency_ms": 90.56460638563993, + "mean_worker_service_time_ms": 90.22241150451042, + "p95_observation_to_action_latency_ms": 92.13036412000656, + "p95_worker_service_time_ms": 91.75596672296524, + "p99_worker_service_time_ms": 110.92562991380692 + } + }, + "provenance": { + "base_config": "/workspace/tasks/20260911T023128Z-p-only4/ant/profile.yaml", + "checkpoint_kind": "best", + "model_artifact": { + "checkpoint": "checkpoints/steps_5000_pytorch_model.pt", + "model_config": "config.full.yaml", + "path_in_repo": "OpenVLA/zero-latency/ant_rgb_state_l0_return_gt3000_100ep_openvla_native_sft_5k", + "repo_id": "latency-sensitive-bench/extra-envs-checkpoints", + "source": "local" + }, + "session_ids": [ + 0, + 1, + 2, + 3, + 4 + ] + }, + "sample_model_type": "hidden_regime", + "source_run_id": "20260911T033037730561Z", + "summary": { + "frame_ms": 100.0, + "max_ms": 130.29119229316711, + "mean_effective_frames": 0.9056460638563993, + "mean_ms": 90.56460638563993, + "min_ms": 87.98952293395996, + "n_samples": 7976, + "p50_frames": 0.8970676946640015, + "p50_ms": 89.70676946640015, + "p90_frames": 0.9074710392951966, + "p90_ms": 90.74710392951965, + "p95_frames": 0.9213036412000656, + "p95_ms": 92.13036412000656, + "p99_frames": 1.1217999053001404, + "p99_ms": 112.17999053001404, + "prob_latency_gt_1_frame": 0.03686058174523571, + "prob_latency_gt_2_frames": 0.0, + "prob_latency_gt_3_frames": 0.0, + "std_ms": 4.164609685850939 + }, + "visualization_path": "latency_profile.png", + "workload_id": "ant" +} diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/provenance.json b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..4db693e0d88bbc3693b93e9284168cb46534653f --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/provenance.json @@ -0,0 +1,97 @@ +{ + "task": "ant", + "protocol": { + "gpu": 2, + "seed_start": 42, + "seed_end": 141, + "env_fps": 10, + "obs_fps": 10, + "max_raw_steps": 1000, + "parallel_envs": 16, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 90.56460638563993, + "profile": "/home/ubuntu/lzj/mean-profiling/reference/checkpoint-configs/latency-aware/ant/small-policy/sample-factory-qwenoft-rtx3090-v1/profile/profile.json", + "sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "checkpoint_weights": { + "bytes": 9785070315, + "sha256": "afb954569065450b9b80a86ca96e3898e3fe8d2f8c38a06f63bf3b11c41163b3" + }, + "evaluation": { + "n_episodes": 100, + "mean_return": 1453.844063807972, + "std_return": 693.7275200567642, + "min_return": 85.64836938561511, + "max_return": 2508.917122342891, + "mean_length": 803.85, + "std_length": 328.8088312378486, + "min_length": 60.0, + "max_length": 1000.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "LatencyBench/AntContinuous-v0", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "ant", + "instance_id": "instance_859cf1e47bca6046", + "source_run_id": "20260911T033037730561Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "ant", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42", + "profile_sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant.yaml", + "execution_audit": { + "issued_action_records": 79573, + "applied_action_records": 79465, + "dropped_action_records": 0, + "nonnoop_issued_records": 79573, + "finite_action_values": true, + "latency_sample_count": 79573, + "latency_mean_ms": 90.00919554158884, + "latency_std_ms": 2.514492574433973, + "latency_p95_ms": 91.11971585797141, + "latency_p99_ms": 102.67108120995428 + } + }, + "source_revision": { + "repo": "c3c6a39365a151e9b7a5e215452fd64e957c2b29", + "starvla": "ccca13c5177fe3d3c884b6e2de4965d916016649", + "runtime_fixes": [ + "mean-profile-preparation.patch", + "gym-language-contract.patch", + "loader-spawn-cache.patch", + "loader-spawn-test.patch", + "doom-mean-reset.patch" + ], + "pytorch3d": { + "revision": "33824be3cbc87a7dd1db0f6a9a9de9ac81b2d0ba", + "build": "transforms-only, no native render extension; QwenOFT uses transforms only" + }, + "decord": { + "version": "0.6.0", + "build": "official source CPU decoder CP310", + "wheel_sha256": "e193b356b1e984b4eff08d23b62e482c2c9e5037a6efdc0b1af47079ae2e4c47" + } + }, + "raw_records_format": "gzip(JSONL), lossless", + "startup_checks_included_in_score": false, + "results_status": "evaluation_complete; acceptance_not_inferred" +} diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json new file mode 100644 index 0000000000000000000000000000000000000000..5ad696f011cab89c6379e365b4158b6515205036 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json @@ -0,0 +1,217 @@ +{ + "checkpoint_path": "/home/ubuntu/lzj/mean-profiling/ant/vla-publication/checkpoints/model.pt", + "experiment_name": "ant-mean5000-profile-simulation-100ep", + "latency": "profile_sample", + "latency_type": "profile_sample", + "lengths": [ + 1000, + 1000, + 177, + 1000, + 937, + 1000, + 429, + 1000, + 210, + 660, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 60, + 143, + 1000, + 1000, + 1000, + 871, + 1000, + 1000, + 543, + 1000, + 1000, + 707, + 1000, + 256, + 1000, + 101, + 1000, + 372, + 1000, + 162, + 1000, + 363, + 1000, + 1000, + 1000, + 680, + 1000, + 160, + 1000, + 248, + 1000, + 1000, + 260, + 316, + 1000, + 281, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 189, + 74, + 1000, + 116, + 1000, + 1000, + 215, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000, + 821, + 271, + 1000, + 72, + 1000, + 1000, + 1000, + 1000, + 1000, + 441, + 1000, + 250, + 1000, + 1000, + 1000, + 1000, + 1000, + 1000 + ], + "mean_length": 803.85, + "mean_return": 1453.8440638079721, + "returns": [ + 1846.1103431567394, + 2415.720790707953, + 457.34421085068755, + 1421.7952163289683, + 2037.7234409469488, + 2330.630175869275, + 1161.643572255748, + 2351.1524624990343, + 513.2964809479813, + 1126.8652528911032, + 1693.436933192597, + 948.3780972955639, + 2322.052445211472, + 960.4026770814776, + 1464.564005196777, + 1110.548792782156, + 2246.207900740156, + 85.64836938561511, + 340.54799067574436, + 2457.088748930458, + 2166.2512677098603, + 2357.957592244385, + 1654.8780938737275, + 1499.367100151414, + 2297.4032619179525, + 1253.360764666355, + 1221.270312709775, + 2389.2476464763376, + 1682.5290233886233, + 2474.676425615127, + 382.9231146443659, + 1837.8126619276347, + 227.19436616673684, + 1700.6312067622644, + 960.9452812639541, + 2290.6720141359438, + 328.5729178056416, + 1180.073938772476, + 817.4190215442345, + 1651.2255208727013, + 1428.174672693164, + 1627.3838925098842, + 1079.756369746183, + 2173.9447393037276, + 409.90633829945847, + 2467.2636019929073, + 657.4084558813478, + 974.7436031610902, + 1510.5184342975385, + 602.2339441184535, + 760.9784375126189, + 1941.172113330597, + 624.3590446196446, + 2163.4347041279893, + 1126.9957963444238, + 1405.131632695366, + 1206.2916757636292, + 2392.7980761515178, + 964.0216541467705, + 2252.192880003706, + 2471.9158497657563, + 1902.8491241623092, + 1435.6661382989703, + 1668.3237703695809, + 1813.291243529155, + 446.72353548541076, + 130.84194814079504, + 2315.857153770824, + 288.3915792961347, + 894.0228631227924, + 2030.322535823717, + 507.9449555916754, + 2377.7373967468293, + 897.3077114027096, + 1454.612590266188, + 2292.457960175467, + 1424.378337790017, + 1441.1111023164538, + 1265.4771503717611, + 1662.8808067819505, + 2508.917122342891, + 1655.3510139158748, + 1387.3843721247736, + 646.4356689469432, + 2172.801064037805, + 165.9213897970373, + 1063.1483912161111, + 1000.135342286622, + 1977.2359176146426, + 1937.1674235355138, + 1344.7729257831547, + 786.3379828975102, + 1391.060299752017, + 503.300235688713, + 2446.7482357041768, + 1171.9102336514923, + 2356.7311711183065, + 2356.12199478712, + 1389.2987977192308, + 967.2335383727841 + ], + "seed": 42, + "source_profile_path": "/home/ubuntu/lzj/mean-profiling/reference/checkpoint-configs/latency-aware/ant/small-policy/sample-factory-qwenoft-rtx3090-v1/profile/profile.json", + "std_return": 693.7275200567642, + "suite_name": "profile_sample", + "timestamp_utc": "2026-10-01T07:11:37.707856+00:00" +} diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..ef4749021e1ad4112c6135ed2010d5afed27d493 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c1354df8561fbd24c78158553cb997343065983d33497883dac41e2177a421cf +size 12313765 diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..5a7e1f834ad0d33fa9cfc3d5d20eef3eaeb0de64 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c43d7986e0bd8cfa33d1fb30dd48832c1c48df9a557aa5d97d3b11bb7be4986a +size 1639178 diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..72466c21d71c0538308944e12144fd4358f91892 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1c791b0600f8690f8d3f36f2656417e4cab6a368173a6f08adb5f23ad71028f9 +size 7869 diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..b9bc03d7a4975dabbcd9fb806be07f05c8d65e5c --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eaa49aa3fbce55063a9c3037d18a6b9ff3bd327981b4a4f35799a99c434189df +size 1162241 diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..2dccd206ab0b344f5a64a1778c5ab75ca75c36f2 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0c09425581b2ad7d7786b9489ce88bfc64eecd577fec18aa66280c27e08b8f18 +size 1162235 diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..00b4695ff4b8b7593309c47004487cbf4430d7ec --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b1a7d5db5a150efea2d3bb76abaa4d5328c50ae0918a01e1e891509155994759 +size 47 diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..f51b6fb2e7d77b13ba0d42677a7851afae565773 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:163339eb50d2c2964191177ea5918ce97b681f08c2efb4694c9279cddb37854e +size 1548 diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..23102325f1fac5c62b40910a00c2344d3817f97e --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0ffffd6f5f47d90c8a79ffbafddbca7c98c15fe48b10563a3a5fd7d5bc5d7db3 +size 16329224 diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d943eb152f3a15f78821feaa8b5eedb381af5ec7 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml @@ -0,0 +1,206 @@ +experiment: + name: ant-mean5000-profile-simulation-100ep + seed: 42 +backend: + type: sample_factory + algo: APPO + device: cuda + train_dir: /mnt/checkpoints/latency-sensitive-bench/small_models/ant + restart_behavior: overwrite + run_mode: eval +executor: + mode: simulated + simulated_worker_capacity: 1 + simulated_inference_pool: true + inference_devices: + - cuda:0 + inference_batch_size: 16 +env: + action_space: + dtype: float32 + high: + - 1.0 + - 1.0 + - 1.0 + - 1.0 + - 1.0 + - 1.0 + - 1.0 + - 1.0 + labels: + - back_right_hip_torque + - back_right_ankle_torque + - front_left_hip_torque + - front_left_ankle_torque + - front_right_hip_torque + - front_right_ankle_torque + - back_left_hip_torque + - back_left_ankle_torque + low: + - -1.0 + - -1.0 + - -1.0 + - -1.0 + - -1.0 + - -1.0 + - -1.0 + - -1.0 + type: box + base_prompt: Make the Ant move forward as fast as possible without falling. Predict + eight continuous torques in [-1, 1] ordered as back right hip, back right ankle, + front left hip, front left ankle, front right hip, front right ankle, back left + hip, and back left ankle. + env_fps: 10.0 + env_id: LatencyBench/AntContinuous-v0 + frame_stack: 1 + make_kwargs: + base_env_id: Ant-v4 + base_make_kwargs: + exclude_current_positions_from_observation: true + use_contact_forces: false + render_mode: rgb_array + noop_action: + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + obs_fps: 10.0 + registration_imports: + - latency_bench.envs.gymnasium_ant + state_space: + labels: + - torso_z + - torso_quaternion_w + - torso_quaternion_x + - torso_quaternion_y + - torso_quaternion_z + - front_left_hip_angle + - front_left_ankle_angle + - front_right_hip_angle + - front_right_ankle_angle + - back_left_hip_angle + - back_left_ankle_angle + - back_right_hip_angle + - back_right_ankle_angle + - torso_x_velocity + - torso_y_velocity + - torso_z_velocity + - torso_angular_velocity_x + - torso_angular_velocity_y + - torso_angular_velocity_z + - front_left_hip_angular_velocity + - front_left_ankle_angular_velocity + - front_right_hip_angular_velocity + - front_right_ankle_angular_velocity + - back_left_hip_angular_velocity + - back_left_ankle_angular_velocity + - back_right_hip_angular_velocity + - back_right_ankle_angular_velocity + task_name: ant_rgb_state + name: gymnasium + obs_resize: + - 224 + - 224 +latency: + method: temporal + profile_path: /home/ubuntu/lzj/mean-profiling/reference/checkpoint-configs/latency-aware/ant/small-policy/sample-factory-qwenoft-rtx3090-v1/profile/profile.json + profile_worker_slot: 0 + seed: 271828 + add_latency_info: false +scheduler: + hold_policy: hold + ordering_policy: issue_order_fifo +policy: + type: starvla + checkpoint_path: /home/ubuntu/lzj/mean-profiling/ant/vla-publication/checkpoints/model.pt + model_config_path: /home/ubuntu/lzj/mean-profiling/ant/vla-publication/config.full.yaml + task_manifest_path: /home/ubuntu/lzj/mean-profiling/ant/vla-publication/manifest.json + device: cuda:0 + latency_prompt_map_path: /home/ubuntu/lzj/mean-profiling/ant/vla-publication/latency_prompt_map.json + latency_prompt_key: 1 + prompt_mode: raw + unnorm_key: new_embodiment + backbone_path: /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct + worker_python_executable: /home/ubuntu/lzj/conda/envs/qwenoft/bin/python + action_prefix: + mode: none +training: + train_for_env_steps: 10000000 + num_workers: 8 + num_envs_per_worker: 8 + worker_num_splits: 2 + num_policies: 1 + batch_size: 1024 + rollout: 64 + recurrence: 1 + num_epochs: 2 + num_batches_per_epoch: 4 + num_batches_to_accumulate: 2 + policy_workers_per_policy: 1 + max_policy_lag: 10000 + learning_rate: 0.00295 + lr_schedule: linear_decay + lr_schedule_kl_threshold: 0.008 + gamma: 0.99 + gae_lambda: 0.95 + ppo_clip_ratio: 0.2 + ppo_clip_value: 1.0 + value_loss_coeff: 1.3 + max_grad_norm: 3.5 + exploration_loss: entropy + exploration_loss_coeff: 0.0 + kl_loss_coeff: 0.1 + reward_scale: 1.0 + reward_clip: 1000.0 + async_rl: false + serial_mode: false + batched_sampling: false + with_vtrace: false + use_rnn: false + encoder_mlp_layers: + - 64 + - 64 + nonlinearity: tanh + adaptive_stddev: false + policy_initialization: torch_default + initial_stddev: 1.0 + actor_critic_share_weights: true + shuffle_minibatches: false + value_bootstrap: true + normalize_input: true + normalize_returns: true + decorrelate_experience_max_seconds: 10 + decorrelate_envs_on_one_worker: true + set_workers_cpu_affinity: true + force_envs_single_thread: true + save_every_sec: 600 + keep_checkpoints: 3 + save_best_every_sec: 60 + save_best_after: 100000 +evaluation: + eval_interval_steps: null + eval_episodes: 100 + eval_parallel_envs: 16 + eval_max_steps: 1000 + eval_deterministic: true + eval_latency_values: null + eval_raw_reward: true + eval_suites: + fixed: [] + normal: [] + uniform: [] +logging: + output_dir: /home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant + video: + enabled: false + save_step_records: true + save_action_records: true + save_latency_records: true + wandb_project: null + wandb_group: null + wandb_job_type: null + simulated_pipeline_profile: false diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/statistics.json b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/statistics.json new file mode 100644 index 0000000000000000000000000000000000000000..348a7391b935df1b16e2f1839d019a7baa310683 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/statistics.json @@ -0,0 +1,36 @@ +{ + "n_episodes": 100, + "mean_return": 1453.844063807972, + "std_return": 693.7275200567642, + "min_return": 85.64836938561511, + "max_return": 2508.917122342891, + "mean_length": 803.85, + "std_length": 328.8088312378486, + "min_length": 60.0, + "max_length": 1000.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "LatencyBench/AntContinuous-v0", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "ant", + "instance_id": "instance_859cf1e47bca6046", + "source_run_id": "20260911T033037730561Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "ant", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42", + "profile_sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant.yaml" +} diff --git a/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/stdout.log b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/stdout.log new file mode 100644 index 0000000000000000000000000000000000000000..76f014378d69b16b9d6cb0d808d6016e278da970 --- /dev/null +++ b/latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/stdout.log @@ -0,0 +1,405 @@ +[bench] run=ant-mean5000-profile-simulation-100ep sweeps=1 episodes_per_sweep=100 total_episode_runs=100 +[bench] sweep 1/1: eval_latency=profile_sample +10/01 [06:50:55] INFO | >> Failed to load library ( ctypesloader.py:70 + 'libOpenGL.so.0' ): libOpenGL.so.0: + cannot open shared object file: No + such file or directory + INFO | >> No OpenGL_accelerate acceleratesupport.py:24 + module loaded: No module named + 'OpenGL_accelerate' + INFO | >> Failed to load library ( ctypesloader.py:70 + 'libOpenGL.so.0' ): libOpenGL.so.0: + cannot open shared object file: No + such file or directory +10/01 [06:50:58] INFO | >> Loaded mixtures from Behavior registry.py:113 + (data_config): ['BEHAVIOR_challenge'] + INFO | >> Loaded data_config from DOMINO: registry.py:107 + ['robotwin'] + INFO | >> Loaded embodiment_tags from registry.py:110 + DOMINO (data_config): [] + INFO | >> Loaded mixtures from DOMINO registry.py:113 + (data_config): ['domino', + 'domino_clean', 'domino_random', + 'domino_cotrain'] + INFO | >> Loaded data_config from Franka: registry.py:107 + ['custom_robot_config', + 'demo_sim_franka_delta_joints', + 'SO101'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Franka (data_config): [] + INFO | >> Loaded mixtures from Franka registry.py:113 + (data_config): ['custom_dataset', + 'custom_dataset_2', + 'demo_sim_pick_place', 'SO101_pick'] + INFO | >> Loaded data_config from LIBERO: registry.py:107 + ['libero_franka'] + INFO | >> Loaded embodiment_tags from registry.py:110 + LIBERO (data_config): [] + INFO | >> Loaded mixtures from LIBERO registry.py:113 + (data_config): ['libero_all', + 'libero_goal', 'multi_robot'] + INFO | >> Loaded data_config from MIKASA: registry.py:107 + ['mikasa_franka_h1'] + INFO | >> Loaded embodiment_tags from registry.py:110 + MIKASA (data_config): + ['mikasa_franka_h1'] + INFO | >> Loaded mixtures from MIKASA registry.py:113 + (data_config): + ['local/intercept_grab_fast_vla_v0_h1_ + train'] + INFO | >> Loaded data_config from registry.py:107 + RoboChallenge_table30v2: + ['ur5_robochallenge', + 'arx5_robochallenge', + 'dosw1_robochallenge'] + INFO | >> Loaded embodiment_tags from registry.py:110 + RoboChallenge_table30v2 (data_config): + ['ur5_robochallenge', + 'arx5_robochallenge', + 'dosw1_robochallenge'] + INFO | >> Loaded mixtures from registry.py:113 + RoboChallenge_table30v2 (data_config): + ['robochallenge_table30v2_shred_paper' + , 'robochallenge_table30v2_ur5_all', + 'robochallenge_table30v2_arx5_all', + 'robochallenge_table30v2_dosw1_all'] + INFO | >> Loaded data_config from registry.py:107 + Robocasa_365: + ['panda_omron_robocasa365'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Robocasa_365 (data_config): [] + INFO | >> Loaded mixtures from registry.py:113 + Robocasa_365 (data_config): + ['robocasa365_open_drawer_target_human + ', + 'robocasa365_atomic_target_human_all', + 'robocasa365_composite_target_human_al + l', 'robocasa365_target_human_all'] + INFO | >> Loaded data_config from registry.py:107 + Robocasa_tabletop: + ['fourier_gr1_arms_waist'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Robocasa_tabletop (data_config): [] + INFO | >> Loaded mixtures from registry.py:113 + Robocasa_tabletop (data_config): + ['fourier_gr1_unified_1000'] + INFO | >> Loaded data_config from registry.py:107 + Robotwin: ['robotwin', 'robotwin50', + 'arx_x5'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Robotwin (data_config): [] + INFO | >> Loaded mixtures from Robotwin registry.py:113 + (data_config): ['robotwin_all', + 'robotwin_all_50', 'robotwin', + 'robotwin_task1', 'robotwin_task2', + 'arx_x5'] + INFO | >> Loaded data_config from registry.py:107 + SimplerEnv: ['oxe_droid', + 'oxe_bridge', 'oxe_rt1'] + INFO | >> Loaded embodiment_tags from registry.py:110 + SimplerEnv (data_config): [] + INFO | >> Loaded mixtures from SimplerEnv registry.py:113 + (data_config): ['bridge', + 'bridge_rt_1'] + INFO | >> Loaded data_config from registry.py:107 + VLA-Arena: ['vla_arena_franka'] + INFO | >> Loaded embodiment_tags from registry.py:110 + VLA-Arena (data_config): [] + INFO | >> Loaded mixtures from VLA-Arena registry.py:113 + (data_config): ['vla_arena_L0_S', + 'vla_arena_L0_M', 'vla_arena_L0_L'] + INFO | >> Loaded data_config from registry.py:107 + rl_games: ['rl_games_flappy', + 'rl_games_demon_attack', + 'rl_games_defend_the_line', + 'rl_games_deadly_corridor', + 'rl_games_asterix', + 'rl_games_atlantis', + 'rl_games_gymnasium', + 'rl_games_gymnasium_discrete', + 'rl_games_gymnasium_native'] + INFO | >> Loaded embodiment_tags from registry.py:110 + rl_games (data_config): + ['rl_games_flappy', + 'rl_games_demon_attack', + 'rl_games_defend_the_line', + 'rl_games_deadly_corridor', + 'rl_games_asterix', + 'rl_games_atlantis', + 'rl_games_gymnasium', + 'rl_games_gymnasium_discrete', + 'rl_games_gymnasium_native'] + INFO | >> Loaded mixtures from rl_games registry.py:113 + (data_config): ['flappy_train', + 'flappy_train__bridge', + 'flappy_mixed_latency_train', + 'flappy_mixed_latency_train__bridge', + 'demon_attack_train', + 'demon_attack_train__bridge', + 'demon_attack_mixed_latency_train', + 'demon_attack_mixed_latency_train__bri + dge', 'defend_the_line_train', + 'defend_the_line_train__bridge', + 'defend_the_line_mixed_latency_train', + 'defend_the_line_mixed_latency_train__ + bridge', 'deadly_corridor_train', + 'deadly_corridor_train__bridge', + 'deadly_corridor_mixed_latency_train', + 'deadly_corridor_mixed_latency_train__ + bridge', 'asterix_train', + 'asterix_train__bridge', + 'asterix_mixed_latency_train', + 'asterix_mixed_latency_train__bridge', + 'atlantis_train', + 'atlantis_train__bridge', + 'atlantis_mixed_latency_train', + 'atlantis_mixed_latency_train__bridge' + , 'h1hand_balance_hard'] + INFO | >> PolicyServerWrapper: loading policy_wrapper.py:73 + framework from + /home/ubuntu/lzj/mean-profiling/a + nt/vla-publication/checkpoints/mo + del.pt + INFO | >> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/an + t/vla-publication/checkpoints/model + .pt` +[QWen3] loading /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct with gradient_checkpointing=True + Loading checkpoint shards: 0%| | 0/2 [00:00> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/an + t/vla-publication/checkpoints/model + .pt` + INFO | >> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/an + t/vla-publication/checkpoints/model + .pt` + INFO | >> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/an + t/vla-publication/checkpoints/model + .pt` + INFO | >> PolicyNormProcessor policy_norm_processor.py:333 + ready: + robot_type=rl_games_gymna + sium, + unnorm_key=new_embodiment + , + action_keys=['action.butt + on'] (dims=[8]), + state_keys=['state.game_s + tate'] + INFO | >> PolicyServerWrapper ready: policy_wrapper.py:126 + action_chunk_size=1, + default_unnorm_key=new_embodimen + t, + available_unnorm_keys=['new_embo + diment'], + action_keys=['action.button'], + state_keys=['state.game_state'] +[bench] sweep 1/1 episode 1/100 done +[bench] sweep 1/1 episode 2/100 done +[bench] sweep 1/1 episode 3/100 done +[bench] sweep 1/1 episode 4/100 done +[bench] sweep 1/1 episode 5/100 done +[bench] sweep 1/1 episode 6/100 done +[bench] sweep 1/1 episode 7/100 done +[bench] sweep 1/1 episode 8/100 done +[bench] sweep 1/1 episode 9/100 done +[bench] sweep 1/1 episode 10/100 done +[bench] sweep 1/1 episode 11/100 done +[bench] sweep 1/1 episode 12/100 done +[bench] sweep 1/1 episode 13/100 done +[bench] sweep 1/1 episode 14/100 done +[bench] sweep 1/1 episode 15/100 done +[bench] sweep 1/1 episode 16/100 done +[bench] sweep 1/1 episode 17/100 done +[bench] sweep 1/1 episode 18/100 done +[bench] sweep 1/1 episode 19/100 done +[bench] sweep 1/1 episode 20/100 done +[bench] sweep 1/1 episode 21/100 done +[bench] sweep 1/1 episode 22/100 done +[bench] sweep 1/1 episode 23/100 done +[bench] sweep 1/1 episode 24/100 done +[bench] sweep 1/1 episode 25/100 done +[bench] sweep 1/1 episode 26/100 done +[bench] sweep 1/1 episode 27/100 done +[bench] sweep 1/1 episode 28/100 done +[bench] sweep 1/1 episode 29/100 done +[bench] sweep 1/1 episode 30/100 done +[bench] sweep 1/1 episode 31/100 done +[bench] sweep 1/1 episode 32/100 done +[bench] sweep 1/1 episode 33/100 done +[bench] sweep 1/1 episode 34/100 done +[bench] sweep 1/1 episode 35/100 done +[bench] sweep 1/1 episode 36/100 done +[bench] sweep 1/1 episode 37/100 done +[bench] sweep 1/1 episode 38/100 done +[bench] sweep 1/1 episode 39/100 done +[bench] sweep 1/1 episode 40/100 done +[bench] sweep 1/1 episode 41/100 done +[bench] sweep 1/1 episode 42/100 done +[bench] sweep 1/1 episode 43/100 done +[bench] sweep 1/1 episode 44/100 done +[bench] sweep 1/1 episode 45/100 done +[bench] sweep 1/1 episode 46/100 done +[bench] sweep 1/1 episode 47/100 done +[bench] sweep 1/1 episode 48/100 done +[bench] sweep 1/1 episode 49/100 done +[bench] sweep 1/1 episode 50/100 done +[bench] sweep 1/1 episode 51/100 done +[bench] sweep 1/1 episode 52/100 done +[bench] sweep 1/1 episode 53/100 done +[bench] sweep 1/1 episode 54/100 done +[bench] sweep 1/1 episode 55/100 done +[bench] sweep 1/1 episode 56/100 done +[bench] sweep 1/1 episode 57/100 done +[bench] sweep 1/1 episode 58/100 done +[bench] sweep 1/1 episode 59/100 done +[bench] sweep 1/1 episode 60/100 done +[bench] sweep 1/1 episode 61/100 done +[bench] sweep 1/1 episode 62/100 done +[bench] sweep 1/1 episode 63/100 done +[bench] sweep 1/1 episode 64/100 done +[bench] sweep 1/1 episode 65/100 done +[bench] sweep 1/1 episode 66/100 done +[bench] sweep 1/1 episode 67/100 done +[bench] sweep 1/1 episode 68/100 done +[bench] sweep 1/1 episode 69/100 done +[bench] sweep 1/1 episode 70/100 done +[bench] sweep 1/1 episode 71/100 done +[bench] sweep 1/1 episode 72/100 done +[bench] sweep 1/1 episode 73/100 done +[bench] sweep 1/1 episode 74/100 done +[bench] sweep 1/1 episode 75/100 done +[bench] sweep 1/1 episode 76/100 done +[bench] sweep 1/1 episode 77/100 done +[bench] sweep 1/1 episode 78/100 done +[bench] sweep 1/1 episode 79/100 done +[bench] sweep 1/1 episode 80/100 done +[bench] sweep 1/1 episode 81/100 done +[bench] sweep 1/1 episode 82/100 done +[bench] sweep 1/1 episode 83/100 done +[bench] sweep 1/1 episode 84/100 done +[bench] sweep 1/1 episode 85/100 done +[bench] sweep 1/1 episode 86/100 done +[bench] sweep 1/1 episode 87/100 done +[bench] sweep 1/1 episode 88/100 done +[bench] sweep 1/1 episode 89/100 done +[bench] sweep 1/1 episode 90/100 done +[bench] sweep 1/1 episode 91/100 done +[bench] sweep 1/1 episode 92/100 done +[bench] sweep 1/1 episode 93/100 done +[bench] sweep 1/1 episode 94/100 done +[bench] sweep 1/1 episode 95/100 done +[bench] sweep 1/1 episode 96/100 done +[bench] sweep 1/1 episode 97/100 done +[bench] sweep 1/1 episode 98/100 done +[bench] sweep 1/1 episode 99/100 done +[bench] sweep 1/1 episode 100/100 done +[bench] sweep 1/1 complete elapsed=1243.1s +episode=0 return=1846.110 steps=1000 mean_latency_ms=89.89614608291177 +episode=1 return=2415.721 steps=1000 mean_latency_ms=90.00308114332259 +episode=2 return=457.344 steps=177 mean_latency_ms=89.83716885697598 +episode=3 return=1421.795 steps=1000 mean_latency_ms=89.87909631338808 +episode=4 return=2037.723 steps=937 mean_latency_ms=89.82685347370092 +episode=5 return=2330.630 steps=1000 mean_latency_ms=90.47193606091501 +episode=6 return=1161.644 steps=429 mean_latency_ms=89.84194070141322 +episode=7 return=2351.152 steps=1000 mean_latency_ms=89.92640891799017 +episode=8 return=513.296 steps=210 mean_latency_ms=89.88895656571908 +episode=9 return=1126.865 steps=660 mean_latency_ms=89.91361550654544 +episode=10 return=1693.437 steps=1000 mean_latency_ms=89.84960962337662 +episode=11 return=948.378 steps=1000 mean_latency_ms=89.94678527711802 +episode=12 return=2322.052 steps=1000 mean_latency_ms=90.11873818885832 +episode=13 return=960.403 steps=1000 mean_latency_ms=90.93377411320307 +episode=14 return=1464.564 steps=1000 mean_latency_ms=89.80893705661644 +episode=15 return=1110.549 steps=1000 mean_latency_ms=89.99466844889166 +episode=16 return=2246.208 steps=1000 mean_latency_ms=90.1624262080728 +episode=17 return=85.648 steps=60 mean_latency_ms=89.87005518664785 +episode=18 return=340.548 steps=143 mean_latency_ms=89.94240076131771 +episode=19 return=2457.089 steps=1000 mean_latency_ms=89.92054036086635 +episode=20 return=2166.251 steps=1000 mean_latency_ms=89.95452553058773 +episode=21 return=2357.958 steps=1000 mean_latency_ms=89.86780458600198 +episode=22 return=1654.878 steps=871 mean_latency_ms=90.08433827425095 +episode=23 return=1499.367 steps=1000 mean_latency_ms=89.89663615668341 +episode=24 return=2297.403 steps=1000 mean_latency_ms=90.09818426014289 +episode=25 return=1253.361 steps=543 mean_latency_ms=89.9390124443734 +episode=26 return=1221.270 steps=1000 mean_latency_ms=89.84986177450952 +episode=27 return=2389.248 steps=1000 mean_latency_ms=89.95772586857817 +episode=28 return=1682.529 steps=707 mean_latency_ms=89.76145439054764 +episode=29 return=2474.676 steps=1000 mean_latency_ms=89.82093759631324 +episode=30 return=382.923 steps=256 mean_latency_ms=90.69916524888657 +episode=31 return=1837.813 steps=1000 mean_latency_ms=90.03642087221974 +episode=32 return=227.194 steps=101 mean_latency_ms=89.8553742761573 +episode=33 return=1700.631 steps=1000 mean_latency_ms=89.80097198453268 +episode=34 return=960.945 steps=372 mean_latency_ms=89.8420903148968 +episode=35 return=2290.672 steps=1000 mean_latency_ms=89.91770573449698 +episode=36 return=328.573 steps=162 mean_latency_ms=90.01187187392946 +episode=37 return=1180.074 steps=1000 mean_latency_ms=89.81938304804656 +episode=38 return=817.419 steps=363 mean_latency_ms=89.85140773938038 +episode=39 return=1651.226 steps=1000 mean_latency_ms=91.17610023451576 +episode=40 return=1428.175 steps=1000 mean_latency_ms=89.8551155619885 +episode=41 return=1627.384 steps=1000 mean_latency_ms=90.55986754698809 +episode=42 return=1079.756 steps=680 mean_latency_ms=90.17098553312343 +episode=43 return=2173.945 steps=1000 mean_latency_ms=89.84319301261918 +episode=44 return=409.906 steps=160 mean_latency_ms=89.66802828269809 +episode=45 return=2467.264 steps=1000 mean_latency_ms=89.90844708827387 +episode=46 return=657.408 steps=248 mean_latency_ms=89.86487149424892 +episode=47 return=974.744 steps=1000 mean_latency_ms=89.76305094278182 +episode=48 return=1510.518 steps=1000 mean_latency_ms=90.24355118464125 +episode=49 return=602.234 steps=260 mean_latency_ms=89.7103209703719 +episode=50 return=760.978 steps=316 mean_latency_ms=89.8206829517188 +episode=51 return=1941.172 steps=1000 mean_latency_ms=90.30785204408768 +episode=52 return=624.359 steps=281 mean_latency_ms=89.94582387208622 +episode=53 return=2163.435 steps=1000 mean_latency_ms=89.84848132390947 +episode=54 return=1126.996 steps=1000 mean_latency_ms=89.84637728060243 +episode=55 return=1405.132 steps=1000 mean_latency_ms=90.18855922596491 +episode=56 return=1206.292 steps=1000 mean_latency_ms=89.78065539051504 +episode=57 return=2392.798 steps=1000 mean_latency_ms=89.76388668266138 +episode=58 return=964.022 steps=1000 mean_latency_ms=89.82618651237911 +episode=59 return=2252.193 steps=1000 mean_latency_ms=89.8197082349776 +episode=60 return=2471.916 steps=1000 mean_latency_ms=89.96642568195992 +episode=61 return=1902.849 steps=1000 mean_latency_ms=89.87542708971246 +episode=62 return=1435.666 steps=1000 mean_latency_ms=90.28644124851098 +episode=63 return=1668.324 steps=1000 mean_latency_ms=89.86433221097877 +episode=64 return=1813.291 steps=1000 mean_latency_ms=89.85118001877315 +episode=65 return=446.724 steps=189 mean_latency_ms=89.8309544306309 +episode=66 return=130.842 steps=74 mean_latency_ms=89.82416773165995 +episode=67 return=2315.857 steps=1000 mean_latency_ms=90.25959750757508 +episode=68 return=288.392 steps=116 mean_latency_ms=90.09271984792927 +episode=69 return=894.023 steps=1000 mean_latency_ms=89.89663691508213 +episode=70 return=2030.323 steps=1000 mean_latency_ms=89.84028619017428 +episode=71 return=507.945 steps=215 mean_latency_ms=90.57267432538549 +episode=72 return=2377.737 steps=1000 mean_latency_ms=89.84919425782105 +episode=73 return=897.308 steps=1000 mean_latency_ms=89.90431472264346 +episode=74 return=1454.613 steps=1000 mean_latency_ms=91.19515970740413 +episode=75 return=2292.458 steps=1000 mean_latency_ms=89.8333901030839 +episode=76 return=1424.378 steps=1000 mean_latency_ms=89.88029014661089 +episode=77 return=1441.111 steps=1000 mean_latency_ms=89.79844243631413 +episode=78 return=1265.477 steps=1000 mean_latency_ms=89.86009503143968 +episode=79 return=1662.881 steps=1000 mean_latency_ms=90.5661722205243 +episode=80 return=2508.917 steps=1000 mean_latency_ms=89.87403626041336 +episode=81 return=1655.351 steps=1000 mean_latency_ms=90.05039760075688 +episode=82 return=1387.384 steps=821 mean_latency_ms=90.10235730111886 +episode=83 return=646.436 steps=271 mean_latency_ms=89.7559653760994 +episode=84 return=2172.801 steps=1000 mean_latency_ms=89.84602989356796 +episode=85 return=165.921 steps=72 mean_latency_ms=89.66756877688618 +episode=86 return=1063.148 steps=1000 mean_latency_ms=89.77409215132576 +episode=87 return=1000.135 steps=1000 mean_latency_ms=89.81900933661238 +episode=88 return=1977.236 steps=1000 mean_latency_ms=89.77653862908736 +episode=89 return=1937.167 steps=1000 mean_latency_ms=90.12773943823525 +episode=90 return=1344.773 steps=1000 mean_latency_ms=90.39620143170467 +episode=91 return=786.338 steps=441 mean_latency_ms=89.82893206036925 +episode=92 return=1391.060 steps=1000 mean_latency_ms=89.86676880070257 +episode=93 return=503.300 steps=250 mean_latency_ms=89.94819176115624 +episode=94 return=2446.748 steps=1000 mean_latency_ms=89.84848658183878 +episode=95 return=1171.910 steps=1000 mean_latency_ms=89.92785850220504 +episode=96 return=2356.731 steps=1000 mean_latency_ms=90.42933754946152 +episode=97 return=2356.122 steps=1000 mean_latency_ms=89.82049779117614 +episode=98 return=1389.299 steps=1000 mean_latency_ms=90.82209581044775 +episode=99 return=967.234 steps=1000 mean_latency_ms=89.80016695371027 +summary latency=profile_sample episodes=100 mean_return=1453.844 std_return=693.728 mean_length=803.9 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/REPORT.md b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/REPORT.md new file mode 100644 index 0000000000000000000000000000000000000000..7f2af02a5f64a7ae2fdb82b2a93c6eaac45fb57d --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/REPORT.md @@ -0,0 +1,20 @@ +# QwenOFT mean-trained checkpoints under profile simulation + +Four final step-5000 H1 checkpoints; two rounds, one evaluation per physical GPU2/3,100 episodes each (400 total). + +The training latency was fixed mean; this evaluation samples the complete archived RTX3090 temporal hidden-regime profile. Simulator FPS, seeds, horizon, limits and model/profile identities are in evaluation-plan.json. Standard deviations below use ddof=0. Returns have task-specific scales. Startup checks are separate and excluded. + +| Task | Episodes | Return mean +/- SD | Length mean +/- SD | Success | Invalid | +|---|---:|---:|---:|---:|---:| +| flappy | 100 | 384.824005 +/- 116.787774 | 3119.31 +/- 939.87 | not provided by task | 0 | +| deadly_corridor | 100 | 1620.798776 +/- 913.624278 | 148.53 +/- 49.46 | not provided by task | 0 | +| ant | 100 | 1453.844064 +/- 693.727520 | 803.85 +/- 328.81 | not provided by task | 0 | +| intercept | 100 | 3.544349 +/- 7.071923 | 60.00 +/- 0.00 | 9/100 | 0 | + +No success metric is invented for Flappy/Deadly/Ant. Intercept reports the native accumulated success flag. No policy-quality acceptance gate is claimed. + +Compatibility repairs: portable robot_type copied from each actual training manifest (weights unchanged); official ViZDoom1.2.4 VizdoomCorridor-v0 uses the same deadly_corridor WAD as SF, preserves render contract and semantic seven-button ordering; public action space is equivalent MultiBinary7. Existing native render/button/history tests passed. Full eval source/patch and original profile assets are archived. + +Flappy/Deadly seeds1000000..1000099; Ant42..141; Intercept4242424242..4242424341. Latency seed271828. Flappy10/10Hz, Deadly35/8.75Hz, Ant10/10Hz, Intercept20/20Hz. Max raw frames3600/3600/1000/60; capacities1. MIKASA H1 holds last chunk action; no prefix, no DAgger. Ant keeps its training prompt label1 while execution latency is sampled. + +Raw JSONL logs are losslessly gzip-compressed for distribution; original uncompressed records remain on the experiment host. Empty observation_attempts files are retained; admission/drop evidence is in steps/actions. Per-task CSV and full400 episode CSV are provided. diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/all_episodes.csv b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/all_episodes.csv new file mode 100644 index 0000000000000000000000000000000000000000..be5e27101cb97f6a2dd0d85a0399d3fb8a5ea839 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/all_episodes.csv @@ -0,0 +1,401 @@ +task,episode_id,seed,return_env,length,mean_latency_ms,success +flappy,0,1000000,444.6000052243471,3600,76.02271694866694, +flappy,1,1000001,444.6000052243471,3600,76.14445348705047, +flappy,2,1000002,444.6000052243471,3600,75.83047266244563, +flappy,3,1000003,444.6000052243471,3600,76.04121221698036, +flappy,4,1000004,444.6000052243471,3600,75.7789115791707, +flappy,5,1000005,228.2000027000904,1861,76.22757676162651, +flappy,6,1000006,444.6000052243471,3600,75.98373978309758, +flappy,7,1000007,444.6000052243471,3600,75.85552109823348, +flappy,8,1000008,444.6000052243471,3600,75.9782303085917, +flappy,9,1000009,444.6000052243471,3600,75.94667987356688, +flappy,10,1000010,444.6000052243471,3600,75.66396359484234, +flappy,11,1000011,444.6000052243471,3600,75.7794525026407, +flappy,12,1000012,444.6000052243471,3600,75.90110110734818, +flappy,13,1000013,444.6000052243471,3600,76.01870178237883, +flappy,14,1000014,444.6000052243471,3600,75.75567207010911, +flappy,15,1000015,444.6000052243471,3600,75.83026036637241, +flappy,16,1000016,444.6000052243471,3600,75.74502908171665, +flappy,17,1000017,444.6000052243471,3600,75.84316844302293, +flappy,18,1000018,444.6000052243471,3600,75.85876738771161, +flappy,19,1000019,265.50000313669443,2162,75.89492798135642, +flappy,20,1000020,444.6000052243471,3600,75.90859756288593, +flappy,21,1000021,444.6000052243471,3600,75.93474621914784, +flappy,22,1000022,444.6000052243471,3600,75.77022360156529, +flappy,23,1000023,444.6000052243471,3600,75.8506098974935, +flappy,24,1000024,444.6000052243471,3600,75.80511776716725, +flappy,25,1000025,116.00000138580799,955,76.07937915327228, +flappy,26,1000026,444.6000052243471,3600,75.77409482659607, +flappy,27,1000027,444.6000052243471,3600,75.82354466933252, +flappy,28,1000028,444.6000052243471,3600,75.92578714415393, +flappy,29,1000029,444.6000052243471,3600,75.77326038618416, +flappy,30,1000030,256.0000030249357,2085,75.8461606092662, +flappy,31,1000031,444.6000052243471,3600,75.87053786258159, +flappy,32,1000032,444.6000052243471,3600,75.90930861144982, +flappy,33,1000033,444.6000052243471,3600,75.80530422686525, +flappy,34,1000034,444.6000052243471,3600,76.05997569829616, +flappy,35,1000035,444.6000052243471,3600,75.67579907153437, +flappy,36,1000036,444.6000052243471,3600,76.07561842170198, +flappy,37,1000037,444.6000052243471,3600,75.87459102177027, +flappy,38,1000038,55.60000067949295,468,75.8887188983619, +flappy,39,1000039,444.6000052243471,3600,75.86536772802552, +flappy,40,1000040,432.900005094707,3512,76.00355652525975, +flappy,41,1000041,274.8000032454729,2237,75.7658282850597, +flappy,42,1000042,264.90000312775373,2156,75.95264956954799, +flappy,43,1000043,265.4000031352043,2161,75.82748305801191, +flappy,44,1000044,444.6000052243471,3600,75.9295822845668, +flappy,45,1000045,143.90000171214342,1180,75.94310218110371, +flappy,46,1000046,444.6000052243471,3600,75.69568531179425, +flappy,47,1000047,93.1000011190772,771,76.0527875505066, +flappy,48,1000048,56.10000068694353,473,76.20882901957174, +flappy,49,1000049,265.2000031322241,2159,76.05401077635972, +flappy,50,1000050,444.6000052243471,3600,75.89333271844873, +flappy,51,1000051,444.6000052243471,3600,75.89090159365671, +flappy,52,1000052,398.80000469088554,3234,75.91218218803246, +flappy,53,1000053,444.6000052243471,3600,75.86400590251726, +flappy,54,1000054,270.2000031918287,2200,76.01590238337654, +flappy,55,1000055,69.70000084489584,582,75.68422480575155, +flappy,56,1000056,444.6000052243471,3600,75.87884524455251, +flappy,57,1000057,444.6000052243471,3600,75.96981187494319, +flappy,58,1000058,444.6000052243471,3600,76.03772455115222, +flappy,59,1000059,437.90000515431166,3553,76.04088529786887, +flappy,60,1000060,348.9000041112304,2834,75.79422825165413, +flappy,61,1000061,444.6000052243471,3600,75.87728099437057, +flappy,62,1000062,78.9000009521842,656,75.96352981662133, +flappy,63,1000063,444.6000052243471,3600,75.80269270184165, +flappy,64,1000064,444.6000052243471,3600,75.88518180564401, +flappy,65,1000065,444.6000052243471,3600,75.87533034544981, +flappy,66,1000066,444.6000052243471,3600,75.94241138050401, +flappy,67,1000067,444.6000052243471,3600,75.95312277771471, +flappy,68,1000068,444.6000052243471,3600,75.8998829764233, +flappy,69,1000069,444.6000052243471,3600,75.98564617573034, +flappy,70,1000070,444.6000052243471,3600,75.68328575087021, +flappy,71,1000071,135.0000016093254,1109,75.99546963217229, +flappy,72,1000072,444.6000052243471,3600,75.9923106611263, +flappy,73,1000073,444.6000052243471,3600,75.80422251719546, +flappy,74,1000074,444.6000052243471,3600,75.95469853250815, +flappy,75,1000075,444.6000052243471,3600,75.74551875442629, +flappy,76,1000076,444.6000052243471,3600,75.93301571087362, +flappy,77,1000077,444.6000052243471,3600,75.98384926019328, +flappy,78,1000078,444.6000052243471,3600,75.85055115368883, +flappy,79,1000079,444.6000052243471,3600,75.97142616222317, +flappy,80,1000080,444.6000052243471,3600,75.97039764106849, +flappy,81,1000081,444.6000052243471,3600,75.74469321422862, +flappy,82,1000082,116.20000138878822,957,76.0366526049804, +flappy,83,1000083,444.6000052243471,3600,75.95924386190674, +flappy,84,1000084,444.6000052243471,3600,76.0310580385874, +flappy,85,1000085,36.60000045597553,314,75.8634823847272, +flappy,86,1000086,260.90000308305025,2125,75.91783880059099, +flappy,87,1000087,444.6000052243471,3600,76.16752514785735, +flappy,88,1000088,444.6000052243471,3600,75.83331254385584, +flappy,89,1000089,444.6000052243471,3600,76.2113387300584, +flappy,90,1000090,444.6000052243471,3600,75.8334932097261, +flappy,91,1000091,225.20000265538692,1831,75.75618859671614, +flappy,92,1000092,444.6000052243471,3600,75.79683788505955, +flappy,93,1000093,180.9000021442771,1478,75.96465307644473, +flappy,94,1000094,305.2000035941601,2478,75.86565754734926, +flappy,95,1000095,444.6000052243471,3600,75.74523644464854, +flappy,96,1000096,444.6000052243471,3600,75.94353591524424, +flappy,97,1000097,444.6000052243471,3600,75.81927739599219, +flappy,98,1000098,444.6000052243471,3600,75.96229410618645, +flappy,99,1000099,444.6000052243471,3600,75.94512877548694, +deadly_corridor,0,1000000,337.47547912597656,72,71.90727374040254, +deadly_corridor,1,1000001,819.0284423828125,143,73.84762082340946, +deadly_corridor,2,1000002,2284.857650756836,182,72.79171012339609, +deadly_corridor,3,1000003,2276.2068634033203,189,76.345275285376, +deadly_corridor,4,1000004,805.2153015136719,150,73.86282581373551, +deadly_corridor,5,1000005,621.8231658935547,115,74.31105893586228, +deadly_corridor,6,1000006,2276.414749145508,176,74.2226331369995, +deadly_corridor,7,1000007,2284.310989379883,176,72.9072057957754, +deadly_corridor,8,1000008,81.07798767089844,49,73.20258272646697, +deadly_corridor,9,1000009,317.2351837158203,75,72.54774919154028, +deadly_corridor,10,1000010,2282.7608489990234,176,72.78293151689127, +deadly_corridor,11,1000011,88.11907958984375,45,72.60486105128022, +deadly_corridor,12,1000012,2281.468536376953,176,72.29193331603048, +deadly_corridor,13,1000013,2276.6868591308594,178,72.73330265771509, +deadly_corridor,14,1000014,2276.1705932617188,178,73.30067987408609, +deadly_corridor,15,1000015,2282.6631622314453,177,72.49405489224537, +deadly_corridor,16,1000016,2280.300033569336,172,72.80884970803692, +deadly_corridor,17,1000017,2280.4182891845703,182,73.03539182090206, +deadly_corridor,18,1000018,2281.2594451904297,177,72.50972089313564, +deadly_corridor,19,1000019,479.8523712158203,99,72.58046231642126, +deadly_corridor,20,1000020,2279.7379455566406,181,72.47468246266928, +deadly_corridor,21,1000021,2284.9097442626953,197,83.18983231769475, +deadly_corridor,22,1000022,2286.2730407714844,172,72.84281562147524, +deadly_corridor,23,1000023,244.51919555664062,74,76.23461799191558, +deadly_corridor,24,1000024,2279.957275390625,195,72.94927214021655, +deadly_corridor,25,1000025,2283.952178955078,179,73.18968843008061, +deadly_corridor,26,1000026,2276.701370239258,178,72.87702909462648, +deadly_corridor,27,1000027,2277.142562866211,190,72.45412386128042, +deadly_corridor,28,1000028,2279.025634765625,177,74.11102172804317, +deadly_corridor,29,1000029,2285.7152099609375,177,71.63189230597281, +deadly_corridor,30,1000030,53.374298095703125,44,72.51418721312025, +deadly_corridor,31,1000031,2279.8080444335938,183,72.72702656843174, +deadly_corridor,32,1000032,2282.307357788086,178,74.33584751930213, +deadly_corridor,33,1000033,2282.834014892578,192,73.95005063555192, +deadly_corridor,34,1000034,2284.200241088867,188,76.29368894499888, +deadly_corridor,35,1000035,2287.2159118652344,179,72.81890806090988, +deadly_corridor,36,1000036,2284.693832397461,183,76.28284599973325, +deadly_corridor,37,1000037,2283.2066650390625,178,72.1797344044525, +deadly_corridor,38,1000038,2281.032196044922,178,73.74343783824916, +deadly_corridor,39,1000039,2282.960678100586,190,73.24816830891406, +deadly_corridor,40,1000040,2287.094253540039,185,72.35711232966574, +deadly_corridor,41,1000041,2279.3030853271484,179,72.42125368367608, +deadly_corridor,42,1000042,440.0892791748047,104,73.92064892672727, +deadly_corridor,43,1000043,2280.8592529296875,177,72.36020918178356, +deadly_corridor,44,1000044,2283.4308471679688,189,75.93658060557208, +deadly_corridor,45,1000045,2282.324264526367,181,73.54224681770178, +deadly_corridor,46,1000046,326.0184631347656,74,73.1983876441008, +deadly_corridor,47,1000047,2279.086135864258,182,73.00958120503027, +deadly_corridor,48,1000048,2280.3804626464844,179,73.17268244992928, +deadly_corridor,49,1000049,2276.215301513672,189,75.47590644230628, +deadly_corridor,50,1000050,2278.132034301758,182,74.50495464842548, +deadly_corridor,51,1000051,2285.6056518554688,181,73.41699294418743, +deadly_corridor,52,1000052,2287.240921020508,173,73.22110809114655, +deadly_corridor,53,1000053,310.81517028808594,73,74.09003681120738, +deadly_corridor,54,1000054,2276.6219787597656,175,72.98605010243534, +deadly_corridor,55,1000055,2276.2769470214844,194,75.17704077845171, +deadly_corridor,56,1000056,2278.861602783203,178,72.97353037051572, +deadly_corridor,57,1000057,2279.728561401367,181,73.96913002154926, +deadly_corridor,58,1000058,2280.544464111328,176,73.02432805290651, +deadly_corridor,59,1000059,487.829833984375,108,78.89398217393664, +deadly_corridor,60,1000060,567.0655517578125,113,72.64874721482185, +deadly_corridor,61,1000061,2278.210220336914,177,72.96068484971086, +deadly_corridor,62,1000062,2281.436721801758,186,75.46710866924751, +deadly_corridor,63,1000063,382.2119903564453,89,81.21157315209366, +deadly_corridor,64,1000064,246.2946014404297,70,73.9736408486285, +deadly_corridor,65,1000065,285.21240234375,76,73.13661133681993, +deadly_corridor,66,1000066,310.6737365722656,75,73.40468658737086, +deadly_corridor,67,1000067,346.1162872314453,75,72.1929723632303, +deadly_corridor,68,1000068,804.7056121826172,150,73.76397959753224, +deadly_corridor,69,1000069,2285.6442108154297,184,75.13255757158333, +deadly_corridor,70,1000070,730.5995788574219,132,73.25446825350764, +deadly_corridor,71,1000071,86.91796875,47,76.28335745963689, +deadly_corridor,72,1000072,60.30122375488281,44,76.83513093208644, +deadly_corridor,73,1000073,768.6264343261719,141,77.27057350071598, +deadly_corridor,74,1000074,2280.1071166992188,172,74.16699734355548, +deadly_corridor,75,1000075,860.9334106445312,151,73.15118478347584, +deadly_corridor,76,1000076,722.9459228515625,143,75.5655785931314, +deadly_corridor,77,1000077,2276.8687438964844,182,72.95102474014934, +deadly_corridor,78,1000078,368.3357238769531,79,71.51096709276341, +deadly_corridor,79,1000079,-76.45918273925781,17,72.24888432102617, +deadly_corridor,80,1000080,2281.5543823242188,183,73.32589540463356, +deadly_corridor,81,1000081,2281.6688842773438,171,73.10600900440717, +deadly_corridor,82,1000082,2277.5223083496094,178,73.55648700566698, +deadly_corridor,83,1000083,42.30937194824219,41,73.52700344736942, +deadly_corridor,84,1000084,2285.8980407714844,176,71.98655161011203, +deadly_corridor,85,1000085,68.90191650390625,45,72.84773487604696, +deadly_corridor,86,1000086,2286.2190551757812,171,72.82303966497733, +deadly_corridor,87,1000087,281.1173553466797,76,72.26983276661764, +deadly_corridor,88,1000088,2283.1607971191406,175,73.49638264342678, +deadly_corridor,89,1000089,2277.888946533203,177,73.44736473371472, +deadly_corridor,90,1000090,429.36326599121094,93,71.86172378947977, +deadly_corridor,91,1000091,252.0751953125,70,72.26459581736903, +deadly_corridor,92,1000092,2278.306442260742,192,80.97328482778371, +deadly_corridor,93,1000093,2285.236801147461,175,74.02717585214627, +deadly_corridor,94,1000094,857.2727355957031,152,85.59110000526613, +deadly_corridor,95,1000095,2275.9288024902344,199,73.62958803645523, +deadly_corridor,96,1000096,2286.8704833984375,179,72.31519682456816, +deadly_corridor,97,1000097,2278.048355102539,181,73.50330330803081, +deadly_corridor,98,1000098,2277.4480743408203,178,76.78472725777, +deadly_corridor,99,1000099,2276.9671478271484,178,77.9678189026336, +ant,0,42,1846.1103431567394,1000,89.89614608291177, +ant,1,43,2415.720790707953,1000,90.00308114332259, +ant,2,44,457.34421085068755,177,89.83716885697598, +ant,3,45,1421.7952163289683,1000,89.87909631338808, +ant,4,46,2037.7234409469488,937,89.82685347370092, +ant,5,47,2330.630175869275,1000,90.47193606091501, +ant,6,48,1161.643572255748,429,89.84194070141322, +ant,7,49,2351.1524624990343,1000,89.92640891799017, +ant,8,50,513.2964809479813,210,89.88895656571908, +ant,9,51,1126.8652528911032,660,89.91361550654544, +ant,10,52,1693.436933192597,1000,89.84960962337662, +ant,11,53,948.3780972955639,1000,89.94678527711802, +ant,12,54,2322.052445211472,1000,90.11873818885832, +ant,13,55,960.4026770814776,1000,90.93377411320307, +ant,14,56,1464.564005196777,1000,89.80893705661644, +ant,15,57,1110.548792782156,1000,89.99466844889166, +ant,16,58,2246.207900740156,1000,90.1624262080728, +ant,17,59,85.64836938561511,60,89.87005518664785, +ant,18,60,340.54799067574436,143,89.94240076131771, +ant,19,61,2457.088748930458,1000,89.92054036086635, +ant,20,62,2166.2512677098603,1000,89.95452553058773, +ant,21,63,2357.957592244385,1000,89.86780458600198, +ant,22,64,1654.8780938737275,871,90.08433827425095, +ant,23,65,1499.367100151414,1000,89.89663615668341, +ant,24,66,2297.4032619179525,1000,90.09818426014289, +ant,25,67,1253.360764666355,543,89.9390124443734, +ant,26,68,1221.270312709775,1000,89.84986177450952, +ant,27,69,2389.2476464763376,1000,89.95772586857817, +ant,28,70,1682.5290233886233,707,89.76145439054764, +ant,29,71,2474.676425615127,1000,89.82093759631324, +ant,30,72,382.9231146443659,256,90.69916524888657, +ant,31,73,1837.8126619276347,1000,90.03642087221974, +ant,32,74,227.19436616673684,101,89.8553742761573, +ant,33,75,1700.6312067622644,1000,89.80097198453268, +ant,34,76,960.9452812639541,372,89.8420903148968, +ant,35,77,2290.6720141359438,1000,89.91770573449698, +ant,36,78,328.5729178056416,162,90.01187187392946, +ant,37,79,1180.073938772476,1000,89.81938304804656, +ant,38,80,817.4190215442345,363,89.85140773938038, +ant,39,81,1651.2255208727013,1000,91.17610023451576, +ant,40,82,1428.174672693164,1000,89.8551155619885, +ant,41,83,1627.3838925098842,1000,90.55986754698809, +ant,42,84,1079.756369746183,680,90.17098553312343, +ant,43,85,2173.9447393037276,1000,89.84319301261918, +ant,44,86,409.90633829945847,160,89.66802828269809, +ant,45,87,2467.2636019929073,1000,89.90844708827387, +ant,46,88,657.4084558813478,248,89.86487149424892, +ant,47,89,974.7436031610902,1000,89.76305094278182, +ant,48,90,1510.5184342975385,1000,90.24355118464125, +ant,49,91,602.2339441184535,260,89.7103209703719, +ant,50,92,760.9784375126189,316,89.8206829517188, +ant,51,93,1941.172113330597,1000,90.30785204408768, +ant,52,94,624.3590446196446,281,89.94582387208622, +ant,53,95,2163.4347041279893,1000,89.84848132390947, +ant,54,96,1126.9957963444238,1000,89.84637728060243, +ant,55,97,1405.131632695366,1000,90.18855922596491, +ant,56,98,1206.2916757636292,1000,89.78065539051504, +ant,57,99,2392.7980761515178,1000,89.76388668266138, +ant,58,100,964.0216541467705,1000,89.82618651237911, +ant,59,101,2252.192880003706,1000,89.8197082349776, +ant,60,102,2471.9158497657563,1000,89.96642568195992, +ant,61,103,1902.8491241623092,1000,89.87542708971246, +ant,62,104,1435.6661382989703,1000,90.28644124851098, +ant,63,105,1668.3237703695809,1000,89.86433221097877, +ant,64,106,1813.291243529155,1000,89.85118001877315, +ant,65,107,446.72353548541076,189,89.8309544306309, +ant,66,108,130.84194814079504,74,89.82416773165995, +ant,67,109,2315.857153770824,1000,90.25959750757508, +ant,68,110,288.3915792961347,116,90.09271984792927, +ant,69,111,894.0228631227924,1000,89.89663691508213, +ant,70,112,2030.322535823717,1000,89.84028619017428, +ant,71,113,507.9449555916754,215,90.57267432538549, +ant,72,114,2377.7373967468293,1000,89.84919425782105, +ant,73,115,897.3077114027096,1000,89.90431472264346, +ant,74,116,1454.612590266188,1000,91.19515970740413, +ant,75,117,2292.457960175467,1000,89.8333901030839, +ant,76,118,1424.378337790017,1000,89.88029014661089, +ant,77,119,1441.1111023164538,1000,89.79844243631413, +ant,78,120,1265.4771503717611,1000,89.86009503143968, +ant,79,121,1662.8808067819505,1000,90.5661722205243, +ant,80,122,2508.917122342891,1000,89.87403626041336, +ant,81,123,1655.3510139158748,1000,90.05039760075688, +ant,82,124,1387.3843721247736,821,90.10235730111886, +ant,83,125,646.4356689469432,271,89.7559653760994, +ant,84,126,2172.801064037805,1000,89.84602989356796, +ant,85,127,165.9213897970373,72,89.66756877688618, +ant,86,128,1063.1483912161111,1000,89.77409215132576, +ant,87,129,1000.135342286622,1000,89.81900933661238, +ant,88,130,1977.2359176146426,1000,89.77653862908736, +ant,89,131,1937.1674235355138,1000,90.12773943823525, +ant,90,132,1344.7729257831547,1000,90.39620143170467, +ant,91,133,786.3379828975102,441,89.82893206036925, +ant,92,134,1391.060299752017,1000,89.86676880070257, +ant,93,135,503.300235688713,250,89.94819176115624, +ant,94,136,2446.7482357041768,1000,89.84848658183878, +ant,95,137,1171.9102336514923,1000,89.92785850220504, +ant,96,138,2356.7311711183065,1000,90.42933754946152, +ant,97,139,2356.12199478712,1000,89.82049779117614, +ant,98,140,1389.2987977192308,1000,90.82209581044775, +ant,99,141,967.2335383727841,1000,89.80016695371027, +intercept,0,4242424242,0.7267571190313902,60,99.89614420497905,0.0 +intercept,1,4242424243,2.9096362272975966,60,99.91707940536706,0.0 +intercept,2,4242424244,3.2060351513209753,60,97.78809018716221,0.0 +intercept,3,4242424245,0.7574528902187012,60,98.54894447730877,0.0 +intercept,4,4242424246,0.6827895979695313,60,99.11745353519741,0.0 +intercept,5,4242424247,29.923812823486514,60,99.04064156549293,1.0 +intercept,6,4242424248,0.7661087726592086,60,98.25342313549518,0.0 +intercept,7,4242424249,0.8284444468154106,60,97.98431264506286,0.0 +intercept,8,4242424250,0.9707721562881488,60,99.10671115977826,0.0 +intercept,9,4242424251,1.0944434545235708,60,99.10142489904808,0.0 +intercept,10,4242424252,0.7526731102407211,60,98.24263629181895,0.0 +intercept,11,4242424253,1.0327306617691647,60,99.90610126116793,0.0 +intercept,12,4242424254,24.08529434411321,60,99.04964452767656,1.0 +intercept,13,4242424255,0.8258126199943945,60,99.06333184347895,0.0 +intercept,14,4242424256,0.6465023508935701,60,99.96017435988418,0.0 +intercept,15,4242424257,1.2059930491086561,60,99.07293754243183,0.0 +intercept,16,4242424258,0.8975468523567542,60,99.10975490804557,0.0 +intercept,17,4242424259,0.638558203499997,60,99.9008234011206,0.0 +intercept,18,4242424260,2.473904824233614,60,99.1007534285042,0.0 +intercept,19,4242424261,0.8594156300532632,60,98.2804424689215,0.0 +intercept,20,4242424262,0.7127419076277874,60,97.42140552034121,0.0 +intercept,21,4242424263,1.1195833964738995,60,99.05356389575846,0.0 +intercept,22,4242424264,1.4589147588121705,60,99.10476263429966,0.0 +intercept,23,4242424265,22.348254217096837,60,98.14243140713285,1.0 +intercept,24,4242424266,27.43761277961312,60,99.03063235183511,1.0 +intercept,25,4242424267,0.797963114338927,60,98.95851806063928,0.0 +intercept,26,4242424268,0.6415987604705151,60,99.1202532952496,0.0 +intercept,27,4242424269,1.502438226743834,60,99.8369766656745,0.0 +intercept,28,4242424270,1.277322537265718,60,99.13638822823135,0.0 +intercept,29,4242424271,0.6413188653605175,60,99.8165233572777,0.0 +intercept,30,4242424272,26.015227647672873,60,99.93359984997578,1.0 +intercept,31,4242424273,0.7568511647114065,60,98.29798580223347,0.0 +intercept,32,4242424274,0.7758818510046694,60,95.9119617819155,0.0 +intercept,33,4242424275,0.743274000211386,60,99.16579733811342,0.0 +intercept,34,4242424276,0.9812663898337632,60,99.96913332715677,0.0 +intercept,35,4242424277,0.7364500367548317,60,98.4144170848438,0.0 +intercept,36,4242424278,0.7676261149172205,60,99.87913624991887,0.0 +intercept,37,4242424279,2.6105462690466084,60,99.0124647390605,0.0 +intercept,38,4242424280,0.8922563010128215,60,99.49582641131909,0.0 +intercept,39,4242424281,0.7909053032053635,60,99.95776157301488,0.0 +intercept,40,4242424282,27.747763212013524,60,99.89232705853966,1.0 +intercept,41,4242424283,2.830903574009426,60,99.11182141335861,0.0 +intercept,42,4242424284,3.749473527306691,60,99.89401411987875,0.0 +intercept,43,4242424285,3.2371535471174866,60,98.58941395009701,0.0 +intercept,44,4242424286,1.141169616690604,60,98.95023432158384,0.0 +intercept,45,4242424287,1.2504711685760412,60,99.8783128676535,0.0 +intercept,46,4242424288,1.1401455145678483,60,99.09364640302553,0.0 +intercept,47,4242424289,1.1743367564631626,60,98.24703755640672,0.0 +intercept,48,4242424290,0.6911400489043444,60,98.98847807253395,0.0 +intercept,49,4242424291,0.966755291854497,60,98.31297463384391,0.0 +intercept,50,4242424292,3.7725237559643574,60,98.1307167401627,0.0 +intercept,51,4242424293,0.7292428385990206,60,99.08520847604322,0.0 +intercept,52,4242424294,2.733719722367823,60,99.8949988335446,0.0 +intercept,53,4242424295,2.7277548569836654,60,99.16542541107671,0.0 +intercept,54,4242424296,0.8013565168366767,60,98.2801475641182,0.0 +intercept,55,4242424297,0.9918300381395966,60,98.74883429246843,0.0 +intercept,56,4242424298,3.8384227409260347,60,98.30485570834159,0.0 +intercept,57,4242424299,2.525593837024644,60,99.08211861473346,0.0 +intercept,58,4242424300,1.1939986812940333,60,99.95562586586023,0.0 +intercept,59,4242424301,1.1946645161951892,60,99.11887142756973,0.0 +intercept,60,4242424302,0.6632764584392135,60,99.92733404817194,0.0 +intercept,61,4242424303,0.7345126099826302,60,99.61093201950378,0.0 +intercept,62,4242424304,1.1547945403144695,60,98.88382479344455,0.0 +intercept,63,4242424305,1.0395031699445099,60,99.10455669644611,0.0 +intercept,64,4242424306,2.7713681719324086,60,99.99607387713222,0.0 +intercept,65,4242424307,3.8083399715833366,60,99.90908449191997,0.0 +intercept,66,4242424308,3.1245881704380736,60,99.86388390473627,0.0 +intercept,67,4242424309,0.9936205917911138,60,99.06530098425861,0.0 +intercept,68,4242424310,0.6479002644773573,60,97.31661851374615,0.0 +intercept,69,4242424311,1.09404552471824,60,99.06482130667098,0.0 +intercept,70,4242424312,0.725047086874838,60,99.39714496924636,0.0 +intercept,71,4242424313,2.085218493710272,60,99.91926924929075,0.0 +intercept,72,4242424314,25.112157980707707,60,99.89495984140663,1.0 +intercept,73,4242424315,0.7960666966973804,60,99.91315720008677,0.0 +intercept,74,4242424316,1.8899870013119653,60,99.8635479883608,0.0 +intercept,75,4242424317,24.77215793245705,60,99.08978442272605,1.0 +intercept,76,4242424318,0.763190906640375,60,99.1181927131107,0.0 +intercept,77,4242424319,0.8356004936795216,60,98.87472332915829,0.0 +intercept,78,4242424320,24.543561146681895,60,98.85109478338812,1.0 +intercept,79,4242424321,0.7962639288743958,60,96.64267992061197,0.0 +intercept,80,4242424322,0.6807828926102957,60,98.70551839611774,0.0 +intercept,81,4242424323,1.1704122956143692,60,99.1162675413269,0.0 +intercept,82,4242424324,0.8024117537715938,60,99.10984056283594,0.0 +intercept,83,4242424325,1.0154686415335163,60,99.90653765962175,0.0 +intercept,84,4242424326,0.6267238368745893,60,99.14313411902761,0.0 +intercept,85,4242424327,1.1180786813492887,60,99.87378109642233,0.0 +intercept,86,4242424328,1.0531825890648179,60,99.90759275984404,0.0 +intercept,87,4242424329,0.7319892354425974,60,99.07934787032669,0.0 +intercept,88,4242424330,1.1460731038823724,60,98.32584786308246,0.0 +intercept,89,4242424331,1.145515855285339,60,99.853580446333,0.0 +intercept,90,4242424332,3.1438898412743583,60,98.23875463665809,0.0 +intercept,91,4242424333,1.1678254807484336,60,99.94300368083988,0.0 +intercept,92,4242424334,1.1468605129048228,60,98.28856657896678,0.0 +intercept,93,4242424335,2.816772125195712,60,99.10609232867152,0.0 +intercept,94,4242424336,1.1577836629003286,60,99.93574594730708,0.0 +intercept,95,4242424337,1.0533778404060286,60,99.92123883389186,0.0 +intercept,96,4242424338,0.8533297177054919,60,99.06174133027585,0.0 +intercept,97,4242424339,0.7617567333800253,60,99.95316521359209,0.0 +intercept,98,4242424340,0.9895546428160742,60,99.11385213912092,0.0 +intercept,99,4242424341,0.770722996792756,60,99.44169788411487,0.0 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.csv b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.csv new file mode 100644 index 0000000000000000000000000000000000000000..1b381df6789eea28ea56149c4b780cf0cade01c0 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.csv @@ -0,0 +1,5 @@ +task,episodes,return_mean,return_sd,length_mean,length_sd,success_count,success_rate,invalid_actions,dropped_actions +flappy,100,384.8240045265853,116.78777394316903,3119.31,939.8693174585497,,,0,64 +deadly_corridor,100,1620.7987757873534,913.6242782186637,148.53,49.455930888013825,,,0,0 +ant,100,1453.844063807972,693.7275200567642,803.85,328.8088312378486,,,0,0 +intercept,100,3.5443485127069287,7.07192296411853,60.0,0.0,9,0.09,0,10 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.json new file mode 100644 index 0000000000000000000000000000000000000000..8486070c582599f0cb0c336c70e6569823f4990f --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.json @@ -0,0 +1,205 @@ +{ + "condition": "profile-latency", + "executor_mode": "simulated", + "latency_method": "temporal/profile_sample", + "episodes_per_checkpoint": 100, + "total_episodes": 400, + "checkpoints_metadata_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "results": { + "flappy": { + "n_episodes": 100, + "mean_return": 384.8240045265853, + "std_return": 116.78777394316903, + "min_return": 36.60000045597553, + "max_return": 444.6000052243471, + "mean_length": 3119.31, + "std_length": 939.8693174585497, + "min_length": 314.0, + "max_length": 3600.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "flappy", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "flappy", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T122201421825Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "flappy", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42", + "profile_sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 64, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy.yaml", + "execution_audit": { + "issued_action_records": 311075, + "applied_action_records": 310911, + "dropped_action_records": 64, + "nonnoop_issued_records": 30817, + "finite_action_values": true, + "latency_sample_count": 311075, + "latency_mean_ms": 75.89784633675906, + "latency_std_ms": 3.799946378622932, + "latency_p95_ms": 81.3960393048375, + "latency_p99_ms": 87.23844517488543 + } + }, + "deadly_corridor": { + "n_episodes": 100, + "mean_return": 1620.7987757873534, + "std_return": 913.6242782186637, + "min_return": -76.45918273925781, + "max_return": 2287.240921020508, + "mean_length": 148.53, + "std_length": 49.455930888013825, + "min_length": 17.0, + "max_length": 199.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "doom_deadly_corridor", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "deadly_corridor", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T171446047509Z", + "profile_ref": null, + "env_fps": 35.0, + "obs_fps": 8.75, + "frame_ms": 28.571428571428573, + "latency_type": "profile_sample", + "task": "deadly_corridor", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42", + "profile_sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor.yaml", + "execution_audit": { + "issued_action_records": 3753, + "applied_action_records": 3673, + "dropped_action_records": 0, + "nonnoop_issued_records": 3753, + "finite_action_values": true, + "latency_sample_count": 3753, + "latency_mean_ms": 74.01999621872471, + "latency_std_ms": 5.5537519652567635, + "latency_p95_ms": 89.54825982614612, + "latency_p99_ms": 95.97310052501227 + } + }, + "ant": { + "n_episodes": 100, + "mean_return": 1453.844063807972, + "std_return": 693.7275200567642, + "min_return": 85.64836938561511, + "max_return": 2508.917122342891, + "mean_length": 803.85, + "std_length": 328.8088312378486, + "min_length": 60.0, + "max_length": 1000.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "LatencyBench/AntContinuous-v0", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "ant", + "instance_id": "instance_859cf1e47bca6046", + "source_run_id": "20260911T033037730561Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "ant", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42", + "profile_sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant.yaml", + "execution_audit": { + "issued_action_records": 79573, + "applied_action_records": 79465, + "dropped_action_records": 0, + "nonnoop_issued_records": 79573, + "finite_action_values": true, + "latency_sample_count": 79573, + "latency_mean_ms": 90.00919554158884, + "latency_std_ms": 2.514492574433973, + "latency_p95_ms": 91.11971585797141, + "latency_p99_ms": 102.67108120995428 + } + }, + "intercept": { + "n_episodes": 100, + "mean_return": 3.5443485127069287, + "std_return": 7.07192296411853, + "min_return": 0.6267238368745893, + "max_return": 29.923812823486514, + "mean_length": 60.0, + "std_length": 0.0, + "min_length": 60.0, + "max_length": 60.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "mikasa_intercept_grab_fast", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "mikasa_intercept_grab_fast", + "instance_id": "instance_3a0d42681a03715c", + "source_run_id": "20260909T044501695676Z", + "profile_ref": null, + "env_fps": 20.0, + "obs_fps": 20.0, + "frame_ms": 50.0, + "latency_type": "profile_sample", + "task": "intercept", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0", + "profile_sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 10, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept.yaml", + "success_count": 9, + "success_rate": 0.09, + "execution_audit": { + "issued_action_records": 2974, + "applied_action_records": 2864, + "dropped_action_records": 10, + "nonnoop_issued_records": 2974, + "finite_action_values": true, + "latency_sample_count": 2974, + "latency_mean_ms": 99.11060319379854, + "latency_std_ms": 4.301543980874005, + "latency_p95_ms": 100.2889407458356, + "latency_p99_ms": 100.64616770379737 + } + } + }, + "quality_acceptance": "not inferred; observed statistics only" +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/episodes.csv b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/episodes.csv new file mode 100644 index 0000000000000000000000000000000000000000..b2bcd64fa67fa91907bbb6010397f02aa525e767 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/episodes.csv @@ -0,0 +1,101 @@ +episode_id,seed,return_env,length,mean_latency_ms,invalid_actions,dropped_actions +0,1000000,337.47547912597656,72,71.90727374040254,0,0 +1,1000001,819.0284423828125,143,73.84762082340946,0,0 +2,1000002,2284.857650756836,182,72.79171012339609,0,0 +3,1000003,2276.2068634033203,189,76.345275285376,0,0 +4,1000004,805.2153015136719,150,73.86282581373551,0,0 +5,1000005,621.8231658935547,115,74.31105893586228,0,0 +6,1000006,2276.414749145508,176,74.2226331369995,0,0 +7,1000007,2284.310989379883,176,72.9072057957754,0,0 +8,1000008,81.07798767089844,49,73.20258272646697,0,0 +9,1000009,317.2351837158203,75,72.54774919154028,0,0 +10,1000010,2282.7608489990234,176,72.78293151689127,0,0 +11,1000011,88.11907958984375,45,72.60486105128022,0,0 +12,1000012,2281.468536376953,176,72.29193331603048,0,0 +13,1000013,2276.6868591308594,178,72.73330265771509,0,0 +14,1000014,2276.1705932617188,178,73.30067987408609,0,0 +15,1000015,2282.6631622314453,177,72.49405489224537,0,0 +16,1000016,2280.300033569336,172,72.80884970803692,0,0 +17,1000017,2280.4182891845703,182,73.03539182090206,0,0 +18,1000018,2281.2594451904297,177,72.50972089313564,0,0 +19,1000019,479.8523712158203,99,72.58046231642126,0,0 +20,1000020,2279.7379455566406,181,72.47468246266928,0,0 +21,1000021,2284.9097442626953,197,83.18983231769475,0,0 +22,1000022,2286.2730407714844,172,72.84281562147524,0,0 +23,1000023,244.51919555664062,74,76.23461799191558,0,0 +24,1000024,2279.957275390625,195,72.94927214021655,0,0 +25,1000025,2283.952178955078,179,73.18968843008061,0,0 +26,1000026,2276.701370239258,178,72.87702909462648,0,0 +27,1000027,2277.142562866211,190,72.45412386128042,0,0 +28,1000028,2279.025634765625,177,74.11102172804317,0,0 +29,1000029,2285.7152099609375,177,71.63189230597281,0,0 +30,1000030,53.374298095703125,44,72.51418721312025,0,0 +31,1000031,2279.8080444335938,183,72.72702656843174,0,0 +32,1000032,2282.307357788086,178,74.33584751930213,0,0 +33,1000033,2282.834014892578,192,73.95005063555192,0,0 +34,1000034,2284.200241088867,188,76.29368894499888,0,0 +35,1000035,2287.2159118652344,179,72.81890806090988,0,0 +36,1000036,2284.693832397461,183,76.28284599973325,0,0 +37,1000037,2283.2066650390625,178,72.1797344044525,0,0 +38,1000038,2281.032196044922,178,73.74343783824916,0,0 +39,1000039,2282.960678100586,190,73.24816830891406,0,0 +40,1000040,2287.094253540039,185,72.35711232966574,0,0 +41,1000041,2279.3030853271484,179,72.42125368367608,0,0 +42,1000042,440.0892791748047,104,73.92064892672727,0,0 +43,1000043,2280.8592529296875,177,72.36020918178356,0,0 +44,1000044,2283.4308471679688,189,75.93658060557208,0,0 +45,1000045,2282.324264526367,181,73.54224681770178,0,0 +46,1000046,326.0184631347656,74,73.1983876441008,0,0 +47,1000047,2279.086135864258,182,73.00958120503027,0,0 +48,1000048,2280.3804626464844,179,73.17268244992928,0,0 +49,1000049,2276.215301513672,189,75.47590644230628,0,0 +50,1000050,2278.132034301758,182,74.50495464842548,0,0 +51,1000051,2285.6056518554688,181,73.41699294418743,0,0 +52,1000052,2287.240921020508,173,73.22110809114655,0,0 +53,1000053,310.81517028808594,73,74.09003681120738,0,0 +54,1000054,2276.6219787597656,175,72.98605010243534,0,0 +55,1000055,2276.2769470214844,194,75.17704077845171,0,0 +56,1000056,2278.861602783203,178,72.97353037051572,0,0 +57,1000057,2279.728561401367,181,73.96913002154926,0,0 +58,1000058,2280.544464111328,176,73.02432805290651,0,0 +59,1000059,487.829833984375,108,78.89398217393664,0,0 +60,1000060,567.0655517578125,113,72.64874721482185,0,0 +61,1000061,2278.210220336914,177,72.96068484971086,0,0 +62,1000062,2281.436721801758,186,75.46710866924751,0,0 +63,1000063,382.2119903564453,89,81.21157315209366,0,0 +64,1000064,246.2946014404297,70,73.9736408486285,0,0 +65,1000065,285.21240234375,76,73.13661133681993,0,0 +66,1000066,310.6737365722656,75,73.40468658737086,0,0 +67,1000067,346.1162872314453,75,72.1929723632303,0,0 +68,1000068,804.7056121826172,150,73.76397959753224,0,0 +69,1000069,2285.6442108154297,184,75.13255757158333,0,0 +70,1000070,730.5995788574219,132,73.25446825350764,0,0 +71,1000071,86.91796875,47,76.28335745963689,0,0 +72,1000072,60.30122375488281,44,76.83513093208644,0,0 +73,1000073,768.6264343261719,141,77.27057350071598,0,0 +74,1000074,2280.1071166992188,172,74.16699734355548,0,0 +75,1000075,860.9334106445312,151,73.15118478347584,0,0 +76,1000076,722.9459228515625,143,75.5655785931314,0,0 +77,1000077,2276.8687438964844,182,72.95102474014934,0,0 +78,1000078,368.3357238769531,79,71.51096709276341,0,0 +79,1000079,-76.45918273925781,17,72.24888432102617,0,0 +80,1000080,2281.5543823242188,183,73.32589540463356,0,0 +81,1000081,2281.6688842773438,171,73.10600900440717,0,0 +82,1000082,2277.5223083496094,178,73.55648700566698,0,0 +83,1000083,42.30937194824219,41,73.52700344736942,0,0 +84,1000084,2285.8980407714844,176,71.98655161011203,0,0 +85,1000085,68.90191650390625,45,72.84773487604696,0,0 +86,1000086,2286.2190551757812,171,72.82303966497733,0,0 +87,1000087,281.1173553466797,76,72.26983276661764,0,0 +88,1000088,2283.1607971191406,175,73.49638264342678,0,0 +89,1000089,2277.888946533203,177,73.44736473371472,0,0 +90,1000090,429.36326599121094,93,71.86172378947977,0,0 +91,1000091,252.0751953125,70,72.26459581736903,0,0 +92,1000092,2278.306442260742,192,80.97328482778371,0,0 +93,1000093,2285.236801147461,175,74.02717585214627,0,0 +94,1000094,857.2727355957031,152,85.59110000526613,0,0 +95,1000095,2275.9288024902344,199,73.62958803645523,0,0 +96,1000096,2286.8704833984375,179,72.31519682456816,0,0 +97,1000097,2278.048355102539,181,73.50330330803081,0,0 +98,1000098,2277.4480743408203,178,76.78472725777,0,0 +99,1000099,2276.9671478271484,178,77.9678189026336,0,0 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/eval_config.yaml b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/eval_config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7833b316cd581921c857de1c220fca6247c0193a --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/eval_config.yaml @@ -0,0 +1,164 @@ +experiment: + name: deadly_corridor-mean5000-profile-simulation-100ep + seed: 1000000 +backend: + type: sample_factory + algo: APPO + device: cpu + train_dir: results/sample_factory + restart_behavior: resume + run_mode: eval +executor: + mode: simulated + simulated_worker_capacity: 1 + simulated_inference_pool: true + inference_devices: + - cuda:0 + inference_batch_size: 32 +env: + name: deadly_corridor + env_id: doom_deadly_corridor + env_fps: 35 + obs_fps: 8.75 + noop_action: + - 0 + - 0 + - 0 + - 0 + frame_stack: 1 + res_w: 128 + res_h: 72 + wide_aspect_ratio: false + simulator: cpu + obs_resize: + - 224 + - 224 + action_map: + noop: + - 0 + - 0 + - 0 + - 0 + move_forward: + - 0 + - 1 + - 0 + - 0 + move_backward: + - 0 + - 2 + - 0 + - 0 + move_left: + - 0 + - 0 + - 1 + - 0 + move_right: + - 0 + - 0 + - 2 + - 0 + turn_left: + - 1 + - 0 + - 0 + - 0 + turn_right: + - 2 + - 0 + - 0 + - 0 + attack: + - 0 + - 0 + - 0 + - 1 + action_history_decisions: 8 + screen_resolution: RES_160X120 + render_hud: true + render_crosshair: false + render_weapon: true + render_decals: false + render_particles: false +latency: + method: temporal + profile_path: /home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/deadly_corridor/instance_a5037b165aa0cedc/profile.json + profile_worker_slot: 0 + seed: 271828 + add_latency_info: false +scheduler: + hold_policy: hold + ordering_policy: latest_ready +policy: + type: starvla + actions: + - MOVE_FORWARD + - MOVE_BACKWARD + - MOVE_LEFT + - MOVE_RIGHT + - TURN_LEFT + - TURN_RIGHT + - ATTACK + checkpoint_path: /home/ubuntu/lzj/mean-profiling/deadly_corridor/vla-publication/checkpoints/model.pt + model_config_path: /home/ubuntu/lzj/mean-profiling/deadly_corridor/vla-publication/config.full.yaml + device: cuda:0 + unnorm_key: new_embodiment + prompt_mode: latency_neutral + action_layout: multibinary_7 + state_source: transport + backbone_path: /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct + worker_python_executable: /home/ubuntu/lzj/conda/envs/qwenoft/bin/python +training: + train_for_env_steps: 25000000 + num_workers: 32 + num_envs_per_worker: 4 + worker_num_splits: 2 + num_policies: 1 + batch_size: 1024 + rollout: 128 + recurrence: 128 + num_epochs: 2 + num_batches_per_epoch: 2 + learning_rate: 0.0001 + gamma: 0.99 + gae_lambda: 0.95 + ppo_clip_ratio: 0.1 + ppo_clip_value: 0.2 + exploration_loss: symmetric_kl + exploration_loss_coeff: 0.001 + value_loss_coeff: 0.5 + max_grad_norm: 4.0 + async_rl: true + use_rnn: true + rnn_type: gru + rnn_size: 512 + normalize_input: true + normalize_returns: true + stats_avg: 100 + experiment_summaries_interval: 1 + save_every_sec: 600 + keep_checkpoints: 5 +evaluation: + eval_interval_steps: 1000000 + eval_episodes: 100 + eval_parallel_envs: 32 + eval_max_steps: 3600 + eval_deterministic: true + eval_raw_reward: true + eval_suites: + fixed: [] + normal: [] + uniform: [] + eval_latency_values: null +logging: + output_dir: /home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor + video: + enabled: false + save_step_records: true + save_action_records: true + save_latency_records: true + wandb_project: null + wandb_group: null + wandb_job_type: null + wandb_tags: '' diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py new file mode 100644 index 0000000000000000000000000000000000000000..d5edbe901e40e8e9cbb2e0763281fdfcd77dbfcc --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py @@ -0,0 +1,702 @@ +from __future__ import annotations + +import time +from collections.abc import Callable, Mapping, Sequence +from dataclasses import dataclass, field +from pathlib import Path + +import numpy as np + +from latency_bench.core.clock import EnvClock +from latency_bench.core.decision_action_history import DecisionActionHistory +from latency_bench.core.timing import StageProfiler, profiler_scope +from latency_bench.core.types import ActionEvent, EpisodeMetrics, LatencyRecord, Observation, StepRecord +from latency_bench.envs.atari import TRUE_EPISODE_END_INFO_KEY +from latency_bench.envs.base import EnvAdapter +from latency_bench.executors._simulated_timeline import ( + SimulatedResultTimeline, + SimulatedWorkerCapacity, + build_simulated_action_event, +) +from latency_bench.executors.base import BatchedExecutor +from latency_bench.executors.env_step_backend import EnvStepBackend, env_action_space +from latency_bench.latency.sample import LatencySample +from latency_bench.latency.samplers import LatencySampler +from latency_bench.logging.metrics import ( + compute_episode_metrics, + compute_episode_metrics_from_aggregates, + episode_raw_fact_metadata, + latency_type_from_source, + profile_metadata_from_source, +) +from latency_bench.logging.records import build_step_record +from latency_bench.logging.trajectory_logger import TrajectoryLogger +from latency_bench.policy.action_prefix import with_action_prefix +from latency_bench.policy.base import PolicyRunner +from latency_bench.scheduler.action_queue import ActionScheduler +from latency_bench.scheduler.decision import DecisionScheduler +from latency_bench.utils.io import write_json +from latency_bench.utils.stats import series_stats + + +@dataclass +class _EpisodeBuffers: + step_records: list[StepRecord] | None = None + action_events: list[ActionEvent] | None = None + latency_records: list[LatencyRecord] | None = None + latency_values_ms: list[float] = field(default_factory=list) + episode_return_env: float = 0.0 + survival_steps: int = 0 + game_score: float | None = None + return_raw: float | None = None + num_actions: int = 0 + num_dropped_actions: int = 0 + num_invalid_actions: int = 0 + submitted_observation_frames: int = 0 + dropped_observation_count: int = 0 + soft_reset_count: int = 0 + final_lives: int | None = None + final_is_true_episode_end: bool | None = None + task_metrics: dict | None = None + task_metric_moments: dict | None = None + + def record_step(self, *, reward: float, info: dict) -> None: + self.episode_return_env += float(reward) + self.survival_steps += 1 + if "invalid_action" in info and info["invalid_action"]: + self.num_invalid_actions += 1 + if "soft_reset" in info and info["soft_reset"]: + self.soft_reset_count += 1 + if "lives" in info: + self.final_lives = info["lives"] + if TRUE_EPISODE_END_INFO_KEY in info: + self.final_is_true_episode_end = info[TRUE_EPISODE_END_INFO_KEY] + if "game_score" in info: + self.game_score = float(info["game_score"]) + if "score" in info: + self.game_score = float(info["score"]) + if "task_metrics" in info: + self.task_metrics = info["task_metrics"] + if "task_metric_moments" in info: + self.task_metric_moments = info["task_metric_moments"] + self._update_return_raw(info) + extra_stats = info["episode_extra_stats"] if "episode_extra_stats" in info else None + if isinstance(extra_stats, dict): + self._update_return_raw(extra_stats) + + def _update_return_raw(self, stats: dict) -> None: + for key in ("return_raw", "raw_return", "episodic_raw_return", "episode/raw_return"): + if key in stats and stats[key] is not None: + self.return_raw = float(stats[key]) + + +@dataclass +class _SlotState: + slot_id: int + env: EnvAdapter + latency_source: LatencySampler + action_scheduler: ActionScheduler + result_timeline: SimulatedResultTimeline + active: bool = False + episode_id: int | None = None + episode_seed: int | None = None + env_step: int = 0 + recent_drop_count: int = 0 + decision_action_history: DecisionActionHistory | None = None + decision_admitted: bool = False + decision_issued_action: object = None + buffers: _EpisodeBuffers = field(default_factory=_EpisodeBuffers) + worker_capacity: SimulatedWorkerCapacity = field( + default_factory=lambda: SimulatedWorkerCapacity(capacity=None, busy_until_by_worker={}) + ) + + +@dataclass +class _PendingPolicyObservation: + slot: _SlotState + observation: Observation + obs_id: int + latency_sample: LatencySample + worker_slot: int + + +class BatchedSimulatedLatencyExecutor(BatchedExecutor): + """Run multiple simulated episodes concurrently with independent slot state. + + The main process owns policy inference, latency scheduling, episode accounting, + and logging. Env stepping can be serial in-process or delegated to worker + subprocesses through env_backend. + """ + + def __init__( + self, + *, + env_backend: EnvStepBackend, + policy: PolicyRunner, + decision_scheduler: DecisionScheduler, + latency_sources: Sequence[LatencySampler], + action_schedulers: Sequence[ActionScheduler], + clock: EnvClock, + logger: TrajectoryLogger | None = None, + episode_latency_source_factory: Callable[[int], LatencySampler] | None = None, + simulated_worker_capacity: int | None = None, + profile_pipeline: bool = False, + inference_pool=None, + action_prefix=None, + action_history_decisions: int | None = None, + ): + slot_count = env_backend.num_slots + self.env_backend = env_backend + self.envs = list(env_backend.slot_handles) + self.policy = policy + self.decision_scheduler = decision_scheduler + self.clock = clock + self.logger = logger + self.profile_pipeline = bool(profile_pipeline) + self.inference_pool = inference_pool + self.action_prefix = action_prefix + self._pipeline_profile_rows: list[dict[str, float]] = [] + self.simulated_worker_capacity = simulated_worker_capacity + self._collect_step_records = bool(logger is not None and logger.save_step_records) + self._collect_action_records = bool(logger is not None and logger.save_action_records) + self._collect_latency_records = bool(logger is not None and logger.save_latency_records) + self.episode_latency_source_factory = episode_latency_source_factory + self.slots = [ + _SlotState( + slot_id=slot_id, + env=self.envs[slot_id], + latency_source=latency_sources[slot_id], + action_scheduler=action_schedulers[slot_id], + result_timeline=SimulatedResultTimeline( + ordering_policy=action_schedulers[slot_id].ordering_policy + ), + decision_action_history=( + DecisionActionHistory( + env_action_space(self.envs[slot_id]), num_envs=1, decisions=action_history_decisions + ) if action_history_decisions is not None else None + ), + buffers=self._new_episode_buffers(), + worker_capacity=SimulatedWorkerCapacity( + capacity=simulated_worker_capacity, + busy_until_by_worker={}, + ), + ) + for slot_id in range(slot_count) + ] + self._next_obs_id = 0 + self._next_action_id = 0 + self.started_episodes = 0 + self.completed_episodes = 0 + self._completed_metrics: dict[int, EpisodeMetrics] = {} + self._completed_buffers: dict[int, _EpisodeBuffers] = {} + self._episode_log_order: list[int] = [] + self._next_episode_log_index = 0 + + @property + def num_slots(self) -> int: + return len(self.slots) + + def close(self) -> None: + if self.inference_pool is not None: + self.inference_pool.close() + self.env_backend.close() + + def run_episodes( + self, + *, + episode_ids: Sequence[int], + seeds: Sequence[int | None], + eval_max_steps: int = 10000, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + ) -> list[EpisodeMetrics]: + if eval_max_steps < 0: + raise ValueError("eval_max_steps must be non-negative") + episode_ids = [int(episode_id) for episode_id in episode_ids] + if len(seeds) != len(episode_ids): + raise ValueError("seeds length must match episode_ids length") + + self._reset_run_state(episode_ids) + if not episode_ids: + return [] + + next_episode_index = 0 + initial_slots = min(self.num_slots, len(episode_ids)) + for slot in self.slots[:initial_slots]: + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + + while self.completed_episodes < len(episode_ids): + active_slots = self._active_slots() + if eval_max_steps == 0: + for slot in active_slots: + self._complete_slot(slot, on_episode_complete=on_episode_complete) + if next_episode_index < len(episode_ids): + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + continue + + observations = [] + observation_slots: list[_SlotState] = [] + step_capacity_info: dict[int, dict[str, int | bool | None]] = {} + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + slot.worker_capacity.release_ready(slot.env_step) + self._deliver_arrived_results(slot, raw_frame=slot.env_step) + observation_submitted = False + observation_dropped = False + if self.decision_scheduler.should_observe(slot.env_step, current_time_ms): + prefix_request_pending = ( + self.action_prefix is not None + and self.action_prefix["mode"] != "none" + and slot.result_timeline.pending_observation_count > 0 + ) + if slot.worker_capacity.can_submit() and not prefix_request_pending: + observation_slots.append(slot) + observation_submitted = True + else: + slot.buffers.dropped_observation_count += 1 + observation_dropped = True + slot.recent_drop_count += 1 + if self.simulated_worker_capacity is not None: + step_capacity_info[slot.slot_id] = { + "observation_submitted": observation_submitted, + "observation_dropped": observation_dropped, + } + if slot.decision_action_history is not None and slot.env_step % self.clock.obs_stride_raw_frames == 0: + slot.decision_admitted = observation_submitted + slot.decision_issued_action = slot.action_scheduler.noop_action.value + + observe_ms = 0.0 + if observation_slots: + observe_start = time.perf_counter() + observations_by_slot = self.env_backend.observe_slots([slot.slot_id for slot in observation_slots]) + observe_ms = (time.perf_counter() - observe_start) * 1000.0 + pending_observations = [ + self._sample_policy_observation( + slot, + self._policy_observation( + slot, + observations_by_slot[slot.slot_id], + transport=( + slot.decision_action_history.observation()[0] + if slot.decision_action_history is not None else None + ), + ), + ) + for slot in observation_slots + ] + observations = [pending.observation for pending in pending_observations] + + profile_row = None + if observations: + profiler = StageProfiler(enabled=self.profile_pipeline) + with profiler_scope(profiler): + policy_outputs = ( + self.inference_pool.predict_batch(observations) + if self.inference_pool is not None + else self.policy.predict_batch(observations) + ) + if len(policy_outputs) != len(observations): + raise RuntimeError("policy.predict_batch returned the wrong number of outputs") + if self.profile_pipeline: + profile_row = { + "active_slots": float(len(active_slots)), + "batch_size": float(len(observations)), + "observe_slots_ms": observe_ms, + **{key: float(value) for key, value in profiler.timings.items()}, + } + for pending, policy_output in zip(pending_observations, policy_outputs): + if pending.slot.decision_action_history is not None: + pending.slot.decision_issued_action = policy_output.action.value + self._enqueue_policy_output( + pending.slot, + pending.observation, + policy_output, + obs_id=pending.obs_id, + latency_sample=pending.latency_sample, + worker_slot=pending.worker_slot, + ) + + actions_by_slot = {} + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + self._deliver_arrived_results(slot, raw_frame=slot.env_step) + active_action = slot.action_scheduler.update(slot.env_step, current_time_ms) + actions_by_slot[slot.slot_id] = active_action + + env_step_start = time.perf_counter() + step_responses = self.env_backend.step_slots(actions_by_slot) + if profile_row is not None: + profile_row["env_step_ms"] = (time.perf_counter() - env_step_start) * 1000.0 + self._pipeline_profile_rows.append(profile_row) + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + active_action = actions_by_slot[slot.slot_id] + if ( + slot.decision_action_history is not None + and (slot.env_step + 1) % self.clock.obs_stride_raw_frames == 0 + ): + slot.decision_action_history.append( + [0], [slot.decision_admitted], + [slot.decision_issued_action], [active_action.value], + ) + result = step_responses[slot.slot_id].result + soft_reset = bool(result.info.get("soft_reset")) if isinstance(result.info, dict) else False + episode_done = bool(result.done or result.truncated) and not soft_reset + if slot.buffers.step_records is not None: + record = build_step_record( + episode_id=int(slot.episode_id), + env_step=slot.env_step, + scheduled_time_ms=current_time_ms, + active_action=active_action, + reward=result.reward, + done=episode_done, + info=result.info, + active_event=slot.action_scheduler.latest_applied_event, + frame_ms=self.clock.frame_ms, + latency_type=latency_type_from_source(slot.latency_source), + ) + slot.buffers.step_records.append(record) + slot.buffers.record_step(reward=float(result.reward), info=record.info) + else: + slot.buffers.record_step(reward=float(result.reward), info=result.info) + if self.simulated_worker_capacity is not None and slot.buffers.step_records is not None: + slot.buffers.step_records[-1].info.update( + { + **step_capacity_info[slot.slot_id], + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + ) + if soft_reset: + slot.buffers.num_dropped_actions += slot.action_scheduler.dropped_events_count + slot.action_scheduler.reset() + slot.result_timeline.reset() + self._reset_policy_state(slot.slot_id) + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + slot.recent_drop_count = 0 + + slot.env_step += 1 + if episode_done or slot.env_step >= eval_max_steps: + self._complete_slot(slot, on_episode_complete=on_episode_complete) + if next_episode_index < len(episode_ids): + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + + self._write_pipeline_profile_summary() + return self._ordered_metrics(episode_ids) + + def _policy_observation( + self, + slot: _SlotState, + observation: Observation, + transport: np.ndarray | None = None, + ) -> Observation: + observation = with_action_prefix(observation, slot.action_scheduler, self.action_prefix) + metadata = dict(observation.metadata) + metadata["slot_id"] = slot.slot_id + metadata["episode_id"] = int(slot.episode_id) + metadata["action_noise_seed"] = slot.episode_seed + data = observation.data + if transport is not None: + data = {**data, "transport": transport} if isinstance(data, Mapping) else {"obs": data, "transport": transport} + return Observation( + data=data, + env_step=observation.env_step, + sim_time_ms=observation.sim_time_ms, + metadata=metadata, + ) + + def _sample_policy_observation( + self, + slot: _SlotState, + observation: Observation, + ) -> _PendingPolicyObservation: + obs_id = self._next_obs_id + self._next_obs_id += 1 + raw_frame = int(slot.env_step) + current_time_ms = self.clock.step_to_time_ms(raw_frame) + worker_slot = slot.worker_capacity.assign_worker() + latency_context = { + "observation": observation, + "obs_id": obs_id, + "env_step": raw_frame, + "raw_frame": raw_frame, + "sim_time_ms": current_time_ms, + "episode_id": slot.episode_id, + "slot_id": slot.slot_id, + "worker_slot": worker_slot, + "recent_drop_count": slot.recent_drop_count, + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + latency_sample = slot.latency_source.sample(latency_context) + metadata = dict(observation.metadata) + metadata["obs_id"] = obs_id + policy_observation = Observation( + data=observation.data, + env_step=observation.env_step, + sim_time_ms=observation.sim_time_ms, + metadata=metadata, + ) + slot.worker_capacity.submit( + worker_slot, raw_frame + latency_sample.worker_service_raw_frames + ) + return _PendingPolicyObservation( + slot=slot, + obs_id=obs_id, + latency_sample=latency_sample, + worker_slot=worker_slot, + observation=policy_observation, + ) + + def _reset_run_state(self, episode_ids: Sequence[int]) -> None: + self.started_episodes = 0 + self.completed_episodes = 0 + self._pipeline_profile_rows.clear() + self._completed_metrics.clear() + self._completed_buffers.clear() + self._episode_log_order = [int(episode_id) for episode_id in episode_ids] + self._next_episode_log_index = 0 + for slot in self.slots: + slot.active = False + slot.episode_id = None + slot.episode_seed = None + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + + def _active_slots(self) -> list[_SlotState]: + return [slot for slot in self.slots if slot.active] + + def _deliver_arrived_results(self, slot: _SlotState, *, raw_frame: int | None) -> None: + released, dropped = slot.result_timeline.release_arrived(raw_frame) + slot.buffers.num_dropped_actions += len(dropped) + for event in released: + slot.action_scheduler.enqueue(event) + + def _start_slot(self, slot: _SlotState, *, episode_id: int, seed: int | None) -> None: + if self.episode_latency_source_factory is not None: + slot.latency_source = self.episode_latency_source_factory(episode_id) + slot.active = True + slot.episode_id = int(episode_id) + slot.episode_seed = None if seed is None else int(seed) + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + self._reset_policy_state(slot.slot_id) + self.env_backend.reset_slot(slot.slot_id, episode_id=episode_id, seed=seed) + self.started_episodes += 1 + + def _reset_policy_state(self, slot_id: int) -> None: + if self.inference_pool is not None: + self.inference_pool.reset_state(slot_id) + else: + self.policy.reset_state(slot_id=slot_id) + + def _complete_slot( + self, + slot: _SlotState, + *, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + ) -> None: + if not slot.active or slot.episode_id is None: + return + episode_id = int(slot.episode_id) + self._deliver_arrived_results(slot, raw_frame=None) + slot.buffers.num_dropped_actions += slot.action_scheduler.dropped_events_count + metrics = self._compute_episode_metrics( + episode_id=episode_id, + buffers=slot.buffers, + metadata=episode_raw_fact_metadata( + mode="simulated", + episode_seed=slot.episode_seed, + env_fps=self.clock.env_fps, + obs_fps=self.clock.obs_fps, + frame_ms=self.clock.frame_ms, + latency_type=latency_type_from_source(slot.latency_source), + latency_source=slot.latency_source, + ) + | slot.action_scheduler.chunk_metrics() + | ( + { + "submitted_observation_frames": slot.buffers.submitted_observation_frames, + "dropped_observation_count": slot.buffers.dropped_observation_count, + "simulated_worker_capacity": self.simulated_worker_capacity, + "inference_worker_count": self.simulated_worker_capacity, + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + if self.simulated_worker_capacity is not None + else {} + ), + ) + self._completed_metrics[episode_id] = metrics + self._completed_buffers[episode_id] = slot.buffers + self.completed_episodes += 1 + slot.active = False + slot.episode_id = None + slot.episode_seed = None + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + self._flush_completed_in_episode_order() + if on_episode_complete is not None: + on_episode_complete(metrics) + + def _enqueue_policy_output( + self, + slot: _SlotState, + observation, + policy_output, + *, + obs_id: int, + latency_sample: LatencySample, + worker_slot: int, + ) -> None: + raw_frame = int(slot.env_step) + latency_ms = latency_sample.latency_ms + ready_raw_frame = raw_frame + latency_sample.action_ready_raw_frames + ready_time_ms = self.clock.step_to_time_ms(ready_raw_frame) + latency_type = latency_type_from_source(slot.latency_source) + profile_metadata = profile_metadata_from_source(slot.latency_source) + slot_metadata = { + "episode_id": int(slot.episode_id), + "slot_id": int(slot.slot_id), + "worker_id": int(worker_slot), + } + latency_record, event = build_simulated_action_event( + action_id=self._next_action_id, + obs_id=obs_id, + policy_output=policy_output, + raw_frame=raw_frame, + ready_raw_frame=ready_raw_frame, + ready_time_ms=ready_time_ms, + latency_sample=latency_sample, + frame_ms=self.clock.frame_ms, + latency_type=latency_type, + profile_metadata=profile_metadata, + latency_record_metadata=slot_metadata, + extra_event_metadata=slot_metadata, + ) + self._next_action_id += 1 + slot.result_timeline.submit(obs_id=obs_id, ready_raw_frame=ready_raw_frame, event=event) + slot.buffers.submitted_observation_frames += 1 + slot.recent_drop_count = 0 + slot.buffers.num_actions += 1 + if slot.buffers.action_events is not None: + slot.buffers.action_events.append(event) + slot.buffers.latency_values_ms.append(latency_ms) + if slot.buffers.latency_records is not None: + slot.buffers.latency_records.append(latency_record) + + def _flush_completed_in_episode_order(self) -> None: + if self.logger is None: + return + while self._next_episode_log_index < len(self._episode_log_order): + episode_id = self._episode_log_order[self._next_episode_log_index] + if episode_id not in self._completed_metrics: + break + buffers = self._completed_buffers[episode_id] + metrics = self._completed_metrics[episode_id] + if buffers.step_records is not None: + for record in buffers.step_records: + self.logger.log_step(record) + if buffers.action_events is not None: + for event in buffers.action_events: + self.logger.log_action_event(event) + if buffers.latency_records is not None: + for latency_record in buffers.latency_records: + self.logger.log_latency(latency_record) + self.logger.log_episode_metrics(metrics) + self._next_episode_log_index += 1 + + def _ordered_metrics(self, episode_ids: Sequence[int]) -> list[EpisodeMetrics]: + return [self._completed_metrics[int(episode_id)] for episode_id in episode_ids] + + def _new_episode_buffers(self) -> _EpisodeBuffers: + return _EpisodeBuffers( + step_records=[] if self._collect_step_records else None, + action_events=[] if self._collect_action_records else None, + latency_records=[] if self._collect_latency_records else None, + ) + + def _write_pipeline_profile_summary(self) -> None: + if not self.profile_pipeline or self.logger is None or not self._pipeline_profile_rows: + return + keys = sorted({key for row in self._pipeline_profile_rows for key in row}) + summary = { + "num_profiled_batches": len(self._pipeline_profile_rows), + **{ + key: series_stats([float(row[key]) for row in self._pipeline_profile_rows if key in row]) + for key in keys + }, + } + write_json(Path(self.logger.output_dir) / "simulated_pipeline_summary.json", summary) + + def _compute_episode_metrics( + self, + *, + episode_id: int, + buffers: _EpisodeBuffers, + metadata: dict, + ) -> EpisodeMetrics: + if buffers.task_metrics is not None: + metadata["task_metrics"] = buffers.task_metrics + if buffers.task_metric_moments is not None: + metadata["task_metric_moments"] = buffers.task_metric_moments + if buffers.final_lives is not None: + metadata["final_lives"] = buffers.final_lives + if buffers.final_is_true_episode_end is not None: + metadata["final_is_true_episode_end"] = buffers.final_is_true_episode_end + metadata["soft_reset_count"] = buffers.soft_reset_count + if buffers.step_records is not None and buffers.action_events is not None: + return compute_episode_metrics( + episode_id=episode_id, + step_records=buffers.step_records, + action_events=buffers.action_events, + latency_values_ms=buffers.latency_values_ms, + metadata=metadata, + frame_ms=self.clock.frame_ms, + ) + return compute_episode_metrics_from_aggregates( + episode_id=episode_id, + episode_return_env=buffers.episode_return_env, + survival_steps=buffers.survival_steps, + return_raw=buffers.return_raw, + game_score=buffers.game_score, + latency_values_ms=buffers.latency_values_ms, + num_actions=buffers.num_actions, + num_dropped_actions=buffers.num_dropped_actions, + num_invalid_actions=buffers.num_invalid_actions, + metadata=metadata, + ) diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch new file mode 100644 index 0000000000000000000000000000000000000000..bdeca64d184242eedaa57463d2e0a242f43e19ba --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch @@ -0,0 +1,99 @@ +diff --git a/latency_bench/envs/deadly_corridor.py b/latency_bench/envs/deadly_corridor.py +index 4dcaa48c..dc4d1186 100644 +--- a/latency_bench/envs/deadly_corridor.py ++++ b/latency_bench/envs/deadly_corridor.py +@@ -5,7 +5,7 @@ from collections import deque + from typing import Any + + import numpy as np +-from gymnasium.spaces import Box, Tuple ++from gymnasium.spaces import Box, MultiBinary, Tuple + + from latency_bench.core.types import Action, Observation, StepResult + from latency_bench.envs.base import EnvAdapter +@@ -346,6 +346,7 @@ class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + export_env_raw_rgb_frames: bool = True, + ): + import gymnasium as gym ++ import vizdoom + import vizdoom.gymnasium_wrapper # noqa: F401 (registers the env ids) + + env_cfg = config["env"] +@@ -360,30 +361,27 @@ class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + ) + if key in env_cfg + } +- attempts = [ +- ("VizdoomDeadlyCorridor-MultiBinary-v1", {}), +- ("VizdoomDeadlyCorridor-MultiBinary-v0", {}), +- ("VizdoomDeadlyCorridor-v1", {"max_buttons_pressed": 0}), +- ("VizdoomDeadlyCorridor-v0", {"max_buttons_pressed": 0}), +- ] +- last_exc: Exception | None = None +- self.gym_env = None +- for env_id, kwargs in attempts: +- try: +- # frame_skip=1: the latency_bench scheduler advances obs_stride raw +- # frames per decision and holds the action between observations. +- self.gym_env = gym.make( +- env_id, render_mode="rgb_array", frame_skip=1, **render_options, **kwargs +- ) +- self.env_id = env_id +- break +- except (gym.error.NameNotFound, gym.error.VersionNotFound, gym.error.NamespaceNotFound) as exc: +- last_exc = exc +- if self.gym_env is None: +- raise RuntimeError(f"Failed to create Deadly Corridor MultiBinary env: {last_exc}") ++ # ViZDoom registers deadly_corridor.cfg under this official Gym ID. ++ self.env_id = "VizdoomCorridor-v0" ++ self.gym_env = gym.make( ++ self.env_id, render_mode="rgb_array", frame_skip=1, max_buttons_pressed=0, ++ ) ++ game = self.gym_env.unwrapped.game ++ game.close() ++ for key, value in render_options.items(): ++ if key == "screen_resolution": ++ value = getattr(vizdoom.ScreenResolution, value) ++ getattr(game, f"set_{key}")(value) ++ game.init() ++ self.gym_env.unwrapped.observation_space.spaces["screen"] = Box( ++ 0, 255, ++ shape=(game.get_screen_height(), game.get_screen_width(), game.get_screen_channels()), ++ dtype=np.uint8, ++ ) + + self._runtime_button_order = _deadly_runtime_button_names(self.gym_env) + self._num_buttons = len(self._runtime_button_order) ++ self.gym_env.unwrapped.action_space = MultiBinary(self._num_buttons) + self.noop_action = noop_action or Action( + value=[0] * self._num_buttons, name="NOOP", is_noop=True + ) +diff --git a/tests/integration/test_deadly_render_contract.py b/tests/integration/test_deadly_render_contract.py +index 535db22a..09894b1b 100644 +--- a/tests/integration/test_deadly_render_contract.py ++++ b/tests/integration/test_deadly_render_contract.py +@@ -5,11 +5,13 @@ import json + import numpy as np + import pytest + +-pytest.importorskip("vizdoom", minversion="1.3.0") ++pytest.importorskip("vizdoom", minversion="1.2.4") + pytest.importorskip("sample_factory") + + from latency_bench.envs.deadly_corridor import DeadlyCorridorEnvAdapter, DeadlyCorridorVlaEnvAdapter + from latency_bench.envs.raw_rgb import ENV_RAW_RGB_FRAME_STACK_INFO_KEY ++from latency_bench.core.types import Action ++from gymnasium.spaces import MultiBinary + from scripts.tasks.decision_history.eval_vla_hist8 import evaluation_config + + +@@ -33,6 +35,9 @@ def test_hist8_deadly_vla_uses_the_teacher_resolution_and_hud(tmp_path): + # The health/ammo panel is stable across the two engine reset paths; + # the animated face and enemies can differ with their RNG streams. + np.testing.assert_array_equal(teacher_frame[-20:, :64], student_frame[-20:, :64]) ++ assert isinstance(student.gym_env.action_space, MultiBinary) ++ step = student.step(Action(value=[1, 0, 0, 0, 0, 0, 1], name="forward_attack")) ++ assert np.isfinite(step.reward) + finally: + teacher.close() + student.close() diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py new file mode 100644 index 0000000000000000000000000000000000000000..1016be20dca9e949c752192edb407b9a84e30e35 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py @@ -0,0 +1,455 @@ +from __future__ import annotations + +import copy +from collections import deque +from typing import Any + +import numpy as np +from gymnasium.spaces import Box, MultiBinary, Tuple + +from latency_bench.core.types import Action, Observation, StepResult +from latency_bench.envs.base import EnvAdapter +from latency_bench.utils.array import looks_chw +from latency_bench.envs.raw_rgb import RawRgbFrameStackBuffer + + +def _noop_action_from_space(space) -> Any: + n = getattr(space, "n", None) + if n is not None: + return 0 + if isinstance(space, Tuple): + return tuple(_noop_action_from_space(subspace) for subspace in space.spaces) + if isinstance(space, Box): + import numpy as np + + return np.zeros(space.shape, dtype=space.dtype) + raise TypeError(f"Unsupported action space for Deadly Corridor no-op action: {space}") + + +def _coerce_noop_action_for_space(value: Any, space) -> Any: + if isinstance(space, Tuple): + if isinstance(value, (list, tuple)): + if len(value) != len(space.spaces): + raise ValueError( + f"Deadly Corridor no-op action length {len(value)} does not match action space {space}" + ) + return tuple( + _coerce_noop_action_for_space(item, subspace) + for item, subspace in zip(value, space.spaces) + ) + if value == 0: + return _noop_action_from_space(space) + return value + + +def _spec_with_reward_scaling(spec: Any, disable_reward_scaling: bool) -> Any: + if not disable_reward_scaling: + return spec + spec_to_use = copy.copy(spec) + spec_to_use.reward_scaling = 1.0 + return spec_to_use + + +def _synchronous_eval_fps_from_config(config: dict[str, Any], default: int = 35) -> int: + env_cfg = config.get("env", {}) + try: + fps = int(float(env_cfg.get("env_fps", default))) + except (TypeError, ValueError) as exc: + raise ValueError("env_fps must be positive") from exc + if fps <= 0: + raise ValueError("env_fps must be positive") + return fps + + +def _build_sample_factory_eval_cfg(config: dict[str, Any]) -> Any: + from training.deadly_corridor_sf import integration + from training.common.utils import maybe_set_cli_override + + integration.register_deadly_corridor_components() + base_cfg = integration.SAMPLE_FACTORY_CONFIG_PARSER.parse_eval( + integration.build_cli_args_from_config(config) + ) + eval_fps = _synchronous_eval_fps_from_config(config) + cfg = copy.deepcopy(base_cfg) + if _requires_sample_factory_checkpoint_config(config): + from sample_factory.cfg.arguments import load_from_checkpoint + + cfg = load_from_checkpoint(cfg) + + for key in ( + "seed", + "res_w", + "res_h", + "wide_aspect_ratio", + ): + if hasattr(base_cfg, key): + maybe_set_cli_override(cfg, key, getattr(base_cfg, key)) + maybe_set_cli_override(cfg, "frame_stack", 1) + explicit_max_episode_steps = int(getattr(base_cfg, "max_episode_steps", 0) or 0) + if explicit_max_episode_steps > 0: + maybe_set_cli_override(cfg, "max_episode_steps", explicit_max_episode_steps) + else: + eval_max_steps = int(getattr(base_cfg, "eval_max_steps", 0) or 0) + if eval_max_steps > 0: + maybe_set_cli_override(cfg, "max_episode_steps", eval_max_steps) + + maybe_set_cli_override(cfg, "mode", "eval") + maybe_set_cli_override(cfg, "latency_type", "zero") + maybe_set_cli_override(cfg, "fixed_latency_ms", 0.0) + maybe_set_cli_override(cfg, "env_frameskip", 1) + maybe_set_cli_override(cfg, "eval_env_frameskip", 1) + maybe_set_cli_override(cfg, "num_envs", 1) + maybe_set_cli_override(cfg, "no_render", True) + maybe_set_cli_override(cfg, "save_video", False) + maybe_set_cli_override(cfg, "fps", eval_fps) + maybe_set_cli_override(cfg, "eval_deterministic", bool(getattr(base_cfg, "eval_deterministic", True))) + maybe_set_cli_override(cfg, "disable_reward_scaling", bool(getattr(base_cfg, "eval_raw_reward", False))) + return cfg + + +def _requires_sample_factory_checkpoint_config(config: dict[str, Any]) -> bool: + policy_type = str(config.get("policy", {}).get("type", "")).strip().lower() + return policy_type == "deadly_corridor_sf" + + +def _seed_initialized_vizdoom_game(env: Any, seed: int) -> bool: + unwrapped = getattr(env, "unwrapped", env) + game = getattr(unwrapped, "game", None) + if game is None: + return False + unwrapped.seed(int(seed)) + game.set_seed(int(unwrapped.curr_seed)) + return True + + +class DeadlyCorridorEnvAdapter(EnvAdapter): + """Latency-bench adapter for ViZDoom Deadly Corridor using the SF Doom env stack.""" + OBSERVATION_TYPE = "vizdoom_frame_v1" + + def __init__( + self, + *, + config: dict[str, Any], + noop_action: Action | None = None, + export_env_raw_rgb_frames: bool = False, + ): + env_cfg = config["env"] + env_id = str(env_cfg.get("env_id", "doom_deadly_corridor")) + env_fps = float(env_cfg.get("env_fps", 35)) + self.frame_stack = int(env_cfg.get("frame_stack", 1) or 1) + + from sample_factory.utils.attr_dict import AttrDict + from sf_examples.vizdoom.doom.doom_utils import DOOM_ENVS, make_doom_env_from_spec + + cfg = _build_sample_factory_eval_cfg(config) + spec = next((item for item in DOOM_ENVS if item.name == str(env_id)), None) + if spec is None: + raise ValueError(f"Unknown ViZDoom env spec: {env_id}") + spec_to_use = _spec_with_reward_scaling( + spec, + disable_reward_scaling=bool(getattr(cfg, "disable_reward_scaling", False)), + ) + self.gym_env = make_doom_env_from_spec( + spec_to_use, + str(env_id), + cfg, + AttrDict(worker_index=0, vector_index=0, env_id=0), + render_mode=None, + ) + self.cfg = cfg + self.env_id = env_id + self.env_fps = float(env_fps) + action_space = self.gym_env.action_space + noop_value = _noop_action_from_space(action_space) + if noop_action is None: + self.noop_action = Action(value=noop_value, name=str(noop_value), is_noop=True) + else: + coerced_noop_value = _coerce_noop_action_for_space(noop_action.value, action_space) + self.noop_action = Action( + value=coerced_noop_value, + name=str(coerced_noop_value), + is_noop=True, + is_oneshot=noop_action.is_oneshot, + ) + self.env_step = 0 + self.export_env_raw_rgb_frames = bool(export_env_raw_rgb_frames) + self._last_info: dict[str, Any] = {} + self._last_frame: Any = None + self._observed_frames: deque[np.ndarray] = deque(maxlen=self.frame_stack) + self._raw_rgb_frames = RawRgbFrameStackBuffer(num_frames=self.frame_stack) + + def reset(self, seed: int | None = None) -> Observation: + self.env_step = 0 + self._observed_frames.clear() + if seed is not None: + if _seed_initialized_vizdoom_game(self.gym_env, int(seed)): + obs, info = self.gym_env.reset() + else: + try: + obs, info = self.gym_env.reset(seed=seed) + except TypeError: + obs, info = self.gym_env.reset() + else: + obs, info = self.gym_env.reset() + self._last_frame = obs + self._last_info = dict(info or {}) + self._reset_frame_stack(obs) + if self.export_env_raw_rgb_frames: + self._reset_raw_rgb_frame_stack() + return self._make_observation(info=self._last_info) + + def step(self, action: Action) -> StepResult: + gym_action = action.value + obs, reward, terminated, truncated, info = self.gym_env.step(gym_action) + self.env_step += 1 + self._last_frame = obs + self._last_info = dict(info or {}) + self._append_frame(obs) + if self.export_env_raw_rgb_frames and not bool(terminated or truncated): + self._append_raw_rgb_frame() + observation = self._make_observation(info=self._last_info) + step_info = dict(self._last_info) + step_info.update( + { + "env_step": self.env_step, + "sim_time_ms": self.env_step * self.frame_ms, + "applied_action": gym_action, + "applied_action_name": action.name, + "observation": "vizdoom_frame_v1", + } + ) + return StepResult( + observation=observation, + reward=float(reward), + done=bool(terminated), + truncated=bool(truncated), + info=step_info, + ) + + def observe(self) -> Observation: + if self._last_frame is None: + raise RuntimeError("DeadlyCorridorEnvAdapter has no current observation; call reset() first") + metadata = self._metadata(self._last_info) + return Observation( + data=self._policy_frame_stack(), + env_step=self.env_step, + sim_time_ms=self.env_step * self.frame_ms, + metadata=metadata, + ) + + def render_game_frame(self) -> np.ndarray: + return np.transpose(self.gym_env.unwrapped.game.get_state().screen_buffer, (1, 2, 0)) + + def close(self) -> None: + self.gym_env.close() + + def _reset_frame_stack(self, frame: Any) -> None: + self._observed_frames.clear() + self._append_frame(frame) + + def _append_frame(self, frame: Any) -> None: + self._observed_frames.append(_single_frame_data(frame)) + + def _policy_frame_stack(self) -> np.ndarray: + frames = list(self._observed_frames) + if not frames: + raise RuntimeError("Deadly Corridor observe() has no current frame; call reset() first") + if len(frames) < self.frame_stack: + frames = [frames[0]] * (self.frame_stack - len(frames)) + frames + frames = [np.asarray(frame, dtype=np.uint8) for frame in frames[-self.frame_stack :]] + if self.frame_stack == 1: + return frames[-1] + axis = 0 if looks_chw(frames[0]) else -1 + return np.concatenate(frames, axis=axis) + + +def _single_frame_data(frame: Any) -> np.ndarray: + value = frame.get("obs") if isinstance(frame, dict) else frame + arr = np.asarray(value, dtype=np.uint8) + if arr.ndim == 2: + return arr[..., None] + if arr.ndim != 3: + raise ValueError(f"Expected Deadly Corridor image frame with 2 or 3 dims, got {arr.shape!r}") + return arr + + +# Fixed semantic button order the StarVLA multibinary head is trained against. +# Mirrors starVLA.training.rl_games.eval_core._semantic_to_runtime_multibinary. +DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER = ( + "MOVE_FORWARD", + "MOVE_BACKWARD", + "MOVE_LEFT", + "MOVE_RIGHT", + "TURN_LEFT", + "TURN_RIGHT", + "ATTACK", +) + + +def _deadly_runtime_button_names(gym_env: Any) -> list[str]: + """Return the live ViZDoom action-button order (ports eval_core helper). + + The MultiBinary action vector is indexed by the game's available-button + order, which is not guaranteed to equal the semantic order the head emits. + """ + + def _button_name(button: Any) -> str: + name = getattr(button, "name", None) + if name is not None: + return str(name) + text = str(button) + return text.split(".")[-1] if "." in text else text + + for candidate in (gym_env, getattr(gym_env, "unwrapped", None)): + if candidate is None: + continue + for attr_name in ("game", "_game"): + game = getattr(candidate, attr_name, None) + if game is None: + continue + getter = getattr(game, "get_available_buttons", None) + if getter is None: + continue + names = [_button_name(button) for button in getter()] + if names: + return names + return list(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + + +def _semantic_to_runtime_multibinary(semantic_values: list[int], runtime_order: list[str]) -> list[int]: + semantic_map = { + name: int(semantic_values[idx]) + for idx, name in enumerate(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + if idx < len(semantic_values) + } + return [semantic_map.get(name, 0) for name in runtime_order] + + +class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + """Deadly Corridor adapter for StarVLA eval, matching eval_core's env. + + Unlike :class:`DeadlyCorridorEnvAdapter` (sample_factory, factorised action + tuple), this uses the gymnasium ``VizdoomDeadlyCorridor-MultiBinary`` env so + the model's multibinary head can fire arbitrary button subsets, exactly like + ``starVLA.training.rl_games.eval_core``. Native ``frame_skip=1`` is used so + latency_bench's observation-cadence scheduler owns the obs_stride stepping + (see ObservationCadenceDecisionScheduler); setting a native skip would + double-count it. + """ + + OBSERVATION_TYPE = "vizdoom_frame_v1" + + def __init__( + self, + *, + config: dict[str, Any], + noop_action: Action | None = None, + export_env_raw_rgb_frames: bool = True, + ): + import gymnasium as gym + import vizdoom + import vizdoom.gymnasium_wrapper # noqa: F401 (registers the env ids) + + env_cfg = config["env"] + self.env_fps = float(env_cfg.get("env_fps", 35)) + self.frame_stack = int(env_cfg.get("frame_stack", 1) or 1) + # The raw teacher view is part of the policy's observation contract. + render_options = { + key: env_cfg[key] + for key in ( + "screen_resolution", "render_hud", "render_crosshair", + "render_weapon", "render_decals", "render_particles", + ) + if key in env_cfg + } + # ViZDoom registers deadly_corridor.cfg under this official Gym ID. + self.env_id = "VizdoomCorridor-v0" + self.gym_env = gym.make( + self.env_id, render_mode="rgb_array", frame_skip=1, max_buttons_pressed=0, + ) + game = self.gym_env.unwrapped.game + game.close() + for key, value in render_options.items(): + if key == "screen_resolution": + value = getattr(vizdoom.ScreenResolution, value) + getattr(game, f"set_{key}")(value) + game.init() + self.gym_env.unwrapped.observation_space.spaces["screen"] = Box( + 0, 255, + shape=(game.get_screen_height(), game.get_screen_width(), game.get_screen_channels()), + dtype=np.uint8, + ) + + self._runtime_button_order = _deadly_runtime_button_names(self.gym_env) + self._num_buttons = len(self._runtime_button_order) + self.gym_env.unwrapped.action_space = MultiBinary(self._num_buttons) + self.noop_action = noop_action or Action( + value=[0] * self._num_buttons, name="NOOP", is_noop=True + ) + self.export_env_raw_rgb_frames = bool(export_env_raw_rgb_frames) + self.env_step = 0 + self._last_info: dict[str, Any] = {} + self._last_frame: Any = None + self._raw_rgb_frames = RawRgbFrameStackBuffer(num_frames=self.frame_stack) + + def reset(self, seed: int | None = None) -> Observation: + self.env_step = 0 + try: + obs, info = self.gym_env.reset(seed=seed) + except TypeError: + obs, info = self.gym_env.reset() + self._last_frame = obs + self._last_info = dict(info or {}) + if self.export_env_raw_rgb_frames: + self._reset_raw_rgb_frame_stack() + return self._make_observation(info=self._last_info) + + def step(self, action: Action) -> StepResult: + # action.value is a 7-dim multibinary vector in semantic order; re-order + # to the live game's button layout before stepping the MultiBinary env. + semantic = [int(v) for v in np.asarray(action.value).reshape(-1).tolist()] + expected = len(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + if len(semantic) != expected: + raise ValueError( + "DeadlyCorridorVlaEnvAdapter expects a " + f"{expected}-dim multibinary action in semantic order, got " + f"{len(semantic)} values ({action.value!r}). This usually means the " + "policy decoded a non-multibinary layout; ensure the deadly head is " + "action_layout=multibinary_7 and reached the multibinary decode path." + ) + runtime_buttons = _semantic_to_runtime_multibinary(semantic, self._runtime_button_order) + gym_action = np.asarray(runtime_buttons, dtype=np.int8) + obs, reward, terminated, truncated, info = self.gym_env.step(gym_action) + self.env_step += 1 + self._last_frame = obs + self._last_info = dict(info or {}) + if self.export_env_raw_rgb_frames and not bool(terminated or truncated): + self._append_raw_rgb_frame() + observation = self._make_observation(info=self._last_info) + step_info = dict(self._last_info) + step_info.update( + { + "env_step": self.env_step, + "sim_time_ms": self.env_step * self.frame_ms, + "applied_action": runtime_buttons, + "applied_action_name": action.name, + "observation": self.OBSERVATION_TYPE, + } + ) + return StepResult( + observation=observation, + reward=float(reward), + done=bool(terminated), + truncated=bool(truncated), + info=step_info, + ) + + def observe(self) -> Observation: + return self._make_observation(info=self._last_info) + + def render_game_frame(self) -> np.ndarray: + frame = self.gym_env.render() + return np.asarray(frame, dtype=np.uint8) + + def close(self) -> None: + self.gym_env.close() diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py new file mode 100644 index 0000000000000000000000000000000000000000..13f64ef81329e0b3c9296a066e17196a7e6c1d56 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py @@ -0,0 +1,60 @@ +"""Causal action history sampled at completed decision boundaries.""" + +from __future__ import annotations + +import numpy as np +from gymnasium.spaces import Discrete, MultiBinary, Tuple + + +class DecisionActionHistory: + """Encode admission, admitted command, and last applied action for each decision.""" + + def __init__(self, action_space, *, num_envs: int, decisions: int): + self._multibinary = isinstance(action_space, MultiBinary) + if isinstance(action_space, Discrete): + self.action_sizes = (action_space.n,) + elif isinstance(action_space, Tuple) and all(isinstance(space, Discrete) for space in action_space.spaces): + self.action_sizes = tuple(space.n for space in action_space.spaces) + elif self._multibinary and action_space.shape == (7,): + self.action_sizes = (3, 3, 3, 2) + else: + raise NotImplementedError(f"Decision action history does not support {action_space!r}") + self.decisions = decisions + self.action_dim = sum(size - 1 for size in self.action_sizes) + self.step_dim = 1 + 2 * self.action_dim + self.data = np.zeros((num_envs, decisions, self.step_dim), dtype=np.float32) + self._basis = tuple(np.eye(size, dtype=np.float32)[:, 1:] for size in self.action_sizes) + + @property + def observation_dim(self) -> int: + return self.decisions * self.step_dim + + def reset(self, indices=None) -> None: + if indices is None: + self.data.fill(0) + else: + self.data[indices] = 0 + + def append(self, indices, admitted, issued_actions, applied_actions) -> None: + admitted = np.asarray(admitted, dtype=np.float32).reshape(-1) + issued = self._encode(issued_actions) * admitted[:, None] + applied = self._encode(applied_actions) + rows = self.data[indices].copy() + rows[:, :-1] = rows[:, 1:] + rows[:, -1, 0] = admitted + rows[:, -1, 1 : 1 + self.action_dim] = issued + rows[:, -1, 1 + self.action_dim :] = applied + self.data[indices] = rows + + def observation(self) -> np.ndarray: + return self.data.reshape(self.data.shape[0], self.observation_dim).copy() + + def _encode(self, actions) -> np.ndarray: + if self._multibinary: + # The VLA button order is move, strafe, turn, attack; teacher history + # encodes turn, move, strafe, attack. Keep both opposing bits if issued. + return np.asarray(actions, dtype=np.float32).reshape(-1, 7)[:, [4, 5, 0, 1, 2, 3, 6]] + values = np.asarray(actions, dtype=np.int64).reshape(-1, len(self.action_sizes)) + return np.concatenate( + [basis[values[:, index]] for index, basis in enumerate(self._basis)], axis=1 + ) diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py new file mode 100644 index 0000000000000000000000000000000000000000..fca10fa388e20c9df4abb2f235ac4ea284a3144c --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py @@ -0,0 +1,216 @@ +"""Single evaluation driver: run one config's episodes and attach metadata. + +This is the core ``run_from_config`` and its episode-side helpers. Sweep/suite +orchestration lives in :mod:`latency_bench.eval.sweeps`; the CLI in +:mod:`latency_bench.run`. +""" +from __future__ import annotations + +from collections.abc import Callable, Sequence +from pathlib import Path +from typing import Any + +import yaml + +from training.common.utils import seed_everything +from latency_bench.core.types import EpisodeMetrics, ExecutorMode +from latency_bench.eval.config import ( + _episode_seed, + _eval_episodes, + _eval_max_steps, + _evaluation_seed, + resolve_evaluation_config, +) +from latency_bench.eval.reporting import _write_non_sweep_summary +from latency_bench.envs.base import EnvAdapter +from latency_bench.executors.base import BatchedExecutor +from latency_bench.executors.factory import build_executor +from latency_bench.executors.realtime_warmup import plot_realtime_eval_latency +from latency_bench.latency.config import latency_type_from_config +from latency_bench.logging.action_trace_replay import record_videos_from_action_trace +from latency_bench.logging.video import select_episode_return_stratified + + +def run_from_config( + config: dict[str, Any], + extra_metadata: dict[str, Any] | None = None, + *, + write_summary: bool = True, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + episode_ids: Sequence[int] | None = None, + policy: Any | None = None, + env: EnvAdapter | None = None, + env_backend: Any | None = None, + inference_devices: list[str] | None = None, +) -> list[EpisodeMetrics]: + eval_max_steps = _eval_max_steps(config) + resolve_evaluation_config(config) + if ( + policy is None + and env is None + and env_backend is None + and config["policy"]["type"] == "starvla" + ): + from latency_bench.policy.starvla import prepare_starvla_checkpoint_input_config + + prepare_starvla_checkpoint_input_config(config) + + experiment_cfg = config["experiment"] + policy_cfg = config["policy"] + logging_cfg = config["logging"] + seed = _evaluation_seed(config) + configured_num_episodes = _eval_episodes(config) + selected_episode_ids = list(range(configured_num_episodes)) if episode_ids is None else list(episode_ids) + seed_everything(seed) + + executor_kwargs = {} + if policy is not None: + executor_kwargs["policy"] = policy + if env is not None: + executor_kwargs["env"] = env + if env_backend is not None: + executor_kwargs["env_backend"] = env_backend + if inference_devices is not None: + executor_kwargs["inference_devices"] = inference_devices + executor = build_executor(config, **executor_kwargs) + metrics = [] + warmup_metadata_by_episode: dict[int, dict[str, Any]] = {} + try: + output_dir = Path(logging_cfg["output_dir"]) + output_dir.mkdir(parents=True, exist_ok=True) + (output_dir / "resolved_config.yaml").write_text( + yaml.safe_dump(config, sort_keys=False), encoding="utf-8" + ) + if isinstance(executor, BatchedExecutor): + warmup_metadata = executor.run_warmup() + run_episodes_kwargs: dict[str, Any] = { + "episode_ids": selected_episode_ids, + "seeds": [_episode_seed(config, episode_id) for episode_id in selected_episode_ids], + "eval_max_steps": eval_max_steps, + } + if on_episode_complete is not None: + run_episodes_kwargs["on_episode_complete"] = on_episode_complete + metrics = list(executor.run_episodes(**run_episodes_kwargs)) + warmup_metadata_by_episode.update( + (episode_id, warmup_metadata) for episode_id in selected_episode_ids + ) + else: + warmup_metadata = executor.run_warmup() + for episode_id in selected_episode_ids: + warmup_metadata_by_episode[episode_id] = warmup_metadata + episode_metrics = executor.run_episode( + episode_id=episode_id, + seed=_episode_seed(config, episode_id), + eval_max_steps=eval_max_steps, + ) + metrics.append(episode_metrics) + if on_episode_complete is not None: + on_episode_complete(episode_metrics) + metrics.sort(key=lambda item: int(item.episode_id)) + for episode_metrics in metrics: + for key, value in _evaluation_raw_fact_metadata(config, int(episode_metrics.episode_id)).items(): + if episode_metrics.metadata.get(key) is None: + episode_metrics.metadata[key] = value + episode_metrics.metadata.update(warmup_metadata_by_episode[int(episode_metrics.episode_id)]) + if "measurement" in config: + episode_metrics.metadata["measurement"] = config["measurement"] + episode_metrics.metadata["config_name"] = experiment_cfg.get("name") + episode_metrics.metadata["run_name"] = experiment_cfg.get("name") + if "checkpoint_path" in policy_cfg: + episode_metrics.metadata["checkpoint_path"] = policy_cfg["checkpoint_path"] + if "profile_path" in config["latency"]: + episode_metrics.metadata["source_profile_path"] = config["latency"]["profile_path"] + if "checkpoint_kind" in policy_cfg: + episode_metrics.metadata["checkpoint_kind"] = str(policy_cfg["checkpoint_kind"]) + episode_metrics.metadata["output_dir"] = str(logging_cfg["output_dir"]) + if "action_prefix" in policy_cfg: + episode_metrics.metadata["action_prefix"] = policy_cfg["action_prefix"] + if extra_metadata: + episode_metrics.metadata.update(extra_metadata) + if executor.logger is not None: + executor.logger.flush() + _record_realtime_eval_latency_plot(config, executor) + if write_summary: + _write_non_sweep_summary(config, metrics) + _record_stratified_replay_videos(config, metrics, seed=seed) + finally: + executor.close() + return metrics + + +def _record_realtime_eval_latency_plot(config: dict[str, Any], executor: Any) -> None: + if ExecutorMode(config["executor"]["mode"]) != ExecutorMode.REALTIME: + return + if not config["logging"]["save_latency_records"]: + return + + latency_values = list(executor.logger.latency_ms_values) + plot_realtime_eval_latency( + latency_values, + Path(config["logging"]["output_dir"]) / "eval_latency_trace.png", + ) + + +def _record_stratified_replay_videos( + config: dict[str, Any], + metrics: list[EpisodeMetrics], + *, + seed: int, +) -> None: + if "video" not in config["logging"]: + return + video_cfg = config["logging"]["video"] + if not video_cfg["enabled"]: + return + if not config["logging"]["save_step_records"]: + # Replay reads steps.jsonl, which is only written when save_step_records is on. + # Without it (e.g. factor-sweep evals) skip video instead of crashing on a missing file. + return + if ExecutorMode(config["executor"]["mode"]) == ExecutorMode.REALTIME: + return + selections = select_episode_return_stratified( + metrics, + num_bins=video_cfg["num_bins"], + seed=seed, + ) + record_videos_from_action_trace(config, selections=selections, metrics=metrics) + + +def _evaluation_raw_fact_metadata(config: dict[str, Any], episode_id: int) -> dict[str, Any]: + env_cfg = config.get("env", {}) + policy_cfg = config.get("policy", {}) + latency_cfg = config.get("latency", {}) + executor_cfg = config.get("executor", {}) + env_fps = float(env_cfg["env_fps"]) if "env_fps" in env_cfg else None + obs_fps = float(env_cfg["obs_fps"]) if "obs_fps" in env_cfg else None + frame_ms = None if env_fps is None or env_fps <= 0 else 1000.0 / env_fps + executor_mode = str(executor_cfg.get("mode", "")).strip().lower() + latency_type = latency_type_from_config(latency_cfg) + if executor_mode == "paused": + latency_type = "zero" + elif executor_mode == "realtime": + latency_type = "measured" + return { + "mode": executor_cfg.get("mode"), + "episode_seed": _episode_seed(config, episode_id), + "policy_id": _metadata_id(policy_cfg, "policy_id", "id", "type"), + "env_id": _metadata_id(env_cfg, "env_id", "id", "name"), + "model_id": latency_cfg.get("model_id"), + "gpu_class": latency_cfg.get("gpu_class"), + "workload_id": latency_cfg.get("workload_id"), + "instance_id": latency_cfg.get("instance_id"), + "source_run_id": latency_cfg.get("source_run_id"), + "profile_ref": latency_cfg.get("profile_ref"), + "env_fps": env_fps, + "obs_fps": obs_fps, + "frame_ms": frame_ms, + "latency_type": latency_type, + } + + +def _metadata_id(config: dict[str, Any], *keys: str) -> str | None: + for key in keys: + value = config.get(key) + if value is not None: + return str(value) + return None diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py new file mode 100644 index 0000000000000000000000000000000000000000..19206ec9dfd77ce730fbaee77fe932956dac0b4d --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py @@ -0,0 +1,341 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path +from typing import Any + +import gymnasium as gym +import numpy as np +import torch +import yaml + + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT)) +sys.path.insert(0, str(ROOT / "third_party" / "MIKASA-Robo")) + +from latency_bench.core.types import Action, Observation # noqa: E402 +from latency_bench.executors.gpu_batched_env_step_backend import ( # noqa: E402 + GpuBatchedEnvStepBackendBase, + SlotStepOutcome, +) +from mikasa_robo_suite.seed_reset import ( # noqa: E402 + reset_seeded_slot as _reset_seeded_slot, + reset_seeded_slots as _reset_seeded_slots, +) + + +ENV_ID = "InterceptGrabFast-VLA-v0" +INSTRUCTION = "Intercept the rolling ball and grasp it to stop it." +START_SEED = 4242424242 +MIKASA_IMAGE_VIEWS_INFO_KEY = "mikasa_image_views" +MIKASA_STATE_INFO_KEY = "mikasa_proprio" + + +def _scalar(value: Any) -> Any: + if torch.is_tensor(value): + return value.detach().reshape(-1)[0].cpu().item() + return np.asarray(value).reshape(-1)[0].item() + + +def _make_raw_env( + obs_mode: str, + num_envs: int = 1, + simulator_device: str = "gpu", +): + import mikasa_robo_suite.vla.memory_envs # noqa: F401 + + return gym.make( + ENV_ID, + num_envs=num_envs, + obs_mode=obs_mode, + control_mode="pd_ee_delta_pose", + render_mode="all", + sim_backend=simulator_device, + render_backend=simulator_device, + reward_mode="normalized_dense", + ) + + +def _make_ppo_env(num_envs: int = 1, simulator_device: str = "gpu"): + from baselines.ppo.ppo_memtasks import FlattenRGBDObservationWrapper + from mani_skill.vector.wrappers.gymnasium import ManiSkillVectorEnv + from mikasa_robo_suite.vla.dataset_collectors.get_mikasa_robo_datasets import ( + env_info, + ) + + env = _make_raw_env( + "state", + num_envs=num_envs, + simulator_device=simulator_device, + ) + wrappers, _ = env_info(ENV_ID) + for wrapper, kwargs in wrappers: + env = wrapper(env, **kwargs) + env = FlattenRGBDObservationWrapper(env, rgb=False, depth=False, state=True) + return ManiSkillVectorEnv( + env, + num_envs, + ignore_terminations=True, + record_metrics=True, + ) + + +def _make_vla_env(num_envs: int = 1, simulator_device: str = "gpu"): + from mikasa_robo_suite.vla.utils.apply_wrappers import apply_mikasa_vla_wrappers + + return apply_mikasa_vla_wrappers( + _make_raw_env( + "rgb", + num_envs=num_envs, + simulator_device=simulator_device, + ), + include_overlays=False, + ) + + +class _PpoPolicy: + def __init__(self, env, checkpoint: Path): + from baselines.ppo.ppo_memtasks import AgentStateOnly + + self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") + self.agent = AgentStateOnly(env).to(self.device) + self.agent.load_state_dict(torch.load(checkpoint, map_location=self.device)) + self.agent.eval() + + def forward(self, observation): + with torch.no_grad(): + return self.agent.get_action( + {key: value.to(self.device) for key, value in observation.items()}, + deterministic=True, + ) + + +class MikasaEnvStepBackend(GpuBatchedEnvStepBackendBase): + """Own the native MIKASA simulator and its 7D action contract.""" + + backend_name = "mikasa_gpu_batched" + + def __init__(self, *, config: dict[str, Any], num_slots: int, env=None): + noop_action = Action( + value=np.asarray(config["env"]["noop_action"], dtype=np.float32), + name="noop", + is_noop=True, + ) + super().__init__( + config=config, + noop_action=noop_action, + num_slots=num_slots, + action_space=gym.spaces.Box(-1.0, 1.0, shape=(7,), dtype=np.float32), + ) + self.env = ( + _make_vla_env( + num_envs=num_slots, + simulator_device=config["env"]["simulator_device"], + ) + if env is None + else env + ) + self._episode_seeds = [0] * num_slots + self._success = np.zeros(num_slots, dtype=np.bool_) + self._observation, _ = self.env.reset(seed=self._episode_seeds) + + def _reset_slot_observation(self, slot_id: int, *, seed: int | None) -> Observation: + if seed is not None: + self._episode_seeds[slot_id] = int(seed) + self._observation, _ = _reset_seeded_slot( + self.env, + slot_id=slot_id, + seed=self._episode_seeds[slot_id], + ) + self._env_steps[slot_id] = 0 + self._success[slot_id] = False + return self._observation_for_slot(slot_id) + + def _observe_slot_observations( + self, + slot_ids: list[int], + ) -> dict[int, Observation]: + return {slot_id: self._observation_for_slot(slot_id) for slot_id in slot_ids} + + def _step_cores( + self, + slot_ids: list[int], + *, + actions: np.ndarray, + active_mask: np.ndarray, + ) -> Any: + del slot_ids, active_mask + tensor_actions = torch.as_tensor( + actions, + dtype=torch.float32, + device=self.env.unwrapped.device, + ) + self._observation, reward, terminated, truncated, info = self.env.step( + tensor_actions + ) + return reward, terminated, truncated, info + + def _slot_step_outcome(self, state: Any, slot_id: int) -> SlotStepOutcome: + reward, terminated, truncated, info = state + success = bool(_slot_value(info["success"], slot_id)) + self._success[slot_id] |= success + return SlotStepOutcome( + reward=float(_slot_value(reward, slot_id)), + done=bool(_slot_value(terminated, slot_id)), + truncated=bool(_slot_value(truncated, slot_id)), + info={ + "success": success, + "task_metrics": {"success": float(self._success[slot_id])}, + }, + ) + + def _observation_for_slot(self, slot_id: int) -> Observation: + rgb = self._observation["rgb"] + if torch.is_tensor(rgb): + rgb = rgb.detach().cpu().numpy() + rgb = np.asarray(rgb) + views = np.stack( + [ + np.asarray(rgb[slot_id, :, :, :3], dtype=np.uint8), + np.asarray(rgb[slot_id, :, :, 3:6], dtype=np.uint8), + ] + ) + metadata = { + MIKASA_IMAGE_VIEWS_INFO_KEY: views, + MIKASA_STATE_INFO_KEY: self._observation["proprio"][slot_id].detach().cpu().numpy(), + "slot_id": slot_id, + } + if "action_prefix_state_key" in self.config["env"]: + metadata["action_prefix_state_key"] = self.config["env"]["action_prefix_state_key"] + if "returned_action_context" in self.config["env"]: + context = self.config["env"]["returned_action_context"] + metadata["returned_action_context"] = { + **context, + "order": np.asarray(context["order"]), + "low": np.asarray(context["low"], dtype=np.float32), + "high": np.asarray(context["high"], dtype=np.float32), + } + return Observation( + data=None, + env_step=int(self._env_steps[slot_id]), + sim_time_ms=float(self._env_steps[slot_id]) * self._frame_ms, + metadata=metadata, + ) + + def close(self) -> None: + if not self.closed: + self.env.close() + super().close() + + +def _slot_value(value: Any, slot_id: int) -> Any: + if torch.is_tensor(value): + return value.detach().reshape(-1)[slot_id].cpu().item() + return np.asarray(value).reshape(-1)[slot_id].item() + + +def _evaluate(args: argparse.Namespace) -> dict[str, Any]: + env = _make_ppo_env() + policy = _PpoPolicy(env, args.checkpoint) + seeds = [] + successes = [] + returns = [] + lengths = [] + try: + for episode_index in range(args.episodes): + seed = START_SEED + episode_index + observation, _ = env.reset(seed=seed) + success_once = False + episode_return = 0.0 + for step in range(60): + action = policy.forward(observation) + observation, reward, terminated, truncated, info = env.step(action) + success_once = success_once or bool(_scalar(info["success"])) + episode_return += float(_scalar(reward)) + if bool(_scalar(terminated)) or bool(_scalar(truncated)): + break + seeds.append(seed) + successes.append(success_once) + returns.append(episode_return) + lengths.append(step + 1) + finally: + env.close() + summary = { + "seeds": seeds, + "successes": successes, + "success_rate": float(np.mean(successes)), + "returns": returns, + "lengths": lengths, + } + (args.output_dir / "summary.json").write_text( + json.dumps(summary, indent=2) + "\n", encoding="utf-8" + ) + return summary + + +def _latency_eval(argv: list[str]) -> None: + from latency_bench.core.config import load_config + from latency_bench.eval.config import apply_runtime_overrides + from latency_bench.eval.driver import run_from_config + + parser = argparse.ArgumentParser() + parser.add_argument("--eval-config", type=Path, required=True) + parser.add_argument("--checkpoint-path", type=Path) + parser.add_argument("--model-config-path", type=Path) + parser.add_argument("--task-contract-path", type=Path) + parser.add_argument("--run-name") + parser.add_argument("--output-dir", type=Path) + parser.add_argument("--latency-method", choices=("zero", "temporal")) + parser.add_argument("--profile-path", type=Path) + parser.add_argument("--latency-seed", type=int) + args = parser.parse_args(argv) + config = load_config(args.eval_config) + apply_runtime_overrides( + config, + checkpoint_path=args.checkpoint_path, + model_config_path=args.model_config_path, + task_contract_path=args.task_contract_path, + run_name=args.run_name, + output_dir=args.output_dir, + latency_method=args.latency_method, + latency_profile_path=args.profile_path, + latency_seed=args.latency_seed, + ) + output_dir = Path(config["logging"]["output_dir"]) + output_dir.mkdir(parents=True, exist_ok=True) + (output_dir / "eval_config.yaml").write_text( + yaml.safe_dump(config, sort_keys=False), encoding="utf-8" + ) + backend = MikasaEnvStepBackend( + config=config, + num_slots=int(config["evaluation"]["eval_parallel_envs"]), + ) + run_from_config( + config, + env_backend=backend, + inference_devices=config["executor"]["inference_devices"], + ) + + +def main() -> None: + if sys.argv[1:2] == ["latency-eval"]: + _latency_eval(sys.argv[2:]) + return + + parser = argparse.ArgumentParser() + parser.add_argument("--policy", choices=("ppo",), required=True) + parser.add_argument("--checkpoint", type=Path) + parser.add_argument("--episodes", type=int, default=50) + parser.add_argument("--output-dir", type=Path, required=True) + args = parser.parse_args() + args.output_dir.mkdir(parents=True, exist_ok=True) + + _evaluate(args) + + +if __name__ == "__main__": + main() diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py new file mode 100644 index 0000000000000000000000000000000000000000..7b9cab3b3fad77d6f6a3b7919ff4a8124c98d011 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py @@ -0,0 +1,1207 @@ +from __future__ import annotations + +import json +import sys +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any + +import numpy as np +import numpy.typing as npt +from PIL import Image + +from latency_bench.core.actions import ActionResolver +from latency_bench.core.clock import EnvClock +from latency_bench.core.timing import current_profiler +from latency_bench.core.types import Action, Observation, PolicyOutput +from latency_bench.data.ghost_trail import GhostTrailConfig, build_flappy_ghost_trail_window +from latency_bench.data.state_normalization import min_max_normalize_state +from latency_bench.envs.raw_rgb import ENV_RAW_RGB_FRAME_STACK_INFO_KEY +from latency_bench.envs.gymnasium_task import ( + gymnasium_action_space_contract, + gymnasium_task_contract, +) +from latency_bench.policy.base import PolicyRunner +from latency_bench.policy.starvla_prompts import load_latency_prompt_map, resolve_starvla_prompt + +from latency_bench.utils.paths import REPO_ROOT + + +STARVLA_ROOT = REPO_ROOT / "third_party" / "starVLA" +STATEFUL_STARVLA_MODEL_IDS: tuple[str, ...] = ( + "pi0", + "pi-0", + "pi05", + "pi-0.5", + "gr00t", + "qwenpi", + "qwenpi_v3", + "qwengr00t", +) +STATELESS_STARVLA_MODEL_IDS: tuple[str, ...] = ( + "openvla", + "qwenoft", +) + +DEMON_ATTACK_ACTION_LABELS: tuple[str, ...] = ( + "NOOP", + "FIRE", + "RIGHT", + "LEFT", + "RIGHTFIRE", + "LEFTFIRE", +) +DEADLY_CORRIDOR_TURN_LABELS: tuple[str, ...] = ( + "TURN_NOOP", + "TURN_LEFT", + "TURN_RIGHT", +) +DEADLY_CORRIDOR_MOVE_LABELS: tuple[str, ...] = ( + "MOVE_NOOP", + "MOVE_FORWARD", + "MOVE_BACKWARD", +) +DEADLY_CORRIDOR_STRAFE_LABELS: tuple[str, ...] = ( + "STRAFE_NOOP", + "MOVE_LEFT", + "MOVE_RIGHT", +) +DEADLY_CORRIDOR_ATTACK_LABELS: tuple[str, ...] = ( + "ATTACK_NOOP", + "ATTACK", +) + + +class StarVlaPolicyRunner(PolicyRunner): + """Translate observations and model outputs using the task action contract.""" + + def __init__( + self, + *, + wrapper: Any, + checkpoint_path: str, + device: str, + unnorm_key: str | None, + env_name: str, + action_resolver: ActionResolver, + action_refs: Sequence[Any], + latency_prompt_map: dict[str, Any] | None = None, + base_prompt: str | None = None, + latency_prompt_key: int | str | None = None, + prompt_mode: str | None = None, + obs_resize: tuple[int, int] | None = None, + image_transform_config: Mapping[str, Any] | None = None, + observation_stride_raw_frames: int, + model_cfg: Mapping[str, Any] | None = None, + state_normalization: Mapping[str, Any] | None = None, + state_source: str | None = None, + image_views_info_key: str | None = None, + action_output_type: str | None = None, + ) -> None: + self._wrapper = wrapper + self._obs_resize = tuple(obs_resize) if obs_resize else None + self._checkpoint_path = checkpoint_path + self._device = device + self._unnorm_key = unnorm_key + self._env_name = env_name + self._action_by_raw_id = { + raw_action_id: action_resolver.resolve(action_ref) + for raw_action_id, action_ref in enumerate(action_refs) + } + self._latency_prompt_map = latency_prompt_map + self._base_prompt = base_prompt + self._latency_prompt_key = latency_prompt_key + self._prompt_mode = str(prompt_mode or "default").strip().lower() + self._image_transform_config = dict(image_transform_config or {"image_transform": "raw_rgb"}) + self._image_transform = str( + self._image_transform_config.get("image_transform", "raw_rgb") or "raw_rgb" + ).strip().lower() + model_cfg = ( + _normalized_model_cfg_from_wrapper(wrapper) + if model_cfg is None + else _normalized_model_cfg(model_cfg) + ) + self._include_state = _include_state_from_model_cfg(model_cfg) + self._state_dim = _state_dim_from_model_cfg(model_cfg) if self._include_state else None + self._state_normalization = dict(state_normalization or {}) + self._state_source = state_source + self._image_views_info_key = image_views_info_key + self._action_output_type = action_output_type + vla_data = (model_cfg.get("datasets", {}) or {}).get("vla_data", {}) or {} + self._pack_image_sequence = ( + bool(vla_data["pack_image_sequence"]) + if "pack_image_sequence" in vla_data + else False + ) + self._image_sequence_length = ( + int(vla_data["image_sequence_length"]) + if self._pack_image_sequence + else 1 + ) + self._observation_stride_raw_frames = int(observation_stride_raw_frames) + self._image_sequence_raw_span = ( + 1 + + (self._image_sequence_length - 1) + * self._observation_stride_raw_frames + ) + self._num_obs_frames = int(vla_data.get("num_obs_frames", 1) or 1) + self._image_mode = str(vla_data.get("image_mode", "single")) + self._stitch_grid = tuple(vla_data.get("stitch_grid", [2, 2])) + framework_cfg = model_cfg["framework"] + kv_cfg = framework_cfg["kv_memory"] if "kv_memory" in framework_cfg else {} + self._kv_memory_enabled = bool(kv_cfg["enabled"]) if "enabled" in kv_cfg else False + + def reset_state(self, slot_id: int | None = None) -> None: + # Clears the model's per-slot KV memory at episode boundaries (req3). + # No-op unless the framework maintains KV memory. + reset = getattr(self._wrapper, "reset_memory", None) + if callable(reset): + reset(slot_id) + + def predict(self, observation: Observation) -> PolicyOutput: + return self.predict_batch([observation])[0] + + def predict_batch(self, observations: Sequence[Observation]) -> list[PolicyOutput]: + profiler = current_profiler() + with profiler.time("policy_build_example_ms"): + examples = [self._build_example(observation) for observation in observations] + with profiler.time("policy_wrapper_predict_action_ms"): + prediction = self._wrapper.predict_action( + examples=examples, unnorm_key=self._unnorm_key, profiler=profiler + ) + with profiler.time("policy_decode_ms"): + outputs = [ + self._decode_prediction( + prediction=prediction, + index=index, + observation=observation, + example=example, + ) + for index, (observation, example) in enumerate(zip(observations, examples)) + ] + return outputs + + def _decode_prediction( + self, + *, + prediction: dict[str, Any], + index: int, + observation: Observation, + example: dict[str, Any], + ) -> PolicyOutput: + actions = np.asarray(prediction["actions"]) + raw_action_scores = ( + np.asarray(prediction["raw_action_scores"]) + if "raw_action_scores" in prediction + else None + ) + return self._policy_output( + observation=observation, + example=example, + action_payload=actions[index, 0], + action_output_type=( + prediction["action_output_type"] + if self._action_output_type is None + else self._action_output_type + ), + raw_action_scores=None if raw_action_scores is None else raw_action_scores[index, 0], + ) + + def _build_example(self, observation: Observation) -> dict[str, Any]: + frame_source = observation.metadata[ + ENV_RAW_RGB_FRAME_STACK_INFO_KEY + if self._image_views_info_key is None + else self._image_views_info_key + ] + frames = observation_data_to_hwc_uint8_frames(frame_source) # oldest .. newest + transformed = self._transformed_frame(frames=frames, observation=observation) + + if self._image_views_info_key is not None: + pass + elif self._pack_image_sequence: + if transformed is not None: + raise ValueError( + "WanOFT packed image sequences require image_transform=raw_rgb" + ) + if len(frames) < self._image_sequence_raw_span: + raise ValueError( + "WanOFT packed image sequence requires " + f"{self._image_sequence_raw_span} raw frames for " + f"{self._image_sequence_length} decision observations at stride " + f"{self._observation_stride_raw_frames}, got {len(frames)}" + ) + frames = frames[ + -self._image_sequence_raw_span + :: self._observation_stride_raw_frames + ] + elif transformed is not None: + frames = [transformed] + elif self._image_mode == "single" or self._kv_memory_enabled: + frames = frames[-1:] + else: + # Select the temporal observation window to match training (_pack_sample). + raw_span = 1 + (self._num_obs_frames - 1) * self._observation_stride_raw_frames + frames = frames[-raw_span :: self._observation_stride_raw_frames] + + prompt = resolve_starvla_prompt( + env_name=self._env_name, + observation_metadata=observation.metadata, + latency_prompt_map=self._latency_prompt_map, + base_prompt=self._base_prompt, + latency_prompt_key=self._latency_prompt_key, + prompt_mode=self._prompt_mode, + ) + + if self._image_mode == "stitch": + if transformed is not None: + raise ValueError("image_transform is not compatible with image_mode=stitch") + # Tile the window into one image; matches _pack_sample's stitch branch + # (raw frames passed to stitch_frames, which resizes each cell to 224). + images = [_get_stitch_frames()(frames, grid=self._stitch_grid, size=(224, 224))] + else: + if self._obs_resize is not None: + height, width = self._obs_resize + # match training preprocessing exactly: gr00t LeRobotSingleDataset._pack_sample + # does `Image.fromarray(image).resize((224, 224))` (PIL default resample = BICUBIC). + frames = [ + np.asarray(Image.fromarray(frame).resize((width, height)), dtype=np.uint8) + for frame in frames + ] + images = [Image.fromarray(frame) for frame in frames] + + example = { + "image": images, + "lang": prompt, + } + if self._kv_memory_enabled: + example["slot_id"] = observation.metadata["slot_id"] + elif "slot_id" in observation.metadata: + example["slot_id"] = observation.metadata["slot_id"] + if self._include_state: + if self._state_source == "transport": + state = np.asarray(observation.data["transport"], dtype=np.float32) + example["state"] = state.reshape(1, self._state_dim) + elif self._state_normalization: + state = np.asarray( + observation.metadata["gymnasium_state"], dtype=np.float32 + ) + state_min = np.asarray(self._state_normalization["min"], dtype=np.float32) + state_max = np.asarray(self._state_normalization["max"], dtype=np.float32) + state = min_max_normalize_state(state, state_min, state_max) + example["state"] = state.reshape(1, self._state_dim) + else: + example["state"] = np.zeros((1, self._state_dim), dtype=np.float32) + return example + + def _transformed_frame( + self, + *, + frames: Sequence[npt.NDArray[np.uint8]], + observation: Observation, + ) -> npt.NDArray[np.uint8] | None: + if self._image_transform in {"", "none", "raw", "raw_rgb"}: + return None + if self._image_transform not in {"flappy_ghost_trail", "demon_attack_ghost_trail"}: + raise ValueError(f"Unsupported StarVLA image_transform={self._image_transform!r}") + if self._image_transform == "flappy_ghost_trail" and self._env_name != "flappy": + raise ValueError("image_transform=flappy_ghost_trail is only supported for env_name=flappy") + if self._image_transform == "demon_attack_ghost_trail" and self._env_name != "demon_attack": + raise ValueError("image_transform=demon_attack_ghost_trail is only supported for env_name=demon_attack") + + config = GhostTrailConfig( + image_transform=self._image_transform, + history_frames=int(self._image_transform_config.get("history_frames", 5)), + gamma=float(self._image_transform_config.get("gamma", 1.3)), + min_alpha=int(self._image_transform_config.get("min_alpha", 35)), + ground_fraction=float(self._image_transform_config.get("ground_fraction", 0.22)), + scroll_px_per_step=float(self._image_transform_config.get("scroll_px_per_step", 4.0)), + ) + if self._image_transform == "demon_attack_ghost_trail": + # env_step counts raw ALE frames (buffer updated 4× per decision step). + # frames[-0:] == frames, so env_step=0 falls back to the full reset-fill buffer. + valid_count = min(len(frames), int(observation.env_step)) + else: + max_frames = max(1, int(config.history_frames) + 1) + valid_count = min(len(frames), max(1, int(observation.env_step) + 1), max_frames) + window = [np.asarray(frame, dtype=np.uint8) for frame in frames[-valid_count:]] + + if self._image_transform == "demon_attack_ghost_trail": + from latency_bench.data.ghost_trail_demon import build_demon_attack_ghost_trail_window + steps_arg = list(range(len(window))) + return build_demon_attack_ghost_trail_window(window, steps_arg, config=config) + + current_step = int(observation.env_step) + start_step = current_step - valid_count + 1 + steps = list(range(start_step, current_step + 1)) + return build_flappy_ghost_trail_window(window, steps, config=config) + + def _policy_output( + self, + *, + observation: Observation, + example: dict[str, Any], + action_payload: npt.NDArray[Any], + action_output_type: str, + raw_action_scores: npt.NDArray[Any] | None, + ) -> PolicyOutput: + payload = np.asarray(action_payload) + action, action_metadata = action_from_starvla_payload( + payload=payload, + env_name=self._env_name, + action_by_raw_id=self._action_by_raw_id, + action_output_type=action_output_type, + ) + metadata = { + "policy_type": "starvla", + "prompt_source": "latency_prompt_map" if self._latency_prompt_map is not None else "base", + "checkpoint_path": self._checkpoint_path, + "unnorm_key": self._unnorm_key, + "device": self._device, + "input_frame_count": len(example["image"]), + "image_transform": self._image_transform, + "action_output_type": action_output_type, + "action_payload": to_jsonable_action_payload(payload), + "kv_memory_enabled": self._kv_memory_enabled, + **action_metadata, + } + if self._pack_image_sequence: + metadata["image_sequence_length"] = self._image_sequence_length + metadata["input_frame_raw_stride"] = self._observation_stride_raw_frames + metadata["input_frame_raw_span"] = self._image_sequence_raw_span + if "slot_id" in example: + metadata["slot_id"] = example["slot_id"] + if raw_action_scores is not None: + metadata["raw_action_scores"] = [ + float(item) for item in np.asarray(raw_action_scores, dtype=np.float32).tolist() + ] + if "latency_raw_frames" in observation.metadata: + metadata["latency_raw_frames"] = observation.metadata["latency_raw_frames"] + if "latency_ms" in observation.metadata: + metadata["latency_ms"] = observation.metadata["latency_ms"] + if self._latency_prompt_key is not None: + metadata["latency_prompt_key"] = self._latency_prompt_key + return PolicyOutput( + action=action, + raw_output=metadata["action_payload"], + metadata=metadata, + ) + + +def observation_data_to_hwc_uint8_frames(data: Any) -> list[npt.NDArray[np.uint8]]: + frame = _extract_observation_array(data) + if frame.ndim == 4 and frame.shape[-1] == 3: + return [_as_uint8_image(item) for item in frame] + if frame.ndim == 4 and frame.shape[1] == 3: + return [_as_uint8_image(np.transpose(item, (1, 2, 0))) for item in frame] + if frame.ndim == 3 and frame.shape[-1] == 3: + return [_as_uint8_image(frame)] + if frame.ndim == 3 and frame.shape[0] == 3: + return [_as_uint8_image(np.transpose(frame, (1, 2, 0)))] + if ( + frame.ndim == 3 + and frame.shape[0] % 3 == 0 + and frame.shape[0] < frame.shape[1] + and frame.shape[0] < frame.shape[2] + ): + return [ + _as_uint8_image(np.transpose(frame[start : start + 3, :, :], (1, 2, 0))) + for start in range(0, frame.shape[0], 3) + ] + if frame.ndim == 3 and frame.shape[-1] % 3 == 0: + return [ + _as_uint8_image(frame[:, :, start : start + 3]) + for start in range(0, frame.shape[-1], 3) + ] + if frame.ndim == 3 and frame.shape[0] % 3 == 0: + return [ + _as_uint8_image(np.transpose(frame[start : start + 3, :, :], (1, 2, 0))) + for start in range(0, frame.shape[0], 3) + ] + return [_as_uint8_image(frame)] + + +def decode_starvla_action( + *, + vector: npt.NDArray[Any], + env_name: str, + action_by_raw_id: Mapping[int, Action], + action_layout: str | None = None, +) -> tuple[Action, dict[str, Any]]: + deadly_layout = None + if str(env_name) == "deadly_corridor": + action_dim = int(np.asarray(vector).shape[-1]) + deadly_layouts = { + 7: "deadly_corridor_semantic_7", + 11: "deadly_corridor_factorized_11", + 54: "deadly_corridor_joint_54", + } + if action_dim not in deadly_layouts: + raise ValueError( + "Deadly Corridor StarVLA action vector expected 7, 11, or 54 " + f"values, got {action_dim}" + ) + deadly_layout = deadly_layouts[action_dim] + asterix_layout = None + if str(env_name) == "asterix": + action_dim = int(np.asarray(vector).shape[-1]) + if action_layout is not None: + asterix_layout = str(action_layout).strip().lower() + else: + asterix_layout = "factorized_6" if action_dim < 9 else "discrete_9" + + decode_rl_games_actions, _, _ = _load_rl_games_action_decode() + prediction = decode_rl_games_actions( + normalized_actions=np.asarray(vector), + env_name=str(env_name), + deadly_action_layout=(deadly_layout.removeprefix("deadly_corridor_") if deadly_layout is not None else None), + asterix_action_layout=asterix_layout, + ) + action, metadata = action_from_starvla_payload( + payload=np.asarray(prediction["actions"]), + env_name=env_name, + action_by_raw_id=action_by_raw_id, + action_output_type=prediction["action_output_type"], + ) + if deadly_layout is not None: + metadata["action_layout"] = deadly_layout + if deadly_layout == "deadly_corridor_joint_54": + turn, move, strafe, attack = action.value + metadata["raw_action_id"] = turn * 18 + move * 6 + strafe * 2 + attack + elif deadly_layout == "deadly_corridor_semantic_7": + semantic_actions = ( + [0, 1, 0, 0], + [0, 2, 0, 0], + [0, 0, 1, 0], + [0, 0, 2, 0], + [1, 0, 0, 0], + [2, 0, 0, 0], + [0, 0, 0, 1], + ) + metadata["raw_action_id"] = semantic_actions.index(action.value) + if asterix_layout is not None: + metadata["action_layout"] = asterix_layout + return action, metadata + + +# Fixed semantic button order the StarVLA multibinary head is trained against. +# Mirrors starVLA.training.rl_games.eval_core._semantic_to_runtime_multibinary; +# the env adapter re-orders this to the live ViZDoom button layout. +DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER = ( + "MOVE_FORWARD", + "MOVE_BACKWARD", + "MOVE_LEFT", + "MOVE_RIGHT", + "TURN_LEFT", + "TURN_RIGHT", + "ATTACK", +) + + +def action_from_starvla_payload( + *, + payload: npt.NDArray[Any], + env_name: str, + action_by_raw_id: Mapping[int, Action], + action_output_type: str = "", +) -> tuple[Action, dict[str, Any]]: + if str(action_output_type) == "rl_games_continuous": + values = [float(item) for item in np.asarray(payload).reshape(-1).tolist()] + return Action( + value=values, + name="continuous_torque", + is_noop=all(value == 0.0 for value in values), + is_oneshot=False, + ), {"continuous_action": values} + if str(env_name) == "demon_attack": + return demon_attack_action_from_id(int(np.asarray(payload).reshape(-1)[0])) + if str(env_name) == "deadly_corridor": + # Multibinary heads emit an already-thresholded 7-dim button vector in + # fixed semantic order; the env adapter re-orders it to the live ViZDoom + # button layout. Keep it as-is rather than reinterpreting it as a + # [turn, move, strafe, attack] categorical tuple. + if str(action_output_type) == "rl_games_deadly_corridor_multibinary": + buttons = [int(item) for item in np.asarray(payload).reshape(-1).tolist()] + active = [ + DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER[idx] + for idx, pressed in enumerate(buttons) + if idx < len(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) and pressed + ] + action_name = "+".join(active) if active else "NOOP" + return Action( + value=buttons, + name=action_name, + is_noop=not any(buttons), + is_oneshot=False, + ), { + "decoded_multibinary_buttons": buttons, + "action_label": action_name, + "action_layout": "deadly_corridor_multibinary_7", + } + return deadly_corridor_action_from_tuple( + action_value=[int(item) for item in np.asarray(payload).reshape(-1).tolist()], + metadata={"action_layout": "deadly_corridor_tuple"}, + ) + raw_action_id = int(np.asarray(payload).reshape(-1)[0]) + return action_by_raw_id[raw_action_id], {"raw_action_id": raw_action_id} + + +def to_jsonable_action_payload(payload: npt.NDArray[Any]) -> Any: + value = np.asarray(payload).tolist() + if isinstance(value, list) and len(value) == 1: + return value[0] + return value + + +def demon_attack_action_from_id(action_id: int) -> tuple[Action, dict[str, Any]]: + action = Action( + value=action_id, + name=DEMON_ATTACK_ACTION_LABELS[action_id], + is_noop=action_id == 0, + is_oneshot=False, + ) + return action, {"raw_action_id": action_id, "action_label": action.name} + + +def deadly_corridor_action_from_tuple( + *, + action_value: list[int], + metadata: dict[str, Any], +) -> tuple[Action, dict[str, Any]]: + turn, move, strafe, attack = action_value + action_value = [turn, move, strafe, attack] + turn_label = DEADLY_CORRIDOR_TURN_LABELS[turn] + move_label = DEADLY_CORRIDOR_MOVE_LABELS[move] + strafe_label = DEADLY_CORRIDOR_STRAFE_LABELS[strafe] + attack_label = DEADLY_CORRIDOR_ATTACK_LABELS[attack] + active_labels = [ + label + for label in (turn_label, move_label, strafe_label, attack_label) + if not label.endswith("_NOOP") + ] + action_name = "+".join(active_labels) if active_labels else "NOOP" + return Action( + value=action_value, + name=action_name, + is_noop=action_value == [0, 0, 0, 0], + is_oneshot=False, + ), { + "decoded_action_tuple": action_value, + "turn_label": turn_label, + "move_label": move_label, + "strafe_label": strafe_label, + "attack_label": attack_label, + "action_label": action_name, + **metadata, + } + + +def _extract_observation_array(data: Any) -> npt.NDArray[Any]: + if isinstance(data, Mapping): + return np.asarray(data["observation"]) + return np.asarray(data) + + +def _as_uint8_image(frame: npt.NDArray[Any]) -> npt.NDArray[np.uint8]: + return np.ascontiguousarray(frame, dtype=np.uint8) + + +def _normalized_model_cfg(model_cfg: Mapping[str, Any]) -> dict[str, Any]: + _ensure_starvla_path() + from omegaconf import OmegaConf + from starVLA.model.framework.share_tools import apply_config_compat + + cfg = OmegaConf.create(model_cfg) + apply_config_compat(cfg) + _apply_model_family_include_state_compat(cfg) + return OmegaConf.to_container(cfg, resolve=True) + + +def _normalized_model_cfg_from_wrapper(wrapper: Any) -> dict[str, Any]: + return _normalized_model_cfg(wrapper._model_cfg) + + +def _load_starvla_model_config(path: str | Path) -> dict[str, Any]: + from omegaconf import OmegaConf + + return _normalized_model_cfg(OmegaConf.load(path)) + + +def _apply_model_family_include_state_compat(cfg: Any) -> None: + from omegaconf import OmegaConf + + if OmegaConf.select(cfg, "datasets.vla_data.include_state") is not None: + return + + model_ids = ( + _normalized_optional_config_string(cfg, ("model",)), + _normalized_optional_config_string(cfg, ("rl_games", "model_alias")), + _normalized_optional_config_string(cfg, ("framework", "name")), + ) + if any(model_id in STATEFUL_STARVLA_MODEL_IDS for model_id in model_ids if model_id is not None): + OmegaConf.update(cfg, "datasets.vla_data.include_state", True, force_add=True) + return + if any(model_id in STATELESS_STARVLA_MODEL_IDS for model_id in model_ids if model_id is not None): + OmegaConf.update(cfg, "datasets.vla_data.include_state", False, force_add=True) + + +def _normalized_optional_config_string(cfg: Any, path: tuple[str, ...]) -> str | None: + from omegaconf import OmegaConf + + value = OmegaConf.select(cfg, ".".join(path)) + if value is None: + return None + return str(value).strip().lower() + + +def _state_dim_from_model_cfg(model_cfg: dict[str, Any]) -> int: + return model_cfg["framework"]["action_model"]["state_dim"] + + +def _include_state_from_model_cfg(model_cfg: dict[str, Any]) -> bool: + return model_cfg["datasets"]["vla_data"]["include_state"] + + +_STITCH_FRAMES = None + + +def _get_stitch_frames(): + """Lazily import starVLA's stitch_frames (starVLA path is added at runtime).""" + global _STITCH_FRAMES + if _STITCH_FRAMES is None: + _ensure_starvla_path() + from starVLA.training.trainer_utils.trainer_tools import stitch_frames + + _STITCH_FRAMES = stitch_frames + return _STITCH_FRAMES + + +def _ensure_starvla_path() -> None: + starvla_root = str(STARVLA_ROOT) + if starvla_root not in sys.path: + sys.path.insert(0, starvla_root) + + +def _observation_stride_raw_frames(config: Mapping[str, Any]) -> int: + env_cfg = config["env"] + return EnvClock( + env_fps=float(env_cfg["env_fps"]), + obs_fps=float(env_cfg["obs_fps"]), + ).obs_stride_raw_frames + + +def apply_starvla_model_input_config( + config: dict[str, Any], + *, + model_cfg: Mapping[str, Any], + image_transform: str = "raw_rgb", +) -> None: + """Match latency_bench's raw frame stack to a saved StarVLA input contract.""" + vla_data = model_cfg["datasets"]["vla_data"] + pack_image_sequence = ( + bool(vla_data["pack_image_sequence"]) + if "pack_image_sequence" in vla_data + else False + ) + normalized_transform = str(image_transform).strip().lower() + raw_image_transform = normalized_transform in {"", "none", "raw", "raw_rgb"} + if pack_image_sequence: + if not raw_image_transform: + raise ValueError( + "WanOFT packed image sequences require image_transform=raw_rgb" + ) + input_frame_count = int(vla_data["image_sequence_length"]) + else: + if not raw_image_transform: + return + framework_cfg = model_cfg["framework"] + kv_cfg = framework_cfg["kv_memory"] if "kv_memory" in framework_cfg else {} + kv_memory_enabled = bool(kv_cfg["enabled"]) if "enabled" in kv_cfg else False + if kv_memory_enabled: + return + image_mode = str(vla_data["image_mode"]) if "image_mode" in vla_data else "single" + if image_mode == "single": + return + input_frame_count = int(vla_data["num_obs_frames"]) + + observation_stride = _observation_stride_raw_frames(config) + required_raw_frames = 1 + (input_frame_count - 1) * observation_stride + config["env"]["frame_stack"] = max( + int(config["env"]["frame_stack"]), + required_raw_frames, + ) + + +def prepare_starvla_checkpoint_input_config(config: dict[str, Any]) -> None: + """Apply the saved checkpoint input contract before env construction.""" + if config["policy"]["type"] != "starvla": + return + + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + if config["env"]["name"] == "gymnasium": + contract = json.loads( + Path(policy_cfg["task_contract_path"]).read_text(encoding="utf-8") + ) + config["env"]["state_space"] = {"labels": contract["state_labels"]} + if contract["robot_type"] in ("latency_balance_profile_h8", "latency_balance_profile_h16"): + config["env"]["name"] = "balance_profile" + config["env"]["action_context_horizon"] = contract["action_horizon"] + config["env"]["frame_stack"] = 1 + return + if "model_config_path" in policy_cfg: + model_cfg = _load_starvla_model_config(policy_cfg["model_config_path"]) + else: + _ensure_starvla_path() + from starVLA.model.framework.share_tools import read_mode_config + + saved_model_cfg, _norm_stats = read_mode_config(policy_cfg["checkpoint_path"]) + model_cfg = _normalized_model_cfg(saved_model_cfg) + if config["env"]["name"] == "gymnasium": + image_size = model_cfg["rl_games"]["env_eval"]["image_size"] + config["env"]["obs_resize"] = [image_size, image_size] + image_transform_cfg = ( + policy_cfg["image_transform_config"] + if "image_transform_config" in policy_cfg + else {} + ) + image_transform = ( + image_transform_cfg["image_transform"] + if "image_transform" in image_transform_cfg + else "raw_rgb" + ) + apply_starvla_model_input_config( + config, + model_cfg=model_cfg, + image_transform=image_transform, + ) + + +def _load_policy_wrapper_class() -> Any: + _ensure_starvla_path() + from deployment.model_server.policy_wrapper import PolicyServerWrapper + + return PolicyServerWrapper + + +def _profiler_stage(profiler: Any, name: str) -> Any: + from contextlib import nullcontext + + return profiler.time(name) if profiler is not None else nullcontext() + + +def _load_rl_games_action_decode() -> tuple[Any, Any, Any]: + _ensure_starvla_path() + from deployment.model_server.rl_games_action_decode import ( + decode_rl_games_actions, + resolve_asterix_action_decode_spec, + resolve_deadly_action_decode_spec, + ) + + return decode_rl_games_actions, resolve_deadly_action_decode_spec, resolve_asterix_action_decode_spec + + +class LiveStarVlaWrapper: + """In-process stand-in for ``PolicyServerWrapper`` over a *live* framework. + + During training the trainer already holds the model in memory + (``accelerator.unwrap_model(self.model)`` — the same object eval_core calls). + This wrapper exposes only the rl_games-mode surface ``StarVlaPolicyRunner`` + uses — ``predict_action`` (framework forward + rl_games decode), + ``reset_memory`` passthrough, and the ``_model_cfg`` attribute — so no + checkpoint reload is needed. The disk-backed ``PolicyNormProcessor`` is never + built because rl_games decoding ignores un-normalization stats. + """ + + def __init__( + self, + *, + framework: Any, + model_cfg: dict[str, Any], + env_name: str, + rl_games_action_env_dim: int | None = None, + gymnasium_action_space_type: str = "discrete", + action_layout: str | None = None, + multibinary_threshold: float | None = None, + ) -> None: + self._framework = framework + self._model_cfg = model_cfg + self._rl_games_env_name = str(env_name) + self._rl_games_action_env_dim = rl_games_action_env_dim + self._gymnasium_action_space_type = gymnasium_action_space_type + ( + self._decode_rl_games_actions, + resolve_deadly_action_decode_spec, + resolve_asterix_action_decode_spec, + ) = _load_rl_games_action_decode() + self._action_layout = action_layout + self._multibinary_threshold = multibinary_threshold + if self._rl_games_env_name == "deadly_corridor": + self._action_layout, self._multibinary_threshold = resolve_deadly_action_decode_spec( + model_cfg, + action_layout=action_layout, + multibinary_threshold=multibinary_threshold, + ) + elif self._rl_games_env_name == "asterix": + self._action_layout = resolve_asterix_action_decode_spec( + model_cfg, + action_layout=action_layout, + ) + + def reset_memory(self, slot_id: int | None = None) -> None: + reset = getattr(self._framework, "reset_memory", None) + if callable(reset): + reset(slot_id) + + def predict_action( + self, + examples: list[dict[str, Any]], + unnorm_key: str | None = None, + **kwargs: Any, + ) -> dict[str, Any]: + # unnorm_key is unused in rl_games mode; kept for interface parity. + del unnorm_key + profiler = kwargs["profiler"] if "profiler" in kwargs else None + out = self._framework.predict_action(examples=examples, **kwargs) + normalized = np.asarray(out["normalized_actions"]) # (B, T, D) + decode_kwargs: dict[str, Any] = {} + if self._rl_games_env_name == "gymnasium": + decode_kwargs["action_env_dim"] = self._rl_games_action_env_dim + if self._gymnasium_action_space_type == "box": + decode_kwargs["gymnasium_action_space_type"] = "box" + with _profiler_stage(profiler, "starvla_wrapper_rl_games_decode_ms"): + return self._decode_rl_games_actions( + normalized_actions=normalized, + env_name=self._rl_games_env_name, + deadly_action_layout=( + self._action_layout + if self._rl_games_env_name == "deadly_corridor" + else None + ), + deadly_multibinary_threshold=( + self._multibinary_threshold + if self._rl_games_env_name == "deadly_corridor" + else None + ), + asterix_action_layout=( + self._action_layout + if self._rl_games_env_name == "asterix" + else None + ), + **decode_kwargs, + ) + + +_LEGACY_GYMNASIUM_TASK_NAMES = { + "ant_rgb_state": "ant", + "half_cheetah_rgb_state": "half_cheetah", + "hopper_rgb_state": "hopper", + "humanoid_rgb_state": "humanoid", + "inverted_pendulum_rgb_state": "inverted_pendulum", + "swimmer_rgb_state": "swimmer", + "walker2d_rgb_state": "walker2d", +} + + +def _canonical_gymnasium_contract_namespace( + contract: Mapping[str, Any], +) -> dict[str, Any]: + canonical = dict(contract) + task_name = canonical["task_name"] + if task_name in _LEGACY_GYMNASIUM_TASK_NAMES: + canonical["task_name"] = _LEGACY_GYMNASIUM_TASK_NAMES[task_name] + if canonical["env_id"] == "LatencyBench/HopperRgbState-v0": + canonical["env_id"] = "LatencyBench/Hopper-v0" + canonical["registration_imports"] = [ + "latency_bench.envs.gymnasium_hopper" + if module == "latency_bench.envs.gymnasium_hopper_rgb_state" + else module + for module in canonical["registration_imports"] + ] + return canonical + + +def _validate_gymnasium_starvla_contract( + *, + env_cfg: Mapping[str, Any], + policy_cfg: Mapping[str, Any], + model_cfg: Mapping[str, Any], + manifest: Mapping[str, Any], +) -> None: + eval_contract = gymnasium_task_contract(env_cfg) + manifest_task = manifest.get("gymnasium_task") + expected = policy_cfg.get( + "gymnasium_training_task_contract", manifest_task or eval_contract + ) + comparable_eval_contract = {**eval_contract, "make_kwargs": expected["make_kwargs"]} + if _canonical_gymnasium_contract_namespace( + comparable_eval_contract + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA training contract or dataset manifest" + ) + if manifest.get("integration_name", "gymnasium") != "gymnasium": + raise ValueError("StarVLA task manifest is not a Gymnasium handoff") + if manifest_task is not None: + if _canonical_gymnasium_contract_namespace( + manifest_task + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA dataset manifest" + ) + model_contract = model_cfg["datasets"]["vla_data"].get("gymnasium_task_contract") + if model_contract is not None: + if _canonical_gymnasium_contract_namespace( + model_contract + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA model config" + ) + action_space = gymnasium_action_space_contract(env_cfg) + action_layout = str(policy_cfg.get("action_layout", "") or "").strip().lower() + is_asterix_factorized = ( + str(env_cfg.get("task_name", "")) == "asterix" + and action_layout in {"factorized_6", "factorized6", "asterix_factorized_6", "asterix_factorized6"} + ) + if not is_asterix_factorized and manifest["active_action_dim"] != len(action_space["labels"]): + raise ValueError( + "StarVLA dataset active_action_dim does not match its Gymnasium action catalog" + ) + if ( + model_cfg["framework"]["action_model"]["action_env_dim"] + != manifest["active_action_dim"] + ): + raise ValueError( + "StarVLA model action_env_dim does not match the dataset manifest" + ) + model_uses_state = bool(model_cfg["datasets"]["vla_data"]["include_state"]) + manifest_has_state_metadata = ( + "uses_state" in manifest or "state_labels" in manifest + ) + manifest_uses_state = bool(manifest.get("uses_state", model_uses_state)) + if manifest_has_state_metadata: + if policy_cfg.get("state_source") != "transport" and manifest_uses_state != ("state_space" in expected): + raise ValueError( + "StarVLA dataset uses_state does not match the Gymnasium state space" + ) + if manifest_uses_state != model_uses_state: + raise ValueError( + "StarVLA dataset uses_state does not match the model include_state" + ) + if manifest_has_state_metadata and manifest_uses_state: + state_labels = manifest["state_labels"] + expected_state_labels = expected["state_space"]["labels"] if policy_cfg.get("state_source") != "transport" else state_labels + if state_labels != expected_state_labels: + raise ValueError( + "StarVLA dataset state_labels do not match the Gymnasium state space" + ) + if manifest["state_dim"] != len(state_labels): + raise ValueError( + "StarVLA dataset state_dim does not match its state_labels" + ) + if ( + model_cfg["framework"]["action_model"]["state_dim"] + != manifest["state_dim"] + ): + raise ValueError( + "StarVLA model state_dim does not match the dataset manifest" + ) + if not manifest["state_normalization"]: + raise ValueError( + "StarVLA state-enabled dataset manifest is missing state_normalization" + ) + + +def _starvla_runner_kwargs( + config: dict[str, Any], + action_resolver: ActionResolver, + model_cfg: Mapping[str, Any] | None, + *, + base_prompt: str | None, +) -> dict[str, Any]: + """Resolve task and input settings shared by checkpoint and resident models.""" + env_cfg = config["env"] + policy_cfg = config["policy"] + if env_cfg["name"] == "gymnasium": + task_manifest = json.loads( + Path(policy_cfg["task_manifest_path"]).read_text(encoding="utf-8") + ) + _validate_gymnasium_starvla_contract( + env_cfg=env_cfg, + policy_cfg=policy_cfg, + model_cfg=model_cfg, + manifest=task_manifest, + ) + semantic_env_name = env_cfg["task_name"] + action_refs = env_cfg.get("action_order", []) + base_prompt = env_cfg["base_prompt"] + state_normalization = task_manifest.get("state_normalization") + else: + semantic_env_name = env_cfg["name"] + action_refs = policy_cfg.get("actions", action_resolver.default_action_refs()) + state_normalization = policy_cfg["state_normalization"] if "state_normalization" in policy_cfg else None + return dict( + unnorm_key=policy_cfg.get("unnorm_key"), + env_name=semantic_env_name, + action_resolver=action_resolver, + action_refs=action_refs, + latency_prompt_map=( + load_latency_prompt_map(policy_cfg["latency_prompt_map_path"]) + if "latency_prompt_map_path" in policy_cfg + else None + ), + base_prompt=base_prompt, + latency_prompt_key=policy_cfg.get("latency_prompt_key"), + prompt_mode=policy_cfg.get("prompt_mode"), + obs_resize=tuple(env_cfg["obs_resize"]) if env_cfg.get("obs_resize") else None, + image_transform_config=policy_cfg.get("image_transform_config"), + observation_stride_raw_frames=_observation_stride_raw_frames(config), + model_cfg=model_cfg, + state_normalization=state_normalization, + state_source=policy_cfg["state_source"] if "state_source" in policy_cfg else None, + ) + + +def build_starvla_policy( + config: dict[str, Any], + action_resolver: ActionResolver, +) -> PolicyRunner: + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + _ensure_starvla_path() + from latency_bench.policy.starvla_tasks import build_task_starvla_policy + + return build_task_starvla_policy(config) + env_cfg = config["env"] + integration_env_name = env_cfg["name"] + model_cfg = ( + _load_starvla_model_config(policy_cfg["model_config_path"]) + if integration_env_name == "gymnasium" or "model_config_path" in policy_cfg + else None + ) + runner_kwargs = _starvla_runner_kwargs( + config, action_resolver, model_cfg, base_prompt=env_cfg.get("base_prompt") + ) + wrapper_cls = _load_policy_wrapper_class() + wrapper_kwargs: dict[str, Any] = dict( + ckpt_path=policy_cfg["checkpoint_path"], + device=policy_cfg["device"], + use_bf16=True, + unnorm_key=runner_kwargs["unnorm_key"], + action_output_mode=( + policy_cfg["action_output_mode"] + if "action_output_mode" in policy_cfg + else "rl_games" + ), + rl_games_env_name=integration_env_name, + rl_games_action_layout=( + policy_cfg["action_layout"] if "action_layout" in policy_cfg else None + ), + rl_games_multibinary_threshold=( + policy_cfg["multibinary_threshold"] + if "multibinary_threshold" in policy_cfg + else None + ), + ) + if "backbone_path" in policy_cfg: + wrapper_kwargs["backbone_path"] = policy_cfg["backbone_path"] + if integration_env_name == "gymnasium": + action_space = gymnasium_action_space_contract(env_cfg) + wrapper_kwargs["rl_games_action_env_dim"] = len(action_space["labels"]) + if action_space["type"] == "box": + wrapper_kwargs["rl_games_gymnasium_action_space_type"] = "box" + wrapper_kwargs["rl_games_env_name"] = integration_env_name + wrapper = wrapper_cls(**wrapper_kwargs) + return StarVlaPolicyRunner( + wrapper=wrapper, + checkpoint_path=policy_cfg["checkpoint_path"], + device=policy_cfg["device"], + **runner_kwargs, + image_views_info_key=( + policy_cfg["image_views_info_key"] + if "image_views_info_key" in policy_cfg + else None + ), + action_output_type=( + policy_cfg["action_output_type"] + if "action_output_type" in policy_cfg + else None + ), + ) + + +def build_live_starvla_policy( + *, + framework: Any, + model_cfg: dict[str, Any], + config: dict[str, Any], + action_resolver: ActionResolver | None = None, +) -> PolicyRunner: + """Build a StarVLA policy around a *live* in-memory framework (no reload). + + Mirrors ``build_starvla_policy`` but swaps the ckpt-loading + ``PolicyServerWrapper`` for :class:`LiveStarVlaWrapper`, so the trainer's + resident model is evaluated directly. ``model_cfg`` is the in-memory model + config (e.g. ``read_mode_config`` output) the wrapper would otherwise read + from disk. + """ + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + from latency_bench.policy.starvla_tasks import TaskStarVlaPolicyRunner + + contract = json.loads( + Path(policy_cfg["task_contract_path"]).read_text(encoding="utf-8") + ) + return TaskStarVlaPolicyRunner( + framework, + policy_config=policy_cfg, + model_config=model_cfg, + contract=contract, + ) + + env_cfg = config["env"] + integration_env_name = env_cfg["name"] + normalized_model_cfg = ( + _normalized_model_cfg(model_cfg) + if integration_env_name == "gymnasium" + else None + ) + # Resident evaluation historically takes non-Gymnasium prompts from the map. + runner_kwargs = _starvla_runner_kwargs( + config, action_resolver, normalized_model_cfg, base_prompt=None + ) + wrapper_kwargs: dict[str, Any] = dict( + framework=framework, + model_cfg=model_cfg, + env_name=integration_env_name, + action_layout=policy_cfg["action_layout"] if "action_layout" in policy_cfg else None, + multibinary_threshold=( + policy_cfg["multibinary_threshold"] + if "multibinary_threshold" in policy_cfg + else None + ), + ) + if integration_env_name == "gymnasium": + action_space = gymnasium_action_space_contract(env_cfg) + wrapper_kwargs["rl_games_action_env_dim"] = len(action_space["labels"]) + if action_space["type"] == "box": + wrapper_kwargs["gymnasium_action_space_type"] = "box" + wrapper_kwargs["env_name"] = integration_env_name + wrapper = LiveStarVlaWrapper(**wrapper_kwargs) + return StarVlaPolicyRunner( + wrapper=wrapper, + checkpoint_path=policy_cfg.get("checkpoint_path", ""), + device=policy_cfg.get("device", "cuda"), + **runner_kwargs, + ) + + +__all__ = [ + "LiveStarVlaWrapper", + "StarVlaPolicyRunner", + "apply_starvla_model_input_config", + "build_live_starvla_policy", + "build_starvla_policy", + "decode_starvla_action", + "observation_data_to_hwc_uint8_frames", + "prepare_starvla_checkpoint_input_config", +] diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py new file mode 100644 index 0000000000000000000000000000000000000000..4a5cc03813f1bc524a30a7b6b7292a4f4fca85d4 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py @@ -0,0 +1,92 @@ +"""StarVLA inference using the task's training observation/action contract.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import numpy as np +from PIL import Image + +from latency_bench.core.types import Action, Observation, PolicyOutput +from latency_bench.data.starvla_tasks import denormalize, normalize +from latency_bench.policy.base import PolicyRunner + + +class TaskStarVlaPolicyRunner(PolicyRunner): + """Map task RGB/state into a StarVLA model and decode its action chunk.""" + + def __init__(self, framework, *, policy_config: dict, model_config: dict, contract: dict): + self.framework = framework + self.policy_config = policy_config + self.model_config = model_config + self.contract = contract + + def _example(self, observation: Observation) -> dict: + cfg = self.policy_config + state = normalize( + observation.metadata[cfg["state_info_key"]], + self.contract["normalization"]["state"], + ).reshape(1, self.contract["state_dim"]) + data_cfg = self.model_config["datasets"]["vla_data"] + height, width = data_cfg["obs_image_size"] + images = [ + Image.fromarray(frame).resize((width, height)) + for frame in observation.metadata[cfg["image_views_info_key"]] + ] + if data_cfg["image_mode"] == "stitch_views": + from starVLA.training.trainer_utils.trainer_tools import stitch_frames + + # MIKASA's two simultaneous views form one Wan observation, not a video. + images = [stitch_frames(images, grid=data_cfg["stitch_grid"], size=(width, height))] + example = {"image": images, "state": state, "lang": self.contract["prompt"]} + if "action_prefix" in observation.metadata: + example["action_prefix"] = normalize( + observation.metadata["action_prefix"], + self.contract["normalization"]["action"], + ) + example["action_prefix_mask"] = observation.metadata["action_prefix_mask"] + return example + + def predict(self, observation: Observation) -> PolicyOutput: + return self.predict_batch([observation])[0] + + def predict_batch(self, observations: list[Observation]) -> list[PolicyOutput]: + prediction = self.framework.predict_action( + examples=[self._example(observation) for observation in observations] + ) + actions = denormalize( + prediction["normalized_actions"], self.contract["normalization"]["action"] + ) + # Prefix heads were excluded from the loss; retain the frozen controller plan. + for chunk, observation in zip(actions, observations): + if "action_prefix" in observation.metadata: + mask = observation.metadata["action_prefix_mask"] + chunk[mask] = observation.metadata["action_prefix"][mask] + return [ + PolicyOutput( + action=Action(value=chunk[0].tolist(), name="task_command"), + action_chunk=chunk, + raw_output=chunk.tolist(), + metadata={"policy_type": "starvla", "task": self.contract["task"]}, + ) + for chunk in actions + ] + + +def build_task_starvla_policy(config: dict) -> TaskStarVlaPolicyRunner: + # StarVLA and torch are optional in the simulator process; workers own them. + import torch + from starVLA.model.framework.base_framework import baseframework + from starVLA.model.framework.share_tools import read_mode_config + + cfg = config["policy"] + model_config, _ = read_mode_config(cfg["checkpoint_path"]) + framework = baseframework.from_pretrained( + cfg["checkpoint_path"], backbone_path=cfg["backbone_path"] + ) + framework = framework.to(device=cfg["device"], dtype=torch.bfloat16).eval() + contract = json.loads(Path(cfg["task_contract_path"]).read_text(encoding="utf-8")) + return TaskStarVlaPolicyRunner( + framework, policy_config=cfg, model_config=model_config, contract=contract + ) diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json new file mode 100644 index 0000000000000000000000000000000000000000..371542eee194a26ff888c211317abe89050b38e0 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json @@ -0,0 +1,105 @@ +{ + "condition": "profile-latency", + "executor_mode": "simulated", + "latency_method": "temporal", + "profile_source": "originalRTX3090immutableprofiles", + "episodes_per_checkpoint": 100, + "total_episodes": 400, + "rounds": [ + [ + "flappy", + "deadly_corridor" + ], + [ + "ant", + "intercept" + ] + ], + "physical_gpu_assignments": { + "flappy": 2, + "deadly_corridor": 3, + "ant": 2, + "intercept": 3 + }, + "single_gpu_per_job": true, + "round2_requires_both_round1_complete": true, + "latency_seed": 271828, + "tasks": { + "flappy": { + "gpu": 2, + "seed_start": 1000000, + "seed_end": 1000099, + "env_fps": 10, + "obs_fps": 10, + "max_raw_steps": 3600, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 75.87417450998383, + "profile": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/flappy/instance_a5037b165aa0cedc/profile.json", + "sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "deadly_corridor": { + "gpu": 3, + "seed_start": 1000000, + "seed_end": 1000099, + "env_fps": 35, + "obs_fps": 8.75, + "max_raw_steps": 3600, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 73.69250777493353, + "profile": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/deadly_corridor/instance_a5037b165aa0cedc/profile.json", + "sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "ant": { + "gpu": 2, + "seed_start": 42, + "seed_end": 141, + "env_fps": 10, + "obs_fps": 10, + "max_raw_steps": 1000, + "parallel_envs": 16, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 90.56460638563993, + "profile": "/home/ubuntu/lzj/mean-profiling/reference/checkpoint-configs/latency-aware/ant/small-policy/sample-factory-qwenoft-rtx3090-v1/profile/profile.json", + "sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "intercept": { + "gpu": 3, + "seed_start": 4242424242, + "seed_end": 4242424341, + "env_fps": 20, + "obs_fps": 20, + "max_raw_steps": 60, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0", + "profile": { + "mean_ms": 99.05021289731565, + "profile": "/home/ubuntu/lzj/profiles/intercept-published/profiles/qwenoft/1x-rtx3090/mikasa_intercept_grab_fast/instance_3a0d42681a03715c/profile.json", + "sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + } + } +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/execution_audit.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/execution_audit.json new file mode 100644 index 0000000000000000000000000000000000000000..f31796ff7f863ded0d604d536b2e134831e3e066 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/execution_audit.json @@ -0,0 +1,12 @@ +{ + "issued_action_records": 3753, + "applied_action_records": 3673, + "dropped_action_records": 0, + "nonnoop_issued_records": 3753, + "finite_action_values": true, + "latency_sample_count": 3753, + "latency_mean_ms": 74.01999621872471, + "latency_std_ms": 5.5537519652567635, + "latency_p95_ms": 89.54825982614612, + "latency_p99_ms": 95.97310052501227 +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/files.sha256.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/files.sha256.json new file mode 100644 index 0000000000000000000000000000000000000000..dba4b5c6a7651a34c5a6868a9277dfb3c05de805 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/files.sha256.json @@ -0,0 +1,130 @@ +{ + "REPORT.md": { + "bytes": 2163, + "sha256": "b2fd63b1cde2a415b2d77daf0184ce5ec3b6ada1aa199c21941fb04031f77db6" + }, + "all_episodes.csv": { + "bytes": 25198, + "sha256": "bd41f35a464250ed6f9bc16e466aa9f1b55a48ac29072bb0ec3559a24129edac" + }, + "comparison.csv": { + "bytes": 438, + "sha256": "a0c7cf191395dd851bd6222bdf61427f15782fea2db4416ddc072a5f5dc8a861" + }, + "comparison.json": { + "bytes": 7826, + "sha256": "0d5dd042439466aee84cd0d96c31a27a951e57965a7e468cb73ec07883f1f751" + }, + "episodes.csv": { + "bytes": 5665, + "sha256": "69e8140b74f407803e15adcf4412ccb7c5a51abd0a9c274ee55a6c51b4e9245d" + }, + "eval_config.yaml": { + "bytes": 3299, + "sha256": "cdebec7e4e48415f530a665a08f2dd98808381ce1d871f75d25385e4a38b54ec" + }, + "evaluation-code/batched_simulated.py": { + "bytes": 31282, + "sha256": "b901f966d911feab7962a32f21095cb90f7880121811f2b4eab2193afe1381db" + }, + "evaluation-code/deadly-compatibility.patch": { + "bytes": 4570, + "sha256": "623676cc4542b1eab6c9395b163b369ddc605353c1de02d17d8f713167ee07fa" + }, + "evaluation-code/deadly_corridor.py": { + "bytes": 17902, + "sha256": "47f7bc65cba9853e66d79ed2a28f844bd2a094f1285458be166045f2db1690dc" + }, + "evaluation-code/decision_action_history.py": { + "bytes": 2746, + "sha256": "14a9d223e775745b6c402dbce9e2a50a1c3f7b5b9fe528150ef8689126fe97cf" + }, + "evaluation-code/eval_driver.py": { + "bytes": 9133, + "sha256": "330030270fbb695bc5f14037ef7349650bd20c53c881c1159ee55ea066408d9e" + }, + "evaluation-code/mikasa_evaluate.py": { + "bytes": 11466, + "sha256": "6cf9ffee25fcfd6f3255c520fc544c48ff2c8f8912e5369c2410a709820c4ffd" + }, + "evaluation-code/starvla.py": { + "bytes": 48378, + "sha256": "6d9988f3a28d39e46c2f6e80da85edebc42cafa629a2b9f75000414324c1065a" + }, + "evaluation-code/starvla_tasks.py": { + "bytes": 4029, + "sha256": "3fc74169d1554d9dc3358ed85e450cca75eb69bc1fff85284c1054a605633a52" + }, + "evaluation-plan.json": { + "bytes": 4698, + "sha256": "b758a5fb72dcdef49d025e2fd168d024ebd8b18b2b00125145b3cde38b16a318" + }, + "execution_audit.json": { + "bytes": 357, + "sha256": "cd445b9926cd0c8b4da3e48d2137bf80ae44e44a0cef30563d3c7a1f2a31b920" + }, + "profile/latency_burst_model.json": { + "bytes": 26211, + "sha256": "2e52774207a61ccc9902ec6c576d75ae0f60925611f1f8a9fc55c3109f7ff34c" + }, + "profile/latency_distribution.json": { + "bytes": 25069, + "sha256": "c6c7926af60218a9b5f8b0fbabb7f10466d224dd739171349650520ef1da9fc8" + }, + "profile/profile.json": { + "bytes": 2115, + "sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba" + }, + "provenance.json": { + "bytes": 3856, + "sha256": "a34689af61c380a5e7a64d3bf25d921a44967aee0c2220c7ede4a36a97b9cae5" + }, + "queue_eval_latency_profile_sample.json": { + "bytes": 3848, + "sha256": "dd4c4be8669838a85c0d7e06c5ccf20f133bb2fb2cacdd9b905972343bbd03b6" + }, + "raw-records/actions.jsonl.gz": { + "bytes": 522227, + "sha256": "53f0b7bb59fb99cdd2917642c9881cb1443f897c78e20f1567929b4b232680fd" + }, + "raw-records/e2e_latencies.jsonl.gz": { + "bytes": 70931, + "sha256": "9c6b0b88fbfb5f6b37ea28ac5c6701c9abc2fa89b2d6b36b7fe747b6c1cf4398" + }, + "raw-records/episode_metrics.jsonl.gz": { + "bytes": 7835, + "sha256": "c96daa6d0d6f5051f6a20701c54ba21ebf5e425740028916cd553e2c8bd7cbd2" + }, + "raw-records/infer_latencies.jsonl.gz": { + "bytes": 60189, + "sha256": "42a899945c8383a4d86d62b77692c1a2f4022a3cec41d5aada23bdc8f45df931" + }, + "raw-records/latencies.jsonl.gz": { + "bytes": 60183, + "sha256": "12d682b1436b118a5058ec2364e2e7ee9cf45035224d7e04e0723689f476b01b" + }, + "raw-records/observation_attempts.jsonl.gz": { + "bytes": 47, + "sha256": "6ce7d0c6fa3086525b7ba82526a5db7d4a26d33a64b16875bdd644c436069469" + }, + "raw-records/queue_eval_results.jsonl.gz": { + "bytes": 1468, + "sha256": "c717edfdd81c255faec3f3f765ecc4d378f353e459e1124a02b3fb070c71a10a" + }, + "raw-records/steps.jsonl.gz": { + "bytes": 495220, + "sha256": "f6a4d4bf35f367394f0754bff9ada16daa229726d0ccbbfede23411cffbdbd74" + }, + "resolved_config.yaml": { + "bytes": 3367, + "sha256": "0ac444e55dcf84745badbb2614d75fc38b23abd09c87aee5fbc2ecba991854f0" + }, + "statistics.json": { + "bytes": 1288, + "sha256": "39d572fde494928e335d2999b2731fc2705e3d98bb260af101e2d1eef2089467" + }, + "stdout.log": { + "bytes": 59183, + "sha256": "3a511f78f66916995fc35c64092732d1e1499a092cb4db441e003c70a6273c2f" + } +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json new file mode 100644 index 0000000000000000000000000000000000000000..d848adf2fd31cb5b54506c688e0e399f75d4401e --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json @@ -0,0 +1,1350 @@ +{ + "burst_dwell_distribution": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.4200000000000004, + 2.889999999999997, + 4.360000000000003, + 5.83, + 7.300000000000006, + 8.769999999999992, + 10.239999999999998, + 11.709999999999996, + 13.180000000000001, + 14.649999999999999, + 16.120000000000005, + 17.59, + 19.06, + 20.529999999999994, + 22.0, + 22.840000000000003, + 23.68, + 24.52, + 25.360000000000003, + 26.200000000000006, + 27.040000000000006, + 27.879999999999995, + 28.719999999999995, + 29.56, + 30.400000000000002, + 31.239999999999995, + 32.08, + 32.92, + 33.760000000000005, + 34.449999999999996, + 35.08, + 35.71, + 36.34, + 36.97, + 37.599999999999994, + 38.23, + 38.86, + 39.489999999999995, + 40.12, + 40.75, + 41.38, + 42.010000000000005, + 42.64, + 43.900000000000006, + 46.000000000000014, + 48.099999999999994, + 50.19999999999998, + 52.29999999999999, + 54.4, + 56.50000000000001, + 58.59999999999999, + 60.699999999999996, + 62.800000000000004, + 64.9, + 67.0, + 69.1, + 71.20000000000002, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0, + 73.0 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "burst_dwell_lengths": [ + 1, + 1, + 1, + 22, + 34, + 43, + 73 + ], + "burst_merge_gap_records": 30, + "burst_rank_processes": [ + { + "draw_count": 175, + "dwell_length_spearman_rho": -0.7307692307692307, + "level_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.8229808807373, + 87.92940280914307, + 88.05356172561646, + 88.17772064208984, + 88.30187955856323, + 88.42603847503662, + 88.55019739151001, + 88.6743563079834, + 88.79851522445679, + 88.92267414093017, + 89.04683305740356, + 89.17099197387695, + 89.29515089035034, + 89.41930980682373, + 89.54346872329712, + 89.63789928436279, + 89.71003357887268, + 89.78216787338256, + 89.85430216789246, + 89.92643646240235, + 89.99857075691223, + 90.07070505142212, + 90.142839345932, + 90.2149736404419, + 90.28710793495178, + 90.35924222946167, + 90.43137652397155, + 90.50351081848144, + 90.57564511299134, + 90.63903362001692, + 90.68055765833174, + 90.72208169664655, + 90.76360573496137, + 90.80512977327619, + 90.84665381159101, + 90.88817784990583, + 90.92970188822065, + 90.97122592653547, + 91.01274996485029, + 91.0542740031651, + 91.09579804147992, + 91.13732207979474, + 91.17884611810956, + 91.22037015642438, + 91.26130477275167, + 91.30223938907895, + 91.34317400540624, + 91.38410862173353, + 91.42504323806081, + 91.4659778543881, + 91.50691247071538, + 91.54784708704267, + 91.58878170336996, + 91.62971631969724, + 91.67065093602453, + 91.71158555235182, + 91.7525201686791, + 91.79345478500639, + 91.98543370366096, + 92.23783034324646, + 92.49022698283196, + 92.74262362241745, + 92.99502026200294, + 93.24741690158844, + 93.49981354117394, + 93.75221018075943, + 94.00460682034492, + 94.25700345993042, + 94.50940009951591, + 94.76179673910141, + 95.01419337868691, + 95.2665900182724, + 95.53913187503815, + 95.83853402137757, + 96.13793616771699, + 96.43733831405639, + 96.73674046039581, + 97.03614260673523, + 97.33554475307464, + 97.63494689941406, + 97.93434904575348, + 98.2337511920929, + 98.53315333843231, + 98.83255548477173, + 99.13195763111115, + 99.43135977745057, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "level_residual_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -5.004209995269775, + -4.747487674440656, + -4.234043032782417, + -3.7205983911241773, + -3.207153749465938, + -2.6937091078076985, + -2.4288561548505454, + -2.412594890594478, + -2.3963336263384107, + -2.3800723620823434, + -2.363811097826276, + -2.3467164397239686, + -2.328788387775421, + -2.310860335826874, + -2.2929322838783266, + -2.275004231929779, + -2.262637364864349, + -2.2558316826820373, + -2.2490260004997253, + -2.2422203183174134, + -2.2354146361351015, + -2.195725427355084, + -2.123152691977363, + -2.050579956599641, + -1.9780072212219195, + -1.9054344858441978, + -1.8612872191837795, + -1.8455654212406643, + -1.829843623297549, + -1.8141218253544338, + -1.7984000274113185, + -1.7361504282270102, + -1.627373027801509, + -1.5185956273760084, + -1.4098182269505077, + -1.3010408265250062, + -1.2078239304678775, + -1.1301675387791204, + -1.0525111470903639, + -0.9748547554016074, + -0.8971983637128501, + -0.772533151081625, + -0.6008591175079296, + -0.42918508393423593, + -0.25751105036054067, + -0.08583701678684841, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.03896442651748643, + 0.1168932795524593, + 0.19482213258743286, + 0.27275098562240574, + 0.3506798386573793, + 0.41955420970916735, + 0.4793740987777712, + 0.5391939878463745, + 0.5990138769149783, + 0.6588337659835817, + 0.7386345624923714, + 0.8384162664413447, + 0.9381979703903198, + 1.0379796743392942, + 1.1377613782882683, + 1.2261867301804683, + 1.3032557300158931, + 1.380324729851317, + 1.457393729686741, + 1.5344627295221658, + 1.663917820794249, + 1.8457590035029887, + 2.0276001862117283, + 2.209441368920471, + 2.3912825516292076, + 2.5239672115870935, + 2.607495348794126, + 2.691023486001157, + 2.774551623208188, + 2.858079760415219, + 3.110280445643842, + 3.5311536788940487, + 3.9520269121442553, + 4.372900145394462, + 4.793773378644676, + 5.087223434448243, + 5.253250312805173, + 5.419277191162109, + 5.585304069519042, + 5.751330947875975, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443, + 5.834344387054443 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "severity": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09090909090909091, + 0.09103594080338266, + 0.09118393234672305, + 0.09133192389006342, + 0.09147991543340381, + 0.09162790697674418, + 0.09177589852008457, + 0.09192389006342495, + 0.09207188160676533, + 0.09221987315010571, + 0.09236786469344609, + 0.09251585623678647, + 0.09266384778012685, + 0.09281183932346723, + 0.0929598308668076, + 0.09313794201975151, + 0.09333864287989806, + 0.09353934374004459, + 0.09374004460019114, + 0.09394074546033769, + 0.09414144632048423, + 0.09434214718063078, + 0.09454284804077731, + 0.09474354890092386, + 0.0949442497610704, + 0.09514495062121694, + 0.09534565148136348, + 0.09554635234151003, + 0.09574705320165658, + 0.0963255439161966, + 0.09784850926672038, + 0.09937147461724416, + 0.10089443996776792, + 0.1024174053182917, + 0.10394037066881547, + 0.10546333601933924, + 0.10698630136986301, + 0.10850926672038678, + 0.11003223207091055, + 0.11155519742143433, + 0.11307816277195809, + 0.11460112812248187, + 0.11612409347300563, + 0.11764705882352941, + 0.1794117647058826, + 0.24117647058823538, + 0.30294117647058816, + 0.3647058823529414, + 0.4264705882352946, + 0.4882352941176473, + 0.5499999999999998, + 0.6117647058823525, + 0.6735294117647058, + 0.735294117647059, + 0.7970588235294114, + 0.8588235294117645, + 0.9205882352941178, + 0.982352941176471, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "spike_count": 20 + } + ], + "burst_slot_rank_templates": [ + { + "dwell_length": 22, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 1, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 1, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 73, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 43, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 1, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 34, + "slot_to_rank": { + "0": 0 + } + } + ], + "model_type": "hidden_regime", + "pre_worker_time_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 0.3265700340270996, + 0.3265700340270996, + 0.3265700340270996, + 0.3266348361968994, + 0.33209981918334963, + 0.33750788784027097, + 0.34137294578552246, + 0.34554224491119384, + 0.35453883743286135, + 0.3629479331970215, + 0.3648665885925293, + 0.3666981601715088, + 0.36780080795288084, + 0.3758285236358643, + 0.3782614135742188, + 0.38045952796936033, + 0.3833635330200195, + 0.3850455951690674, + 0.38792415618896486, + 0.3914265727996826, + 0.3956317520141602, + 0.399367094039917, + 0.403819580078125, + 0.40693872451782226, + 0.408432674407959, + 0.41067914009094236, + 0.4121252059936523, + 0.41337799072265624, + 0.4151019287109375, + 0.4162428665161133, + 0.41752376556396487, + 0.4190552234649658, + 0.4234558296203613, + 0.42556562423706057, + 0.4273490905761719, + 0.43012102127075197, + 0.4335489273071289, + 0.4404067230224609, + 0.4457015228271484, + 0.45634961128234863, + 0.4604278755187988, + 0.46451854705810547, + 0.4700064468383789, + 0.4807524108886719, + 0.4847769546508789, + 0.49133274078369144, + 0.4975248336791992, + 0.5144282913208008, + 0.5163888931274414, + 0.5209720325469971, + 0.5230713844299316, + 0.5237809181213379, + 0.5248561859130859, + 0.5257608032226563, + 0.5276785278320313, + 0.5288626766204834, + 0.5306490898132324, + 0.532410717010498, + 0.5346214103698731, + 0.5375547790527344, + 0.5419289016723633, + 0.5428438186645508, + 0.5441124725341797, + 0.5461745071411133, + 0.5476796531677246, + 0.5515498542785645, + 0.5529769897460938, + 0.5566401863098145, + 0.559661808013916, + 0.5664050483703613, + 0.5725405502319335, + 0.5828640460968015, + 0.6016114807128906, + 0.6067960834503173, + 0.6122808837890625, + 0.6175057888031008, + 0.6284547805786135, + 0.6340862274169922, + 0.6432880783081055, + 0.6471994781494141, + 0.6558640098571776, + 0.6636908531188965, + 0.6708590888977051, + 0.6754708290100098, + 0.6785876846313477, + 0.68269926071167, + 0.6843280792236328, + 0.6872262668609619, + 0.6915739631652832, + 0.6929038906097412, + 0.6961851119995117, + 0.6998775482177735, + 0.704066162109375, + 0.7050773525238037, + 0.7067256546020507, + 0.7119001483917237, + 0.7176529884338378, + 0.7246382045745848, + 0.7501127815246582, + 0.7531718635559083, + 0.7789012145996096, + 0.9071441173553495, + 1.9022739410400396, + 1.9581221103668216, + 2.016346549987793, + 2.0252209758758544, + 2.0309144973754885, + 2.03552077293396, + 2.0404930877685548, + 2.048240556716919, + 2.0645600891113283, + 2.065875786781311, + 2.0663459587097166, + 2.0667151165008546, + 2.080239059448242, + 2.09495373725891, + 2.0978348731994627, + 2.0999701976776124, + 2.1539656677246093, + 2.209874011993405, + 2.2105369567871094, + 2.2105369567871094, + 2.2105369567871094 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "regime_step_counts": { + "burst": 175, + "calm": 331 + }, + "regime_transition_counts": { + "burst": { + "burst": 168, + "calm": 7 + }, + "calm": { + "burst": 7, + "calm": 319 + } + }, + "reset_scope": "session", + "schema_version": 12, + "spike_median_multiplier": 1.25, + "spike_threshold_ms_by_worker_slot": { + "0": 85.61681151390076 + }, + "worker_count": 1 +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json new file mode 100644 index 0000000000000000000000000000000000000000..0b73af821d6f723577e5aa38f7627e57466517cc --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json @@ -0,0 +1,1027 @@ +{ + "schema_version": 3, + "worker_slots": { + "0": { + "all": { + "count": 506, + "observation_to_action_latency_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 68.33836698532104, + 68.33836698532104, + 68.33836698532104, + 68.34015856647491, + 68.49124857711791, + 68.63975530719758, + 68.71822643089294, + 68.79475389003754, + 68.84044187164307, + 68.88313028526306, + 68.89268061447143, + 68.9019778289795, + 68.90915637969971, + 69.01510049819946, + 69.11269155502319, + 69.14986647605896, + 69.20432386398315, + 69.24449661254883, + 69.30604753494262, + 69.33822512626648, + 69.44921379089355, + 69.54977664947509, + 69.63656955718994, + 69.7220088672638, + 69.79161666870117, + 69.89952602386475, + 69.96432800292969, + 70.09671997070312, + 70.16183603286743, + 70.23581919670104, + 70.44279062271119, + 70.56701078414918, + 70.65245677947998, + 70.70222663879395, + 70.76899917602539, + 70.8354319190979, + 70.92913150787354, + 70.96646771430969, + 71.02234704971313, + 71.05060875892639, + 71.11913133621216, + 71.1626944065094, + 71.20811561584473, + 71.26960675239563, + 71.38850318908692, + 71.56245389938354, + 71.67323970794678, + 71.7874120426178, + 71.84040683746338, + 71.99340531349182, + 72.04899557113647, + 72.1723009109497, + 72.24124952316284, + 72.37159567832947, + 72.47112649917602, + 72.55402723312378, + 72.71883096694947, + 72.74751964569091, + 72.79546329498291, + 72.88825476646423, + 72.93945390701293, + 72.99058198928833, + 73.05161338806153, + 73.07740741729737, + 73.12291696548462, + 73.16261117935181, + 73.23563585281372, + 73.32217198371887, + 73.41925914764404, + 73.53195141792297, + 73.63284660339356, + 73.82795042991638, + 73.89701509475708, + 74.1129421710968, + 74.23751605987549, + 74.36097875595092, + 74.42798328399658, + 74.46784362792968, + 74.4889419555664, + 74.61622435569763, + 74.66285207748413, + 74.70065207481385, + 74.73514791488647, + 74.79485809326172, + 74.85059669494629, + 74.88692506790161, + 74.99262762069702, + 75.05492301940917, + 75.08110652923584, + 75.14116109848023, + 75.17678977966308, + 75.25878009796142, + 75.37594177246093, + 75.49222368240356, + 75.62521614074707, + 75.76543350219727, + 75.79632778167725, + 75.94279739379883, + 76.03301671981812, + 76.19886515617371, + 76.38941028594971, + 76.78739504814148, + 76.92105403900146, + 77.07036893844605, + 77.29421939849854, + 77.43884038925171, + 78.16265077590941, + 86.27572008132933, + 90.50055501937867, + 93.04555112838744, + 95.90279457092285, + 96.10439598369597, + 96.73130201721192, + 97.4090187797546, + 97.54707008361817, + 97.63627235889435, + 97.65129278945923, + 97.66163789367675, + 99.63866498565673, + 101.68823362636554, + 101.71253681182861, + 101.71253681182861, + 101.71253681182861 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "spearman_rho": 0.994997379851183, + "worker_service_time_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 67.66233825683594, + 67.66233825683594, + 67.66233825683594, + 67.6640593996048, + 67.80920910644531, + 67.95179605484009, + 68.02490377426147, + 68.09708329200744, + 68.15453534317017, + 68.20866811275482, + 68.22613072395325, + 68.24237302875518, + 68.24840091705322, + 68.36891282081604, + 68.43633535385132, + 68.47792721748353, + 68.52287883758545, + 68.56143615722657, + 68.59744075775147, + 68.68872938156127, + 68.80711641311646, + 68.98208785057068, + 69.07777425765991, + 69.15221024513245, + 69.2176176071167, + 69.31227882385254, + 69.40780115127563, + 69.44678987503052, + 69.53036285400391, + 69.69970578193664, + 69.79847900390625, + 70.02766709327697, + 70.20742176055909, + 70.25490957260132, + 70.31409641265869, + 70.36884372711182, + 70.39411878585815, + 70.45976868629455, + 70.51949659347534, + 70.56677808761597, + 70.64457504272461, + 70.67562685012817, + 70.71087995529174, + 70.81224910736084, + 70.87421159744262, + 71.12893177032471, + 71.2300579071045, + 71.31576017379761, + 71.43313732147217, + 71.511494846344, + 71.55659273147583, + 71.64283022880554, + 71.67869539260865, + 71.82054615020752, + 71.91304916381836, + 72.00108870506287, + 72.1095682144165, + 72.28666694641113, + 72.34726026535034, + 72.39202815055847, + 72.42830923080444, + 72.46535301208496, + 72.51612672805786, + 72.57685908317566, + 72.63773469924926, + 72.67097898483276, + 72.74358606338501, + 72.8142272758484, + 72.90145801544189, + 72.96141745567321, + 73.08834177017212, + 73.15234274864197, + 73.32698850631714, + 73.46449827194213, + 73.54818227767944, + 73.72015203475952, + 73.84991865158081, + 73.91435957908631, + 73.9749273109436, + 74.03779411315918, + 74.08248973846436, + 74.13354544639587, + 74.1493603515625, + 74.19900452613831, + 74.25045185089111, + 74.31624401092529, + 74.38311100006104, + 74.422039270401, + 74.48597532272339, + 74.52979334831238, + 74.56393516540527, + 74.61002497673034, + 74.70035507202148, + 74.82858920097351, + 74.8776898574829, + 74.93662805557251, + 75.10793037414551, + 75.1927700805664, + 75.26206663131714, + 75.31744284629822, + 75.4060235786438, + 75.4754737854004, + 75.63023115158082, + 75.72989033699037, + 76.04676944732667, + 76.41455778121949, + 76.65793142318725, + 84.21108266830443, + 88.48870141983032, + 91.0584619617462, + 93.88633520126342, + 94.0976613883972, + 94.70356639862061, + 95.35661135673519, + 95.42019290542602, + 95.43041785240173, + 95.52639330673217, + 95.62777320480346, + 97.611005569458, + 99.66365052509296, + 99.68799018859863, + 99.68799018859863, + 99.68799018859863 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + } + }, + "steady": { + "count": 331, + "observation_to_action_latency_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 68.33836698532104, + 68.33836698532104, + 68.33836698532104, + 68.33836698532104, + 68.38673967647553, + 68.48557523679733, + 68.58441079711915, + 68.66100144147873, + 68.71233334255218, + 68.76366524362564, + 68.80820177650452, + 68.84433592844009, + 68.88047008037567, + 68.97942893981934, + 69.05328789710998, + 69.11279309272766, + 69.14725177288055, + 69.19221296310425, + 69.23277371883393, + 69.30906360626221, + 69.34171880245209, + 69.46581592559815, + 69.56281235694885, + 69.66370379447937, + 69.71114492416382, + 69.76014339447022, + 69.89628529548645, + 69.94386335372924, + 70.05395168304443, + 70.11957138061524, + 70.22768272876739, + 70.44380888938903, + 70.5693311882019, + 70.63999856948853, + 70.68795063972473, + 70.73425602912903, + 70.78683614730835, + 70.84482963562012, + 70.92785057067871, + 70.96238609313964, + 71.01126155853271, + 71.04333734512329, + 71.1141190481186, + 71.14226526260376, + 71.1897448015213, + 71.21613080978393, + 71.27579526901245, + 71.43687132835389, + 71.5723529624939, + 71.75288021087647, + 71.81975130081177, + 71.93823480606079, + 71.99483989238739, + 72.04133581161499, + 72.09813795566559, + 72.22221650123596, + 72.33224978446961, + 72.39444343566895, + 72.49450887203217, + 72.66334087371825, + 72.72290792942047, + 72.74462699890137, + 72.78119706153869, + 72.83898482322692, + 72.90189939022065, + 72.93732753753662, + 72.96202266216278, + 72.99534456253052, + 73.04351170063019, + 73.07116641998292, + 73.07783915519714, + 73.1248429775238, + 73.16131078720093, + 73.20008373260498, + 73.27307075500488, + 73.32595911026002, + 73.3897257566452, + 73.45877138137817, + 73.5550972700119, + 73.61690210342407, + 73.69131558418273, + 73.86406087875366, + 73.89704430103302, + 74.08576889038086, + 74.2045013999939, + 74.30898332595825, + 74.38751924037933, + 74.44016815185547, + 74.47647937774659, + 74.5085566329956, + 74.61676844120025, + 74.65941438674926, + 74.6824433517456, + 74.74643013000488, + 74.82909977912902, + 74.8688258266449, + 75.02692987918853, + 75.08692609786988, + 75.17494223117828, + 75.28712387084961, + 75.47549918174744, + 75.68471546173096, + 75.88692374229431, + 76.18196946144104, + 76.53914757728579, + 76.92910305023193, + 77.11558890342712, + 77.23786160469055, + 77.36318332195282, + 77.56394807815552, + 79.15258515357971, + 80.31217767429357, + 81.47177019500724, + 82.01180147695541, + 82.0507668390274, + 82.0897322010994, + 82.7520042648315, + 83.96321120786669, + 85.17441815090187, + 85.76721429824829, + 85.76721429824829, + 85.76721429824829, + 85.76721429824829 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "spearman_rho": 0.994186689079857, + "worker_service_time_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 67.66233825683594, + 67.66233825683594, + 67.66233825683594, + 67.66233825683594, + 67.70880911159516, + 67.80375882101059, + 67.89870853042602, + 67.97159004211426, + 68.01941347122192, + 68.06723690032959, + 68.11558884239197, + 68.16459428358078, + 68.2135997247696, + 68.30200378417969, + 68.37872838020324, + 68.43697241783143, + 68.47426755428314, + 68.51040445327759, + 68.56436436653138, + 68.61998393058776, + 68.78204969406129, + 68.95151624679565, + 69.05305458545685, + 69.09098007202148, + 69.16180240154266, + 69.26782484054566, + 69.3177341222763, + 69.40859308242798, + 69.45521615505218, + 69.53301364898681, + 69.70026827335357, + 69.84933495521545, + 70.0730836725235, + 70.22769279479981, + 70.25313691139222, + 70.30696405410767, + 70.31654298305511, + 70.36379844665527, + 70.40023928165436, + 70.4633428478241, + 70.53131168842316, + 70.58284902572632, + 70.64107429504395, + 70.65948798179626, + 70.69257098674774, + 70.71016096115112, + 70.81642253398896, + 70.87442145347595, + 71.1899939107895, + 71.28837493896485, + 71.36452118873596, + 71.4814284324646, + 71.53837106704712, + 71.55598100662232, + 71.6545636510849, + 71.67584495544433, + 71.82915835380554, + 71.91059803009033, + 71.96637691020966, + 72.03601422309876, + 72.16488472461701, + 72.27936887741089, + 72.3193176651001, + 72.36654196739197, + 72.38955458641053, + 72.4268406677246, + 72.45613436698913, + 72.48574446678161, + 72.51820910930634, + 72.54866027832031, + 72.59626186370849, + 72.6399115562439, + 72.6689730834961, + 72.72170774459839, + 72.76321669578552, + 72.79088823318482, + 72.86669962406158, + 72.95336965560914, + 72.9765655040741, + 73.06378131866455, + 73.1284744644165, + 73.19226808547974, + 73.30640823364257, + 73.44490498542785, + 73.48044187545776, + 73.64104009628296, + 73.78375005722046, + 73.85479689598084, + 73.91497821331023, + 73.98051874160767, + 74.0367184638977, + 74.07894439697266, + 74.12331788539886, + 74.13735425949096, + 74.1667393732071, + 74.23392400741577, + 74.31342358589173, + 74.38415325164794, + 74.49680510520935, + 74.56476838111878, + 74.76452895641327, + 74.83822364807129, + 74.90903673171998, + 74.94802674293518, + 75.13320108890534, + 75.27831100463867, + 75.37183067798614, + 75.47173568725586, + 75.6483791065216, + 76.26545589447021, + 77.267600440979, + 78.36891561508183, + 79.47023078918448, + 79.97171858310699, + 79.98810008049011, + 80.00448157787322, + 80.6638569335937, + 81.88950528955462, + 83.11515364551552, + 83.71501779556274, + 83.71501779556274, + 83.71501779556274, + 83.71501779556274 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + } + } + } + } +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/profile.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/profile.json new file mode 100644 index 0000000000000000000000000000000000000000..333844eb05ec67a5791a289ee39c227c95991641 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/profile.json @@ -0,0 +1,66 @@ +{ + "burst_model_path": "latency_burst_model.json", + "distribution_path": "latency_distribution.json", + "env_fps": 60, + "frame_ms": 16.666666666666668, + "gpu_class": "1x-rtx3090", + "instance_id": "instance_a5037b165aa0cedc", + "latency_kind": "observation_to_action_latency", + "latency_method": "temporal", + "model_id": "openvla", + "n_admitted_observations": 506, + "n_capacity_drops": 493, + "n_observation_attempts": 999, + "per_slot_summary": { + "0": { + "admitted_count": 506, + "mean_observation_to_action_latency_ms": 73.69250777493353, + "mean_worker_service_time_ms": 73.00619814047229, + "p95_observation_to_action_latency_ms": 78.0192643404007, + "p95_worker_service_time_ms": 76.60398542881012, + "p99_worker_service_time_ms": 93.66708476543427 + } + }, + "provenance": { + "base_config": "configs/experiments/deadly_corridor/starvla_profile.yaml", + "checkpoint_kind": "best", + "model_artifact": { + "checkpoint": "checkpoints/steps_500_state/model.safetensors", + "model_config": "config.full.yaml", + "path_in_repo": ".", + "repo_id": "talha15032/openvla_bridge_deadly_corridor_single_latency_clean_data_bce_exp2", + "source": "local" + }, + "session_ids": [ + 0, + 1, + 2, + 3, + 4 + ] + }, + "sample_model_type": "hidden_regime", + "source_run_id": "20260914T171446047509Z", + "summary": { + "frame_ms": 16.666666666666668, + "max_ms": 101.71253681182861, + "mean_effective_frames": 4.4215504664960115, + "mean_ms": 73.69250777493353, + "min_ms": 68.33836698532104, + "n_samples": 506, + "p50_frames": 4.3794349193572994, + "p50_ms": 72.99058198928833, + "p90_frames": 4.606898102760314, + "p90_ms": 76.78163504600525, + "p95_frames": 4.681155860424042, + "p95_ms": 78.0192643404007, + "p99_frames": 5.741582728385924, + "p99_ms": 95.69304547309875, + "prob_latency_gt_1_frame": 1.0, + "prob_latency_gt_2_frames": 1.0, + "prob_latency_gt_3_frames": 1.0, + "std_ms": 4.766097168829327 + }, + "visualization_path": "latency_profile.png", + "workload_id": "deadly_corridor" +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/provenance.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..3fd256ee48d5caba5dd1af86179e478bf9212765 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/provenance.json @@ -0,0 +1,97 @@ +{ + "task": "deadly_corridor", + "protocol": { + "gpu": 3, + "seed_start": 1000000, + "seed_end": 1000099, + "env_fps": 35, + "obs_fps": 8.75, + "max_raw_steps": 3600, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 73.69250777493353, + "profile": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/deadly_corridor/instance_a5037b165aa0cedc/profile.json", + "sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "checkpoint_weights": { + "bytes": 9785680289, + "sha256": "a756924b1f9a93cc4f278e03535dd61535977037f9484957672336e6b9c9265c" + }, + "evaluation": { + "n_episodes": 100, + "mean_return": 1620.7987757873534, + "std_return": 913.6242782186637, + "min_return": -76.45918273925781, + "max_return": 2287.240921020508, + "mean_length": 148.53, + "std_length": 49.455930888013825, + "min_length": 17.0, + "max_length": 199.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "doom_deadly_corridor", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "deadly_corridor", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T171446047509Z", + "profile_ref": null, + "env_fps": 35.0, + "obs_fps": 8.75, + "frame_ms": 28.571428571428573, + "latency_type": "profile_sample", + "task": "deadly_corridor", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42", + "profile_sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor.yaml", + "execution_audit": { + "issued_action_records": 3753, + "applied_action_records": 3673, + "dropped_action_records": 0, + "nonnoop_issued_records": 3753, + "finite_action_values": true, + "latency_sample_count": 3753, + "latency_mean_ms": 74.01999621872471, + "latency_std_ms": 5.5537519652567635, + "latency_p95_ms": 89.54825982614612, + "latency_p99_ms": 95.97310052501227 + } + }, + "source_revision": { + "repo": "c3c6a39365a151e9b7a5e215452fd64e957c2b29", + "starvla": "ccca13c5177fe3d3c884b6e2de4965d916016649", + "runtime_fixes": [ + "mean-profile-preparation.patch", + "gym-language-contract.patch", + "loader-spawn-cache.patch", + "loader-spawn-test.patch", + "doom-mean-reset.patch" + ], + "pytorch3d": { + "revision": "33824be3cbc87a7dd1db0f6a9a9de9ac81b2d0ba", + "build": "transforms-only, no native render extension; QwenOFT uses transforms only" + }, + "decord": { + "version": "0.6.0", + "build": "official source CPU decoder CP310", + "wheel_sha256": "e193b356b1e984b4eff08d23b62e482c2c9e5037a6efdc0b1af47079ae2e4c47" + } + }, + "raw_records_format": "gzip(JSONL), lossless", + "startup_checks_included_in_score": false, + "results_status": "evaluation_complete; acceptance_not_inferred" +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json new file mode 100644 index 0000000000000000000000000000000000000000..9a96fc304bcba5747159ca9c32bf1ca8ff753a55 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json @@ -0,0 +1,217 @@ +{ + "checkpoint_path": "/home/ubuntu/lzj/mean-profiling/deadly_corridor/vla-publication/checkpoints/model.pt", + "experiment_name": "deadly_corridor-mean5000-profile-simulation-100ep", + "latency": "profile_sample", + "latency_type": "profile_sample", + "lengths": [ + 72, + 143, + 182, + 189, + 150, + 115, + 176, + 176, + 49, + 75, + 176, + 45, + 176, + 178, + 178, + 177, + 172, + 182, + 177, + 99, + 181, + 197, + 172, + 74, + 195, + 179, + 178, + 190, + 177, + 177, + 44, + 183, + 178, + 192, + 188, + 179, + 183, + 178, + 178, + 190, + 185, + 179, + 104, + 177, + 189, + 181, + 74, + 182, + 179, + 189, + 182, + 181, + 173, + 73, + 175, + 194, + 178, + 181, + 176, + 108, + 113, + 177, + 186, + 89, + 70, + 76, + 75, + 75, + 150, + 184, + 132, + 47, + 44, + 141, + 172, + 151, + 143, + 182, + 79, + 17, + 183, + 171, + 178, + 41, + 176, + 45, + 171, + 76, + 175, + 177, + 93, + 70, + 192, + 175, + 152, + 199, + 179, + 181, + 178, + 178 + ], + "mean_length": 148.53, + "mean_return": 1620.7987757873534, + "returns": [ + 337.47547912597656, + 819.0284423828125, + 2284.857650756836, + 2276.2068634033203, + 805.2153015136719, + 621.8231658935547, + 2276.414749145508, + 2284.310989379883, + 81.07798767089844, + 317.2351837158203, + 2282.7608489990234, + 88.11907958984375, + 2281.468536376953, + 2276.6868591308594, + 2276.1705932617188, + 2282.6631622314453, + 2280.300033569336, + 2280.4182891845703, + 2281.2594451904297, + 479.8523712158203, + 2279.7379455566406, + 2284.9097442626953, + 2286.2730407714844, + 244.51919555664062, + 2279.957275390625, + 2283.952178955078, + 2276.701370239258, + 2277.142562866211, + 2279.025634765625, + 2285.7152099609375, + 53.374298095703125, + 2279.8080444335938, + 2282.307357788086, + 2282.834014892578, + 2284.200241088867, + 2287.2159118652344, + 2284.693832397461, + 2283.2066650390625, + 2281.032196044922, + 2282.960678100586, + 2287.094253540039, + 2279.3030853271484, + 440.0892791748047, + 2280.8592529296875, + 2283.4308471679688, + 2282.324264526367, + 326.0184631347656, + 2279.086135864258, + 2280.3804626464844, + 2276.215301513672, + 2278.132034301758, + 2285.6056518554688, + 2287.240921020508, + 310.81517028808594, + 2276.6219787597656, + 2276.2769470214844, + 2278.861602783203, + 2279.728561401367, + 2280.544464111328, + 487.829833984375, + 567.0655517578125, + 2278.210220336914, + 2281.436721801758, + 382.2119903564453, + 246.2946014404297, + 285.21240234375, + 310.6737365722656, + 346.1162872314453, + 804.7056121826172, + 2285.6442108154297, + 730.5995788574219, + 86.91796875, + 60.30122375488281, + 768.6264343261719, + 2280.1071166992188, + 860.9334106445312, + 722.9459228515625, + 2276.8687438964844, + 368.3357238769531, + -76.45918273925781, + 2281.5543823242188, + 2281.6688842773438, + 2277.5223083496094, + 42.30937194824219, + 2285.8980407714844, + 68.90191650390625, + 2286.2190551757812, + 281.1173553466797, + 2283.1607971191406, + 2277.888946533203, + 429.36326599121094, + 252.0751953125, + 2278.306442260742, + 2285.236801147461, + 857.2727355957031, + 2275.9288024902344, + 2286.8704833984375, + 2278.048355102539, + 2277.4480743408203, + 2276.9671478271484 + ], + "seed": 1000000, + "source_profile_path": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/deadly_corridor/instance_a5037b165aa0cedc/profile.json", + "std_return": 913.6242782186637, + "suite_name": "profile_sample", + "timestamp_utc": "2026-10-01T06:01:50.313923+00:00" +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..dc7821ad2e4ec900e5606b84345af50d4e930ac9 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:53f0b7bb59fb99cdd2917642c9881cb1443f897c78e20f1567929b4b232680fd +size 522227 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..bd3c62786bc0d211955058981962aaed59ba3a9b --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9c6b0b88fbfb5f6b37ea28ac5c6701c9abc2fa89b2d6b36b7fe747b6c1cf4398 +size 70931 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..388dd48100599be6be0ceebbe5b1e3dc0012aae4 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c96daa6d0d6f5051f6a20701c54ba21ebf5e425740028916cd553e2c8bd7cbd2 +size 7835 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..193e52f83e022b0e69dca956e4093eaba678a077 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:42a899945c8383a4d86d62b77692c1a2f4022a3cec41d5aada23bdc8f45df931 +size 60189 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..e95b8662bae711a300e4998ce7a353511181c410 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:12d682b1436b118a5058ec2364e2e7ee9cf45035224d7e04e0723689f476b01b +size 60183 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..be5e9be8a761ec4a0373493e67d344c7023af475 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6ce7d0c6fa3086525b7ba82526a5db7d4a26d33a64b16875bdd644c436069469 +size 47 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..060a0ed9236903265d909e81bb93c03f4800307c --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c717edfdd81c255faec3f3f765ecc4d378f353e459e1124a02b3fb070c71a10a +size 1468 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..98756624aeb96d342d1c4c47c092ca63550495f9 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f6a4d4bf35f367394f0754bff9ada16daa229726d0ccbbfede23411cffbdbd74 +size 495220 diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d1b6e62937ea54d166fedb8adbad1cb92c1f9d0e --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml @@ -0,0 +1,167 @@ +experiment: + name: deadly_corridor-mean5000-profile-simulation-100ep + seed: 1000000 +backend: + type: sample_factory + algo: APPO + device: cpu + train_dir: results/sample_factory + restart_behavior: resume + run_mode: eval +executor: + mode: simulated + simulated_worker_capacity: 1 + simulated_inference_pool: true + inference_devices: + - cuda:0 + inference_batch_size: 32 +env: + name: deadly_corridor + env_id: doom_deadly_corridor + env_fps: 35 + obs_fps: 8.75 + noop_action: + - 0 + - 0 + - 0 + - 0 + frame_stack: 1 + res_w: 128 + res_h: 72 + wide_aspect_ratio: false + simulator: cpu + obs_resize: + - 224 + - 224 + action_map: + noop: + - 0 + - 0 + - 0 + - 0 + move_forward: + - 0 + - 1 + - 0 + - 0 + move_backward: + - 0 + - 2 + - 0 + - 0 + move_left: + - 0 + - 0 + - 1 + - 0 + move_right: + - 0 + - 0 + - 2 + - 0 + turn_left: + - 1 + - 0 + - 0 + - 0 + turn_right: + - 2 + - 0 + - 0 + - 0 + attack: + - 0 + - 0 + - 0 + - 1 + action_history_decisions: 8 + screen_resolution: RES_160X120 + render_hud: true + render_crosshair: false + render_weapon: true + render_decals: false + render_particles: false +latency: + method: temporal + profile_path: /home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/deadly_corridor/instance_a5037b165aa0cedc/profile.json + profile_worker_slot: 0 + seed: 271828 + add_latency_info: false +scheduler: + hold_policy: hold + ordering_policy: latest_ready +policy: + type: starvla + actions: + - MOVE_FORWARD + - MOVE_BACKWARD + - MOVE_LEFT + - MOVE_RIGHT + - TURN_LEFT + - TURN_RIGHT + - ATTACK + checkpoint_path: /home/ubuntu/lzj/mean-profiling/deadly_corridor/vla-publication/checkpoints/model.pt + model_config_path: /home/ubuntu/lzj/mean-profiling/deadly_corridor/vla-publication/config.full.yaml + device: cuda:0 + unnorm_key: new_embodiment + prompt_mode: latency_neutral + action_layout: multibinary_7 + state_source: transport + backbone_path: /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct + worker_python_executable: /home/ubuntu/lzj/conda/envs/qwenoft/bin/python + action_prefix: + mode: none +training: + train_for_env_steps: 25000000 + num_workers: 32 + num_envs_per_worker: 4 + worker_num_splits: 2 + num_policies: 1 + batch_size: 1024 + rollout: 128 + recurrence: 128 + num_epochs: 2 + num_batches_per_epoch: 2 + learning_rate: 0.0001 + gamma: 0.99 + gae_lambda: 0.95 + ppo_clip_ratio: 0.1 + ppo_clip_value: 0.2 + exploration_loss: symmetric_kl + exploration_loss_coeff: 0.001 + value_loss_coeff: 0.5 + max_grad_norm: 4.0 + async_rl: true + use_rnn: true + rnn_type: gru + rnn_size: 512 + normalize_input: true + normalize_returns: true + stats_avg: 100 + experiment_summaries_interval: 1 + save_every_sec: 600 + keep_checkpoints: 5 +evaluation: + eval_interval_steps: 1000000 + eval_episodes: 100 + eval_parallel_envs: 32 + eval_max_steps: 3600 + eval_deterministic: true + eval_raw_reward: true + eval_suites: + fixed: [] + normal: [] + uniform: [] + eval_latency_values: null +logging: + output_dir: /home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor + video: + enabled: false + save_step_records: true + save_action_records: true + save_latency_records: true + wandb_project: null + wandb_group: null + wandb_job_type: null + wandb_tags: '' + simulated_pipeline_profile: false diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/statistics.json b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/statistics.json new file mode 100644 index 0000000000000000000000000000000000000000..6006972937324919f6c4927a61d2d455567acf16 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/statistics.json @@ -0,0 +1,36 @@ +{ + "n_episodes": 100, + "mean_return": 1620.7987757873534, + "std_return": 913.6242782186637, + "min_return": -76.45918273925781, + "max_return": 2287.240921020508, + "mean_length": 148.53, + "std_length": 49.455930888013825, + "min_length": 17.0, + "max_length": 199.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "doom_deadly_corridor", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "deadly_corridor", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T171446047509Z", + "profile_ref": null, + "env_fps": 35.0, + "obs_fps": 8.75, + "frame_ms": 28.571428571428573, + "latency_type": "profile_sample", + "task": "deadly_corridor", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42", + "profile_sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor.yaml" +} diff --git a/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/stdout.log b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/stdout.log new file mode 100644 index 0000000000000000000000000000000000000000..8b9cf0c2e1533c8c9c379412ff0e6bb0f710c047 --- /dev/null +++ b/latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/stdout.log @@ -0,0 +1,586 @@ +[bench] run=deadly_corridor-mean5000-profile-simulation-100ep sweeps=1 episodes_per_sweep=100 total_episode_runs=100 +[bench] sweep 1/1: eval_latency=profile_sample +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/vizdoom/gymnasium_wrapper/base_gymnasium_env.py:84: UserWarning: Detected screen format CRCGCB. Only RGB24 and GRAY8 are supported in the Gymnasium wrapper. Forcing RGB24. + warnings.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/gymnasium/core.py:311: UserWarning: WARN: env.game to get variables from other wrappers is deprecated and will be removed in v1.0, to get this variable you can do `env.unwrapped.game` for environment variables or `env.get_wrapper_attr('game')` that will search the reminding wrappers. + logger.warn( +10/01 [06:01:06] INFO | >> Loaded mixtures from Behavior registry.py:113 + (data_config): ['BEHAVIOR_challenge'] + INFO | >> Loaded data_config from DOMINO: registry.py:107 + ['robotwin'] + INFO | >> Loaded embodiment_tags from registry.py:110 + DOMINO (data_config): [] + INFO | >> Loaded mixtures from DOMINO registry.py:113 + (data_config): ['domino', + 'domino_clean', 'domino_random', + 'domino_cotrain'] + INFO | >> Loaded data_config from Franka: registry.py:107 + ['custom_robot_config', + 'demo_sim_franka_delta_joints', + 'SO101'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Franka (data_config): [] + INFO | >> Loaded mixtures from Franka registry.py:113 + (data_config): ['custom_dataset', + 'custom_dataset_2', + 'demo_sim_pick_place', 'SO101_pick'] + INFO | >> Loaded data_config from LIBERO: registry.py:107 + ['libero_franka'] + INFO | >> Loaded embodiment_tags from registry.py:110 + LIBERO (data_config): [] + INFO | >> Loaded mixtures from LIBERO registry.py:113 + (data_config): ['libero_all', + 'libero_goal', 'multi_robot'] + INFO | >> Loaded data_config from MIKASA: registry.py:107 + ['mikasa_franka_h1'] + INFO | >> Loaded embodiment_tags from registry.py:110 + MIKASA (data_config): + ['mikasa_franka_h1'] + INFO | >> Loaded mixtures from MIKASA registry.py:113 + (data_config): + ['local/intercept_grab_fast_vla_v0_h1_ + train'] + INFO | >> Loaded data_config from registry.py:107 + RoboChallenge_table30v2: + ['ur5_robochallenge', + 'arx5_robochallenge', + 'dosw1_robochallenge'] + INFO | >> Loaded embodiment_tags from registry.py:110 + RoboChallenge_table30v2 (data_config): + ['ur5_robochallenge', + 'arx5_robochallenge', + 'dosw1_robochallenge'] + INFO | >> Loaded mixtures from registry.py:113 + RoboChallenge_table30v2 (data_config): + ['robochallenge_table30v2_shred_paper' + , 'robochallenge_table30v2_ur5_all', + 'robochallenge_table30v2_arx5_all', + 'robochallenge_table30v2_dosw1_all'] + INFO | >> Loaded data_config from registry.py:107 + Robocasa_365: + ['panda_omron_robocasa365'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Robocasa_365 (data_config): [] + INFO | >> Loaded mixtures from registry.py:113 + Robocasa_365 (data_config): + ['robocasa365_open_drawer_target_human + ', + 'robocasa365_atomic_target_human_all', + 'robocasa365_composite_target_human_al + l', 'robocasa365_target_human_all'] + INFO | >> Loaded data_config from registry.py:107 + Robocasa_tabletop: + ['fourier_gr1_arms_waist'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Robocasa_tabletop (data_config): [] + INFO | >> Loaded mixtures from registry.py:113 + Robocasa_tabletop (data_config): + ['fourier_gr1_unified_1000'] + INFO | >> Loaded data_config from registry.py:107 + Robotwin: ['robotwin', 'robotwin50', + 'arx_x5'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Robotwin (data_config): [] + INFO | >> Loaded mixtures from Robotwin registry.py:113 + (data_config): ['robotwin_all', + 'robotwin_all_50', 'robotwin', + 'robotwin_task1', 'robotwin_task2', + 'arx_x5'] + INFO | >> Loaded data_config from registry.py:107 + SimplerEnv: ['oxe_droid', + 'oxe_bridge', 'oxe_rt1'] + INFO | >> Loaded embodiment_tags from registry.py:110 + SimplerEnv (data_config): [] + INFO | >> Loaded mixtures from SimplerEnv registry.py:113 + (data_config): ['bridge', + 'bridge_rt_1'] + INFO | >> Loaded data_config from registry.py:107 + VLA-Arena: ['vla_arena_franka'] + INFO | >> Loaded embodiment_tags from registry.py:110 + VLA-Arena (data_config): [] + INFO | >> Loaded mixtures from VLA-Arena registry.py:113 + (data_config): ['vla_arena_L0_S', + 'vla_arena_L0_M', 'vla_arena_L0_L'] + INFO | >> Loaded data_config from registry.py:107 + rl_games: ['rl_games_flappy', + 'rl_games_demon_attack', + 'rl_games_defend_the_line', + 'rl_games_deadly_corridor', + 'rl_games_asterix', + 'rl_games_atlantis', + 'rl_games_gymnasium', + 'rl_games_gymnasium_discrete', + 'rl_games_gymnasium_native'] + INFO | >> Loaded embodiment_tags from registry.py:110 + rl_games (data_config): + ['rl_games_flappy', + 'rl_games_demon_attack', + 'rl_games_defend_the_line', + 'rl_games_deadly_corridor', + 'rl_games_asterix', + 'rl_games_atlantis', + 'rl_games_gymnasium', + 'rl_games_gymnasium_discrete', + 'rl_games_gymnasium_native'] + INFO | >> Loaded mixtures from rl_games registry.py:113 + (data_config): ['flappy_train', + 'flappy_train__bridge', + 'flappy_mixed_latency_train', + 'flappy_mixed_latency_train__bridge', + 'demon_attack_train', + 'demon_attack_train__bridge', + 'demon_attack_mixed_latency_train', + 'demon_attack_mixed_latency_train__bri + dge', 'defend_the_line_train', + 'defend_the_line_train__bridge', + 'defend_the_line_mixed_latency_train', + 'defend_the_line_mixed_latency_train__ + bridge', 'deadly_corridor_train', + 'deadly_corridor_train__bridge', + 'deadly_corridor_mixed_latency_train', + 'deadly_corridor_mixed_latency_train__ + bridge', 'asterix_train', + 'asterix_train__bridge', + 'asterix_mixed_latency_train', + 'asterix_mixed_latency_train__bridge', + 'atlantis_train', + 'atlantis_train__bridge', + 'atlantis_mixed_latency_train', + 'atlantis_mixed_latency_train__bridge' + , 'h1hand_balance_hard'] + INFO | >> PolicyServerWrapper: loading policy_wrapper.py:73 + framework from + /home/ubuntu/lzj/mean-profiling/d + eadly_corridor/vla-publication/ch + eckpoints/model.pt + INFO | >> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/de + adly_corridor/vla-publication/check + points/model.pt` +[QWen3] loading /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct with gradient_checkpointing=True + Loading checkpoint shards: 0%| | 0/2 [00:00> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/de + adly_corridor/vla-publication/check + points/model.pt` + INFO | >> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/de + adly_corridor/vla-publication/check + points/model.pt` + INFO | >> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/de + adly_corridor/vla-publication/check + points/model.pt` + INFO | >> PolicyNormProcessor policy_norm_processor.py:333 + ready: + robot_type=rl_games_deadl + y_corridor, + unnorm_key=new_embodiment + , + action_keys=['action.butt + on'] (dims=[7]), + state_keys=['state.game_s + tate'] + INFO | >> PolicyServerWrapper ready: policy_wrapper.py:126 + action_chunk_size=1, + default_unnorm_key=new_embodimen + t, + available_unnorm_keys=['new_embo + diment'], + action_keys=['action.button'], + state_keys=['state.game_state'] +[bench] sweep 1/1 episode 1/100 done +[bench] sweep 1/1 episode 2/100 done +[bench] sweep 1/1 episode 3/100 done +[bench] sweep 1/1 episode 4/100 done +[bench] sweep 1/1 episode 5/100 done +[bench] sweep 1/1 episode 6/100 done +[bench] sweep 1/1 episode 7/100 done +[bench] sweep 1/1 episode 8/100 done +[bench] sweep 1/1 episode 9/100 done +[bench] sweep 1/1 episode 10/100 done +[bench] sweep 1/1 episode 11/100 done +[bench] sweep 1/1 episode 12/100 done +[bench] sweep 1/1 episode 13/100 done +[bench] sweep 1/1 episode 14/100 done +[bench] sweep 1/1 episode 15/100 done +[bench] sweep 1/1 episode 16/100 done +[bench] sweep 1/1 episode 17/100 done +[bench] sweep 1/1 episode 18/100 done +[bench] sweep 1/1 episode 19/100 done +[bench] sweep 1/1 episode 20/100 done +[bench] sweep 1/1 episode 21/100 done +[bench] sweep 1/1 episode 22/100 done +[bench] sweep 1/1 episode 23/100 done +[bench] sweep 1/1 episode 24/100 done +[bench] sweep 1/1 episode 25/100 done +[bench] sweep 1/1 episode 26/100 done +[bench] sweep 1/1 episode 27/100 done +[bench] sweep 1/1 episode 28/100 done +[bench] sweep 1/1 episode 29/100 done +[bench] sweep 1/1 episode 30/100 done +[bench] sweep 1/1 episode 31/100 done +[bench] sweep 1/1 episode 32/100 done +[bench] sweep 1/1 episode 33/100 done +[bench] sweep 1/1 episode 34/100 done +[bench] sweep 1/1 episode 35/100 done +[bench] sweep 1/1 episode 36/100 done +[bench] sweep 1/1 episode 37/100 done +[bench] sweep 1/1 episode 38/100 done +[bench] sweep 1/1 episode 39/100 done +[bench] sweep 1/1 episode 40/100 done +[bench] sweep 1/1 episode 41/100 done +[bench] sweep 1/1 episode 42/100 done +[bench] sweep 1/1 episode 43/100 done +[bench] sweep 1/1 episode 44/100 done +[bench] sweep 1/1 episode 45/100 done +[bench] sweep 1/1 episode 46/100 done +[bench] sweep 1/1 episode 47/100 done +[bench] sweep 1/1 episode 48/100 done +[bench] sweep 1/1 episode 49/100 done +[bench] sweep 1/1 episode 50/100 done +[bench] sweep 1/1 episode 51/100 done +[bench] sweep 1/1 episode 52/100 done +[bench] sweep 1/1 episode 53/100 done +[bench] sweep 1/1 episode 54/100 done +[bench] sweep 1/1 episode 55/100 done +[bench] sweep 1/1 episode 56/100 done +[bench] sweep 1/1 episode 57/100 done +[bench] sweep 1/1 episode 58/100 done +[bench] sweep 1/1 episode 59/100 done +[bench] sweep 1/1 episode 60/100 done +[bench] sweep 1/1 episode 61/100 done +[bench] sweep 1/1 episode 62/100 done +[bench] sweep 1/1 episode 63/100 done +[bench] sweep 1/1 episode 64/100 done +[bench] sweep 1/1 episode 65/100 done +[bench] sweep 1/1 episode 66/100 done +[bench] sweep 1/1 episode 67/100 done +[bench] sweep 1/1 episode 68/100 done +[bench] sweep 1/1 episode 69/100 done +[bench] sweep 1/1 episode 70/100 done +[bench] sweep 1/1 episode 71/100 done +[bench] sweep 1/1 episode 72/100 done +[bench] sweep 1/1 episode 73/100 done +[bench] sweep 1/1 episode 74/100 done +[bench] sweep 1/1 episode 75/100 done +[bench] sweep 1/1 episode 76/100 done +[bench] sweep 1/1 episode 77/100 done +[bench] sweep 1/1 episode 78/100 done +[bench] sweep 1/1 episode 79/100 done +[bench] sweep 1/1 episode 80/100 done +[bench] sweep 1/1 episode 81/100 done +[bench] sweep 1/1 episode 82/100 done +[bench] sweep 1/1 episode 83/100 done +[bench] sweep 1/1 episode 84/100 done +[bench] sweep 1/1 episode 85/100 done +[bench] sweep 1/1 episode 86/100 done +[bench] sweep 1/1 episode 87/100 done +[bench] sweep 1/1 episode 88/100 done +[bench] sweep 1/1 episode 89/100 done +[bench] sweep 1/1 episode 90/100 done +[bench] sweep 1/1 episode 91/100 done +[bench] sweep 1/1 episode 92/100 done +[bench] sweep 1/1 episode 93/100 done +[bench] sweep 1/1 episode 94/100 done +[bench] sweep 1/1 episode 95/100 done +[bench] sweep 1/1 episode 96/100 done +[bench] sweep 1/1 episode 97/100 done +[bench] sweep 1/1 episode 98/100 done +[bench] sweep 1/1 episode 99/100 done +[bench] sweep 1/1 episode 100/100 done +[bench] sweep 1/1 complete elapsed=48.3s +episode=0 return=337.475 steps=72 mean_latency_ms=71.90727374040254 +episode=1 return=819.028 steps=143 mean_latency_ms=73.84762082340946 +episode=2 return=2284.858 steps=182 mean_latency_ms=72.79171012339609 +episode=3 return=2276.207 steps=189 mean_latency_ms=76.345275285376 +episode=4 return=805.215 steps=150 mean_latency_ms=73.86282581373551 +episode=5 return=621.823 steps=115 mean_latency_ms=74.31105893586228 +episode=6 return=2276.415 steps=176 mean_latency_ms=74.2226331369995 +episode=7 return=2284.311 steps=176 mean_latency_ms=72.9072057957754 +episode=8 return=81.078 steps=49 mean_latency_ms=73.20258272646697 +episode=9 return=317.235 steps=75 mean_latency_ms=72.54774919154028 +episode=10 return=2282.761 steps=176 mean_latency_ms=72.78293151689127 +episode=11 return=88.119 steps=45 mean_latency_ms=72.60486105128022 +episode=12 return=2281.469 steps=176 mean_latency_ms=72.29193331603048 +episode=13 return=2276.687 steps=178 mean_latency_ms=72.73330265771509 +episode=14 return=2276.171 steps=178 mean_latency_ms=73.30067987408609 +episode=15 return=2282.663 steps=177 mean_latency_ms=72.49405489224537 +episode=16 return=2280.300 steps=172 mean_latency_ms=72.80884970803692 +episode=17 return=2280.418 steps=182 mean_latency_ms=73.03539182090206 +episode=18 return=2281.259 steps=177 mean_latency_ms=72.50972089313564 +episode=19 return=479.852 steps=99 mean_latency_ms=72.58046231642126 +episode=20 return=2279.738 steps=181 mean_latency_ms=72.47468246266928 +episode=21 return=2284.910 steps=197 mean_latency_ms=83.18983231769475 +episode=22 return=2286.273 steps=172 mean_latency_ms=72.84281562147524 +episode=23 return=244.519 steps=74 mean_latency_ms=76.23461799191558 +episode=24 return=2279.957 steps=195 mean_latency_ms=72.94927214021655 +episode=25 return=2283.952 steps=179 mean_latency_ms=73.18968843008061 +episode=26 return=2276.701 steps=178 mean_latency_ms=72.87702909462648 +episode=27 return=2277.143 steps=190 mean_latency_ms=72.45412386128042 +episode=28 return=2279.026 steps=177 mean_latency_ms=74.11102172804317 +episode=29 return=2285.715 steps=177 mean_latency_ms=71.63189230597281 +episode=30 return=53.374 steps=44 mean_latency_ms=72.51418721312025 +episode=31 return=2279.808 steps=183 mean_latency_ms=72.72702656843174 +episode=32 return=2282.307 steps=178 mean_latency_ms=74.33584751930213 +episode=33 return=2282.834 steps=192 mean_latency_ms=73.95005063555192 +episode=34 return=2284.200 steps=188 mean_latency_ms=76.29368894499888 +episode=35 return=2287.216 steps=179 mean_latency_ms=72.81890806090988 +episode=36 return=2284.694 steps=183 mean_latency_ms=76.28284599973325 +episode=37 return=2283.207 steps=178 mean_latency_ms=72.1797344044525 +episode=38 return=2281.032 steps=178 mean_latency_ms=73.74343783824916 +episode=39 return=2282.961 steps=190 mean_latency_ms=73.24816830891406 +episode=40 return=2287.094 steps=185 mean_latency_ms=72.35711232966574 +episode=41 return=2279.303 steps=179 mean_latency_ms=72.42125368367608 +episode=42 return=440.089 steps=104 mean_latency_ms=73.92064892672727 +episode=43 return=2280.859 steps=177 mean_latency_ms=72.36020918178356 +episode=44 return=2283.431 steps=189 mean_latency_ms=75.93658060557208 +episode=45 return=2282.324 steps=181 mean_latency_ms=73.54224681770178 +episode=46 return=326.018 steps=74 mean_latency_ms=73.1983876441008 +episode=47 return=2279.086 steps=182 mean_latency_ms=73.00958120503027 +episode=48 return=2280.380 steps=179 mean_latency_ms=73.17268244992928 +episode=49 return=2276.215 steps=189 mean_latency_ms=75.47590644230628 +episode=50 return=2278.132 steps=182 mean_latency_ms=74.50495464842548 +episode=51 return=2285.606 steps=181 mean_latency_ms=73.41699294418743 +episode=52 return=2287.241 steps=173 mean_latency_ms=73.22110809114655 +episode=53 return=310.815 steps=73 mean_latency_ms=74.09003681120738 +episode=54 return=2276.622 steps=175 mean_latency_ms=72.98605010243534 +episode=55 return=2276.277 steps=194 mean_latency_ms=75.17704077845171 +episode=56 return=2278.862 steps=178 mean_latency_ms=72.97353037051572 +episode=57 return=2279.729 steps=181 mean_latency_ms=73.96913002154926 +episode=58 return=2280.544 steps=176 mean_latency_ms=73.02432805290651 +episode=59 return=487.830 steps=108 mean_latency_ms=78.89398217393664 +episode=60 return=567.066 steps=113 mean_latency_ms=72.64874721482185 +episode=61 return=2278.210 steps=177 mean_latency_ms=72.96068484971086 +episode=62 return=2281.437 steps=186 mean_latency_ms=75.46710866924751 +episode=63 return=382.212 steps=89 mean_latency_ms=81.21157315209366 +episode=64 return=246.295 steps=70 mean_latency_ms=73.9736408486285 +episode=65 return=285.212 steps=76 mean_latency_ms=73.13661133681993 +episode=66 return=310.674 steps=75 mean_latency_ms=73.40468658737086 +episode=67 return=346.116 steps=75 mean_latency_ms=72.1929723632303 +episode=68 return=804.706 steps=150 mean_latency_ms=73.76397959753224 +episode=69 return=2285.644 steps=184 mean_latency_ms=75.13255757158333 +episode=70 return=730.600 steps=132 mean_latency_ms=73.25446825350764 +episode=71 return=86.918 steps=47 mean_latency_ms=76.28335745963689 +episode=72 return=60.301 steps=44 mean_latency_ms=76.83513093208644 +episode=73 return=768.626 steps=141 mean_latency_ms=77.27057350071598 +episode=74 return=2280.107 steps=172 mean_latency_ms=74.16699734355548 +episode=75 return=860.933 steps=151 mean_latency_ms=73.15118478347584 +episode=76 return=722.946 steps=143 mean_latency_ms=75.5655785931314 +episode=77 return=2276.869 steps=182 mean_latency_ms=72.95102474014934 +episode=78 return=368.336 steps=79 mean_latency_ms=71.51096709276341 +episode=79 return=-76.459 steps=17 mean_latency_ms=72.24888432102617 +episode=80 return=2281.554 steps=183 mean_latency_ms=73.32589540463356 +episode=81 return=2281.669 steps=171 mean_latency_ms=73.10600900440717 +episode=82 return=2277.522 steps=178 mean_latency_ms=73.55648700566698 +episode=83 return=42.309 steps=41 mean_latency_ms=73.52700344736942 +episode=84 return=2285.898 steps=176 mean_latency_ms=71.98655161011203 +episode=85 return=68.902 steps=45 mean_latency_ms=72.84773487604696 +episode=86 return=2286.219 steps=171 mean_latency_ms=72.82303966497733 +episode=87 return=281.117 steps=76 mean_latency_ms=72.26983276661764 +episode=88 return=2283.161 steps=175 mean_latency_ms=73.49638264342678 +episode=89 return=2277.889 steps=177 mean_latency_ms=73.44736473371472 +episode=90 return=429.363 steps=93 mean_latency_ms=71.86172378947977 +episode=91 return=252.075 steps=70 mean_latency_ms=72.26459581736903 +episode=92 return=2278.306 steps=192 mean_latency_ms=80.97328482778371 +episode=93 return=2285.237 steps=175 mean_latency_ms=74.02717585214627 +episode=94 return=857.273 steps=152 mean_latency_ms=85.59110000526613 +episode=95 return=2275.929 steps=199 mean_latency_ms=73.62958803645523 +episode=96 return=2286.870 steps=179 mean_latency_ms=72.31519682456816 +episode=97 return=2278.048 steps=181 mean_latency_ms=73.50330330803081 +episode=98 return=2277.448 steps=178 mean_latency_ms=76.78472725777 +episode=99 return=2276.967 steps=178 mean_latency_ms=77.9678189026336 +summary latency=profile_sample episodes=100 mean_return=1620.799 std_return=913.624 mean_length=148.5 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/REPORT.md b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/REPORT.md new file mode 100644 index 0000000000000000000000000000000000000000..7f2af02a5f64a7ae2fdb82b2a93c6eaac45fb57d --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/REPORT.md @@ -0,0 +1,20 @@ +# QwenOFT mean-trained checkpoints under profile simulation + +Four final step-5000 H1 checkpoints; two rounds, one evaluation per physical GPU2/3,100 episodes each (400 total). + +The training latency was fixed mean; this evaluation samples the complete archived RTX3090 temporal hidden-regime profile. Simulator FPS, seeds, horizon, limits and model/profile identities are in evaluation-plan.json. Standard deviations below use ddof=0. Returns have task-specific scales. Startup checks are separate and excluded. + +| Task | Episodes | Return mean +/- SD | Length mean +/- SD | Success | Invalid | +|---|---:|---:|---:|---:|---:| +| flappy | 100 | 384.824005 +/- 116.787774 | 3119.31 +/- 939.87 | not provided by task | 0 | +| deadly_corridor | 100 | 1620.798776 +/- 913.624278 | 148.53 +/- 49.46 | not provided by task | 0 | +| ant | 100 | 1453.844064 +/- 693.727520 | 803.85 +/- 328.81 | not provided by task | 0 | +| intercept | 100 | 3.544349 +/- 7.071923 | 60.00 +/- 0.00 | 9/100 | 0 | + +No success metric is invented for Flappy/Deadly/Ant. Intercept reports the native accumulated success flag. No policy-quality acceptance gate is claimed. + +Compatibility repairs: portable robot_type copied from each actual training manifest (weights unchanged); official ViZDoom1.2.4 VizdoomCorridor-v0 uses the same deadly_corridor WAD as SF, preserves render contract and semantic seven-button ordering; public action space is equivalent MultiBinary7. Existing native render/button/history tests passed. Full eval source/patch and original profile assets are archived. + +Flappy/Deadly seeds1000000..1000099; Ant42..141; Intercept4242424242..4242424341. Latency seed271828. Flappy10/10Hz, Deadly35/8.75Hz, Ant10/10Hz, Intercept20/20Hz. Max raw frames3600/3600/1000/60; capacities1. MIKASA H1 holds last chunk action; no prefix, no DAgger. Ant keeps its training prompt label1 while execution latency is sampled. + +Raw JSONL logs are losslessly gzip-compressed for distribution; original uncompressed records remain on the experiment host. Empty observation_attempts files are retained; admission/drop evidence is in steps/actions. Per-task CSV and full400 episode CSV are provided. diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/all_episodes.csv b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/all_episodes.csv new file mode 100644 index 0000000000000000000000000000000000000000..be5e27101cb97f6a2dd0d85a0399d3fb8a5ea839 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/all_episodes.csv @@ -0,0 +1,401 @@ +task,episode_id,seed,return_env,length,mean_latency_ms,success +flappy,0,1000000,444.6000052243471,3600,76.02271694866694, +flappy,1,1000001,444.6000052243471,3600,76.14445348705047, +flappy,2,1000002,444.6000052243471,3600,75.83047266244563, +flappy,3,1000003,444.6000052243471,3600,76.04121221698036, +flappy,4,1000004,444.6000052243471,3600,75.7789115791707, +flappy,5,1000005,228.2000027000904,1861,76.22757676162651, +flappy,6,1000006,444.6000052243471,3600,75.98373978309758, +flappy,7,1000007,444.6000052243471,3600,75.85552109823348, +flappy,8,1000008,444.6000052243471,3600,75.9782303085917, +flappy,9,1000009,444.6000052243471,3600,75.94667987356688, +flappy,10,1000010,444.6000052243471,3600,75.66396359484234, +flappy,11,1000011,444.6000052243471,3600,75.7794525026407, +flappy,12,1000012,444.6000052243471,3600,75.90110110734818, +flappy,13,1000013,444.6000052243471,3600,76.01870178237883, +flappy,14,1000014,444.6000052243471,3600,75.75567207010911, +flappy,15,1000015,444.6000052243471,3600,75.83026036637241, +flappy,16,1000016,444.6000052243471,3600,75.74502908171665, +flappy,17,1000017,444.6000052243471,3600,75.84316844302293, +flappy,18,1000018,444.6000052243471,3600,75.85876738771161, +flappy,19,1000019,265.50000313669443,2162,75.89492798135642, +flappy,20,1000020,444.6000052243471,3600,75.90859756288593, +flappy,21,1000021,444.6000052243471,3600,75.93474621914784, +flappy,22,1000022,444.6000052243471,3600,75.77022360156529, +flappy,23,1000023,444.6000052243471,3600,75.8506098974935, +flappy,24,1000024,444.6000052243471,3600,75.80511776716725, +flappy,25,1000025,116.00000138580799,955,76.07937915327228, +flappy,26,1000026,444.6000052243471,3600,75.77409482659607, +flappy,27,1000027,444.6000052243471,3600,75.82354466933252, +flappy,28,1000028,444.6000052243471,3600,75.92578714415393, +flappy,29,1000029,444.6000052243471,3600,75.77326038618416, +flappy,30,1000030,256.0000030249357,2085,75.8461606092662, +flappy,31,1000031,444.6000052243471,3600,75.87053786258159, +flappy,32,1000032,444.6000052243471,3600,75.90930861144982, +flappy,33,1000033,444.6000052243471,3600,75.80530422686525, +flappy,34,1000034,444.6000052243471,3600,76.05997569829616, +flappy,35,1000035,444.6000052243471,3600,75.67579907153437, +flappy,36,1000036,444.6000052243471,3600,76.07561842170198, +flappy,37,1000037,444.6000052243471,3600,75.87459102177027, +flappy,38,1000038,55.60000067949295,468,75.8887188983619, +flappy,39,1000039,444.6000052243471,3600,75.86536772802552, +flappy,40,1000040,432.900005094707,3512,76.00355652525975, +flappy,41,1000041,274.8000032454729,2237,75.7658282850597, +flappy,42,1000042,264.90000312775373,2156,75.95264956954799, +flappy,43,1000043,265.4000031352043,2161,75.82748305801191, +flappy,44,1000044,444.6000052243471,3600,75.9295822845668, +flappy,45,1000045,143.90000171214342,1180,75.94310218110371, +flappy,46,1000046,444.6000052243471,3600,75.69568531179425, +flappy,47,1000047,93.1000011190772,771,76.0527875505066, +flappy,48,1000048,56.10000068694353,473,76.20882901957174, +flappy,49,1000049,265.2000031322241,2159,76.05401077635972, +flappy,50,1000050,444.6000052243471,3600,75.89333271844873, +flappy,51,1000051,444.6000052243471,3600,75.89090159365671, +flappy,52,1000052,398.80000469088554,3234,75.91218218803246, +flappy,53,1000053,444.6000052243471,3600,75.86400590251726, +flappy,54,1000054,270.2000031918287,2200,76.01590238337654, +flappy,55,1000055,69.70000084489584,582,75.68422480575155, +flappy,56,1000056,444.6000052243471,3600,75.87884524455251, +flappy,57,1000057,444.6000052243471,3600,75.96981187494319, +flappy,58,1000058,444.6000052243471,3600,76.03772455115222, +flappy,59,1000059,437.90000515431166,3553,76.04088529786887, +flappy,60,1000060,348.9000041112304,2834,75.79422825165413, +flappy,61,1000061,444.6000052243471,3600,75.87728099437057, +flappy,62,1000062,78.9000009521842,656,75.96352981662133, +flappy,63,1000063,444.6000052243471,3600,75.80269270184165, +flappy,64,1000064,444.6000052243471,3600,75.88518180564401, +flappy,65,1000065,444.6000052243471,3600,75.87533034544981, +flappy,66,1000066,444.6000052243471,3600,75.94241138050401, +flappy,67,1000067,444.6000052243471,3600,75.95312277771471, +flappy,68,1000068,444.6000052243471,3600,75.8998829764233, +flappy,69,1000069,444.6000052243471,3600,75.98564617573034, +flappy,70,1000070,444.6000052243471,3600,75.68328575087021, +flappy,71,1000071,135.0000016093254,1109,75.99546963217229, +flappy,72,1000072,444.6000052243471,3600,75.9923106611263, +flappy,73,1000073,444.6000052243471,3600,75.80422251719546, +flappy,74,1000074,444.6000052243471,3600,75.95469853250815, +flappy,75,1000075,444.6000052243471,3600,75.74551875442629, +flappy,76,1000076,444.6000052243471,3600,75.93301571087362, +flappy,77,1000077,444.6000052243471,3600,75.98384926019328, +flappy,78,1000078,444.6000052243471,3600,75.85055115368883, +flappy,79,1000079,444.6000052243471,3600,75.97142616222317, +flappy,80,1000080,444.6000052243471,3600,75.97039764106849, +flappy,81,1000081,444.6000052243471,3600,75.74469321422862, +flappy,82,1000082,116.20000138878822,957,76.0366526049804, +flappy,83,1000083,444.6000052243471,3600,75.95924386190674, +flappy,84,1000084,444.6000052243471,3600,76.0310580385874, +flappy,85,1000085,36.60000045597553,314,75.8634823847272, +flappy,86,1000086,260.90000308305025,2125,75.91783880059099, +flappy,87,1000087,444.6000052243471,3600,76.16752514785735, +flappy,88,1000088,444.6000052243471,3600,75.83331254385584, +flappy,89,1000089,444.6000052243471,3600,76.2113387300584, +flappy,90,1000090,444.6000052243471,3600,75.8334932097261, +flappy,91,1000091,225.20000265538692,1831,75.75618859671614, +flappy,92,1000092,444.6000052243471,3600,75.79683788505955, +flappy,93,1000093,180.9000021442771,1478,75.96465307644473, +flappy,94,1000094,305.2000035941601,2478,75.86565754734926, +flappy,95,1000095,444.6000052243471,3600,75.74523644464854, +flappy,96,1000096,444.6000052243471,3600,75.94353591524424, +flappy,97,1000097,444.6000052243471,3600,75.81927739599219, +flappy,98,1000098,444.6000052243471,3600,75.96229410618645, +flappy,99,1000099,444.6000052243471,3600,75.94512877548694, +deadly_corridor,0,1000000,337.47547912597656,72,71.90727374040254, +deadly_corridor,1,1000001,819.0284423828125,143,73.84762082340946, +deadly_corridor,2,1000002,2284.857650756836,182,72.79171012339609, +deadly_corridor,3,1000003,2276.2068634033203,189,76.345275285376, +deadly_corridor,4,1000004,805.2153015136719,150,73.86282581373551, +deadly_corridor,5,1000005,621.8231658935547,115,74.31105893586228, +deadly_corridor,6,1000006,2276.414749145508,176,74.2226331369995, +deadly_corridor,7,1000007,2284.310989379883,176,72.9072057957754, +deadly_corridor,8,1000008,81.07798767089844,49,73.20258272646697, +deadly_corridor,9,1000009,317.2351837158203,75,72.54774919154028, +deadly_corridor,10,1000010,2282.7608489990234,176,72.78293151689127, +deadly_corridor,11,1000011,88.11907958984375,45,72.60486105128022, +deadly_corridor,12,1000012,2281.468536376953,176,72.29193331603048, +deadly_corridor,13,1000013,2276.6868591308594,178,72.73330265771509, +deadly_corridor,14,1000014,2276.1705932617188,178,73.30067987408609, +deadly_corridor,15,1000015,2282.6631622314453,177,72.49405489224537, +deadly_corridor,16,1000016,2280.300033569336,172,72.80884970803692, +deadly_corridor,17,1000017,2280.4182891845703,182,73.03539182090206, +deadly_corridor,18,1000018,2281.2594451904297,177,72.50972089313564, +deadly_corridor,19,1000019,479.8523712158203,99,72.58046231642126, +deadly_corridor,20,1000020,2279.7379455566406,181,72.47468246266928, +deadly_corridor,21,1000021,2284.9097442626953,197,83.18983231769475, +deadly_corridor,22,1000022,2286.2730407714844,172,72.84281562147524, +deadly_corridor,23,1000023,244.51919555664062,74,76.23461799191558, +deadly_corridor,24,1000024,2279.957275390625,195,72.94927214021655, +deadly_corridor,25,1000025,2283.952178955078,179,73.18968843008061, +deadly_corridor,26,1000026,2276.701370239258,178,72.87702909462648, +deadly_corridor,27,1000027,2277.142562866211,190,72.45412386128042, +deadly_corridor,28,1000028,2279.025634765625,177,74.11102172804317, +deadly_corridor,29,1000029,2285.7152099609375,177,71.63189230597281, +deadly_corridor,30,1000030,53.374298095703125,44,72.51418721312025, +deadly_corridor,31,1000031,2279.8080444335938,183,72.72702656843174, +deadly_corridor,32,1000032,2282.307357788086,178,74.33584751930213, +deadly_corridor,33,1000033,2282.834014892578,192,73.95005063555192, +deadly_corridor,34,1000034,2284.200241088867,188,76.29368894499888, +deadly_corridor,35,1000035,2287.2159118652344,179,72.81890806090988, +deadly_corridor,36,1000036,2284.693832397461,183,76.28284599973325, +deadly_corridor,37,1000037,2283.2066650390625,178,72.1797344044525, +deadly_corridor,38,1000038,2281.032196044922,178,73.74343783824916, +deadly_corridor,39,1000039,2282.960678100586,190,73.24816830891406, +deadly_corridor,40,1000040,2287.094253540039,185,72.35711232966574, +deadly_corridor,41,1000041,2279.3030853271484,179,72.42125368367608, +deadly_corridor,42,1000042,440.0892791748047,104,73.92064892672727, +deadly_corridor,43,1000043,2280.8592529296875,177,72.36020918178356, +deadly_corridor,44,1000044,2283.4308471679688,189,75.93658060557208, +deadly_corridor,45,1000045,2282.324264526367,181,73.54224681770178, +deadly_corridor,46,1000046,326.0184631347656,74,73.1983876441008, +deadly_corridor,47,1000047,2279.086135864258,182,73.00958120503027, +deadly_corridor,48,1000048,2280.3804626464844,179,73.17268244992928, +deadly_corridor,49,1000049,2276.215301513672,189,75.47590644230628, +deadly_corridor,50,1000050,2278.132034301758,182,74.50495464842548, +deadly_corridor,51,1000051,2285.6056518554688,181,73.41699294418743, +deadly_corridor,52,1000052,2287.240921020508,173,73.22110809114655, +deadly_corridor,53,1000053,310.81517028808594,73,74.09003681120738, +deadly_corridor,54,1000054,2276.6219787597656,175,72.98605010243534, +deadly_corridor,55,1000055,2276.2769470214844,194,75.17704077845171, +deadly_corridor,56,1000056,2278.861602783203,178,72.97353037051572, +deadly_corridor,57,1000057,2279.728561401367,181,73.96913002154926, +deadly_corridor,58,1000058,2280.544464111328,176,73.02432805290651, +deadly_corridor,59,1000059,487.829833984375,108,78.89398217393664, +deadly_corridor,60,1000060,567.0655517578125,113,72.64874721482185, +deadly_corridor,61,1000061,2278.210220336914,177,72.96068484971086, +deadly_corridor,62,1000062,2281.436721801758,186,75.46710866924751, +deadly_corridor,63,1000063,382.2119903564453,89,81.21157315209366, +deadly_corridor,64,1000064,246.2946014404297,70,73.9736408486285, +deadly_corridor,65,1000065,285.21240234375,76,73.13661133681993, +deadly_corridor,66,1000066,310.6737365722656,75,73.40468658737086, +deadly_corridor,67,1000067,346.1162872314453,75,72.1929723632303, +deadly_corridor,68,1000068,804.7056121826172,150,73.76397959753224, +deadly_corridor,69,1000069,2285.6442108154297,184,75.13255757158333, +deadly_corridor,70,1000070,730.5995788574219,132,73.25446825350764, +deadly_corridor,71,1000071,86.91796875,47,76.28335745963689, +deadly_corridor,72,1000072,60.30122375488281,44,76.83513093208644, +deadly_corridor,73,1000073,768.6264343261719,141,77.27057350071598, +deadly_corridor,74,1000074,2280.1071166992188,172,74.16699734355548, +deadly_corridor,75,1000075,860.9334106445312,151,73.15118478347584, +deadly_corridor,76,1000076,722.9459228515625,143,75.5655785931314, +deadly_corridor,77,1000077,2276.8687438964844,182,72.95102474014934, +deadly_corridor,78,1000078,368.3357238769531,79,71.51096709276341, +deadly_corridor,79,1000079,-76.45918273925781,17,72.24888432102617, +deadly_corridor,80,1000080,2281.5543823242188,183,73.32589540463356, +deadly_corridor,81,1000081,2281.6688842773438,171,73.10600900440717, +deadly_corridor,82,1000082,2277.5223083496094,178,73.55648700566698, +deadly_corridor,83,1000083,42.30937194824219,41,73.52700344736942, +deadly_corridor,84,1000084,2285.8980407714844,176,71.98655161011203, +deadly_corridor,85,1000085,68.90191650390625,45,72.84773487604696, +deadly_corridor,86,1000086,2286.2190551757812,171,72.82303966497733, +deadly_corridor,87,1000087,281.1173553466797,76,72.26983276661764, +deadly_corridor,88,1000088,2283.1607971191406,175,73.49638264342678, +deadly_corridor,89,1000089,2277.888946533203,177,73.44736473371472, +deadly_corridor,90,1000090,429.36326599121094,93,71.86172378947977, +deadly_corridor,91,1000091,252.0751953125,70,72.26459581736903, +deadly_corridor,92,1000092,2278.306442260742,192,80.97328482778371, +deadly_corridor,93,1000093,2285.236801147461,175,74.02717585214627, +deadly_corridor,94,1000094,857.2727355957031,152,85.59110000526613, +deadly_corridor,95,1000095,2275.9288024902344,199,73.62958803645523, +deadly_corridor,96,1000096,2286.8704833984375,179,72.31519682456816, +deadly_corridor,97,1000097,2278.048355102539,181,73.50330330803081, +deadly_corridor,98,1000098,2277.4480743408203,178,76.78472725777, +deadly_corridor,99,1000099,2276.9671478271484,178,77.9678189026336, +ant,0,42,1846.1103431567394,1000,89.89614608291177, +ant,1,43,2415.720790707953,1000,90.00308114332259, +ant,2,44,457.34421085068755,177,89.83716885697598, +ant,3,45,1421.7952163289683,1000,89.87909631338808, +ant,4,46,2037.7234409469488,937,89.82685347370092, +ant,5,47,2330.630175869275,1000,90.47193606091501, +ant,6,48,1161.643572255748,429,89.84194070141322, +ant,7,49,2351.1524624990343,1000,89.92640891799017, +ant,8,50,513.2964809479813,210,89.88895656571908, +ant,9,51,1126.8652528911032,660,89.91361550654544, +ant,10,52,1693.436933192597,1000,89.84960962337662, +ant,11,53,948.3780972955639,1000,89.94678527711802, +ant,12,54,2322.052445211472,1000,90.11873818885832, +ant,13,55,960.4026770814776,1000,90.93377411320307, +ant,14,56,1464.564005196777,1000,89.80893705661644, +ant,15,57,1110.548792782156,1000,89.99466844889166, +ant,16,58,2246.207900740156,1000,90.1624262080728, +ant,17,59,85.64836938561511,60,89.87005518664785, +ant,18,60,340.54799067574436,143,89.94240076131771, +ant,19,61,2457.088748930458,1000,89.92054036086635, +ant,20,62,2166.2512677098603,1000,89.95452553058773, +ant,21,63,2357.957592244385,1000,89.86780458600198, +ant,22,64,1654.8780938737275,871,90.08433827425095, +ant,23,65,1499.367100151414,1000,89.89663615668341, +ant,24,66,2297.4032619179525,1000,90.09818426014289, +ant,25,67,1253.360764666355,543,89.9390124443734, +ant,26,68,1221.270312709775,1000,89.84986177450952, +ant,27,69,2389.2476464763376,1000,89.95772586857817, +ant,28,70,1682.5290233886233,707,89.76145439054764, +ant,29,71,2474.676425615127,1000,89.82093759631324, +ant,30,72,382.9231146443659,256,90.69916524888657, +ant,31,73,1837.8126619276347,1000,90.03642087221974, +ant,32,74,227.19436616673684,101,89.8553742761573, +ant,33,75,1700.6312067622644,1000,89.80097198453268, +ant,34,76,960.9452812639541,372,89.8420903148968, +ant,35,77,2290.6720141359438,1000,89.91770573449698, +ant,36,78,328.5729178056416,162,90.01187187392946, +ant,37,79,1180.073938772476,1000,89.81938304804656, +ant,38,80,817.4190215442345,363,89.85140773938038, +ant,39,81,1651.2255208727013,1000,91.17610023451576, +ant,40,82,1428.174672693164,1000,89.8551155619885, +ant,41,83,1627.3838925098842,1000,90.55986754698809, +ant,42,84,1079.756369746183,680,90.17098553312343, +ant,43,85,2173.9447393037276,1000,89.84319301261918, +ant,44,86,409.90633829945847,160,89.66802828269809, +ant,45,87,2467.2636019929073,1000,89.90844708827387, +ant,46,88,657.4084558813478,248,89.86487149424892, +ant,47,89,974.7436031610902,1000,89.76305094278182, +ant,48,90,1510.5184342975385,1000,90.24355118464125, +ant,49,91,602.2339441184535,260,89.7103209703719, +ant,50,92,760.9784375126189,316,89.8206829517188, +ant,51,93,1941.172113330597,1000,90.30785204408768, +ant,52,94,624.3590446196446,281,89.94582387208622, +ant,53,95,2163.4347041279893,1000,89.84848132390947, +ant,54,96,1126.9957963444238,1000,89.84637728060243, +ant,55,97,1405.131632695366,1000,90.18855922596491, +ant,56,98,1206.2916757636292,1000,89.78065539051504, +ant,57,99,2392.7980761515178,1000,89.76388668266138, +ant,58,100,964.0216541467705,1000,89.82618651237911, +ant,59,101,2252.192880003706,1000,89.8197082349776, +ant,60,102,2471.9158497657563,1000,89.96642568195992, +ant,61,103,1902.8491241623092,1000,89.87542708971246, +ant,62,104,1435.6661382989703,1000,90.28644124851098, +ant,63,105,1668.3237703695809,1000,89.86433221097877, +ant,64,106,1813.291243529155,1000,89.85118001877315, +ant,65,107,446.72353548541076,189,89.8309544306309, +ant,66,108,130.84194814079504,74,89.82416773165995, +ant,67,109,2315.857153770824,1000,90.25959750757508, +ant,68,110,288.3915792961347,116,90.09271984792927, +ant,69,111,894.0228631227924,1000,89.89663691508213, +ant,70,112,2030.322535823717,1000,89.84028619017428, +ant,71,113,507.9449555916754,215,90.57267432538549, +ant,72,114,2377.7373967468293,1000,89.84919425782105, +ant,73,115,897.3077114027096,1000,89.90431472264346, +ant,74,116,1454.612590266188,1000,91.19515970740413, +ant,75,117,2292.457960175467,1000,89.8333901030839, +ant,76,118,1424.378337790017,1000,89.88029014661089, +ant,77,119,1441.1111023164538,1000,89.79844243631413, +ant,78,120,1265.4771503717611,1000,89.86009503143968, +ant,79,121,1662.8808067819505,1000,90.5661722205243, +ant,80,122,2508.917122342891,1000,89.87403626041336, +ant,81,123,1655.3510139158748,1000,90.05039760075688, +ant,82,124,1387.3843721247736,821,90.10235730111886, +ant,83,125,646.4356689469432,271,89.7559653760994, +ant,84,126,2172.801064037805,1000,89.84602989356796, +ant,85,127,165.9213897970373,72,89.66756877688618, +ant,86,128,1063.1483912161111,1000,89.77409215132576, +ant,87,129,1000.135342286622,1000,89.81900933661238, +ant,88,130,1977.2359176146426,1000,89.77653862908736, +ant,89,131,1937.1674235355138,1000,90.12773943823525, +ant,90,132,1344.7729257831547,1000,90.39620143170467, +ant,91,133,786.3379828975102,441,89.82893206036925, +ant,92,134,1391.060299752017,1000,89.86676880070257, +ant,93,135,503.300235688713,250,89.94819176115624, +ant,94,136,2446.7482357041768,1000,89.84848658183878, +ant,95,137,1171.9102336514923,1000,89.92785850220504, +ant,96,138,2356.7311711183065,1000,90.42933754946152, +ant,97,139,2356.12199478712,1000,89.82049779117614, +ant,98,140,1389.2987977192308,1000,90.82209581044775, +ant,99,141,967.2335383727841,1000,89.80016695371027, +intercept,0,4242424242,0.7267571190313902,60,99.89614420497905,0.0 +intercept,1,4242424243,2.9096362272975966,60,99.91707940536706,0.0 +intercept,2,4242424244,3.2060351513209753,60,97.78809018716221,0.0 +intercept,3,4242424245,0.7574528902187012,60,98.54894447730877,0.0 +intercept,4,4242424246,0.6827895979695313,60,99.11745353519741,0.0 +intercept,5,4242424247,29.923812823486514,60,99.04064156549293,1.0 +intercept,6,4242424248,0.7661087726592086,60,98.25342313549518,0.0 +intercept,7,4242424249,0.8284444468154106,60,97.98431264506286,0.0 +intercept,8,4242424250,0.9707721562881488,60,99.10671115977826,0.0 +intercept,9,4242424251,1.0944434545235708,60,99.10142489904808,0.0 +intercept,10,4242424252,0.7526731102407211,60,98.24263629181895,0.0 +intercept,11,4242424253,1.0327306617691647,60,99.90610126116793,0.0 +intercept,12,4242424254,24.08529434411321,60,99.04964452767656,1.0 +intercept,13,4242424255,0.8258126199943945,60,99.06333184347895,0.0 +intercept,14,4242424256,0.6465023508935701,60,99.96017435988418,0.0 +intercept,15,4242424257,1.2059930491086561,60,99.07293754243183,0.0 +intercept,16,4242424258,0.8975468523567542,60,99.10975490804557,0.0 +intercept,17,4242424259,0.638558203499997,60,99.9008234011206,0.0 +intercept,18,4242424260,2.473904824233614,60,99.1007534285042,0.0 +intercept,19,4242424261,0.8594156300532632,60,98.2804424689215,0.0 +intercept,20,4242424262,0.7127419076277874,60,97.42140552034121,0.0 +intercept,21,4242424263,1.1195833964738995,60,99.05356389575846,0.0 +intercept,22,4242424264,1.4589147588121705,60,99.10476263429966,0.0 +intercept,23,4242424265,22.348254217096837,60,98.14243140713285,1.0 +intercept,24,4242424266,27.43761277961312,60,99.03063235183511,1.0 +intercept,25,4242424267,0.797963114338927,60,98.95851806063928,0.0 +intercept,26,4242424268,0.6415987604705151,60,99.1202532952496,0.0 +intercept,27,4242424269,1.502438226743834,60,99.8369766656745,0.0 +intercept,28,4242424270,1.277322537265718,60,99.13638822823135,0.0 +intercept,29,4242424271,0.6413188653605175,60,99.8165233572777,0.0 +intercept,30,4242424272,26.015227647672873,60,99.93359984997578,1.0 +intercept,31,4242424273,0.7568511647114065,60,98.29798580223347,0.0 +intercept,32,4242424274,0.7758818510046694,60,95.9119617819155,0.0 +intercept,33,4242424275,0.743274000211386,60,99.16579733811342,0.0 +intercept,34,4242424276,0.9812663898337632,60,99.96913332715677,0.0 +intercept,35,4242424277,0.7364500367548317,60,98.4144170848438,0.0 +intercept,36,4242424278,0.7676261149172205,60,99.87913624991887,0.0 +intercept,37,4242424279,2.6105462690466084,60,99.0124647390605,0.0 +intercept,38,4242424280,0.8922563010128215,60,99.49582641131909,0.0 +intercept,39,4242424281,0.7909053032053635,60,99.95776157301488,0.0 +intercept,40,4242424282,27.747763212013524,60,99.89232705853966,1.0 +intercept,41,4242424283,2.830903574009426,60,99.11182141335861,0.0 +intercept,42,4242424284,3.749473527306691,60,99.89401411987875,0.0 +intercept,43,4242424285,3.2371535471174866,60,98.58941395009701,0.0 +intercept,44,4242424286,1.141169616690604,60,98.95023432158384,0.0 +intercept,45,4242424287,1.2504711685760412,60,99.8783128676535,0.0 +intercept,46,4242424288,1.1401455145678483,60,99.09364640302553,0.0 +intercept,47,4242424289,1.1743367564631626,60,98.24703755640672,0.0 +intercept,48,4242424290,0.6911400489043444,60,98.98847807253395,0.0 +intercept,49,4242424291,0.966755291854497,60,98.31297463384391,0.0 +intercept,50,4242424292,3.7725237559643574,60,98.1307167401627,0.0 +intercept,51,4242424293,0.7292428385990206,60,99.08520847604322,0.0 +intercept,52,4242424294,2.733719722367823,60,99.8949988335446,0.0 +intercept,53,4242424295,2.7277548569836654,60,99.16542541107671,0.0 +intercept,54,4242424296,0.8013565168366767,60,98.2801475641182,0.0 +intercept,55,4242424297,0.9918300381395966,60,98.74883429246843,0.0 +intercept,56,4242424298,3.8384227409260347,60,98.30485570834159,0.0 +intercept,57,4242424299,2.525593837024644,60,99.08211861473346,0.0 +intercept,58,4242424300,1.1939986812940333,60,99.95562586586023,0.0 +intercept,59,4242424301,1.1946645161951892,60,99.11887142756973,0.0 +intercept,60,4242424302,0.6632764584392135,60,99.92733404817194,0.0 +intercept,61,4242424303,0.7345126099826302,60,99.61093201950378,0.0 +intercept,62,4242424304,1.1547945403144695,60,98.88382479344455,0.0 +intercept,63,4242424305,1.0395031699445099,60,99.10455669644611,0.0 +intercept,64,4242424306,2.7713681719324086,60,99.99607387713222,0.0 +intercept,65,4242424307,3.8083399715833366,60,99.90908449191997,0.0 +intercept,66,4242424308,3.1245881704380736,60,99.86388390473627,0.0 +intercept,67,4242424309,0.9936205917911138,60,99.06530098425861,0.0 +intercept,68,4242424310,0.6479002644773573,60,97.31661851374615,0.0 +intercept,69,4242424311,1.09404552471824,60,99.06482130667098,0.0 +intercept,70,4242424312,0.725047086874838,60,99.39714496924636,0.0 +intercept,71,4242424313,2.085218493710272,60,99.91926924929075,0.0 +intercept,72,4242424314,25.112157980707707,60,99.89495984140663,1.0 +intercept,73,4242424315,0.7960666966973804,60,99.91315720008677,0.0 +intercept,74,4242424316,1.8899870013119653,60,99.8635479883608,0.0 +intercept,75,4242424317,24.77215793245705,60,99.08978442272605,1.0 +intercept,76,4242424318,0.763190906640375,60,99.1181927131107,0.0 +intercept,77,4242424319,0.8356004936795216,60,98.87472332915829,0.0 +intercept,78,4242424320,24.543561146681895,60,98.85109478338812,1.0 +intercept,79,4242424321,0.7962639288743958,60,96.64267992061197,0.0 +intercept,80,4242424322,0.6807828926102957,60,98.70551839611774,0.0 +intercept,81,4242424323,1.1704122956143692,60,99.1162675413269,0.0 +intercept,82,4242424324,0.8024117537715938,60,99.10984056283594,0.0 +intercept,83,4242424325,1.0154686415335163,60,99.90653765962175,0.0 +intercept,84,4242424326,0.6267238368745893,60,99.14313411902761,0.0 +intercept,85,4242424327,1.1180786813492887,60,99.87378109642233,0.0 +intercept,86,4242424328,1.0531825890648179,60,99.90759275984404,0.0 +intercept,87,4242424329,0.7319892354425974,60,99.07934787032669,0.0 +intercept,88,4242424330,1.1460731038823724,60,98.32584786308246,0.0 +intercept,89,4242424331,1.145515855285339,60,99.853580446333,0.0 +intercept,90,4242424332,3.1438898412743583,60,98.23875463665809,0.0 +intercept,91,4242424333,1.1678254807484336,60,99.94300368083988,0.0 +intercept,92,4242424334,1.1468605129048228,60,98.28856657896678,0.0 +intercept,93,4242424335,2.816772125195712,60,99.10609232867152,0.0 +intercept,94,4242424336,1.1577836629003286,60,99.93574594730708,0.0 +intercept,95,4242424337,1.0533778404060286,60,99.92123883389186,0.0 +intercept,96,4242424338,0.8533297177054919,60,99.06174133027585,0.0 +intercept,97,4242424339,0.7617567333800253,60,99.95316521359209,0.0 +intercept,98,4242424340,0.9895546428160742,60,99.11385213912092,0.0 +intercept,99,4242424341,0.770722996792756,60,99.44169788411487,0.0 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.csv b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.csv new file mode 100644 index 0000000000000000000000000000000000000000..1b381df6789eea28ea56149c4b780cf0cade01c0 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.csv @@ -0,0 +1,5 @@ +task,episodes,return_mean,return_sd,length_mean,length_sd,success_count,success_rate,invalid_actions,dropped_actions +flappy,100,384.8240045265853,116.78777394316903,3119.31,939.8693174585497,,,0,64 +deadly_corridor,100,1620.7987757873534,913.6242782186637,148.53,49.455930888013825,,,0,0 +ant,100,1453.844063807972,693.7275200567642,803.85,328.8088312378486,,,0,0 +intercept,100,3.5443485127069287,7.07192296411853,60.0,0.0,9,0.09,0,10 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.json new file mode 100644 index 0000000000000000000000000000000000000000..8486070c582599f0cb0c336c70e6569823f4990f --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/comparison.json @@ -0,0 +1,205 @@ +{ + "condition": "profile-latency", + "executor_mode": "simulated", + "latency_method": "temporal/profile_sample", + "episodes_per_checkpoint": 100, + "total_episodes": 400, + "checkpoints_metadata_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "results": { + "flappy": { + "n_episodes": 100, + "mean_return": 384.8240045265853, + "std_return": 116.78777394316903, + "min_return": 36.60000045597553, + "max_return": 444.6000052243471, + "mean_length": 3119.31, + "std_length": 939.8693174585497, + "min_length": 314.0, + "max_length": 3600.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "flappy", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "flappy", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T122201421825Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "flappy", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42", + "profile_sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 64, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy.yaml", + "execution_audit": { + "issued_action_records": 311075, + "applied_action_records": 310911, + "dropped_action_records": 64, + "nonnoop_issued_records": 30817, + "finite_action_values": true, + "latency_sample_count": 311075, + "latency_mean_ms": 75.89784633675906, + "latency_std_ms": 3.799946378622932, + "latency_p95_ms": 81.3960393048375, + "latency_p99_ms": 87.23844517488543 + } + }, + "deadly_corridor": { + "n_episodes": 100, + "mean_return": 1620.7987757873534, + "std_return": 913.6242782186637, + "min_return": -76.45918273925781, + "max_return": 2287.240921020508, + "mean_length": 148.53, + "std_length": 49.455930888013825, + "min_length": 17.0, + "max_length": 199.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "doom_deadly_corridor", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "deadly_corridor", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T171446047509Z", + "profile_ref": null, + "env_fps": 35.0, + "obs_fps": 8.75, + "frame_ms": 28.571428571428573, + "latency_type": "profile_sample", + "task": "deadly_corridor", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42", + "profile_sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor.yaml", + "execution_audit": { + "issued_action_records": 3753, + "applied_action_records": 3673, + "dropped_action_records": 0, + "nonnoop_issued_records": 3753, + "finite_action_values": true, + "latency_sample_count": 3753, + "latency_mean_ms": 74.01999621872471, + "latency_std_ms": 5.5537519652567635, + "latency_p95_ms": 89.54825982614612, + "latency_p99_ms": 95.97310052501227 + } + }, + "ant": { + "n_episodes": 100, + "mean_return": 1453.844063807972, + "std_return": 693.7275200567642, + "min_return": 85.64836938561511, + "max_return": 2508.917122342891, + "mean_length": 803.85, + "std_length": 328.8088312378486, + "min_length": 60.0, + "max_length": 1000.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "LatencyBench/AntContinuous-v0", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "ant", + "instance_id": "instance_859cf1e47bca6046", + "source_run_id": "20260911T033037730561Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "ant", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42", + "profile_sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant.yaml", + "execution_audit": { + "issued_action_records": 79573, + "applied_action_records": 79465, + "dropped_action_records": 0, + "nonnoop_issued_records": 79573, + "finite_action_values": true, + "latency_sample_count": 79573, + "latency_mean_ms": 90.00919554158884, + "latency_std_ms": 2.514492574433973, + "latency_p95_ms": 91.11971585797141, + "latency_p99_ms": 102.67108120995428 + } + }, + "intercept": { + "n_episodes": 100, + "mean_return": 3.5443485127069287, + "std_return": 7.07192296411853, + "min_return": 0.6267238368745893, + "max_return": 29.923812823486514, + "mean_length": 60.0, + "std_length": 0.0, + "min_length": 60.0, + "max_length": 60.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "mikasa_intercept_grab_fast", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "mikasa_intercept_grab_fast", + "instance_id": "instance_3a0d42681a03715c", + "source_run_id": "20260909T044501695676Z", + "profile_ref": null, + "env_fps": 20.0, + "obs_fps": 20.0, + "frame_ms": 50.0, + "latency_type": "profile_sample", + "task": "intercept", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0", + "profile_sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 10, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept.yaml", + "success_count": 9, + "success_rate": 0.09, + "execution_audit": { + "issued_action_records": 2974, + "applied_action_records": 2864, + "dropped_action_records": 10, + "nonnoop_issued_records": 2974, + "finite_action_values": true, + "latency_sample_count": 2974, + "latency_mean_ms": 99.11060319379854, + "latency_std_ms": 4.301543980874005, + "latency_p95_ms": 100.2889407458356, + "latency_p99_ms": 100.64616770379737 + } + } + }, + "quality_acceptance": "not inferred; observed statistics only" +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/episodes.csv b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/episodes.csv new file mode 100644 index 0000000000000000000000000000000000000000..1febb012a47b2112b19d73367a5f3616e7bd210e --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/episodes.csv @@ -0,0 +1,101 @@ +episode_id,seed,return_env,length,mean_latency_ms,invalid_actions,dropped_actions +0,1000000,444.6000052243471,3600,76.02271694866694,0,0 +1,1000001,444.6000052243471,3600,76.14445348705047,0,1 +2,1000002,444.6000052243471,3600,75.83047266244563,0,0 +3,1000003,444.6000052243471,3600,76.04121221698036,0,2 +4,1000004,444.6000052243471,3600,75.7789115791707,0,1 +5,1000005,228.2000027000904,1861,76.22757676162651,0,0 +6,1000006,444.6000052243471,3600,75.98373978309758,0,0 +7,1000007,444.6000052243471,3600,75.85552109823348,0,2 +8,1000008,444.6000052243471,3600,75.9782303085917,0,2 +9,1000009,444.6000052243471,3600,75.94667987356688,0,1 +10,1000010,444.6000052243471,3600,75.66396359484234,0,0 +11,1000011,444.6000052243471,3600,75.7794525026407,0,0 +12,1000012,444.6000052243471,3600,75.90110110734818,0,0 +13,1000013,444.6000052243471,3600,76.01870178237883,0,0 +14,1000014,444.6000052243471,3600,75.75567207010911,0,0 +15,1000015,444.6000052243471,3600,75.83026036637241,0,0 +16,1000016,444.6000052243471,3600,75.74502908171665,0,1 +17,1000017,444.6000052243471,3600,75.84316844302293,0,1 +18,1000018,444.6000052243471,3600,75.85876738771161,0,2 +19,1000019,265.50000313669443,2162,75.89492798135642,0,0 +20,1000020,444.6000052243471,3600,75.90859756288593,0,0 +21,1000021,444.6000052243471,3600,75.93474621914784,0,0 +22,1000022,444.6000052243471,3600,75.77022360156529,0,0 +23,1000023,444.6000052243471,3600,75.8506098974935,0,0 +24,1000024,444.6000052243471,3600,75.80511776716725,0,0 +25,1000025,116.00000138580799,955,76.07937915327228,0,1 +26,1000026,444.6000052243471,3600,75.77409482659607,0,0 +27,1000027,444.6000052243471,3600,75.82354466933252,0,0 +28,1000028,444.6000052243471,3600,75.92578714415393,0,2 +29,1000029,444.6000052243471,3600,75.77326038618416,0,0 +30,1000030,256.0000030249357,2085,75.8461606092662,0,1 +31,1000031,444.6000052243471,3600,75.87053786258159,0,1 +32,1000032,444.6000052243471,3600,75.90930861144982,0,1 +33,1000033,444.6000052243471,3600,75.80530422686525,0,0 +34,1000034,444.6000052243471,3600,76.05997569829616,0,1 +35,1000035,444.6000052243471,3600,75.67579907153437,0,0 +36,1000036,444.6000052243471,3600,76.07561842170198,0,3 +37,1000037,444.6000052243471,3600,75.87459102177027,0,0 +38,1000038,55.60000067949295,468,75.8887188983619,0,0 +39,1000039,444.6000052243471,3600,75.86536772802552,0,0 +40,1000040,432.900005094707,3512,76.00355652525975,0,0 +41,1000041,274.8000032454729,2237,75.7658282850597,0,0 +42,1000042,264.90000312775373,2156,75.95264956954799,0,0 +43,1000043,265.4000031352043,2161,75.82748305801191,0,1 +44,1000044,444.6000052243471,3600,75.9295822845668,0,1 +45,1000045,143.90000171214342,1180,75.94310218110371,0,0 +46,1000046,444.6000052243471,3600,75.69568531179425,0,0 +47,1000047,93.1000011190772,771,76.0527875505066,0,0 +48,1000048,56.10000068694353,473,76.20882901957174,0,0 +49,1000049,265.2000031322241,2159,76.05401077635972,0,0 +50,1000050,444.6000052243471,3600,75.89333271844873,0,1 +51,1000051,444.6000052243471,3600,75.89090159365671,0,0 +52,1000052,398.80000469088554,3234,75.91218218803246,0,1 +53,1000053,444.6000052243471,3600,75.86400590251726,0,0 +54,1000054,270.2000031918287,2200,76.01590238337654,0,2 +55,1000055,69.70000084489584,582,75.68422480575155,0,0 +56,1000056,444.6000052243471,3600,75.87884524455251,0,0 +57,1000057,444.6000052243471,3600,75.96981187494319,0,1 +58,1000058,444.6000052243471,3600,76.03772455115222,0,0 +59,1000059,437.90000515431166,3553,76.04088529786887,0,1 +60,1000060,348.9000041112304,2834,75.79422825165413,0,0 +61,1000061,444.6000052243471,3600,75.87728099437057,0,1 +62,1000062,78.9000009521842,656,75.96352981662133,0,0 +63,1000063,444.6000052243471,3600,75.80269270184165,0,1 +64,1000064,444.6000052243471,3600,75.88518180564401,0,0 +65,1000065,444.6000052243471,3600,75.87533034544981,0,1 +66,1000066,444.6000052243471,3600,75.94241138050401,0,0 +67,1000067,444.6000052243471,3600,75.95312277771471,0,0 +68,1000068,444.6000052243471,3600,75.8998829764233,0,0 +69,1000069,444.6000052243471,3600,75.98564617573034,0,1 +70,1000070,444.6000052243471,3600,75.68328575087021,0,0 +71,1000071,135.0000016093254,1109,75.99546963217229,0,4 +72,1000072,444.6000052243471,3600,75.9923106611263,0,0 +73,1000073,444.6000052243471,3600,75.80422251719546,0,0 +74,1000074,444.6000052243471,3600,75.95469853250815,0,0 +75,1000075,444.6000052243471,3600,75.74551875442629,0,1 +76,1000076,444.6000052243471,3600,75.93301571087362,0,3 +77,1000077,444.6000052243471,3600,75.98384926019328,0,1 +78,1000078,444.6000052243471,3600,75.85055115368883,0,1 +79,1000079,444.6000052243471,3600,75.97142616222317,0,2 +80,1000080,444.6000052243471,3600,75.97039764106849,0,0 +81,1000081,444.6000052243471,3600,75.74469321422862,0,0 +82,1000082,116.20000138878822,957,76.0366526049804,0,1 +83,1000083,444.6000052243471,3600,75.95924386190674,0,0 +84,1000084,444.6000052243471,3600,76.0310580385874,0,0 +85,1000085,36.60000045597553,314,75.8634823847272,0,0 +86,1000086,260.90000308305025,2125,75.91783880059099,0,2 +87,1000087,444.6000052243471,3600,76.16752514785735,0,0 +88,1000088,444.6000052243471,3600,75.83331254385584,0,1 +89,1000089,444.6000052243471,3600,76.2113387300584,0,7 +90,1000090,444.6000052243471,3600,75.8334932097261,0,1 +91,1000091,225.20000265538692,1831,75.75618859671614,0,0 +92,1000092,444.6000052243471,3600,75.79683788505955,0,1 +93,1000093,180.9000021442771,1478,75.96465307644473,0,0 +94,1000094,305.2000035941601,2478,75.86565754734926,0,0 +95,1000095,444.6000052243471,3600,75.74523644464854,0,1 +96,1000096,444.6000052243471,3600,75.94353591524424,0,1 +97,1000097,444.6000052243471,3600,75.81927739599219,0,1 +98,1000098,444.6000052243471,3600,75.96229410618645,0,0 +99,1000099,444.6000052243471,3600,75.94512877548694,0,1 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/eval_config.yaml b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/eval_config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..91e25ca724de4d0e3b48900a0afb7c517b3b439a --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/eval_config.yaml @@ -0,0 +1,82 @@ +experiment: + name: flappy-mean5000-profile-simulation-100ep + seed: 1000000 +backend: + run_mode: eval +executor: + mode: simulated + simulated_worker_capacity: 1 + simulated_inference_pool: true + inference_devices: + - cuda:0 + inference_batch_size: 32 +env: + name: flappy + gym_id: FlappyBird-v0 + env_fps: 10 + obs_fps: 10 + render_mode: rgb_array + observation_mode: image + use_lidar: false + normalize_obs: true + audio_on: false + frame_stack: 1 + obs_resize: + - 224 + - 224 + simulator: gpu + use_gpu_render: false + gpu_render_device: auto + gpu_render_profile: false + gpu_render_profile_interval: 200 + noop_action: noop + action_map: + noop: 0 + flap: 1 + oneshot_actions: + - flap + action_history_decisions: 8 +latency: + method: temporal + profile_path: /home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/flappy/instance_a5037b165aa0cedc/profile.json + profile_worker_slot: 0 + seed: 271828 + add_latency_info: false +scheduler: + hold_policy: one_frame_then_noop + ordering_policy: latest_ready +policy: + type: starvla + checkpoint_path: /home/ubuntu/lzj/mean-profiling/flappy/vla-publication/checkpoints/model.pt + model_config_path: /home/ubuntu/lzj/mean-profiling/flappy/vla-publication/config.full.yaml + device: cuda:0 + unnorm_key: new_embodiment + prompt_mode: latency_neutral + actions: + - noop + - flap + state_source: transport + backbone_path: /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct + worker_python_executable: /home/ubuntu/lzj/conda/envs/qwenoft/bin/python +evaluation: + eval_episodes: 100 + eval_parallel_envs: 32 + latency_bench_env_backend: flappy_gpu_batched + eval_latency_values: null + eval_max_steps: 3600 + eval_deterministic: true + eval_raw_reward: true + eval_suites: + fixed: [] + normal: [] + uniform: [] +logging: + output_dir: /home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy + save_step_records: true + save_action_records: true + save_latency_records: true + video: + enabled: false + wandb_project: null + wandb_group: null + wandb_job_type: null diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py new file mode 100644 index 0000000000000000000000000000000000000000..d5edbe901e40e8e9cbb2e0763281fdfcd77dbfcc --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py @@ -0,0 +1,702 @@ +from __future__ import annotations + +import time +from collections.abc import Callable, Mapping, Sequence +from dataclasses import dataclass, field +from pathlib import Path + +import numpy as np + +from latency_bench.core.clock import EnvClock +from latency_bench.core.decision_action_history import DecisionActionHistory +from latency_bench.core.timing import StageProfiler, profiler_scope +from latency_bench.core.types import ActionEvent, EpisodeMetrics, LatencyRecord, Observation, StepRecord +from latency_bench.envs.atari import TRUE_EPISODE_END_INFO_KEY +from latency_bench.envs.base import EnvAdapter +from latency_bench.executors._simulated_timeline import ( + SimulatedResultTimeline, + SimulatedWorkerCapacity, + build_simulated_action_event, +) +from latency_bench.executors.base import BatchedExecutor +from latency_bench.executors.env_step_backend import EnvStepBackend, env_action_space +from latency_bench.latency.sample import LatencySample +from latency_bench.latency.samplers import LatencySampler +from latency_bench.logging.metrics import ( + compute_episode_metrics, + compute_episode_metrics_from_aggregates, + episode_raw_fact_metadata, + latency_type_from_source, + profile_metadata_from_source, +) +from latency_bench.logging.records import build_step_record +from latency_bench.logging.trajectory_logger import TrajectoryLogger +from latency_bench.policy.action_prefix import with_action_prefix +from latency_bench.policy.base import PolicyRunner +from latency_bench.scheduler.action_queue import ActionScheduler +from latency_bench.scheduler.decision import DecisionScheduler +from latency_bench.utils.io import write_json +from latency_bench.utils.stats import series_stats + + +@dataclass +class _EpisodeBuffers: + step_records: list[StepRecord] | None = None + action_events: list[ActionEvent] | None = None + latency_records: list[LatencyRecord] | None = None + latency_values_ms: list[float] = field(default_factory=list) + episode_return_env: float = 0.0 + survival_steps: int = 0 + game_score: float | None = None + return_raw: float | None = None + num_actions: int = 0 + num_dropped_actions: int = 0 + num_invalid_actions: int = 0 + submitted_observation_frames: int = 0 + dropped_observation_count: int = 0 + soft_reset_count: int = 0 + final_lives: int | None = None + final_is_true_episode_end: bool | None = None + task_metrics: dict | None = None + task_metric_moments: dict | None = None + + def record_step(self, *, reward: float, info: dict) -> None: + self.episode_return_env += float(reward) + self.survival_steps += 1 + if "invalid_action" in info and info["invalid_action"]: + self.num_invalid_actions += 1 + if "soft_reset" in info and info["soft_reset"]: + self.soft_reset_count += 1 + if "lives" in info: + self.final_lives = info["lives"] + if TRUE_EPISODE_END_INFO_KEY in info: + self.final_is_true_episode_end = info[TRUE_EPISODE_END_INFO_KEY] + if "game_score" in info: + self.game_score = float(info["game_score"]) + if "score" in info: + self.game_score = float(info["score"]) + if "task_metrics" in info: + self.task_metrics = info["task_metrics"] + if "task_metric_moments" in info: + self.task_metric_moments = info["task_metric_moments"] + self._update_return_raw(info) + extra_stats = info["episode_extra_stats"] if "episode_extra_stats" in info else None + if isinstance(extra_stats, dict): + self._update_return_raw(extra_stats) + + def _update_return_raw(self, stats: dict) -> None: + for key in ("return_raw", "raw_return", "episodic_raw_return", "episode/raw_return"): + if key in stats and stats[key] is not None: + self.return_raw = float(stats[key]) + + +@dataclass +class _SlotState: + slot_id: int + env: EnvAdapter + latency_source: LatencySampler + action_scheduler: ActionScheduler + result_timeline: SimulatedResultTimeline + active: bool = False + episode_id: int | None = None + episode_seed: int | None = None + env_step: int = 0 + recent_drop_count: int = 0 + decision_action_history: DecisionActionHistory | None = None + decision_admitted: bool = False + decision_issued_action: object = None + buffers: _EpisodeBuffers = field(default_factory=_EpisodeBuffers) + worker_capacity: SimulatedWorkerCapacity = field( + default_factory=lambda: SimulatedWorkerCapacity(capacity=None, busy_until_by_worker={}) + ) + + +@dataclass +class _PendingPolicyObservation: + slot: _SlotState + observation: Observation + obs_id: int + latency_sample: LatencySample + worker_slot: int + + +class BatchedSimulatedLatencyExecutor(BatchedExecutor): + """Run multiple simulated episodes concurrently with independent slot state. + + The main process owns policy inference, latency scheduling, episode accounting, + and logging. Env stepping can be serial in-process or delegated to worker + subprocesses through env_backend. + """ + + def __init__( + self, + *, + env_backend: EnvStepBackend, + policy: PolicyRunner, + decision_scheduler: DecisionScheduler, + latency_sources: Sequence[LatencySampler], + action_schedulers: Sequence[ActionScheduler], + clock: EnvClock, + logger: TrajectoryLogger | None = None, + episode_latency_source_factory: Callable[[int], LatencySampler] | None = None, + simulated_worker_capacity: int | None = None, + profile_pipeline: bool = False, + inference_pool=None, + action_prefix=None, + action_history_decisions: int | None = None, + ): + slot_count = env_backend.num_slots + self.env_backend = env_backend + self.envs = list(env_backend.slot_handles) + self.policy = policy + self.decision_scheduler = decision_scheduler + self.clock = clock + self.logger = logger + self.profile_pipeline = bool(profile_pipeline) + self.inference_pool = inference_pool + self.action_prefix = action_prefix + self._pipeline_profile_rows: list[dict[str, float]] = [] + self.simulated_worker_capacity = simulated_worker_capacity + self._collect_step_records = bool(logger is not None and logger.save_step_records) + self._collect_action_records = bool(logger is not None and logger.save_action_records) + self._collect_latency_records = bool(logger is not None and logger.save_latency_records) + self.episode_latency_source_factory = episode_latency_source_factory + self.slots = [ + _SlotState( + slot_id=slot_id, + env=self.envs[slot_id], + latency_source=latency_sources[slot_id], + action_scheduler=action_schedulers[slot_id], + result_timeline=SimulatedResultTimeline( + ordering_policy=action_schedulers[slot_id].ordering_policy + ), + decision_action_history=( + DecisionActionHistory( + env_action_space(self.envs[slot_id]), num_envs=1, decisions=action_history_decisions + ) if action_history_decisions is not None else None + ), + buffers=self._new_episode_buffers(), + worker_capacity=SimulatedWorkerCapacity( + capacity=simulated_worker_capacity, + busy_until_by_worker={}, + ), + ) + for slot_id in range(slot_count) + ] + self._next_obs_id = 0 + self._next_action_id = 0 + self.started_episodes = 0 + self.completed_episodes = 0 + self._completed_metrics: dict[int, EpisodeMetrics] = {} + self._completed_buffers: dict[int, _EpisodeBuffers] = {} + self._episode_log_order: list[int] = [] + self._next_episode_log_index = 0 + + @property + def num_slots(self) -> int: + return len(self.slots) + + def close(self) -> None: + if self.inference_pool is not None: + self.inference_pool.close() + self.env_backend.close() + + def run_episodes( + self, + *, + episode_ids: Sequence[int], + seeds: Sequence[int | None], + eval_max_steps: int = 10000, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + ) -> list[EpisodeMetrics]: + if eval_max_steps < 0: + raise ValueError("eval_max_steps must be non-negative") + episode_ids = [int(episode_id) for episode_id in episode_ids] + if len(seeds) != len(episode_ids): + raise ValueError("seeds length must match episode_ids length") + + self._reset_run_state(episode_ids) + if not episode_ids: + return [] + + next_episode_index = 0 + initial_slots = min(self.num_slots, len(episode_ids)) + for slot in self.slots[:initial_slots]: + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + + while self.completed_episodes < len(episode_ids): + active_slots = self._active_slots() + if eval_max_steps == 0: + for slot in active_slots: + self._complete_slot(slot, on_episode_complete=on_episode_complete) + if next_episode_index < len(episode_ids): + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + continue + + observations = [] + observation_slots: list[_SlotState] = [] + step_capacity_info: dict[int, dict[str, int | bool | None]] = {} + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + slot.worker_capacity.release_ready(slot.env_step) + self._deliver_arrived_results(slot, raw_frame=slot.env_step) + observation_submitted = False + observation_dropped = False + if self.decision_scheduler.should_observe(slot.env_step, current_time_ms): + prefix_request_pending = ( + self.action_prefix is not None + and self.action_prefix["mode"] != "none" + and slot.result_timeline.pending_observation_count > 0 + ) + if slot.worker_capacity.can_submit() and not prefix_request_pending: + observation_slots.append(slot) + observation_submitted = True + else: + slot.buffers.dropped_observation_count += 1 + observation_dropped = True + slot.recent_drop_count += 1 + if self.simulated_worker_capacity is not None: + step_capacity_info[slot.slot_id] = { + "observation_submitted": observation_submitted, + "observation_dropped": observation_dropped, + } + if slot.decision_action_history is not None and slot.env_step % self.clock.obs_stride_raw_frames == 0: + slot.decision_admitted = observation_submitted + slot.decision_issued_action = slot.action_scheduler.noop_action.value + + observe_ms = 0.0 + if observation_slots: + observe_start = time.perf_counter() + observations_by_slot = self.env_backend.observe_slots([slot.slot_id for slot in observation_slots]) + observe_ms = (time.perf_counter() - observe_start) * 1000.0 + pending_observations = [ + self._sample_policy_observation( + slot, + self._policy_observation( + slot, + observations_by_slot[slot.slot_id], + transport=( + slot.decision_action_history.observation()[0] + if slot.decision_action_history is not None else None + ), + ), + ) + for slot in observation_slots + ] + observations = [pending.observation for pending in pending_observations] + + profile_row = None + if observations: + profiler = StageProfiler(enabled=self.profile_pipeline) + with profiler_scope(profiler): + policy_outputs = ( + self.inference_pool.predict_batch(observations) + if self.inference_pool is not None + else self.policy.predict_batch(observations) + ) + if len(policy_outputs) != len(observations): + raise RuntimeError("policy.predict_batch returned the wrong number of outputs") + if self.profile_pipeline: + profile_row = { + "active_slots": float(len(active_slots)), + "batch_size": float(len(observations)), + "observe_slots_ms": observe_ms, + **{key: float(value) for key, value in profiler.timings.items()}, + } + for pending, policy_output in zip(pending_observations, policy_outputs): + if pending.slot.decision_action_history is not None: + pending.slot.decision_issued_action = policy_output.action.value + self._enqueue_policy_output( + pending.slot, + pending.observation, + policy_output, + obs_id=pending.obs_id, + latency_sample=pending.latency_sample, + worker_slot=pending.worker_slot, + ) + + actions_by_slot = {} + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + self._deliver_arrived_results(slot, raw_frame=slot.env_step) + active_action = slot.action_scheduler.update(slot.env_step, current_time_ms) + actions_by_slot[slot.slot_id] = active_action + + env_step_start = time.perf_counter() + step_responses = self.env_backend.step_slots(actions_by_slot) + if profile_row is not None: + profile_row["env_step_ms"] = (time.perf_counter() - env_step_start) * 1000.0 + self._pipeline_profile_rows.append(profile_row) + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + active_action = actions_by_slot[slot.slot_id] + if ( + slot.decision_action_history is not None + and (slot.env_step + 1) % self.clock.obs_stride_raw_frames == 0 + ): + slot.decision_action_history.append( + [0], [slot.decision_admitted], + [slot.decision_issued_action], [active_action.value], + ) + result = step_responses[slot.slot_id].result + soft_reset = bool(result.info.get("soft_reset")) if isinstance(result.info, dict) else False + episode_done = bool(result.done or result.truncated) and not soft_reset + if slot.buffers.step_records is not None: + record = build_step_record( + episode_id=int(slot.episode_id), + env_step=slot.env_step, + scheduled_time_ms=current_time_ms, + active_action=active_action, + reward=result.reward, + done=episode_done, + info=result.info, + active_event=slot.action_scheduler.latest_applied_event, + frame_ms=self.clock.frame_ms, + latency_type=latency_type_from_source(slot.latency_source), + ) + slot.buffers.step_records.append(record) + slot.buffers.record_step(reward=float(result.reward), info=record.info) + else: + slot.buffers.record_step(reward=float(result.reward), info=result.info) + if self.simulated_worker_capacity is not None and slot.buffers.step_records is not None: + slot.buffers.step_records[-1].info.update( + { + **step_capacity_info[slot.slot_id], + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + ) + if soft_reset: + slot.buffers.num_dropped_actions += slot.action_scheduler.dropped_events_count + slot.action_scheduler.reset() + slot.result_timeline.reset() + self._reset_policy_state(slot.slot_id) + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + slot.recent_drop_count = 0 + + slot.env_step += 1 + if episode_done or slot.env_step >= eval_max_steps: + self._complete_slot(slot, on_episode_complete=on_episode_complete) + if next_episode_index < len(episode_ids): + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + + self._write_pipeline_profile_summary() + return self._ordered_metrics(episode_ids) + + def _policy_observation( + self, + slot: _SlotState, + observation: Observation, + transport: np.ndarray | None = None, + ) -> Observation: + observation = with_action_prefix(observation, slot.action_scheduler, self.action_prefix) + metadata = dict(observation.metadata) + metadata["slot_id"] = slot.slot_id + metadata["episode_id"] = int(slot.episode_id) + metadata["action_noise_seed"] = slot.episode_seed + data = observation.data + if transport is not None: + data = {**data, "transport": transport} if isinstance(data, Mapping) else {"obs": data, "transport": transport} + return Observation( + data=data, + env_step=observation.env_step, + sim_time_ms=observation.sim_time_ms, + metadata=metadata, + ) + + def _sample_policy_observation( + self, + slot: _SlotState, + observation: Observation, + ) -> _PendingPolicyObservation: + obs_id = self._next_obs_id + self._next_obs_id += 1 + raw_frame = int(slot.env_step) + current_time_ms = self.clock.step_to_time_ms(raw_frame) + worker_slot = slot.worker_capacity.assign_worker() + latency_context = { + "observation": observation, + "obs_id": obs_id, + "env_step": raw_frame, + "raw_frame": raw_frame, + "sim_time_ms": current_time_ms, + "episode_id": slot.episode_id, + "slot_id": slot.slot_id, + "worker_slot": worker_slot, + "recent_drop_count": slot.recent_drop_count, + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + latency_sample = slot.latency_source.sample(latency_context) + metadata = dict(observation.metadata) + metadata["obs_id"] = obs_id + policy_observation = Observation( + data=observation.data, + env_step=observation.env_step, + sim_time_ms=observation.sim_time_ms, + metadata=metadata, + ) + slot.worker_capacity.submit( + worker_slot, raw_frame + latency_sample.worker_service_raw_frames + ) + return _PendingPolicyObservation( + slot=slot, + obs_id=obs_id, + latency_sample=latency_sample, + worker_slot=worker_slot, + observation=policy_observation, + ) + + def _reset_run_state(self, episode_ids: Sequence[int]) -> None: + self.started_episodes = 0 + self.completed_episodes = 0 + self._pipeline_profile_rows.clear() + self._completed_metrics.clear() + self._completed_buffers.clear() + self._episode_log_order = [int(episode_id) for episode_id in episode_ids] + self._next_episode_log_index = 0 + for slot in self.slots: + slot.active = False + slot.episode_id = None + slot.episode_seed = None + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + + def _active_slots(self) -> list[_SlotState]: + return [slot for slot in self.slots if slot.active] + + def _deliver_arrived_results(self, slot: _SlotState, *, raw_frame: int | None) -> None: + released, dropped = slot.result_timeline.release_arrived(raw_frame) + slot.buffers.num_dropped_actions += len(dropped) + for event in released: + slot.action_scheduler.enqueue(event) + + def _start_slot(self, slot: _SlotState, *, episode_id: int, seed: int | None) -> None: + if self.episode_latency_source_factory is not None: + slot.latency_source = self.episode_latency_source_factory(episode_id) + slot.active = True + slot.episode_id = int(episode_id) + slot.episode_seed = None if seed is None else int(seed) + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + self._reset_policy_state(slot.slot_id) + self.env_backend.reset_slot(slot.slot_id, episode_id=episode_id, seed=seed) + self.started_episodes += 1 + + def _reset_policy_state(self, slot_id: int) -> None: + if self.inference_pool is not None: + self.inference_pool.reset_state(slot_id) + else: + self.policy.reset_state(slot_id=slot_id) + + def _complete_slot( + self, + slot: _SlotState, + *, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + ) -> None: + if not slot.active or slot.episode_id is None: + return + episode_id = int(slot.episode_id) + self._deliver_arrived_results(slot, raw_frame=None) + slot.buffers.num_dropped_actions += slot.action_scheduler.dropped_events_count + metrics = self._compute_episode_metrics( + episode_id=episode_id, + buffers=slot.buffers, + metadata=episode_raw_fact_metadata( + mode="simulated", + episode_seed=slot.episode_seed, + env_fps=self.clock.env_fps, + obs_fps=self.clock.obs_fps, + frame_ms=self.clock.frame_ms, + latency_type=latency_type_from_source(slot.latency_source), + latency_source=slot.latency_source, + ) + | slot.action_scheduler.chunk_metrics() + | ( + { + "submitted_observation_frames": slot.buffers.submitted_observation_frames, + "dropped_observation_count": slot.buffers.dropped_observation_count, + "simulated_worker_capacity": self.simulated_worker_capacity, + "inference_worker_count": self.simulated_worker_capacity, + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + if self.simulated_worker_capacity is not None + else {} + ), + ) + self._completed_metrics[episode_id] = metrics + self._completed_buffers[episode_id] = slot.buffers + self.completed_episodes += 1 + slot.active = False + slot.episode_id = None + slot.episode_seed = None + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + self._flush_completed_in_episode_order() + if on_episode_complete is not None: + on_episode_complete(metrics) + + def _enqueue_policy_output( + self, + slot: _SlotState, + observation, + policy_output, + *, + obs_id: int, + latency_sample: LatencySample, + worker_slot: int, + ) -> None: + raw_frame = int(slot.env_step) + latency_ms = latency_sample.latency_ms + ready_raw_frame = raw_frame + latency_sample.action_ready_raw_frames + ready_time_ms = self.clock.step_to_time_ms(ready_raw_frame) + latency_type = latency_type_from_source(slot.latency_source) + profile_metadata = profile_metadata_from_source(slot.latency_source) + slot_metadata = { + "episode_id": int(slot.episode_id), + "slot_id": int(slot.slot_id), + "worker_id": int(worker_slot), + } + latency_record, event = build_simulated_action_event( + action_id=self._next_action_id, + obs_id=obs_id, + policy_output=policy_output, + raw_frame=raw_frame, + ready_raw_frame=ready_raw_frame, + ready_time_ms=ready_time_ms, + latency_sample=latency_sample, + frame_ms=self.clock.frame_ms, + latency_type=latency_type, + profile_metadata=profile_metadata, + latency_record_metadata=slot_metadata, + extra_event_metadata=slot_metadata, + ) + self._next_action_id += 1 + slot.result_timeline.submit(obs_id=obs_id, ready_raw_frame=ready_raw_frame, event=event) + slot.buffers.submitted_observation_frames += 1 + slot.recent_drop_count = 0 + slot.buffers.num_actions += 1 + if slot.buffers.action_events is not None: + slot.buffers.action_events.append(event) + slot.buffers.latency_values_ms.append(latency_ms) + if slot.buffers.latency_records is not None: + slot.buffers.latency_records.append(latency_record) + + def _flush_completed_in_episode_order(self) -> None: + if self.logger is None: + return + while self._next_episode_log_index < len(self._episode_log_order): + episode_id = self._episode_log_order[self._next_episode_log_index] + if episode_id not in self._completed_metrics: + break + buffers = self._completed_buffers[episode_id] + metrics = self._completed_metrics[episode_id] + if buffers.step_records is not None: + for record in buffers.step_records: + self.logger.log_step(record) + if buffers.action_events is not None: + for event in buffers.action_events: + self.logger.log_action_event(event) + if buffers.latency_records is not None: + for latency_record in buffers.latency_records: + self.logger.log_latency(latency_record) + self.logger.log_episode_metrics(metrics) + self._next_episode_log_index += 1 + + def _ordered_metrics(self, episode_ids: Sequence[int]) -> list[EpisodeMetrics]: + return [self._completed_metrics[int(episode_id)] for episode_id in episode_ids] + + def _new_episode_buffers(self) -> _EpisodeBuffers: + return _EpisodeBuffers( + step_records=[] if self._collect_step_records else None, + action_events=[] if self._collect_action_records else None, + latency_records=[] if self._collect_latency_records else None, + ) + + def _write_pipeline_profile_summary(self) -> None: + if not self.profile_pipeline or self.logger is None or not self._pipeline_profile_rows: + return + keys = sorted({key for row in self._pipeline_profile_rows for key in row}) + summary = { + "num_profiled_batches": len(self._pipeline_profile_rows), + **{ + key: series_stats([float(row[key]) for row in self._pipeline_profile_rows if key in row]) + for key in keys + }, + } + write_json(Path(self.logger.output_dir) / "simulated_pipeline_summary.json", summary) + + def _compute_episode_metrics( + self, + *, + episode_id: int, + buffers: _EpisodeBuffers, + metadata: dict, + ) -> EpisodeMetrics: + if buffers.task_metrics is not None: + metadata["task_metrics"] = buffers.task_metrics + if buffers.task_metric_moments is not None: + metadata["task_metric_moments"] = buffers.task_metric_moments + if buffers.final_lives is not None: + metadata["final_lives"] = buffers.final_lives + if buffers.final_is_true_episode_end is not None: + metadata["final_is_true_episode_end"] = buffers.final_is_true_episode_end + metadata["soft_reset_count"] = buffers.soft_reset_count + if buffers.step_records is not None and buffers.action_events is not None: + return compute_episode_metrics( + episode_id=episode_id, + step_records=buffers.step_records, + action_events=buffers.action_events, + latency_values_ms=buffers.latency_values_ms, + metadata=metadata, + frame_ms=self.clock.frame_ms, + ) + return compute_episode_metrics_from_aggregates( + episode_id=episode_id, + episode_return_env=buffers.episode_return_env, + survival_steps=buffers.survival_steps, + return_raw=buffers.return_raw, + game_score=buffers.game_score, + latency_values_ms=buffers.latency_values_ms, + num_actions=buffers.num_actions, + num_dropped_actions=buffers.num_dropped_actions, + num_invalid_actions=buffers.num_invalid_actions, + metadata=metadata, + ) diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch new file mode 100644 index 0000000000000000000000000000000000000000..bdeca64d184242eedaa57463d2e0a242f43e19ba --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch @@ -0,0 +1,99 @@ +diff --git a/latency_bench/envs/deadly_corridor.py b/latency_bench/envs/deadly_corridor.py +index 4dcaa48c..dc4d1186 100644 +--- a/latency_bench/envs/deadly_corridor.py ++++ b/latency_bench/envs/deadly_corridor.py +@@ -5,7 +5,7 @@ from collections import deque + from typing import Any + + import numpy as np +-from gymnasium.spaces import Box, Tuple ++from gymnasium.spaces import Box, MultiBinary, Tuple + + from latency_bench.core.types import Action, Observation, StepResult + from latency_bench.envs.base import EnvAdapter +@@ -346,6 +346,7 @@ class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + export_env_raw_rgb_frames: bool = True, + ): + import gymnasium as gym ++ import vizdoom + import vizdoom.gymnasium_wrapper # noqa: F401 (registers the env ids) + + env_cfg = config["env"] +@@ -360,30 +361,27 @@ class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + ) + if key in env_cfg + } +- attempts = [ +- ("VizdoomDeadlyCorridor-MultiBinary-v1", {}), +- ("VizdoomDeadlyCorridor-MultiBinary-v0", {}), +- ("VizdoomDeadlyCorridor-v1", {"max_buttons_pressed": 0}), +- ("VizdoomDeadlyCorridor-v0", {"max_buttons_pressed": 0}), +- ] +- last_exc: Exception | None = None +- self.gym_env = None +- for env_id, kwargs in attempts: +- try: +- # frame_skip=1: the latency_bench scheduler advances obs_stride raw +- # frames per decision and holds the action between observations. +- self.gym_env = gym.make( +- env_id, render_mode="rgb_array", frame_skip=1, **render_options, **kwargs +- ) +- self.env_id = env_id +- break +- except (gym.error.NameNotFound, gym.error.VersionNotFound, gym.error.NamespaceNotFound) as exc: +- last_exc = exc +- if self.gym_env is None: +- raise RuntimeError(f"Failed to create Deadly Corridor MultiBinary env: {last_exc}") ++ # ViZDoom registers deadly_corridor.cfg under this official Gym ID. ++ self.env_id = "VizdoomCorridor-v0" ++ self.gym_env = gym.make( ++ self.env_id, render_mode="rgb_array", frame_skip=1, max_buttons_pressed=0, ++ ) ++ game = self.gym_env.unwrapped.game ++ game.close() ++ for key, value in render_options.items(): ++ if key == "screen_resolution": ++ value = getattr(vizdoom.ScreenResolution, value) ++ getattr(game, f"set_{key}")(value) ++ game.init() ++ self.gym_env.unwrapped.observation_space.spaces["screen"] = Box( ++ 0, 255, ++ shape=(game.get_screen_height(), game.get_screen_width(), game.get_screen_channels()), ++ dtype=np.uint8, ++ ) + + self._runtime_button_order = _deadly_runtime_button_names(self.gym_env) + self._num_buttons = len(self._runtime_button_order) ++ self.gym_env.unwrapped.action_space = MultiBinary(self._num_buttons) + self.noop_action = noop_action or Action( + value=[0] * self._num_buttons, name="NOOP", is_noop=True + ) +diff --git a/tests/integration/test_deadly_render_contract.py b/tests/integration/test_deadly_render_contract.py +index 535db22a..09894b1b 100644 +--- a/tests/integration/test_deadly_render_contract.py ++++ b/tests/integration/test_deadly_render_contract.py +@@ -5,11 +5,13 @@ import json + import numpy as np + import pytest + +-pytest.importorskip("vizdoom", minversion="1.3.0") ++pytest.importorskip("vizdoom", minversion="1.2.4") + pytest.importorskip("sample_factory") + + from latency_bench.envs.deadly_corridor import DeadlyCorridorEnvAdapter, DeadlyCorridorVlaEnvAdapter + from latency_bench.envs.raw_rgb import ENV_RAW_RGB_FRAME_STACK_INFO_KEY ++from latency_bench.core.types import Action ++from gymnasium.spaces import MultiBinary + from scripts.tasks.decision_history.eval_vla_hist8 import evaluation_config + + +@@ -33,6 +35,9 @@ def test_hist8_deadly_vla_uses_the_teacher_resolution_and_hud(tmp_path): + # The health/ammo panel is stable across the two engine reset paths; + # the animated face and enemies can differ with their RNG streams. + np.testing.assert_array_equal(teacher_frame[-20:, :64], student_frame[-20:, :64]) ++ assert isinstance(student.gym_env.action_space, MultiBinary) ++ step = student.step(Action(value=[1, 0, 0, 0, 0, 0, 1], name="forward_attack")) ++ assert np.isfinite(step.reward) + finally: + teacher.close() + student.close() diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py new file mode 100644 index 0000000000000000000000000000000000000000..1016be20dca9e949c752192edb407b9a84e30e35 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py @@ -0,0 +1,455 @@ +from __future__ import annotations + +import copy +from collections import deque +from typing import Any + +import numpy as np +from gymnasium.spaces import Box, MultiBinary, Tuple + +from latency_bench.core.types import Action, Observation, StepResult +from latency_bench.envs.base import EnvAdapter +from latency_bench.utils.array import looks_chw +from latency_bench.envs.raw_rgb import RawRgbFrameStackBuffer + + +def _noop_action_from_space(space) -> Any: + n = getattr(space, "n", None) + if n is not None: + return 0 + if isinstance(space, Tuple): + return tuple(_noop_action_from_space(subspace) for subspace in space.spaces) + if isinstance(space, Box): + import numpy as np + + return np.zeros(space.shape, dtype=space.dtype) + raise TypeError(f"Unsupported action space for Deadly Corridor no-op action: {space}") + + +def _coerce_noop_action_for_space(value: Any, space) -> Any: + if isinstance(space, Tuple): + if isinstance(value, (list, tuple)): + if len(value) != len(space.spaces): + raise ValueError( + f"Deadly Corridor no-op action length {len(value)} does not match action space {space}" + ) + return tuple( + _coerce_noop_action_for_space(item, subspace) + for item, subspace in zip(value, space.spaces) + ) + if value == 0: + return _noop_action_from_space(space) + return value + + +def _spec_with_reward_scaling(spec: Any, disable_reward_scaling: bool) -> Any: + if not disable_reward_scaling: + return spec + spec_to_use = copy.copy(spec) + spec_to_use.reward_scaling = 1.0 + return spec_to_use + + +def _synchronous_eval_fps_from_config(config: dict[str, Any], default: int = 35) -> int: + env_cfg = config.get("env", {}) + try: + fps = int(float(env_cfg.get("env_fps", default))) + except (TypeError, ValueError) as exc: + raise ValueError("env_fps must be positive") from exc + if fps <= 0: + raise ValueError("env_fps must be positive") + return fps + + +def _build_sample_factory_eval_cfg(config: dict[str, Any]) -> Any: + from training.deadly_corridor_sf import integration + from training.common.utils import maybe_set_cli_override + + integration.register_deadly_corridor_components() + base_cfg = integration.SAMPLE_FACTORY_CONFIG_PARSER.parse_eval( + integration.build_cli_args_from_config(config) + ) + eval_fps = _synchronous_eval_fps_from_config(config) + cfg = copy.deepcopy(base_cfg) + if _requires_sample_factory_checkpoint_config(config): + from sample_factory.cfg.arguments import load_from_checkpoint + + cfg = load_from_checkpoint(cfg) + + for key in ( + "seed", + "res_w", + "res_h", + "wide_aspect_ratio", + ): + if hasattr(base_cfg, key): + maybe_set_cli_override(cfg, key, getattr(base_cfg, key)) + maybe_set_cli_override(cfg, "frame_stack", 1) + explicit_max_episode_steps = int(getattr(base_cfg, "max_episode_steps", 0) or 0) + if explicit_max_episode_steps > 0: + maybe_set_cli_override(cfg, "max_episode_steps", explicit_max_episode_steps) + else: + eval_max_steps = int(getattr(base_cfg, "eval_max_steps", 0) or 0) + if eval_max_steps > 0: + maybe_set_cli_override(cfg, "max_episode_steps", eval_max_steps) + + maybe_set_cli_override(cfg, "mode", "eval") + maybe_set_cli_override(cfg, "latency_type", "zero") + maybe_set_cli_override(cfg, "fixed_latency_ms", 0.0) + maybe_set_cli_override(cfg, "env_frameskip", 1) + maybe_set_cli_override(cfg, "eval_env_frameskip", 1) + maybe_set_cli_override(cfg, "num_envs", 1) + maybe_set_cli_override(cfg, "no_render", True) + maybe_set_cli_override(cfg, "save_video", False) + maybe_set_cli_override(cfg, "fps", eval_fps) + maybe_set_cli_override(cfg, "eval_deterministic", bool(getattr(base_cfg, "eval_deterministic", True))) + maybe_set_cli_override(cfg, "disable_reward_scaling", bool(getattr(base_cfg, "eval_raw_reward", False))) + return cfg + + +def _requires_sample_factory_checkpoint_config(config: dict[str, Any]) -> bool: + policy_type = str(config.get("policy", {}).get("type", "")).strip().lower() + return policy_type == "deadly_corridor_sf" + + +def _seed_initialized_vizdoom_game(env: Any, seed: int) -> bool: + unwrapped = getattr(env, "unwrapped", env) + game = getattr(unwrapped, "game", None) + if game is None: + return False + unwrapped.seed(int(seed)) + game.set_seed(int(unwrapped.curr_seed)) + return True + + +class DeadlyCorridorEnvAdapter(EnvAdapter): + """Latency-bench adapter for ViZDoom Deadly Corridor using the SF Doom env stack.""" + OBSERVATION_TYPE = "vizdoom_frame_v1" + + def __init__( + self, + *, + config: dict[str, Any], + noop_action: Action | None = None, + export_env_raw_rgb_frames: bool = False, + ): + env_cfg = config["env"] + env_id = str(env_cfg.get("env_id", "doom_deadly_corridor")) + env_fps = float(env_cfg.get("env_fps", 35)) + self.frame_stack = int(env_cfg.get("frame_stack", 1) or 1) + + from sample_factory.utils.attr_dict import AttrDict + from sf_examples.vizdoom.doom.doom_utils import DOOM_ENVS, make_doom_env_from_spec + + cfg = _build_sample_factory_eval_cfg(config) + spec = next((item for item in DOOM_ENVS if item.name == str(env_id)), None) + if spec is None: + raise ValueError(f"Unknown ViZDoom env spec: {env_id}") + spec_to_use = _spec_with_reward_scaling( + spec, + disable_reward_scaling=bool(getattr(cfg, "disable_reward_scaling", False)), + ) + self.gym_env = make_doom_env_from_spec( + spec_to_use, + str(env_id), + cfg, + AttrDict(worker_index=0, vector_index=0, env_id=0), + render_mode=None, + ) + self.cfg = cfg + self.env_id = env_id + self.env_fps = float(env_fps) + action_space = self.gym_env.action_space + noop_value = _noop_action_from_space(action_space) + if noop_action is None: + self.noop_action = Action(value=noop_value, name=str(noop_value), is_noop=True) + else: + coerced_noop_value = _coerce_noop_action_for_space(noop_action.value, action_space) + self.noop_action = Action( + value=coerced_noop_value, + name=str(coerced_noop_value), + is_noop=True, + is_oneshot=noop_action.is_oneshot, + ) + self.env_step = 0 + self.export_env_raw_rgb_frames = bool(export_env_raw_rgb_frames) + self._last_info: dict[str, Any] = {} + self._last_frame: Any = None + self._observed_frames: deque[np.ndarray] = deque(maxlen=self.frame_stack) + self._raw_rgb_frames = RawRgbFrameStackBuffer(num_frames=self.frame_stack) + + def reset(self, seed: int | None = None) -> Observation: + self.env_step = 0 + self._observed_frames.clear() + if seed is not None: + if _seed_initialized_vizdoom_game(self.gym_env, int(seed)): + obs, info = self.gym_env.reset() + else: + try: + obs, info = self.gym_env.reset(seed=seed) + except TypeError: + obs, info = self.gym_env.reset() + else: + obs, info = self.gym_env.reset() + self._last_frame = obs + self._last_info = dict(info or {}) + self._reset_frame_stack(obs) + if self.export_env_raw_rgb_frames: + self._reset_raw_rgb_frame_stack() + return self._make_observation(info=self._last_info) + + def step(self, action: Action) -> StepResult: + gym_action = action.value + obs, reward, terminated, truncated, info = self.gym_env.step(gym_action) + self.env_step += 1 + self._last_frame = obs + self._last_info = dict(info or {}) + self._append_frame(obs) + if self.export_env_raw_rgb_frames and not bool(terminated or truncated): + self._append_raw_rgb_frame() + observation = self._make_observation(info=self._last_info) + step_info = dict(self._last_info) + step_info.update( + { + "env_step": self.env_step, + "sim_time_ms": self.env_step * self.frame_ms, + "applied_action": gym_action, + "applied_action_name": action.name, + "observation": "vizdoom_frame_v1", + } + ) + return StepResult( + observation=observation, + reward=float(reward), + done=bool(terminated), + truncated=bool(truncated), + info=step_info, + ) + + def observe(self) -> Observation: + if self._last_frame is None: + raise RuntimeError("DeadlyCorridorEnvAdapter has no current observation; call reset() first") + metadata = self._metadata(self._last_info) + return Observation( + data=self._policy_frame_stack(), + env_step=self.env_step, + sim_time_ms=self.env_step * self.frame_ms, + metadata=metadata, + ) + + def render_game_frame(self) -> np.ndarray: + return np.transpose(self.gym_env.unwrapped.game.get_state().screen_buffer, (1, 2, 0)) + + def close(self) -> None: + self.gym_env.close() + + def _reset_frame_stack(self, frame: Any) -> None: + self._observed_frames.clear() + self._append_frame(frame) + + def _append_frame(self, frame: Any) -> None: + self._observed_frames.append(_single_frame_data(frame)) + + def _policy_frame_stack(self) -> np.ndarray: + frames = list(self._observed_frames) + if not frames: + raise RuntimeError("Deadly Corridor observe() has no current frame; call reset() first") + if len(frames) < self.frame_stack: + frames = [frames[0]] * (self.frame_stack - len(frames)) + frames + frames = [np.asarray(frame, dtype=np.uint8) for frame in frames[-self.frame_stack :]] + if self.frame_stack == 1: + return frames[-1] + axis = 0 if looks_chw(frames[0]) else -1 + return np.concatenate(frames, axis=axis) + + +def _single_frame_data(frame: Any) -> np.ndarray: + value = frame.get("obs") if isinstance(frame, dict) else frame + arr = np.asarray(value, dtype=np.uint8) + if arr.ndim == 2: + return arr[..., None] + if arr.ndim != 3: + raise ValueError(f"Expected Deadly Corridor image frame with 2 or 3 dims, got {arr.shape!r}") + return arr + + +# Fixed semantic button order the StarVLA multibinary head is trained against. +# Mirrors starVLA.training.rl_games.eval_core._semantic_to_runtime_multibinary. +DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER = ( + "MOVE_FORWARD", + "MOVE_BACKWARD", + "MOVE_LEFT", + "MOVE_RIGHT", + "TURN_LEFT", + "TURN_RIGHT", + "ATTACK", +) + + +def _deadly_runtime_button_names(gym_env: Any) -> list[str]: + """Return the live ViZDoom action-button order (ports eval_core helper). + + The MultiBinary action vector is indexed by the game's available-button + order, which is not guaranteed to equal the semantic order the head emits. + """ + + def _button_name(button: Any) -> str: + name = getattr(button, "name", None) + if name is not None: + return str(name) + text = str(button) + return text.split(".")[-1] if "." in text else text + + for candidate in (gym_env, getattr(gym_env, "unwrapped", None)): + if candidate is None: + continue + for attr_name in ("game", "_game"): + game = getattr(candidate, attr_name, None) + if game is None: + continue + getter = getattr(game, "get_available_buttons", None) + if getter is None: + continue + names = [_button_name(button) for button in getter()] + if names: + return names + return list(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + + +def _semantic_to_runtime_multibinary(semantic_values: list[int], runtime_order: list[str]) -> list[int]: + semantic_map = { + name: int(semantic_values[idx]) + for idx, name in enumerate(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + if idx < len(semantic_values) + } + return [semantic_map.get(name, 0) for name in runtime_order] + + +class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + """Deadly Corridor adapter for StarVLA eval, matching eval_core's env. + + Unlike :class:`DeadlyCorridorEnvAdapter` (sample_factory, factorised action + tuple), this uses the gymnasium ``VizdoomDeadlyCorridor-MultiBinary`` env so + the model's multibinary head can fire arbitrary button subsets, exactly like + ``starVLA.training.rl_games.eval_core``. Native ``frame_skip=1`` is used so + latency_bench's observation-cadence scheduler owns the obs_stride stepping + (see ObservationCadenceDecisionScheduler); setting a native skip would + double-count it. + """ + + OBSERVATION_TYPE = "vizdoom_frame_v1" + + def __init__( + self, + *, + config: dict[str, Any], + noop_action: Action | None = None, + export_env_raw_rgb_frames: bool = True, + ): + import gymnasium as gym + import vizdoom + import vizdoom.gymnasium_wrapper # noqa: F401 (registers the env ids) + + env_cfg = config["env"] + self.env_fps = float(env_cfg.get("env_fps", 35)) + self.frame_stack = int(env_cfg.get("frame_stack", 1) or 1) + # The raw teacher view is part of the policy's observation contract. + render_options = { + key: env_cfg[key] + for key in ( + "screen_resolution", "render_hud", "render_crosshair", + "render_weapon", "render_decals", "render_particles", + ) + if key in env_cfg + } + # ViZDoom registers deadly_corridor.cfg under this official Gym ID. + self.env_id = "VizdoomCorridor-v0" + self.gym_env = gym.make( + self.env_id, render_mode="rgb_array", frame_skip=1, max_buttons_pressed=0, + ) + game = self.gym_env.unwrapped.game + game.close() + for key, value in render_options.items(): + if key == "screen_resolution": + value = getattr(vizdoom.ScreenResolution, value) + getattr(game, f"set_{key}")(value) + game.init() + self.gym_env.unwrapped.observation_space.spaces["screen"] = Box( + 0, 255, + shape=(game.get_screen_height(), game.get_screen_width(), game.get_screen_channels()), + dtype=np.uint8, + ) + + self._runtime_button_order = _deadly_runtime_button_names(self.gym_env) + self._num_buttons = len(self._runtime_button_order) + self.gym_env.unwrapped.action_space = MultiBinary(self._num_buttons) + self.noop_action = noop_action or Action( + value=[0] * self._num_buttons, name="NOOP", is_noop=True + ) + self.export_env_raw_rgb_frames = bool(export_env_raw_rgb_frames) + self.env_step = 0 + self._last_info: dict[str, Any] = {} + self._last_frame: Any = None + self._raw_rgb_frames = RawRgbFrameStackBuffer(num_frames=self.frame_stack) + + def reset(self, seed: int | None = None) -> Observation: + self.env_step = 0 + try: + obs, info = self.gym_env.reset(seed=seed) + except TypeError: + obs, info = self.gym_env.reset() + self._last_frame = obs + self._last_info = dict(info or {}) + if self.export_env_raw_rgb_frames: + self._reset_raw_rgb_frame_stack() + return self._make_observation(info=self._last_info) + + def step(self, action: Action) -> StepResult: + # action.value is a 7-dim multibinary vector in semantic order; re-order + # to the live game's button layout before stepping the MultiBinary env. + semantic = [int(v) for v in np.asarray(action.value).reshape(-1).tolist()] + expected = len(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + if len(semantic) != expected: + raise ValueError( + "DeadlyCorridorVlaEnvAdapter expects a " + f"{expected}-dim multibinary action in semantic order, got " + f"{len(semantic)} values ({action.value!r}). This usually means the " + "policy decoded a non-multibinary layout; ensure the deadly head is " + "action_layout=multibinary_7 and reached the multibinary decode path." + ) + runtime_buttons = _semantic_to_runtime_multibinary(semantic, self._runtime_button_order) + gym_action = np.asarray(runtime_buttons, dtype=np.int8) + obs, reward, terminated, truncated, info = self.gym_env.step(gym_action) + self.env_step += 1 + self._last_frame = obs + self._last_info = dict(info or {}) + if self.export_env_raw_rgb_frames and not bool(terminated or truncated): + self._append_raw_rgb_frame() + observation = self._make_observation(info=self._last_info) + step_info = dict(self._last_info) + step_info.update( + { + "env_step": self.env_step, + "sim_time_ms": self.env_step * self.frame_ms, + "applied_action": runtime_buttons, + "applied_action_name": action.name, + "observation": self.OBSERVATION_TYPE, + } + ) + return StepResult( + observation=observation, + reward=float(reward), + done=bool(terminated), + truncated=bool(truncated), + info=step_info, + ) + + def observe(self) -> Observation: + return self._make_observation(info=self._last_info) + + def render_game_frame(self) -> np.ndarray: + frame = self.gym_env.render() + return np.asarray(frame, dtype=np.uint8) + + def close(self) -> None: + self.gym_env.close() diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py new file mode 100644 index 0000000000000000000000000000000000000000..13f64ef81329e0b3c9296a066e17196a7e6c1d56 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py @@ -0,0 +1,60 @@ +"""Causal action history sampled at completed decision boundaries.""" + +from __future__ import annotations + +import numpy as np +from gymnasium.spaces import Discrete, MultiBinary, Tuple + + +class DecisionActionHistory: + """Encode admission, admitted command, and last applied action for each decision.""" + + def __init__(self, action_space, *, num_envs: int, decisions: int): + self._multibinary = isinstance(action_space, MultiBinary) + if isinstance(action_space, Discrete): + self.action_sizes = (action_space.n,) + elif isinstance(action_space, Tuple) and all(isinstance(space, Discrete) for space in action_space.spaces): + self.action_sizes = tuple(space.n for space in action_space.spaces) + elif self._multibinary and action_space.shape == (7,): + self.action_sizes = (3, 3, 3, 2) + else: + raise NotImplementedError(f"Decision action history does not support {action_space!r}") + self.decisions = decisions + self.action_dim = sum(size - 1 for size in self.action_sizes) + self.step_dim = 1 + 2 * self.action_dim + self.data = np.zeros((num_envs, decisions, self.step_dim), dtype=np.float32) + self._basis = tuple(np.eye(size, dtype=np.float32)[:, 1:] for size in self.action_sizes) + + @property + def observation_dim(self) -> int: + return self.decisions * self.step_dim + + def reset(self, indices=None) -> None: + if indices is None: + self.data.fill(0) + else: + self.data[indices] = 0 + + def append(self, indices, admitted, issued_actions, applied_actions) -> None: + admitted = np.asarray(admitted, dtype=np.float32).reshape(-1) + issued = self._encode(issued_actions) * admitted[:, None] + applied = self._encode(applied_actions) + rows = self.data[indices].copy() + rows[:, :-1] = rows[:, 1:] + rows[:, -1, 0] = admitted + rows[:, -1, 1 : 1 + self.action_dim] = issued + rows[:, -1, 1 + self.action_dim :] = applied + self.data[indices] = rows + + def observation(self) -> np.ndarray: + return self.data.reshape(self.data.shape[0], self.observation_dim).copy() + + def _encode(self, actions) -> np.ndarray: + if self._multibinary: + # The VLA button order is move, strafe, turn, attack; teacher history + # encodes turn, move, strafe, attack. Keep both opposing bits if issued. + return np.asarray(actions, dtype=np.float32).reshape(-1, 7)[:, [4, 5, 0, 1, 2, 3, 6]] + values = np.asarray(actions, dtype=np.int64).reshape(-1, len(self.action_sizes)) + return np.concatenate( + [basis[values[:, index]] for index, basis in enumerate(self._basis)], axis=1 + ) diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py new file mode 100644 index 0000000000000000000000000000000000000000..fca10fa388e20c9df4abb2f235ac4ea284a3144c --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py @@ -0,0 +1,216 @@ +"""Single evaluation driver: run one config's episodes and attach metadata. + +This is the core ``run_from_config`` and its episode-side helpers. Sweep/suite +orchestration lives in :mod:`latency_bench.eval.sweeps`; the CLI in +:mod:`latency_bench.run`. +""" +from __future__ import annotations + +from collections.abc import Callable, Sequence +from pathlib import Path +from typing import Any + +import yaml + +from training.common.utils import seed_everything +from latency_bench.core.types import EpisodeMetrics, ExecutorMode +from latency_bench.eval.config import ( + _episode_seed, + _eval_episodes, + _eval_max_steps, + _evaluation_seed, + resolve_evaluation_config, +) +from latency_bench.eval.reporting import _write_non_sweep_summary +from latency_bench.envs.base import EnvAdapter +from latency_bench.executors.base import BatchedExecutor +from latency_bench.executors.factory import build_executor +from latency_bench.executors.realtime_warmup import plot_realtime_eval_latency +from latency_bench.latency.config import latency_type_from_config +from latency_bench.logging.action_trace_replay import record_videos_from_action_trace +from latency_bench.logging.video import select_episode_return_stratified + + +def run_from_config( + config: dict[str, Any], + extra_metadata: dict[str, Any] | None = None, + *, + write_summary: bool = True, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + episode_ids: Sequence[int] | None = None, + policy: Any | None = None, + env: EnvAdapter | None = None, + env_backend: Any | None = None, + inference_devices: list[str] | None = None, +) -> list[EpisodeMetrics]: + eval_max_steps = _eval_max_steps(config) + resolve_evaluation_config(config) + if ( + policy is None + and env is None + and env_backend is None + and config["policy"]["type"] == "starvla" + ): + from latency_bench.policy.starvla import prepare_starvla_checkpoint_input_config + + prepare_starvla_checkpoint_input_config(config) + + experiment_cfg = config["experiment"] + policy_cfg = config["policy"] + logging_cfg = config["logging"] + seed = _evaluation_seed(config) + configured_num_episodes = _eval_episodes(config) + selected_episode_ids = list(range(configured_num_episodes)) if episode_ids is None else list(episode_ids) + seed_everything(seed) + + executor_kwargs = {} + if policy is not None: + executor_kwargs["policy"] = policy + if env is not None: + executor_kwargs["env"] = env + if env_backend is not None: + executor_kwargs["env_backend"] = env_backend + if inference_devices is not None: + executor_kwargs["inference_devices"] = inference_devices + executor = build_executor(config, **executor_kwargs) + metrics = [] + warmup_metadata_by_episode: dict[int, dict[str, Any]] = {} + try: + output_dir = Path(logging_cfg["output_dir"]) + output_dir.mkdir(parents=True, exist_ok=True) + (output_dir / "resolved_config.yaml").write_text( + yaml.safe_dump(config, sort_keys=False), encoding="utf-8" + ) + if isinstance(executor, BatchedExecutor): + warmup_metadata = executor.run_warmup() + run_episodes_kwargs: dict[str, Any] = { + "episode_ids": selected_episode_ids, + "seeds": [_episode_seed(config, episode_id) for episode_id in selected_episode_ids], + "eval_max_steps": eval_max_steps, + } + if on_episode_complete is not None: + run_episodes_kwargs["on_episode_complete"] = on_episode_complete + metrics = list(executor.run_episodes(**run_episodes_kwargs)) + warmup_metadata_by_episode.update( + (episode_id, warmup_metadata) for episode_id in selected_episode_ids + ) + else: + warmup_metadata = executor.run_warmup() + for episode_id in selected_episode_ids: + warmup_metadata_by_episode[episode_id] = warmup_metadata + episode_metrics = executor.run_episode( + episode_id=episode_id, + seed=_episode_seed(config, episode_id), + eval_max_steps=eval_max_steps, + ) + metrics.append(episode_metrics) + if on_episode_complete is not None: + on_episode_complete(episode_metrics) + metrics.sort(key=lambda item: int(item.episode_id)) + for episode_metrics in metrics: + for key, value in _evaluation_raw_fact_metadata(config, int(episode_metrics.episode_id)).items(): + if episode_metrics.metadata.get(key) is None: + episode_metrics.metadata[key] = value + episode_metrics.metadata.update(warmup_metadata_by_episode[int(episode_metrics.episode_id)]) + if "measurement" in config: + episode_metrics.metadata["measurement"] = config["measurement"] + episode_metrics.metadata["config_name"] = experiment_cfg.get("name") + episode_metrics.metadata["run_name"] = experiment_cfg.get("name") + if "checkpoint_path" in policy_cfg: + episode_metrics.metadata["checkpoint_path"] = policy_cfg["checkpoint_path"] + if "profile_path" in config["latency"]: + episode_metrics.metadata["source_profile_path"] = config["latency"]["profile_path"] + if "checkpoint_kind" in policy_cfg: + episode_metrics.metadata["checkpoint_kind"] = str(policy_cfg["checkpoint_kind"]) + episode_metrics.metadata["output_dir"] = str(logging_cfg["output_dir"]) + if "action_prefix" in policy_cfg: + episode_metrics.metadata["action_prefix"] = policy_cfg["action_prefix"] + if extra_metadata: + episode_metrics.metadata.update(extra_metadata) + if executor.logger is not None: + executor.logger.flush() + _record_realtime_eval_latency_plot(config, executor) + if write_summary: + _write_non_sweep_summary(config, metrics) + _record_stratified_replay_videos(config, metrics, seed=seed) + finally: + executor.close() + return metrics + + +def _record_realtime_eval_latency_plot(config: dict[str, Any], executor: Any) -> None: + if ExecutorMode(config["executor"]["mode"]) != ExecutorMode.REALTIME: + return + if not config["logging"]["save_latency_records"]: + return + + latency_values = list(executor.logger.latency_ms_values) + plot_realtime_eval_latency( + latency_values, + Path(config["logging"]["output_dir"]) / "eval_latency_trace.png", + ) + + +def _record_stratified_replay_videos( + config: dict[str, Any], + metrics: list[EpisodeMetrics], + *, + seed: int, +) -> None: + if "video" not in config["logging"]: + return + video_cfg = config["logging"]["video"] + if not video_cfg["enabled"]: + return + if not config["logging"]["save_step_records"]: + # Replay reads steps.jsonl, which is only written when save_step_records is on. + # Without it (e.g. factor-sweep evals) skip video instead of crashing on a missing file. + return + if ExecutorMode(config["executor"]["mode"]) == ExecutorMode.REALTIME: + return + selections = select_episode_return_stratified( + metrics, + num_bins=video_cfg["num_bins"], + seed=seed, + ) + record_videos_from_action_trace(config, selections=selections, metrics=metrics) + + +def _evaluation_raw_fact_metadata(config: dict[str, Any], episode_id: int) -> dict[str, Any]: + env_cfg = config.get("env", {}) + policy_cfg = config.get("policy", {}) + latency_cfg = config.get("latency", {}) + executor_cfg = config.get("executor", {}) + env_fps = float(env_cfg["env_fps"]) if "env_fps" in env_cfg else None + obs_fps = float(env_cfg["obs_fps"]) if "obs_fps" in env_cfg else None + frame_ms = None if env_fps is None or env_fps <= 0 else 1000.0 / env_fps + executor_mode = str(executor_cfg.get("mode", "")).strip().lower() + latency_type = latency_type_from_config(latency_cfg) + if executor_mode == "paused": + latency_type = "zero" + elif executor_mode == "realtime": + latency_type = "measured" + return { + "mode": executor_cfg.get("mode"), + "episode_seed": _episode_seed(config, episode_id), + "policy_id": _metadata_id(policy_cfg, "policy_id", "id", "type"), + "env_id": _metadata_id(env_cfg, "env_id", "id", "name"), + "model_id": latency_cfg.get("model_id"), + "gpu_class": latency_cfg.get("gpu_class"), + "workload_id": latency_cfg.get("workload_id"), + "instance_id": latency_cfg.get("instance_id"), + "source_run_id": latency_cfg.get("source_run_id"), + "profile_ref": latency_cfg.get("profile_ref"), + "env_fps": env_fps, + "obs_fps": obs_fps, + "frame_ms": frame_ms, + "latency_type": latency_type, + } + + +def _metadata_id(config: dict[str, Any], *keys: str) -> str | None: + for key in keys: + value = config.get(key) + if value is not None: + return str(value) + return None diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py new file mode 100644 index 0000000000000000000000000000000000000000..19206ec9dfd77ce730fbaee77fe932956dac0b4d --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py @@ -0,0 +1,341 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path +from typing import Any + +import gymnasium as gym +import numpy as np +import torch +import yaml + + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT)) +sys.path.insert(0, str(ROOT / "third_party" / "MIKASA-Robo")) + +from latency_bench.core.types import Action, Observation # noqa: E402 +from latency_bench.executors.gpu_batched_env_step_backend import ( # noqa: E402 + GpuBatchedEnvStepBackendBase, + SlotStepOutcome, +) +from mikasa_robo_suite.seed_reset import ( # noqa: E402 + reset_seeded_slot as _reset_seeded_slot, + reset_seeded_slots as _reset_seeded_slots, +) + + +ENV_ID = "InterceptGrabFast-VLA-v0" +INSTRUCTION = "Intercept the rolling ball and grasp it to stop it." +START_SEED = 4242424242 +MIKASA_IMAGE_VIEWS_INFO_KEY = "mikasa_image_views" +MIKASA_STATE_INFO_KEY = "mikasa_proprio" + + +def _scalar(value: Any) -> Any: + if torch.is_tensor(value): + return value.detach().reshape(-1)[0].cpu().item() + return np.asarray(value).reshape(-1)[0].item() + + +def _make_raw_env( + obs_mode: str, + num_envs: int = 1, + simulator_device: str = "gpu", +): + import mikasa_robo_suite.vla.memory_envs # noqa: F401 + + return gym.make( + ENV_ID, + num_envs=num_envs, + obs_mode=obs_mode, + control_mode="pd_ee_delta_pose", + render_mode="all", + sim_backend=simulator_device, + render_backend=simulator_device, + reward_mode="normalized_dense", + ) + + +def _make_ppo_env(num_envs: int = 1, simulator_device: str = "gpu"): + from baselines.ppo.ppo_memtasks import FlattenRGBDObservationWrapper + from mani_skill.vector.wrappers.gymnasium import ManiSkillVectorEnv + from mikasa_robo_suite.vla.dataset_collectors.get_mikasa_robo_datasets import ( + env_info, + ) + + env = _make_raw_env( + "state", + num_envs=num_envs, + simulator_device=simulator_device, + ) + wrappers, _ = env_info(ENV_ID) + for wrapper, kwargs in wrappers: + env = wrapper(env, **kwargs) + env = FlattenRGBDObservationWrapper(env, rgb=False, depth=False, state=True) + return ManiSkillVectorEnv( + env, + num_envs, + ignore_terminations=True, + record_metrics=True, + ) + + +def _make_vla_env(num_envs: int = 1, simulator_device: str = "gpu"): + from mikasa_robo_suite.vla.utils.apply_wrappers import apply_mikasa_vla_wrappers + + return apply_mikasa_vla_wrappers( + _make_raw_env( + "rgb", + num_envs=num_envs, + simulator_device=simulator_device, + ), + include_overlays=False, + ) + + +class _PpoPolicy: + def __init__(self, env, checkpoint: Path): + from baselines.ppo.ppo_memtasks import AgentStateOnly + + self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") + self.agent = AgentStateOnly(env).to(self.device) + self.agent.load_state_dict(torch.load(checkpoint, map_location=self.device)) + self.agent.eval() + + def forward(self, observation): + with torch.no_grad(): + return self.agent.get_action( + {key: value.to(self.device) for key, value in observation.items()}, + deterministic=True, + ) + + +class MikasaEnvStepBackend(GpuBatchedEnvStepBackendBase): + """Own the native MIKASA simulator and its 7D action contract.""" + + backend_name = "mikasa_gpu_batched" + + def __init__(self, *, config: dict[str, Any], num_slots: int, env=None): + noop_action = Action( + value=np.asarray(config["env"]["noop_action"], dtype=np.float32), + name="noop", + is_noop=True, + ) + super().__init__( + config=config, + noop_action=noop_action, + num_slots=num_slots, + action_space=gym.spaces.Box(-1.0, 1.0, shape=(7,), dtype=np.float32), + ) + self.env = ( + _make_vla_env( + num_envs=num_slots, + simulator_device=config["env"]["simulator_device"], + ) + if env is None + else env + ) + self._episode_seeds = [0] * num_slots + self._success = np.zeros(num_slots, dtype=np.bool_) + self._observation, _ = self.env.reset(seed=self._episode_seeds) + + def _reset_slot_observation(self, slot_id: int, *, seed: int | None) -> Observation: + if seed is not None: + self._episode_seeds[slot_id] = int(seed) + self._observation, _ = _reset_seeded_slot( + self.env, + slot_id=slot_id, + seed=self._episode_seeds[slot_id], + ) + self._env_steps[slot_id] = 0 + self._success[slot_id] = False + return self._observation_for_slot(slot_id) + + def _observe_slot_observations( + self, + slot_ids: list[int], + ) -> dict[int, Observation]: + return {slot_id: self._observation_for_slot(slot_id) for slot_id in slot_ids} + + def _step_cores( + self, + slot_ids: list[int], + *, + actions: np.ndarray, + active_mask: np.ndarray, + ) -> Any: + del slot_ids, active_mask + tensor_actions = torch.as_tensor( + actions, + dtype=torch.float32, + device=self.env.unwrapped.device, + ) + self._observation, reward, terminated, truncated, info = self.env.step( + tensor_actions + ) + return reward, terminated, truncated, info + + def _slot_step_outcome(self, state: Any, slot_id: int) -> SlotStepOutcome: + reward, terminated, truncated, info = state + success = bool(_slot_value(info["success"], slot_id)) + self._success[slot_id] |= success + return SlotStepOutcome( + reward=float(_slot_value(reward, slot_id)), + done=bool(_slot_value(terminated, slot_id)), + truncated=bool(_slot_value(truncated, slot_id)), + info={ + "success": success, + "task_metrics": {"success": float(self._success[slot_id])}, + }, + ) + + def _observation_for_slot(self, slot_id: int) -> Observation: + rgb = self._observation["rgb"] + if torch.is_tensor(rgb): + rgb = rgb.detach().cpu().numpy() + rgb = np.asarray(rgb) + views = np.stack( + [ + np.asarray(rgb[slot_id, :, :, :3], dtype=np.uint8), + np.asarray(rgb[slot_id, :, :, 3:6], dtype=np.uint8), + ] + ) + metadata = { + MIKASA_IMAGE_VIEWS_INFO_KEY: views, + MIKASA_STATE_INFO_KEY: self._observation["proprio"][slot_id].detach().cpu().numpy(), + "slot_id": slot_id, + } + if "action_prefix_state_key" in self.config["env"]: + metadata["action_prefix_state_key"] = self.config["env"]["action_prefix_state_key"] + if "returned_action_context" in self.config["env"]: + context = self.config["env"]["returned_action_context"] + metadata["returned_action_context"] = { + **context, + "order": np.asarray(context["order"]), + "low": np.asarray(context["low"], dtype=np.float32), + "high": np.asarray(context["high"], dtype=np.float32), + } + return Observation( + data=None, + env_step=int(self._env_steps[slot_id]), + sim_time_ms=float(self._env_steps[slot_id]) * self._frame_ms, + metadata=metadata, + ) + + def close(self) -> None: + if not self.closed: + self.env.close() + super().close() + + +def _slot_value(value: Any, slot_id: int) -> Any: + if torch.is_tensor(value): + return value.detach().reshape(-1)[slot_id].cpu().item() + return np.asarray(value).reshape(-1)[slot_id].item() + + +def _evaluate(args: argparse.Namespace) -> dict[str, Any]: + env = _make_ppo_env() + policy = _PpoPolicy(env, args.checkpoint) + seeds = [] + successes = [] + returns = [] + lengths = [] + try: + for episode_index in range(args.episodes): + seed = START_SEED + episode_index + observation, _ = env.reset(seed=seed) + success_once = False + episode_return = 0.0 + for step in range(60): + action = policy.forward(observation) + observation, reward, terminated, truncated, info = env.step(action) + success_once = success_once or bool(_scalar(info["success"])) + episode_return += float(_scalar(reward)) + if bool(_scalar(terminated)) or bool(_scalar(truncated)): + break + seeds.append(seed) + successes.append(success_once) + returns.append(episode_return) + lengths.append(step + 1) + finally: + env.close() + summary = { + "seeds": seeds, + "successes": successes, + "success_rate": float(np.mean(successes)), + "returns": returns, + "lengths": lengths, + } + (args.output_dir / "summary.json").write_text( + json.dumps(summary, indent=2) + "\n", encoding="utf-8" + ) + return summary + + +def _latency_eval(argv: list[str]) -> None: + from latency_bench.core.config import load_config + from latency_bench.eval.config import apply_runtime_overrides + from latency_bench.eval.driver import run_from_config + + parser = argparse.ArgumentParser() + parser.add_argument("--eval-config", type=Path, required=True) + parser.add_argument("--checkpoint-path", type=Path) + parser.add_argument("--model-config-path", type=Path) + parser.add_argument("--task-contract-path", type=Path) + parser.add_argument("--run-name") + parser.add_argument("--output-dir", type=Path) + parser.add_argument("--latency-method", choices=("zero", "temporal")) + parser.add_argument("--profile-path", type=Path) + parser.add_argument("--latency-seed", type=int) + args = parser.parse_args(argv) + config = load_config(args.eval_config) + apply_runtime_overrides( + config, + checkpoint_path=args.checkpoint_path, + model_config_path=args.model_config_path, + task_contract_path=args.task_contract_path, + run_name=args.run_name, + output_dir=args.output_dir, + latency_method=args.latency_method, + latency_profile_path=args.profile_path, + latency_seed=args.latency_seed, + ) + output_dir = Path(config["logging"]["output_dir"]) + output_dir.mkdir(parents=True, exist_ok=True) + (output_dir / "eval_config.yaml").write_text( + yaml.safe_dump(config, sort_keys=False), encoding="utf-8" + ) + backend = MikasaEnvStepBackend( + config=config, + num_slots=int(config["evaluation"]["eval_parallel_envs"]), + ) + run_from_config( + config, + env_backend=backend, + inference_devices=config["executor"]["inference_devices"], + ) + + +def main() -> None: + if sys.argv[1:2] == ["latency-eval"]: + _latency_eval(sys.argv[2:]) + return + + parser = argparse.ArgumentParser() + parser.add_argument("--policy", choices=("ppo",), required=True) + parser.add_argument("--checkpoint", type=Path) + parser.add_argument("--episodes", type=int, default=50) + parser.add_argument("--output-dir", type=Path, required=True) + args = parser.parse_args() + args.output_dir.mkdir(parents=True, exist_ok=True) + + _evaluate(args) + + +if __name__ == "__main__": + main() diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py new file mode 100644 index 0000000000000000000000000000000000000000..7b9cab3b3fad77d6f6a3b7919ff4a8124c98d011 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py @@ -0,0 +1,1207 @@ +from __future__ import annotations + +import json +import sys +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any + +import numpy as np +import numpy.typing as npt +from PIL import Image + +from latency_bench.core.actions import ActionResolver +from latency_bench.core.clock import EnvClock +from latency_bench.core.timing import current_profiler +from latency_bench.core.types import Action, Observation, PolicyOutput +from latency_bench.data.ghost_trail import GhostTrailConfig, build_flappy_ghost_trail_window +from latency_bench.data.state_normalization import min_max_normalize_state +from latency_bench.envs.raw_rgb import ENV_RAW_RGB_FRAME_STACK_INFO_KEY +from latency_bench.envs.gymnasium_task import ( + gymnasium_action_space_contract, + gymnasium_task_contract, +) +from latency_bench.policy.base import PolicyRunner +from latency_bench.policy.starvla_prompts import load_latency_prompt_map, resolve_starvla_prompt + +from latency_bench.utils.paths import REPO_ROOT + + +STARVLA_ROOT = REPO_ROOT / "third_party" / "starVLA" +STATEFUL_STARVLA_MODEL_IDS: tuple[str, ...] = ( + "pi0", + "pi-0", + "pi05", + "pi-0.5", + "gr00t", + "qwenpi", + "qwenpi_v3", + "qwengr00t", +) +STATELESS_STARVLA_MODEL_IDS: tuple[str, ...] = ( + "openvla", + "qwenoft", +) + +DEMON_ATTACK_ACTION_LABELS: tuple[str, ...] = ( + "NOOP", + "FIRE", + "RIGHT", + "LEFT", + "RIGHTFIRE", + "LEFTFIRE", +) +DEADLY_CORRIDOR_TURN_LABELS: tuple[str, ...] = ( + "TURN_NOOP", + "TURN_LEFT", + "TURN_RIGHT", +) +DEADLY_CORRIDOR_MOVE_LABELS: tuple[str, ...] = ( + "MOVE_NOOP", + "MOVE_FORWARD", + "MOVE_BACKWARD", +) +DEADLY_CORRIDOR_STRAFE_LABELS: tuple[str, ...] = ( + "STRAFE_NOOP", + "MOVE_LEFT", + "MOVE_RIGHT", +) +DEADLY_CORRIDOR_ATTACK_LABELS: tuple[str, ...] = ( + "ATTACK_NOOP", + "ATTACK", +) + + +class StarVlaPolicyRunner(PolicyRunner): + """Translate observations and model outputs using the task action contract.""" + + def __init__( + self, + *, + wrapper: Any, + checkpoint_path: str, + device: str, + unnorm_key: str | None, + env_name: str, + action_resolver: ActionResolver, + action_refs: Sequence[Any], + latency_prompt_map: dict[str, Any] | None = None, + base_prompt: str | None = None, + latency_prompt_key: int | str | None = None, + prompt_mode: str | None = None, + obs_resize: tuple[int, int] | None = None, + image_transform_config: Mapping[str, Any] | None = None, + observation_stride_raw_frames: int, + model_cfg: Mapping[str, Any] | None = None, + state_normalization: Mapping[str, Any] | None = None, + state_source: str | None = None, + image_views_info_key: str | None = None, + action_output_type: str | None = None, + ) -> None: + self._wrapper = wrapper + self._obs_resize = tuple(obs_resize) if obs_resize else None + self._checkpoint_path = checkpoint_path + self._device = device + self._unnorm_key = unnorm_key + self._env_name = env_name + self._action_by_raw_id = { + raw_action_id: action_resolver.resolve(action_ref) + for raw_action_id, action_ref in enumerate(action_refs) + } + self._latency_prompt_map = latency_prompt_map + self._base_prompt = base_prompt + self._latency_prompt_key = latency_prompt_key + self._prompt_mode = str(prompt_mode or "default").strip().lower() + self._image_transform_config = dict(image_transform_config or {"image_transform": "raw_rgb"}) + self._image_transform = str( + self._image_transform_config.get("image_transform", "raw_rgb") or "raw_rgb" + ).strip().lower() + model_cfg = ( + _normalized_model_cfg_from_wrapper(wrapper) + if model_cfg is None + else _normalized_model_cfg(model_cfg) + ) + self._include_state = _include_state_from_model_cfg(model_cfg) + self._state_dim = _state_dim_from_model_cfg(model_cfg) if self._include_state else None + self._state_normalization = dict(state_normalization or {}) + self._state_source = state_source + self._image_views_info_key = image_views_info_key + self._action_output_type = action_output_type + vla_data = (model_cfg.get("datasets", {}) or {}).get("vla_data", {}) or {} + self._pack_image_sequence = ( + bool(vla_data["pack_image_sequence"]) + if "pack_image_sequence" in vla_data + else False + ) + self._image_sequence_length = ( + int(vla_data["image_sequence_length"]) + if self._pack_image_sequence + else 1 + ) + self._observation_stride_raw_frames = int(observation_stride_raw_frames) + self._image_sequence_raw_span = ( + 1 + + (self._image_sequence_length - 1) + * self._observation_stride_raw_frames + ) + self._num_obs_frames = int(vla_data.get("num_obs_frames", 1) or 1) + self._image_mode = str(vla_data.get("image_mode", "single")) + self._stitch_grid = tuple(vla_data.get("stitch_grid", [2, 2])) + framework_cfg = model_cfg["framework"] + kv_cfg = framework_cfg["kv_memory"] if "kv_memory" in framework_cfg else {} + self._kv_memory_enabled = bool(kv_cfg["enabled"]) if "enabled" in kv_cfg else False + + def reset_state(self, slot_id: int | None = None) -> None: + # Clears the model's per-slot KV memory at episode boundaries (req3). + # No-op unless the framework maintains KV memory. + reset = getattr(self._wrapper, "reset_memory", None) + if callable(reset): + reset(slot_id) + + def predict(self, observation: Observation) -> PolicyOutput: + return self.predict_batch([observation])[0] + + def predict_batch(self, observations: Sequence[Observation]) -> list[PolicyOutput]: + profiler = current_profiler() + with profiler.time("policy_build_example_ms"): + examples = [self._build_example(observation) for observation in observations] + with profiler.time("policy_wrapper_predict_action_ms"): + prediction = self._wrapper.predict_action( + examples=examples, unnorm_key=self._unnorm_key, profiler=profiler + ) + with profiler.time("policy_decode_ms"): + outputs = [ + self._decode_prediction( + prediction=prediction, + index=index, + observation=observation, + example=example, + ) + for index, (observation, example) in enumerate(zip(observations, examples)) + ] + return outputs + + def _decode_prediction( + self, + *, + prediction: dict[str, Any], + index: int, + observation: Observation, + example: dict[str, Any], + ) -> PolicyOutput: + actions = np.asarray(prediction["actions"]) + raw_action_scores = ( + np.asarray(prediction["raw_action_scores"]) + if "raw_action_scores" in prediction + else None + ) + return self._policy_output( + observation=observation, + example=example, + action_payload=actions[index, 0], + action_output_type=( + prediction["action_output_type"] + if self._action_output_type is None + else self._action_output_type + ), + raw_action_scores=None if raw_action_scores is None else raw_action_scores[index, 0], + ) + + def _build_example(self, observation: Observation) -> dict[str, Any]: + frame_source = observation.metadata[ + ENV_RAW_RGB_FRAME_STACK_INFO_KEY + if self._image_views_info_key is None + else self._image_views_info_key + ] + frames = observation_data_to_hwc_uint8_frames(frame_source) # oldest .. newest + transformed = self._transformed_frame(frames=frames, observation=observation) + + if self._image_views_info_key is not None: + pass + elif self._pack_image_sequence: + if transformed is not None: + raise ValueError( + "WanOFT packed image sequences require image_transform=raw_rgb" + ) + if len(frames) < self._image_sequence_raw_span: + raise ValueError( + "WanOFT packed image sequence requires " + f"{self._image_sequence_raw_span} raw frames for " + f"{self._image_sequence_length} decision observations at stride " + f"{self._observation_stride_raw_frames}, got {len(frames)}" + ) + frames = frames[ + -self._image_sequence_raw_span + :: self._observation_stride_raw_frames + ] + elif transformed is not None: + frames = [transformed] + elif self._image_mode == "single" or self._kv_memory_enabled: + frames = frames[-1:] + else: + # Select the temporal observation window to match training (_pack_sample). + raw_span = 1 + (self._num_obs_frames - 1) * self._observation_stride_raw_frames + frames = frames[-raw_span :: self._observation_stride_raw_frames] + + prompt = resolve_starvla_prompt( + env_name=self._env_name, + observation_metadata=observation.metadata, + latency_prompt_map=self._latency_prompt_map, + base_prompt=self._base_prompt, + latency_prompt_key=self._latency_prompt_key, + prompt_mode=self._prompt_mode, + ) + + if self._image_mode == "stitch": + if transformed is not None: + raise ValueError("image_transform is not compatible with image_mode=stitch") + # Tile the window into one image; matches _pack_sample's stitch branch + # (raw frames passed to stitch_frames, which resizes each cell to 224). + images = [_get_stitch_frames()(frames, grid=self._stitch_grid, size=(224, 224))] + else: + if self._obs_resize is not None: + height, width = self._obs_resize + # match training preprocessing exactly: gr00t LeRobotSingleDataset._pack_sample + # does `Image.fromarray(image).resize((224, 224))` (PIL default resample = BICUBIC). + frames = [ + np.asarray(Image.fromarray(frame).resize((width, height)), dtype=np.uint8) + for frame in frames + ] + images = [Image.fromarray(frame) for frame in frames] + + example = { + "image": images, + "lang": prompt, + } + if self._kv_memory_enabled: + example["slot_id"] = observation.metadata["slot_id"] + elif "slot_id" in observation.metadata: + example["slot_id"] = observation.metadata["slot_id"] + if self._include_state: + if self._state_source == "transport": + state = np.asarray(observation.data["transport"], dtype=np.float32) + example["state"] = state.reshape(1, self._state_dim) + elif self._state_normalization: + state = np.asarray( + observation.metadata["gymnasium_state"], dtype=np.float32 + ) + state_min = np.asarray(self._state_normalization["min"], dtype=np.float32) + state_max = np.asarray(self._state_normalization["max"], dtype=np.float32) + state = min_max_normalize_state(state, state_min, state_max) + example["state"] = state.reshape(1, self._state_dim) + else: + example["state"] = np.zeros((1, self._state_dim), dtype=np.float32) + return example + + def _transformed_frame( + self, + *, + frames: Sequence[npt.NDArray[np.uint8]], + observation: Observation, + ) -> npt.NDArray[np.uint8] | None: + if self._image_transform in {"", "none", "raw", "raw_rgb"}: + return None + if self._image_transform not in {"flappy_ghost_trail", "demon_attack_ghost_trail"}: + raise ValueError(f"Unsupported StarVLA image_transform={self._image_transform!r}") + if self._image_transform == "flappy_ghost_trail" and self._env_name != "flappy": + raise ValueError("image_transform=flappy_ghost_trail is only supported for env_name=flappy") + if self._image_transform == "demon_attack_ghost_trail" and self._env_name != "demon_attack": + raise ValueError("image_transform=demon_attack_ghost_trail is only supported for env_name=demon_attack") + + config = GhostTrailConfig( + image_transform=self._image_transform, + history_frames=int(self._image_transform_config.get("history_frames", 5)), + gamma=float(self._image_transform_config.get("gamma", 1.3)), + min_alpha=int(self._image_transform_config.get("min_alpha", 35)), + ground_fraction=float(self._image_transform_config.get("ground_fraction", 0.22)), + scroll_px_per_step=float(self._image_transform_config.get("scroll_px_per_step", 4.0)), + ) + if self._image_transform == "demon_attack_ghost_trail": + # env_step counts raw ALE frames (buffer updated 4× per decision step). + # frames[-0:] == frames, so env_step=0 falls back to the full reset-fill buffer. + valid_count = min(len(frames), int(observation.env_step)) + else: + max_frames = max(1, int(config.history_frames) + 1) + valid_count = min(len(frames), max(1, int(observation.env_step) + 1), max_frames) + window = [np.asarray(frame, dtype=np.uint8) for frame in frames[-valid_count:]] + + if self._image_transform == "demon_attack_ghost_trail": + from latency_bench.data.ghost_trail_demon import build_demon_attack_ghost_trail_window + steps_arg = list(range(len(window))) + return build_demon_attack_ghost_trail_window(window, steps_arg, config=config) + + current_step = int(observation.env_step) + start_step = current_step - valid_count + 1 + steps = list(range(start_step, current_step + 1)) + return build_flappy_ghost_trail_window(window, steps, config=config) + + def _policy_output( + self, + *, + observation: Observation, + example: dict[str, Any], + action_payload: npt.NDArray[Any], + action_output_type: str, + raw_action_scores: npt.NDArray[Any] | None, + ) -> PolicyOutput: + payload = np.asarray(action_payload) + action, action_metadata = action_from_starvla_payload( + payload=payload, + env_name=self._env_name, + action_by_raw_id=self._action_by_raw_id, + action_output_type=action_output_type, + ) + metadata = { + "policy_type": "starvla", + "prompt_source": "latency_prompt_map" if self._latency_prompt_map is not None else "base", + "checkpoint_path": self._checkpoint_path, + "unnorm_key": self._unnorm_key, + "device": self._device, + "input_frame_count": len(example["image"]), + "image_transform": self._image_transform, + "action_output_type": action_output_type, + "action_payload": to_jsonable_action_payload(payload), + "kv_memory_enabled": self._kv_memory_enabled, + **action_metadata, + } + if self._pack_image_sequence: + metadata["image_sequence_length"] = self._image_sequence_length + metadata["input_frame_raw_stride"] = self._observation_stride_raw_frames + metadata["input_frame_raw_span"] = self._image_sequence_raw_span + if "slot_id" in example: + metadata["slot_id"] = example["slot_id"] + if raw_action_scores is not None: + metadata["raw_action_scores"] = [ + float(item) for item in np.asarray(raw_action_scores, dtype=np.float32).tolist() + ] + if "latency_raw_frames" in observation.metadata: + metadata["latency_raw_frames"] = observation.metadata["latency_raw_frames"] + if "latency_ms" in observation.metadata: + metadata["latency_ms"] = observation.metadata["latency_ms"] + if self._latency_prompt_key is not None: + metadata["latency_prompt_key"] = self._latency_prompt_key + return PolicyOutput( + action=action, + raw_output=metadata["action_payload"], + metadata=metadata, + ) + + +def observation_data_to_hwc_uint8_frames(data: Any) -> list[npt.NDArray[np.uint8]]: + frame = _extract_observation_array(data) + if frame.ndim == 4 and frame.shape[-1] == 3: + return [_as_uint8_image(item) for item in frame] + if frame.ndim == 4 and frame.shape[1] == 3: + return [_as_uint8_image(np.transpose(item, (1, 2, 0))) for item in frame] + if frame.ndim == 3 and frame.shape[-1] == 3: + return [_as_uint8_image(frame)] + if frame.ndim == 3 and frame.shape[0] == 3: + return [_as_uint8_image(np.transpose(frame, (1, 2, 0)))] + if ( + frame.ndim == 3 + and frame.shape[0] % 3 == 0 + and frame.shape[0] < frame.shape[1] + and frame.shape[0] < frame.shape[2] + ): + return [ + _as_uint8_image(np.transpose(frame[start : start + 3, :, :], (1, 2, 0))) + for start in range(0, frame.shape[0], 3) + ] + if frame.ndim == 3 and frame.shape[-1] % 3 == 0: + return [ + _as_uint8_image(frame[:, :, start : start + 3]) + for start in range(0, frame.shape[-1], 3) + ] + if frame.ndim == 3 and frame.shape[0] % 3 == 0: + return [ + _as_uint8_image(np.transpose(frame[start : start + 3, :, :], (1, 2, 0))) + for start in range(0, frame.shape[0], 3) + ] + return [_as_uint8_image(frame)] + + +def decode_starvla_action( + *, + vector: npt.NDArray[Any], + env_name: str, + action_by_raw_id: Mapping[int, Action], + action_layout: str | None = None, +) -> tuple[Action, dict[str, Any]]: + deadly_layout = None + if str(env_name) == "deadly_corridor": + action_dim = int(np.asarray(vector).shape[-1]) + deadly_layouts = { + 7: "deadly_corridor_semantic_7", + 11: "deadly_corridor_factorized_11", + 54: "deadly_corridor_joint_54", + } + if action_dim not in deadly_layouts: + raise ValueError( + "Deadly Corridor StarVLA action vector expected 7, 11, or 54 " + f"values, got {action_dim}" + ) + deadly_layout = deadly_layouts[action_dim] + asterix_layout = None + if str(env_name) == "asterix": + action_dim = int(np.asarray(vector).shape[-1]) + if action_layout is not None: + asterix_layout = str(action_layout).strip().lower() + else: + asterix_layout = "factorized_6" if action_dim < 9 else "discrete_9" + + decode_rl_games_actions, _, _ = _load_rl_games_action_decode() + prediction = decode_rl_games_actions( + normalized_actions=np.asarray(vector), + env_name=str(env_name), + deadly_action_layout=(deadly_layout.removeprefix("deadly_corridor_") if deadly_layout is not None else None), + asterix_action_layout=asterix_layout, + ) + action, metadata = action_from_starvla_payload( + payload=np.asarray(prediction["actions"]), + env_name=env_name, + action_by_raw_id=action_by_raw_id, + action_output_type=prediction["action_output_type"], + ) + if deadly_layout is not None: + metadata["action_layout"] = deadly_layout + if deadly_layout == "deadly_corridor_joint_54": + turn, move, strafe, attack = action.value + metadata["raw_action_id"] = turn * 18 + move * 6 + strafe * 2 + attack + elif deadly_layout == "deadly_corridor_semantic_7": + semantic_actions = ( + [0, 1, 0, 0], + [0, 2, 0, 0], + [0, 0, 1, 0], + [0, 0, 2, 0], + [1, 0, 0, 0], + [2, 0, 0, 0], + [0, 0, 0, 1], + ) + metadata["raw_action_id"] = semantic_actions.index(action.value) + if asterix_layout is not None: + metadata["action_layout"] = asterix_layout + return action, metadata + + +# Fixed semantic button order the StarVLA multibinary head is trained against. +# Mirrors starVLA.training.rl_games.eval_core._semantic_to_runtime_multibinary; +# the env adapter re-orders this to the live ViZDoom button layout. +DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER = ( + "MOVE_FORWARD", + "MOVE_BACKWARD", + "MOVE_LEFT", + "MOVE_RIGHT", + "TURN_LEFT", + "TURN_RIGHT", + "ATTACK", +) + + +def action_from_starvla_payload( + *, + payload: npt.NDArray[Any], + env_name: str, + action_by_raw_id: Mapping[int, Action], + action_output_type: str = "", +) -> tuple[Action, dict[str, Any]]: + if str(action_output_type) == "rl_games_continuous": + values = [float(item) for item in np.asarray(payload).reshape(-1).tolist()] + return Action( + value=values, + name="continuous_torque", + is_noop=all(value == 0.0 for value in values), + is_oneshot=False, + ), {"continuous_action": values} + if str(env_name) == "demon_attack": + return demon_attack_action_from_id(int(np.asarray(payload).reshape(-1)[0])) + if str(env_name) == "deadly_corridor": + # Multibinary heads emit an already-thresholded 7-dim button vector in + # fixed semantic order; the env adapter re-orders it to the live ViZDoom + # button layout. Keep it as-is rather than reinterpreting it as a + # [turn, move, strafe, attack] categorical tuple. + if str(action_output_type) == "rl_games_deadly_corridor_multibinary": + buttons = [int(item) for item in np.asarray(payload).reshape(-1).tolist()] + active = [ + DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER[idx] + for idx, pressed in enumerate(buttons) + if idx < len(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) and pressed + ] + action_name = "+".join(active) if active else "NOOP" + return Action( + value=buttons, + name=action_name, + is_noop=not any(buttons), + is_oneshot=False, + ), { + "decoded_multibinary_buttons": buttons, + "action_label": action_name, + "action_layout": "deadly_corridor_multibinary_7", + } + return deadly_corridor_action_from_tuple( + action_value=[int(item) for item in np.asarray(payload).reshape(-1).tolist()], + metadata={"action_layout": "deadly_corridor_tuple"}, + ) + raw_action_id = int(np.asarray(payload).reshape(-1)[0]) + return action_by_raw_id[raw_action_id], {"raw_action_id": raw_action_id} + + +def to_jsonable_action_payload(payload: npt.NDArray[Any]) -> Any: + value = np.asarray(payload).tolist() + if isinstance(value, list) and len(value) == 1: + return value[0] + return value + + +def demon_attack_action_from_id(action_id: int) -> tuple[Action, dict[str, Any]]: + action = Action( + value=action_id, + name=DEMON_ATTACK_ACTION_LABELS[action_id], + is_noop=action_id == 0, + is_oneshot=False, + ) + return action, {"raw_action_id": action_id, "action_label": action.name} + + +def deadly_corridor_action_from_tuple( + *, + action_value: list[int], + metadata: dict[str, Any], +) -> tuple[Action, dict[str, Any]]: + turn, move, strafe, attack = action_value + action_value = [turn, move, strafe, attack] + turn_label = DEADLY_CORRIDOR_TURN_LABELS[turn] + move_label = DEADLY_CORRIDOR_MOVE_LABELS[move] + strafe_label = DEADLY_CORRIDOR_STRAFE_LABELS[strafe] + attack_label = DEADLY_CORRIDOR_ATTACK_LABELS[attack] + active_labels = [ + label + for label in (turn_label, move_label, strafe_label, attack_label) + if not label.endswith("_NOOP") + ] + action_name = "+".join(active_labels) if active_labels else "NOOP" + return Action( + value=action_value, + name=action_name, + is_noop=action_value == [0, 0, 0, 0], + is_oneshot=False, + ), { + "decoded_action_tuple": action_value, + "turn_label": turn_label, + "move_label": move_label, + "strafe_label": strafe_label, + "attack_label": attack_label, + "action_label": action_name, + **metadata, + } + + +def _extract_observation_array(data: Any) -> npt.NDArray[Any]: + if isinstance(data, Mapping): + return np.asarray(data["observation"]) + return np.asarray(data) + + +def _as_uint8_image(frame: npt.NDArray[Any]) -> npt.NDArray[np.uint8]: + return np.ascontiguousarray(frame, dtype=np.uint8) + + +def _normalized_model_cfg(model_cfg: Mapping[str, Any]) -> dict[str, Any]: + _ensure_starvla_path() + from omegaconf import OmegaConf + from starVLA.model.framework.share_tools import apply_config_compat + + cfg = OmegaConf.create(model_cfg) + apply_config_compat(cfg) + _apply_model_family_include_state_compat(cfg) + return OmegaConf.to_container(cfg, resolve=True) + + +def _normalized_model_cfg_from_wrapper(wrapper: Any) -> dict[str, Any]: + return _normalized_model_cfg(wrapper._model_cfg) + + +def _load_starvla_model_config(path: str | Path) -> dict[str, Any]: + from omegaconf import OmegaConf + + return _normalized_model_cfg(OmegaConf.load(path)) + + +def _apply_model_family_include_state_compat(cfg: Any) -> None: + from omegaconf import OmegaConf + + if OmegaConf.select(cfg, "datasets.vla_data.include_state") is not None: + return + + model_ids = ( + _normalized_optional_config_string(cfg, ("model",)), + _normalized_optional_config_string(cfg, ("rl_games", "model_alias")), + _normalized_optional_config_string(cfg, ("framework", "name")), + ) + if any(model_id in STATEFUL_STARVLA_MODEL_IDS for model_id in model_ids if model_id is not None): + OmegaConf.update(cfg, "datasets.vla_data.include_state", True, force_add=True) + return + if any(model_id in STATELESS_STARVLA_MODEL_IDS for model_id in model_ids if model_id is not None): + OmegaConf.update(cfg, "datasets.vla_data.include_state", False, force_add=True) + + +def _normalized_optional_config_string(cfg: Any, path: tuple[str, ...]) -> str | None: + from omegaconf import OmegaConf + + value = OmegaConf.select(cfg, ".".join(path)) + if value is None: + return None + return str(value).strip().lower() + + +def _state_dim_from_model_cfg(model_cfg: dict[str, Any]) -> int: + return model_cfg["framework"]["action_model"]["state_dim"] + + +def _include_state_from_model_cfg(model_cfg: dict[str, Any]) -> bool: + return model_cfg["datasets"]["vla_data"]["include_state"] + + +_STITCH_FRAMES = None + + +def _get_stitch_frames(): + """Lazily import starVLA's stitch_frames (starVLA path is added at runtime).""" + global _STITCH_FRAMES + if _STITCH_FRAMES is None: + _ensure_starvla_path() + from starVLA.training.trainer_utils.trainer_tools import stitch_frames + + _STITCH_FRAMES = stitch_frames + return _STITCH_FRAMES + + +def _ensure_starvla_path() -> None: + starvla_root = str(STARVLA_ROOT) + if starvla_root not in sys.path: + sys.path.insert(0, starvla_root) + + +def _observation_stride_raw_frames(config: Mapping[str, Any]) -> int: + env_cfg = config["env"] + return EnvClock( + env_fps=float(env_cfg["env_fps"]), + obs_fps=float(env_cfg["obs_fps"]), + ).obs_stride_raw_frames + + +def apply_starvla_model_input_config( + config: dict[str, Any], + *, + model_cfg: Mapping[str, Any], + image_transform: str = "raw_rgb", +) -> None: + """Match latency_bench's raw frame stack to a saved StarVLA input contract.""" + vla_data = model_cfg["datasets"]["vla_data"] + pack_image_sequence = ( + bool(vla_data["pack_image_sequence"]) + if "pack_image_sequence" in vla_data + else False + ) + normalized_transform = str(image_transform).strip().lower() + raw_image_transform = normalized_transform in {"", "none", "raw", "raw_rgb"} + if pack_image_sequence: + if not raw_image_transform: + raise ValueError( + "WanOFT packed image sequences require image_transform=raw_rgb" + ) + input_frame_count = int(vla_data["image_sequence_length"]) + else: + if not raw_image_transform: + return + framework_cfg = model_cfg["framework"] + kv_cfg = framework_cfg["kv_memory"] if "kv_memory" in framework_cfg else {} + kv_memory_enabled = bool(kv_cfg["enabled"]) if "enabled" in kv_cfg else False + if kv_memory_enabled: + return + image_mode = str(vla_data["image_mode"]) if "image_mode" in vla_data else "single" + if image_mode == "single": + return + input_frame_count = int(vla_data["num_obs_frames"]) + + observation_stride = _observation_stride_raw_frames(config) + required_raw_frames = 1 + (input_frame_count - 1) * observation_stride + config["env"]["frame_stack"] = max( + int(config["env"]["frame_stack"]), + required_raw_frames, + ) + + +def prepare_starvla_checkpoint_input_config(config: dict[str, Any]) -> None: + """Apply the saved checkpoint input contract before env construction.""" + if config["policy"]["type"] != "starvla": + return + + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + if config["env"]["name"] == "gymnasium": + contract = json.loads( + Path(policy_cfg["task_contract_path"]).read_text(encoding="utf-8") + ) + config["env"]["state_space"] = {"labels": contract["state_labels"]} + if contract["robot_type"] in ("latency_balance_profile_h8", "latency_balance_profile_h16"): + config["env"]["name"] = "balance_profile" + config["env"]["action_context_horizon"] = contract["action_horizon"] + config["env"]["frame_stack"] = 1 + return + if "model_config_path" in policy_cfg: + model_cfg = _load_starvla_model_config(policy_cfg["model_config_path"]) + else: + _ensure_starvla_path() + from starVLA.model.framework.share_tools import read_mode_config + + saved_model_cfg, _norm_stats = read_mode_config(policy_cfg["checkpoint_path"]) + model_cfg = _normalized_model_cfg(saved_model_cfg) + if config["env"]["name"] == "gymnasium": + image_size = model_cfg["rl_games"]["env_eval"]["image_size"] + config["env"]["obs_resize"] = [image_size, image_size] + image_transform_cfg = ( + policy_cfg["image_transform_config"] + if "image_transform_config" in policy_cfg + else {} + ) + image_transform = ( + image_transform_cfg["image_transform"] + if "image_transform" in image_transform_cfg + else "raw_rgb" + ) + apply_starvla_model_input_config( + config, + model_cfg=model_cfg, + image_transform=image_transform, + ) + + +def _load_policy_wrapper_class() -> Any: + _ensure_starvla_path() + from deployment.model_server.policy_wrapper import PolicyServerWrapper + + return PolicyServerWrapper + + +def _profiler_stage(profiler: Any, name: str) -> Any: + from contextlib import nullcontext + + return profiler.time(name) if profiler is not None else nullcontext() + + +def _load_rl_games_action_decode() -> tuple[Any, Any, Any]: + _ensure_starvla_path() + from deployment.model_server.rl_games_action_decode import ( + decode_rl_games_actions, + resolve_asterix_action_decode_spec, + resolve_deadly_action_decode_spec, + ) + + return decode_rl_games_actions, resolve_deadly_action_decode_spec, resolve_asterix_action_decode_spec + + +class LiveStarVlaWrapper: + """In-process stand-in for ``PolicyServerWrapper`` over a *live* framework. + + During training the trainer already holds the model in memory + (``accelerator.unwrap_model(self.model)`` — the same object eval_core calls). + This wrapper exposes only the rl_games-mode surface ``StarVlaPolicyRunner`` + uses — ``predict_action`` (framework forward + rl_games decode), + ``reset_memory`` passthrough, and the ``_model_cfg`` attribute — so no + checkpoint reload is needed. The disk-backed ``PolicyNormProcessor`` is never + built because rl_games decoding ignores un-normalization stats. + """ + + def __init__( + self, + *, + framework: Any, + model_cfg: dict[str, Any], + env_name: str, + rl_games_action_env_dim: int | None = None, + gymnasium_action_space_type: str = "discrete", + action_layout: str | None = None, + multibinary_threshold: float | None = None, + ) -> None: + self._framework = framework + self._model_cfg = model_cfg + self._rl_games_env_name = str(env_name) + self._rl_games_action_env_dim = rl_games_action_env_dim + self._gymnasium_action_space_type = gymnasium_action_space_type + ( + self._decode_rl_games_actions, + resolve_deadly_action_decode_spec, + resolve_asterix_action_decode_spec, + ) = _load_rl_games_action_decode() + self._action_layout = action_layout + self._multibinary_threshold = multibinary_threshold + if self._rl_games_env_name == "deadly_corridor": + self._action_layout, self._multibinary_threshold = resolve_deadly_action_decode_spec( + model_cfg, + action_layout=action_layout, + multibinary_threshold=multibinary_threshold, + ) + elif self._rl_games_env_name == "asterix": + self._action_layout = resolve_asterix_action_decode_spec( + model_cfg, + action_layout=action_layout, + ) + + def reset_memory(self, slot_id: int | None = None) -> None: + reset = getattr(self._framework, "reset_memory", None) + if callable(reset): + reset(slot_id) + + def predict_action( + self, + examples: list[dict[str, Any]], + unnorm_key: str | None = None, + **kwargs: Any, + ) -> dict[str, Any]: + # unnorm_key is unused in rl_games mode; kept for interface parity. + del unnorm_key + profiler = kwargs["profiler"] if "profiler" in kwargs else None + out = self._framework.predict_action(examples=examples, **kwargs) + normalized = np.asarray(out["normalized_actions"]) # (B, T, D) + decode_kwargs: dict[str, Any] = {} + if self._rl_games_env_name == "gymnasium": + decode_kwargs["action_env_dim"] = self._rl_games_action_env_dim + if self._gymnasium_action_space_type == "box": + decode_kwargs["gymnasium_action_space_type"] = "box" + with _profiler_stage(profiler, "starvla_wrapper_rl_games_decode_ms"): + return self._decode_rl_games_actions( + normalized_actions=normalized, + env_name=self._rl_games_env_name, + deadly_action_layout=( + self._action_layout + if self._rl_games_env_name == "deadly_corridor" + else None + ), + deadly_multibinary_threshold=( + self._multibinary_threshold + if self._rl_games_env_name == "deadly_corridor" + else None + ), + asterix_action_layout=( + self._action_layout + if self._rl_games_env_name == "asterix" + else None + ), + **decode_kwargs, + ) + + +_LEGACY_GYMNASIUM_TASK_NAMES = { + "ant_rgb_state": "ant", + "half_cheetah_rgb_state": "half_cheetah", + "hopper_rgb_state": "hopper", + "humanoid_rgb_state": "humanoid", + "inverted_pendulum_rgb_state": "inverted_pendulum", + "swimmer_rgb_state": "swimmer", + "walker2d_rgb_state": "walker2d", +} + + +def _canonical_gymnasium_contract_namespace( + contract: Mapping[str, Any], +) -> dict[str, Any]: + canonical = dict(contract) + task_name = canonical["task_name"] + if task_name in _LEGACY_GYMNASIUM_TASK_NAMES: + canonical["task_name"] = _LEGACY_GYMNASIUM_TASK_NAMES[task_name] + if canonical["env_id"] == "LatencyBench/HopperRgbState-v0": + canonical["env_id"] = "LatencyBench/Hopper-v0" + canonical["registration_imports"] = [ + "latency_bench.envs.gymnasium_hopper" + if module == "latency_bench.envs.gymnasium_hopper_rgb_state" + else module + for module in canonical["registration_imports"] + ] + return canonical + + +def _validate_gymnasium_starvla_contract( + *, + env_cfg: Mapping[str, Any], + policy_cfg: Mapping[str, Any], + model_cfg: Mapping[str, Any], + manifest: Mapping[str, Any], +) -> None: + eval_contract = gymnasium_task_contract(env_cfg) + manifest_task = manifest.get("gymnasium_task") + expected = policy_cfg.get( + "gymnasium_training_task_contract", manifest_task or eval_contract + ) + comparable_eval_contract = {**eval_contract, "make_kwargs": expected["make_kwargs"]} + if _canonical_gymnasium_contract_namespace( + comparable_eval_contract + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA training contract or dataset manifest" + ) + if manifest.get("integration_name", "gymnasium") != "gymnasium": + raise ValueError("StarVLA task manifest is not a Gymnasium handoff") + if manifest_task is not None: + if _canonical_gymnasium_contract_namespace( + manifest_task + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA dataset manifest" + ) + model_contract = model_cfg["datasets"]["vla_data"].get("gymnasium_task_contract") + if model_contract is not None: + if _canonical_gymnasium_contract_namespace( + model_contract + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA model config" + ) + action_space = gymnasium_action_space_contract(env_cfg) + action_layout = str(policy_cfg.get("action_layout", "") or "").strip().lower() + is_asterix_factorized = ( + str(env_cfg.get("task_name", "")) == "asterix" + and action_layout in {"factorized_6", "factorized6", "asterix_factorized_6", "asterix_factorized6"} + ) + if not is_asterix_factorized and manifest["active_action_dim"] != len(action_space["labels"]): + raise ValueError( + "StarVLA dataset active_action_dim does not match its Gymnasium action catalog" + ) + if ( + model_cfg["framework"]["action_model"]["action_env_dim"] + != manifest["active_action_dim"] + ): + raise ValueError( + "StarVLA model action_env_dim does not match the dataset manifest" + ) + model_uses_state = bool(model_cfg["datasets"]["vla_data"]["include_state"]) + manifest_has_state_metadata = ( + "uses_state" in manifest or "state_labels" in manifest + ) + manifest_uses_state = bool(manifest.get("uses_state", model_uses_state)) + if manifest_has_state_metadata: + if policy_cfg.get("state_source") != "transport" and manifest_uses_state != ("state_space" in expected): + raise ValueError( + "StarVLA dataset uses_state does not match the Gymnasium state space" + ) + if manifest_uses_state != model_uses_state: + raise ValueError( + "StarVLA dataset uses_state does not match the model include_state" + ) + if manifest_has_state_metadata and manifest_uses_state: + state_labels = manifest["state_labels"] + expected_state_labels = expected["state_space"]["labels"] if policy_cfg.get("state_source") != "transport" else state_labels + if state_labels != expected_state_labels: + raise ValueError( + "StarVLA dataset state_labels do not match the Gymnasium state space" + ) + if manifest["state_dim"] != len(state_labels): + raise ValueError( + "StarVLA dataset state_dim does not match its state_labels" + ) + if ( + model_cfg["framework"]["action_model"]["state_dim"] + != manifest["state_dim"] + ): + raise ValueError( + "StarVLA model state_dim does not match the dataset manifest" + ) + if not manifest["state_normalization"]: + raise ValueError( + "StarVLA state-enabled dataset manifest is missing state_normalization" + ) + + +def _starvla_runner_kwargs( + config: dict[str, Any], + action_resolver: ActionResolver, + model_cfg: Mapping[str, Any] | None, + *, + base_prompt: str | None, +) -> dict[str, Any]: + """Resolve task and input settings shared by checkpoint and resident models.""" + env_cfg = config["env"] + policy_cfg = config["policy"] + if env_cfg["name"] == "gymnasium": + task_manifest = json.loads( + Path(policy_cfg["task_manifest_path"]).read_text(encoding="utf-8") + ) + _validate_gymnasium_starvla_contract( + env_cfg=env_cfg, + policy_cfg=policy_cfg, + model_cfg=model_cfg, + manifest=task_manifest, + ) + semantic_env_name = env_cfg["task_name"] + action_refs = env_cfg.get("action_order", []) + base_prompt = env_cfg["base_prompt"] + state_normalization = task_manifest.get("state_normalization") + else: + semantic_env_name = env_cfg["name"] + action_refs = policy_cfg.get("actions", action_resolver.default_action_refs()) + state_normalization = policy_cfg["state_normalization"] if "state_normalization" in policy_cfg else None + return dict( + unnorm_key=policy_cfg.get("unnorm_key"), + env_name=semantic_env_name, + action_resolver=action_resolver, + action_refs=action_refs, + latency_prompt_map=( + load_latency_prompt_map(policy_cfg["latency_prompt_map_path"]) + if "latency_prompt_map_path" in policy_cfg + else None + ), + base_prompt=base_prompt, + latency_prompt_key=policy_cfg.get("latency_prompt_key"), + prompt_mode=policy_cfg.get("prompt_mode"), + obs_resize=tuple(env_cfg["obs_resize"]) if env_cfg.get("obs_resize") else None, + image_transform_config=policy_cfg.get("image_transform_config"), + observation_stride_raw_frames=_observation_stride_raw_frames(config), + model_cfg=model_cfg, + state_normalization=state_normalization, + state_source=policy_cfg["state_source"] if "state_source" in policy_cfg else None, + ) + + +def build_starvla_policy( + config: dict[str, Any], + action_resolver: ActionResolver, +) -> PolicyRunner: + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + _ensure_starvla_path() + from latency_bench.policy.starvla_tasks import build_task_starvla_policy + + return build_task_starvla_policy(config) + env_cfg = config["env"] + integration_env_name = env_cfg["name"] + model_cfg = ( + _load_starvla_model_config(policy_cfg["model_config_path"]) + if integration_env_name == "gymnasium" or "model_config_path" in policy_cfg + else None + ) + runner_kwargs = _starvla_runner_kwargs( + config, action_resolver, model_cfg, base_prompt=env_cfg.get("base_prompt") + ) + wrapper_cls = _load_policy_wrapper_class() + wrapper_kwargs: dict[str, Any] = dict( + ckpt_path=policy_cfg["checkpoint_path"], + device=policy_cfg["device"], + use_bf16=True, + unnorm_key=runner_kwargs["unnorm_key"], + action_output_mode=( + policy_cfg["action_output_mode"] + if "action_output_mode" in policy_cfg + else "rl_games" + ), + rl_games_env_name=integration_env_name, + rl_games_action_layout=( + policy_cfg["action_layout"] if "action_layout" in policy_cfg else None + ), + rl_games_multibinary_threshold=( + policy_cfg["multibinary_threshold"] + if "multibinary_threshold" in policy_cfg + else None + ), + ) + if "backbone_path" in policy_cfg: + wrapper_kwargs["backbone_path"] = policy_cfg["backbone_path"] + if integration_env_name == "gymnasium": + action_space = gymnasium_action_space_contract(env_cfg) + wrapper_kwargs["rl_games_action_env_dim"] = len(action_space["labels"]) + if action_space["type"] == "box": + wrapper_kwargs["rl_games_gymnasium_action_space_type"] = "box" + wrapper_kwargs["rl_games_env_name"] = integration_env_name + wrapper = wrapper_cls(**wrapper_kwargs) + return StarVlaPolicyRunner( + wrapper=wrapper, + checkpoint_path=policy_cfg["checkpoint_path"], + device=policy_cfg["device"], + **runner_kwargs, + image_views_info_key=( + policy_cfg["image_views_info_key"] + if "image_views_info_key" in policy_cfg + else None + ), + action_output_type=( + policy_cfg["action_output_type"] + if "action_output_type" in policy_cfg + else None + ), + ) + + +def build_live_starvla_policy( + *, + framework: Any, + model_cfg: dict[str, Any], + config: dict[str, Any], + action_resolver: ActionResolver | None = None, +) -> PolicyRunner: + """Build a StarVLA policy around a *live* in-memory framework (no reload). + + Mirrors ``build_starvla_policy`` but swaps the ckpt-loading + ``PolicyServerWrapper`` for :class:`LiveStarVlaWrapper`, so the trainer's + resident model is evaluated directly. ``model_cfg`` is the in-memory model + config (e.g. ``read_mode_config`` output) the wrapper would otherwise read + from disk. + """ + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + from latency_bench.policy.starvla_tasks import TaskStarVlaPolicyRunner + + contract = json.loads( + Path(policy_cfg["task_contract_path"]).read_text(encoding="utf-8") + ) + return TaskStarVlaPolicyRunner( + framework, + policy_config=policy_cfg, + model_config=model_cfg, + contract=contract, + ) + + env_cfg = config["env"] + integration_env_name = env_cfg["name"] + normalized_model_cfg = ( + _normalized_model_cfg(model_cfg) + if integration_env_name == "gymnasium" + else None + ) + # Resident evaluation historically takes non-Gymnasium prompts from the map. + runner_kwargs = _starvla_runner_kwargs( + config, action_resolver, normalized_model_cfg, base_prompt=None + ) + wrapper_kwargs: dict[str, Any] = dict( + framework=framework, + model_cfg=model_cfg, + env_name=integration_env_name, + action_layout=policy_cfg["action_layout"] if "action_layout" in policy_cfg else None, + multibinary_threshold=( + policy_cfg["multibinary_threshold"] + if "multibinary_threshold" in policy_cfg + else None + ), + ) + if integration_env_name == "gymnasium": + action_space = gymnasium_action_space_contract(env_cfg) + wrapper_kwargs["rl_games_action_env_dim"] = len(action_space["labels"]) + if action_space["type"] == "box": + wrapper_kwargs["gymnasium_action_space_type"] = "box" + wrapper_kwargs["env_name"] = integration_env_name + wrapper = LiveStarVlaWrapper(**wrapper_kwargs) + return StarVlaPolicyRunner( + wrapper=wrapper, + checkpoint_path=policy_cfg.get("checkpoint_path", ""), + device=policy_cfg.get("device", "cuda"), + **runner_kwargs, + ) + + +__all__ = [ + "LiveStarVlaWrapper", + "StarVlaPolicyRunner", + "apply_starvla_model_input_config", + "build_live_starvla_policy", + "build_starvla_policy", + "decode_starvla_action", + "observation_data_to_hwc_uint8_frames", + "prepare_starvla_checkpoint_input_config", +] diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py new file mode 100644 index 0000000000000000000000000000000000000000..4a5cc03813f1bc524a30a7b6b7292a4f4fca85d4 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py @@ -0,0 +1,92 @@ +"""StarVLA inference using the task's training observation/action contract.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import numpy as np +from PIL import Image + +from latency_bench.core.types import Action, Observation, PolicyOutput +from latency_bench.data.starvla_tasks import denormalize, normalize +from latency_bench.policy.base import PolicyRunner + + +class TaskStarVlaPolicyRunner(PolicyRunner): + """Map task RGB/state into a StarVLA model and decode its action chunk.""" + + def __init__(self, framework, *, policy_config: dict, model_config: dict, contract: dict): + self.framework = framework + self.policy_config = policy_config + self.model_config = model_config + self.contract = contract + + def _example(self, observation: Observation) -> dict: + cfg = self.policy_config + state = normalize( + observation.metadata[cfg["state_info_key"]], + self.contract["normalization"]["state"], + ).reshape(1, self.contract["state_dim"]) + data_cfg = self.model_config["datasets"]["vla_data"] + height, width = data_cfg["obs_image_size"] + images = [ + Image.fromarray(frame).resize((width, height)) + for frame in observation.metadata[cfg["image_views_info_key"]] + ] + if data_cfg["image_mode"] == "stitch_views": + from starVLA.training.trainer_utils.trainer_tools import stitch_frames + + # MIKASA's two simultaneous views form one Wan observation, not a video. + images = [stitch_frames(images, grid=data_cfg["stitch_grid"], size=(width, height))] + example = {"image": images, "state": state, "lang": self.contract["prompt"]} + if "action_prefix" in observation.metadata: + example["action_prefix"] = normalize( + observation.metadata["action_prefix"], + self.contract["normalization"]["action"], + ) + example["action_prefix_mask"] = observation.metadata["action_prefix_mask"] + return example + + def predict(self, observation: Observation) -> PolicyOutput: + return self.predict_batch([observation])[0] + + def predict_batch(self, observations: list[Observation]) -> list[PolicyOutput]: + prediction = self.framework.predict_action( + examples=[self._example(observation) for observation in observations] + ) + actions = denormalize( + prediction["normalized_actions"], self.contract["normalization"]["action"] + ) + # Prefix heads were excluded from the loss; retain the frozen controller plan. + for chunk, observation in zip(actions, observations): + if "action_prefix" in observation.metadata: + mask = observation.metadata["action_prefix_mask"] + chunk[mask] = observation.metadata["action_prefix"][mask] + return [ + PolicyOutput( + action=Action(value=chunk[0].tolist(), name="task_command"), + action_chunk=chunk, + raw_output=chunk.tolist(), + metadata={"policy_type": "starvla", "task": self.contract["task"]}, + ) + for chunk in actions + ] + + +def build_task_starvla_policy(config: dict) -> TaskStarVlaPolicyRunner: + # StarVLA and torch are optional in the simulator process; workers own them. + import torch + from starVLA.model.framework.base_framework import baseframework + from starVLA.model.framework.share_tools import read_mode_config + + cfg = config["policy"] + model_config, _ = read_mode_config(cfg["checkpoint_path"]) + framework = baseframework.from_pretrained( + cfg["checkpoint_path"], backbone_path=cfg["backbone_path"] + ) + framework = framework.to(device=cfg["device"], dtype=torch.bfloat16).eval() + contract = json.loads(Path(cfg["task_contract_path"]).read_text(encoding="utf-8")) + return TaskStarVlaPolicyRunner( + framework, policy_config=cfg, model_config=model_config, contract=contract + ) diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json new file mode 100644 index 0000000000000000000000000000000000000000..371542eee194a26ff888c211317abe89050b38e0 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json @@ -0,0 +1,105 @@ +{ + "condition": "profile-latency", + "executor_mode": "simulated", + "latency_method": "temporal", + "profile_source": "originalRTX3090immutableprofiles", + "episodes_per_checkpoint": 100, + "total_episodes": 400, + "rounds": [ + [ + "flappy", + "deadly_corridor" + ], + [ + "ant", + "intercept" + ] + ], + "physical_gpu_assignments": { + "flappy": 2, + "deadly_corridor": 3, + "ant": 2, + "intercept": 3 + }, + "single_gpu_per_job": true, + "round2_requires_both_round1_complete": true, + "latency_seed": 271828, + "tasks": { + "flappy": { + "gpu": 2, + "seed_start": 1000000, + "seed_end": 1000099, + "env_fps": 10, + "obs_fps": 10, + "max_raw_steps": 3600, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 75.87417450998383, + "profile": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/flappy/instance_a5037b165aa0cedc/profile.json", + "sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "deadly_corridor": { + "gpu": 3, + "seed_start": 1000000, + "seed_end": 1000099, + "env_fps": 35, + "obs_fps": 8.75, + "max_raw_steps": 3600, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 73.69250777493353, + "profile": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/deadly_corridor/instance_a5037b165aa0cedc/profile.json", + "sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "ant": { + "gpu": 2, + "seed_start": 42, + "seed_end": 141, + "env_fps": 10, + "obs_fps": 10, + "max_raw_steps": 1000, + "parallel_envs": 16, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 90.56460638563993, + "profile": "/home/ubuntu/lzj/mean-profiling/reference/checkpoint-configs/latency-aware/ant/small-policy/sample-factory-qwenoft-rtx3090-v1/profile/profile.json", + "sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "intercept": { + "gpu": 3, + "seed_start": 4242424242, + "seed_end": 4242424341, + "env_fps": 20, + "obs_fps": 20, + "max_raw_steps": 60, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0", + "profile": { + "mean_ms": 99.05021289731565, + "profile": "/home/ubuntu/lzj/profiles/intercept-published/profiles/qwenoft/1x-rtx3090/mikasa_intercept_grab_fast/instance_3a0d42681a03715c/profile.json", + "sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + } + } +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/execution_audit.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/execution_audit.json new file mode 100644 index 0000000000000000000000000000000000000000..8b0f90c02aad8843102296664b294f5e1e34bc15 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/execution_audit.json @@ -0,0 +1,12 @@ +{ + "issued_action_records": 311075, + "applied_action_records": 310911, + "dropped_action_records": 64, + "nonnoop_issued_records": 30817, + "finite_action_values": true, + "latency_sample_count": 311075, + "latency_mean_ms": 75.89784633675906, + "latency_std_ms": 3.799946378622932, + "latency_p95_ms": 81.3960393048375, + "latency_p99_ms": 87.23844517488543 +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/files.sha256.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/files.sha256.json new file mode 100644 index 0000000000000000000000000000000000000000..915ce79e26f5741fe4309d3091c4407129ee424c --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/files.sha256.json @@ -0,0 +1,130 @@ +{ + "REPORT.md": { + "bytes": 2163, + "sha256": "b2fd63b1cde2a415b2d77daf0184ce5ec3b6ada1aa199c21941fb04031f77db6" + }, + "all_episodes.csv": { + "bytes": 25198, + "sha256": "bd41f35a464250ed6f9bc16e466aa9f1b55a48ac29072bb0ec3559a24129edac" + }, + "comparison.csv": { + "bytes": 438, + "sha256": "a0c7cf191395dd851bd6222bdf61427f15782fea2db4416ddc072a5f5dc8a861" + }, + "comparison.json": { + "bytes": 7826, + "sha256": "0d5dd042439466aee84cd0d96c31a27a951e57965a7e468cb73ec07883f1f751" + }, + "episodes.csv": { + "bytes": 5755, + "sha256": "92b258a93b02c0511f19d7b50975e2a167dd0ca796834f581be77245037ff135" + }, + "eval_config.yaml": { + "bytes": 2056, + "sha256": "9ad297a08d96c66a4ed57c2d66083faf36fdd585a7955faa1d39d9deefa43794" + }, + "evaluation-code/batched_simulated.py": { + "bytes": 31282, + "sha256": "b901f966d911feab7962a32f21095cb90f7880121811f2b4eab2193afe1381db" + }, + "evaluation-code/deadly-compatibility.patch": { + "bytes": 4570, + "sha256": "623676cc4542b1eab6c9395b163b369ddc605353c1de02d17d8f713167ee07fa" + }, + "evaluation-code/deadly_corridor.py": { + "bytes": 17902, + "sha256": "47f7bc65cba9853e66d79ed2a28f844bd2a094f1285458be166045f2db1690dc" + }, + "evaluation-code/decision_action_history.py": { + "bytes": 2746, + "sha256": "14a9d223e775745b6c402dbce9e2a50a1c3f7b5b9fe528150ef8689126fe97cf" + }, + "evaluation-code/eval_driver.py": { + "bytes": 9133, + "sha256": "330030270fbb695bc5f14037ef7349650bd20c53c881c1159ee55ea066408d9e" + }, + "evaluation-code/mikasa_evaluate.py": { + "bytes": 11466, + "sha256": "6cf9ffee25fcfd6f3255c520fc544c48ff2c8f8912e5369c2410a709820c4ffd" + }, + "evaluation-code/starvla.py": { + "bytes": 48378, + "sha256": "6d9988f3a28d39e46c2f6e80da85edebc42cafa629a2b9f75000414324c1065a" + }, + "evaluation-code/starvla_tasks.py": { + "bytes": 4029, + "sha256": "3fc74169d1554d9dc3358ed85e450cca75eb69bc1fff85284c1054a605633a52" + }, + "evaluation-plan.json": { + "bytes": 4698, + "sha256": "b758a5fb72dcdef49d025e2fd168d024ebd8b18b2b00125145b3cde38b16a318" + }, + "execution_audit.json": { + "bytes": 363, + "sha256": "ae7af259c3c9c6c1f8c9686b831e1c6692231af9f005ae862af9e14760b1dbef" + }, + "profile/latency_burst_model.json": { + "bytes": 24143, + "sha256": "8dbbcb87d41bf25c3ca421d80427445d29c31b323e306e8ada99c845746187b3" + }, + "profile/latency_distribution.json": { + "bytes": 25054, + "sha256": "e568e466bdec6ab73d4b5c595bc02e7c7047bdb3f39324254a3aa6f660900250" + }, + "profile/profile.json": { + "bytes": 2140, + "sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc" + }, + "provenance.json": { + "bytes": 3736, + "sha256": "21329d6e66227f37d21d6e78676038d151e6f99f472b30288792b4ff0d8171f3" + }, + "queue_eval_latency_profile_sample.json": { + "bytes": 3915, + "sha256": "3a18d8878aeb1a459aaf716db04b930b7d1bc9d61273d9a5532beb829c984aa8" + }, + "raw-records/actions.jsonl.gz": { + "bytes": 23634236, + "sha256": "783f69617b6eeba129eadb2d698719018b185e13f9558e2fd644fd584ea5471e" + }, + "raw-records/e2e_latencies.jsonl.gz": { + "bytes": 6608982, + "sha256": "c273135c50f8244c182e67fc63203c3bb9f95cc4d8654f26ea5a0008895c2350" + }, + "raw-records/episode_metrics.jsonl.gz": { + "bytes": 6794, + "sha256": "ead85d36c65e4badc29e9e9d97f3a0b0899c7ad0116354adc91b364cfcf52a43" + }, + "raw-records/infer_latencies.jsonl.gz": { + "bytes": 4678442, + "sha256": "6046f2601da69be9c8849b5ca4f69a89e3ad2354d17a8788708c0b2a20ddb47e" + }, + "raw-records/latencies.jsonl.gz": { + "bytes": 4678436, + "sha256": "b79be007401a136bb21ad8e25c4932b567c61dab18e3a052ae0f21ec790a15ea" + }, + "raw-records/observation_attempts.jsonl.gz": { + "bytes": 47, + "sha256": "8b9659fba7af43449720211464ae93a9fa76d044dcceae45b6c39d556ca9f404" + }, + "raw-records/queue_eval_results.jsonl.gz": { + "bytes": 794, + "sha256": "8b7dcbafff1492142f70180b7c6c63cc53b8d866c0247569ebe7498200717e4e" + }, + "raw-records/steps.jsonl.gz": { + "bytes": 7221452, + "sha256": "5bdf7b2ae8b367dd6412faaca477cdb70d3f6bdbc1142a1688d6f48f9801abd8" + }, + "resolved_config.yaml": { + "bytes": 2124, + "sha256": "5ada3462c4b88aff8bfb7604eaa003b6621167dfc3f35675e001ee0725fd52dc" + }, + "statistics.json": { + "bytes": 1218, + "sha256": "232db020427660a6a0f8721175f3433c23b0e2cc51c3769ffffb55a6a96e8aac" + }, + "stdout.log": { + "bytes": 27195, + "sha256": "580e80297e5ea307aad4668e1a25996e033ec23e021d1dff12a0e80e6d23e52c" + } +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json new file mode 100644 index 0000000000000000000000000000000000000000..1d3ce7afc3b937ad7f530fa299d69c46467e8bf5 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json @@ -0,0 +1,1343 @@ +{ + "burst_dwell_distribution": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 2.1400000000000095, + 3.280000000000002, + 4.419999999999995, + 5.560000000000004, + 6.7000000000000135, + 7.840000000000006, + 8.979999999999999, + 10.11999999999999, + 11.260000000000002, + 12.399999999999993, + 13.540000000000003, + 14.679999999999996, + 15.820000000000004, + 16.959999999999997, + 18.10000000000001, + 19.24, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0, + 20.0 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "burst_dwell_lengths": [ + 1, + 1, + 1, + 1, + 1, + 20 + ], + "burst_merge_gap_records": 30, + "burst_rank_processes": [ + { + "draw_count": 25, + "dwell_length_spearman_rho": -1.0, + "level_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.86154556274414, + 89.9292633342743, + 90.03083999156952, + 90.13241664886475, + 90.23399330615997, + 90.3355699634552, + 90.43714662075043, + 90.53872327804565, + 90.64029993534088, + 90.7418765926361, + 90.84345324993133, + 90.94502990722657, + 91.0466065645218, + 91.14818322181702, + 91.24975987911225, + 91.35133653640747, + 91.4529131937027, + 91.55448985099792, + 91.72152845859527, + 91.88856706619262, + 92.05560567378998, + 92.22264428138733, + 92.38968288898468, + 92.55672149658203, + 92.72376010417938, + 92.89079871177674, + 93.05783731937409, + 93.22487592697144, + 93.39191453456878, + 93.55895314216613, + 93.7259917497635, + 93.89303035736084, + 94.06006896495819, + 94.22710757255554, + 94.40493988990784, + 94.60435962677002, + 94.8037793636322, + 95.00319910049438, + 95.20261883735657, + 95.40203857421875, + 95.60145831108093, + 95.80087804794312, + 96.0002977848053, + 96.19971752166748, + 96.39913725852966, + 96.59855699539185, + 96.79797673225403, + 96.99739646911621, + 97.1968162059784, + 97.39623594284058, + 97.59565567970276, + 97.80684516906739, + 98.02391953468323, + 98.24099390029907, + 98.45806826591492, + 98.67514263153076, + 98.8922169971466, + 99.10929136276245, + 99.32636572837829, + 99.54344009399414, + 99.76051445960998, + 99.97758882522582, + 100.19466319084167, + 100.41173755645752, + 100.62881192207337, + 100.84588628768921, + 101.06296065330505, + 101.2800350189209, + 101.36654417037964, + 101.45305332183838, + 101.53956247329712, + 101.62607162475587, + 101.7125807762146, + 101.79908992767334, + 101.88559907913208, + 101.97210823059082, + 102.05861738204956, + 102.1451265335083, + 102.23163568496705, + 102.31814483642579, + 102.40465398788452, + 102.49116313934326, + 102.577672290802, + 102.66418144226074, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "level_residual_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.3340394496917725, + -2.193997082710266, + -2.030614321231842, + -1.8672315597534177, + -1.703848798274994, + -1.54046603679657, + -1.3770832753181457, + -1.2137005138397214, + -1.0503177523612974, + -0.8869349908828732, + -0.723552229404449, + -0.5601694679260254, + -0.39678670644760117, + -0.2334039449691767, + -0.07002118349075337, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.07002118349075376, + 0.2334039449691785, + 0.39678670644760117, + 0.5601694679260238, + 0.7235522294044485, + 0.8869349908828733, + 1.050317752361298, + 1.2137005138397208, + 1.3770832753181454, + 1.5404660367965701, + 1.7038487982749948, + 1.8672315597534175, + 2.0306143212318424, + 2.193997082710267, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725, + 2.3340394496917725 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "severity": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.1, + 0.13600000000000004, + 0.19000000000000009, + 0.24400000000000005, + 0.298, + 0.3520000000000001, + 0.40600000000000014, + 0.45999999999999996, + 0.514, + 0.5680000000000001, + 0.6220000000000001, + 0.6760000000000002, + 0.7300000000000002, + 0.784, + 0.8380000000000001, + 0.8920000000000001, + 0.946, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "spike_count": 7 + } + ], + "burst_slot_rank_templates": [ + { + "dwell_length": 20, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 1, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 1, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 1, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 1, + "slot_to_rank": { + "0": 0 + } + }, + { + "dwell_length": 1, + "slot_to_rank": { + "0": 0 + } + } + ], + "model_type": "hidden_regime", + "pre_worker_time_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 0.2047891616821289, + 0.2047891616821289, + 0.2047891616821289, + 0.20532842445373536, + 0.20854279708862306, + 0.21274469852447508, + 0.21318495559692383, + 0.2145397424697876, + 0.21521625900268554, + 0.2155177412033081, + 0.21618946838378905, + 0.21632096672058104, + 0.21762361526489257, + 0.23855551719665527, + 0.2459665298461914, + 0.26547898292541505, + 0.269596004486084, + 0.27336691856384276, + 0.2765643119812012, + 0.2792212963104248, + 0.2838506317138672, + 0.2880555152893066, + 0.2918656349182129, + 0.2992979049682617, + 0.3044628143310547, + 0.311277494430542, + 0.3166473388671875, + 0.3191685676574707, + 0.3213308334350586, + 0.3233044242858887, + 0.3250684928894043, + 0.3266937255859375, + 0.32917425155639646, + 0.33290016174316406, + 0.3347792625427246, + 0.33614530563354494, + 0.3384699821472168, + 0.34012059211730955, + 0.3432632637023926, + 0.3451723384857178, + 0.3476134490966797, + 0.34894747734069825, + 0.35086851119995116, + 0.35228058815002444, + 0.3533748054504395, + 0.35460223197937013, + 0.3575477600097656, + 0.3603792190551758, + 0.3620700645446777, + 0.36475080490112305, + 0.3671477508544922, + 0.3716339111328125, + 0.37473144531249997, + 0.3802911376953125, + 0.3840335464477539, + 0.38729562759399416, + 0.39225730895996097, + 0.39461712837219237, + 0.3967057418823242, + 0.3984334468841553, + 0.3996261978149414, + 0.4016141891479492, + 0.4027920150756836, + 0.40396700859069823, + 0.4046883392333984, + 0.4059607219696045, + 0.40665283203125, + 0.4082965850830078, + 0.4097418212890625, + 0.4103153991699219, + 0.4111045265197754, + 0.41308178901672366, + 0.41454967498779294, + 0.4162317752838135, + 0.41797222137451173, + 0.4192102909088135, + 0.4206089973449707, + 0.4219807529449463, + 0.4226674461364746, + 0.425172233581543, + 0.42700037002563473, + 0.4291196823120117, + 0.4314889907836914, + 0.434225959777832, + 0.4353223991394043, + 0.4372860336303711, + 0.43889951705932617, + 0.4425575542449951, + 0.44569433212280274, + 0.4494281673431396, + 0.4549727439880371, + 0.46193933486938477, + 0.46680646896362316, + 0.4753505516052246, + 0.4836859893798828, + 0.491743278503418, + 0.5080690383911132, + 0.5304245281219482, + 0.5384305953979492, + 0.5668252277374268, + 0.5818485450744629, + 0.6024106502532963, + 0.6333318519592286, + 0.6780055236816408, + 0.7588464927673341, + 0.9670760154724117, + 1.1565387725830074, + 1.2728116035461425, + 1.2875932693481444, + 1.3012917518615723, + 1.3147638320922852, + 1.316931776046753, + 1.3239655952453613, + 1.3250430908203126, + 1.333270156860351, + 1.334588599205017, + 1.3354324951171876, + 1.338087242126465, + 1.3629943504333506, + 1.4335690317153942, + 1.4773869514465332, + 1.4773869514465332, + 1.4773869514465332 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "regime_step_counts": { + "burst": 25, + "calm": 913 + }, + "regime_transition_counts": { + "burst": { + "burst": 19, + "calm": 6 + }, + "calm": { + "burst": 5, + "calm": 903 + } + }, + "reset_scope": "session", + "schema_version": 12, + "spike_median_multiplier": 1.25, + "spike_threshold_ms_by_worker_slot": { + "0": 87.94450879096985 + }, + "worker_count": 1 +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json new file mode 100644 index 0000000000000000000000000000000000000000..c41773a53e85acbab7648fffd76d9b1998aab64f --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json @@ -0,0 +1,1027 @@ +{ + "schema_version": 3, + "worker_slots": { + "0": { + "all": { + "count": 938, + "observation_to_action_latency_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 69.08178091049194, + 69.08178091049194, + 69.08178091049194, + 69.3522614994049, + 69.70705702209473, + 69.73708504867554, + 69.78681921768188, + 69.83019891738891, + 69.86153617858886, + 69.87217982673646, + 69.95185136604309, + 69.96146729373932, + 69.9623488998413, + 70.1965622997284, + 70.4496435546875, + 70.65893580436706, + 70.80613803863525, + 70.9744321346283, + 71.06177293777466, + 71.26533513069153, + 71.3854751586914, + 71.50117712020874, + 71.6162088394165, + 71.72474594116211, + 71.8356183052063, + 71.97291188240051, + 72.05821580886841, + 72.18015824317932, + 72.2925217628479, + 72.38883204460144, + 72.51062061309814, + 72.59757180213929, + 72.74044584274291, + 72.90701542854309, + 73.09148342132569, + 73.34760197639466, + 73.5559868812561, + 73.7169365787506, + 73.83404542922973, + 73.97230345726014, + 74.05031507492066, + 74.1158531665802, + 74.20066675186158, + 74.28788143157959, + 74.34544937133789, + 74.42585638046265, + 74.5021183013916, + 74.59316722869873, + 74.64286708831787, + 74.72254509925843, + 74.78180068969726, + 74.86512093544006, + 74.9555677986145, + 75.01526986122131, + 75.05677150726318, + 75.08256605148316, + 75.1190312385559, + 75.1481228351593, + 75.18350809097291, + 75.25976671218872, + 75.31729890823364, + 75.36110377311707, + 75.395055103302, + 75.42647839546204, + 75.47659118652344, + 75.51591772079468, + 75.53838272094727, + 75.56156175613404, + 75.62323127746582, + 75.72485078811646, + 75.8093360710144, + 75.91970076560973, + 76.0030605506897, + 76.07787865638733, + 76.19754167556763, + 76.29926692008972, + 76.5935975074768, + 76.8529163646698, + 77.14129537582397, + 77.27078147888183, + 77.4041700553894, + 77.52268018722535, + 77.67947263717652, + 77.86368671417236, + 78.08059213638306, + 78.20550627708435, + 78.33564949035645, + 78.47313025474548, + 78.58401529312134, + 78.75006255149842, + 78.88746383666992, + 79.01072158813477, + 79.22483276367188, + 79.37009027481079, + 79.43116065979004, + 79.51448454856873, + 79.58098840713501, + 79.64467880249023, + 79.83353368759155, + 79.90991855621338, + 80.01278017044068, + 80.16442475318908, + 80.35230888366699, + 80.45341986656189, + 80.65504892349243, + 81.01053869247437, + 81.33792181015015, + 81.83728136062622, + 82.09776624679566, + 82.88440424919129, + 85.80679029464721, + 86.47118320560455, + 88.17492262077329, + 90.34708720684044, + 91.05079241180417, + 94.43715539455391, + 95.42594339179995, + 97.76248225688946, + 101.14307580184945, + 103.30021444034578, + 103.8806209564209, + 103.8806209564209, + 103.8806209564209 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "spearman_rho": 0.9983983283197062, + "worker_service_time_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 68.75959205627441, + 68.75959205627441, + 68.75959205627441, + 69.00946328353882, + 69.33122481918335, + 69.340614528656, + 69.36211964035034, + 69.37948598384857, + 69.39005448150635, + 69.40352217102051, + 69.41496569061279, + 69.5075655670166, + 69.52358875274658, + 69.67593017578125, + 70.01167224884033, + 70.26080693244934, + 70.3821382522583, + 70.48221622467041, + 70.62546207427978, + 70.7868817615509, + 70.91883989334106, + 71.00235118865967, + 71.15950798034667, + 71.23610876083374, + 71.41498142242432, + 71.54023916244506, + 71.64315576553345, + 71.77561540603638, + 71.91856380462646, + 71.97557046890259, + 72.10452894210816, + 72.22118029594421, + 72.30684946060181, + 72.47460773468018, + 72.67994302749634, + 72.96932061195373, + 73.17147970199585, + 73.30587714195252, + 73.46305452346802, + 73.5724040031433, + 73.69833086013794, + 73.75194005966186, + 73.81824583053589, + 73.89441362380981, + 73.94912057876587, + 74.02918689727784, + 74.09208641052246, + 74.14808616638183, + 74.22865966796876, + 74.3110915184021, + 74.39030626296997, + 74.47650032043457, + 74.55701307296754, + 74.63550645828246, + 74.67433490753174, + 74.72970756530762, + 74.75590114593506, + 74.78372334480285, + 74.84009609222412, + 74.87447578430175, + 74.91398078918456, + 74.96205759048462, + 75.0261568069458, + 75.04786426544189, + 75.07474729537964, + 75.10028100013733, + 75.14237432479858, + 75.17450904846191, + 75.23760179519654, + 75.28151608467103, + 75.36534164428711, + 75.50974855422973, + 75.57104387283326, + 75.67517771720887, + 75.77710733413696, + 75.8984913635254, + 76.11688394546509, + 76.40118645668029, + 76.58011302947999, + 76.83199727058411, + 76.96488168716431, + 77.09794030189514, + 77.22499988555909, + 77.38846775054931, + 77.58174741744995, + 77.85108305931091, + 77.97168064117432, + 78.05904194831848, + 78.17130668640137, + 78.34864307403565, + 78.49651319503783, + 78.62861022949218, + 78.75943649291993, + 78.88904364585876, + 79.01704561233521, + 79.09859545707702, + 79.1674017906189, + 79.24642925262451, + 79.36339258193969, + 79.4634185886383, + 79.56384315490723, + 79.68323440551758, + 79.83302282333374, + 80.01404458999635, + 80.14517425537109, + 80.41443992614745, + 80.71330451965332, + 80.94748831748963, + 81.28469449996949, + 81.74042127609253, + 84.93557418823242, + 85.38462333011626, + 87.05774599647519, + 89.07818281078332, + 89.77948538208005, + 93.12340239047982, + 94.22508243370058, + 96.61849896907819, + 99.91970232772836, + 102.09033740425112, + 102.7218542098999, + 102.7218542098999, + 102.7218542098999 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + } + }, + "steady": { + "count": 913, + "observation_to_action_latency_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 69.08178091049194, + 69.08178091049194, + 69.08178091049194, + 69.33682310962676, + 69.70602769756317, + 69.73298105430602, + 69.78197941589356, + 69.82547656059265, + 69.85996865463257, + 69.86592574834823, + 69.9350998840332, + 69.95916070127487, + 69.96226736068725, + 70.18701538085938, + 70.42333960056305, + 70.65646963119507, + 70.77730486392974, + 70.91751949310303, + 71.05068702220916, + 71.2463477897644, + 71.35197869300842, + 71.48028535842896, + 71.60071415901184, + 71.66944967269897, + 71.83421282768249, + 71.97153978347778, + 72.05503916740417, + 72.16232189178467, + 72.27710499286651, + 72.36795484542847, + 72.4969009590149, + 72.58834071159363, + 72.68100485801696, + 72.8787705230713, + 73.06083678245544, + 73.28979613304138, + 73.50630414485931, + 73.65355167388915, + 73.78796582698823, + 73.9308990764618, + 74.02597165107727, + 74.10450706481933, + 74.15856179237366, + 74.2300809764862, + 74.31587416648864, + 74.382629737854, + 74.46606838703156, + 74.51586047172546, + 74.60264430046081, + 74.65282180786133, + 74.74318698406219, + 74.80846590995789, + 74.90220269680023, + 74.95917541503906, + 75.01957015037537, + 75.06663251876832, + 75.10325560569763, + 75.12833253860474, + 75.16516513347625, + 75.20677936553955, + 75.27058818817139, + 75.31999588012695, + 75.37381466388702, + 75.40834631919861, + 75.44362805366517, + 75.49281471252442, + 75.5199806213379, + 75.54890730857849, + 75.581811876297, + 75.67140531539917, + 75.75319175243378, + 75.84056982994079, + 75.99374648094177, + 76.05999645233155, + 76.1609637928009, + 76.28250997543336, + 76.54175560474395, + 76.81956731796265, + 77.02320156097413, + 77.22324830055237, + 77.3714802980423, + 77.48301849365234, + 77.65079073905945, + 77.77116039276123, + 77.95271341323853, + 78.1702041053772, + 78.287269115448, + 78.40774132728576, + 78.52459082126617, + 78.66848516464233, + 78.7810197687149, + 78.95290679931641, + 79.09251703262329, + 79.32884536743164, + 79.38602262496948, + 79.46924849510192, + 79.56182599067688, + 79.6104863166809, + 79.68763805389405, + 79.86168578147888, + 79.94084632873535, + 80.04360647201538, + 80.27118618011475, + 80.39313398361206, + 80.54387353897094, + 80.81621770858764, + 81.05243911743165, + 81.47305136680603, + 81.9590122127533, + 82.24108312606812, + 83.134890294075, + 83.21320182132722, + 83.472487323761, + 83.80236521148682, + 83.88565205574037, + 84.17694194316863, + 84.36015990066528, + 85.40999264430995, + 86.15873416709898, + 87.43203411245344, + 88.18218803405762, + 88.18218803405762, + 88.18218803405762 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "spearman_rho": 0.9983820564891248, + "worker_service_time_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 68.75959205627441, + 68.75959205627441, + 68.75959205627441, + 68.9952012271881, + 69.33107182598114, + 69.33882781982422, + 69.36007095718384, + 69.37814243555069, + 69.38795571231842, + 69.4011555633545, + 69.41225225067139, + 69.48535344314575, + 69.520663356781, + 69.6434458732605, + 69.98708435535431, + 70.25511850357056, + 70.37471828460693, + 70.46561038970947, + 70.58880788326263, + 70.7325645160675, + 70.90581192970276, + 70.99077920913696, + 71.12853382587433, + 71.20477669715882, + 71.36176012992858, + 71.52387050628663, + 71.6368390083313, + 71.76177242279053, + 71.86840916156768, + 71.9571689414978, + 72.07422236442567, + 72.20868577957154, + 72.28431457519531, + 72.39973854064941, + 72.6086862707138, + 72.83428075790405, + 73.08615458011627, + 73.2752686882019, + 73.39492557525635, + 73.52648917198181, + 73.64443270683289, + 73.7143747329712, + 73.78998221874237, + 73.86660785675049, + 73.91925894737244, + 73.99132339477539, + 74.04697244167328, + 74.1324462890625, + 74.16984744548797, + 74.25121559143066, + 74.35599178314209, + 74.41620349884033, + 74.49587069034577, + 74.59911281585693, + 74.63859403133392, + 74.6903220462799, + 74.7385401725769, + 74.76270219802856, + 74.7884992647171, + 74.85078914642334, + 74.87601968765259, + 74.92103958129883, + 74.97096348285675, + 75.03208468437195, + 75.05556582450866, + 75.08175855636597, + 75.1202612876892, + 75.14634769439698, + 75.2041501712799, + 75.24591962814331, + 75.32219695568085, + 75.47021465301513, + 75.56161201000214, + 75.65174047470093, + 75.75194797992707, + 75.88618678092956, + 76.02752676010132, + 76.22177825927734, + 76.46858201980591, + 76.72656226158142, + 76.88455163478851, + 77.05107126235961, + 77.18098442554474, + 77.32378499984742, + 77.56891134738922, + 77.81127361297608, + 77.93552327156067, + 78.02126794815064, + 78.11919279575348, + 78.26314960479736, + 78.41598585605621, + 78.5708372592926, + 78.68496776580811, + 78.82959077835083, + 78.94930918693542, + 79.04689074516297, + 79.13465294837951, + 79.19665607452393, + 79.26221225738526, + 79.41456558227539, + 79.51796751022339, + 79.63442335128784, + 79.76517954826355, + 79.93546268463135, + 80.07000221252441, + 80.24641067504882, + 80.62730677127838, + 80.8230856513977, + 81.05952773571015, + 81.43655324935914, + 81.95316825389862, + 82.17854516839982, + 82.40586045837402, + 82.52450770950318, + 82.82636683940889, + 83.0767314314842, + 83.19135388183594, + 84.48698911523813, + 85.15314119148253, + 86.32820743513105, + 87.06488084793091, + 87.06488084793091, + 87.06488084793091 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + } + } + } + } +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/profile.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/profile.json new file mode 100644 index 0000000000000000000000000000000000000000..1dc1af8afd0740a8e69efc4ad41b1e3173bdf3fc --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/profile/profile.json @@ -0,0 +1,66 @@ +{ + "burst_model_path": "latency_burst_model.json", + "distribution_path": "latency_distribution.json", + "env_fps": 30, + "frame_ms": 33.333333333333336, + "gpu_class": "1x-rtx3090", + "instance_id": "instance_a5037b165aa0cedc", + "latency_kind": "observation_to_action_latency", + "latency_method": "temporal", + "model_id": "openvla", + "n_admitted_observations": 938, + "n_capacity_drops": 1849, + "n_observation_attempts": 2787, + "per_slot_summary": { + "0": { + "admitted_count": 938, + "mean_observation_to_action_latency_ms": 75.87417450998383, + "mean_worker_service_time_ms": 75.42596991280757, + "p95_observation_to_action_latency_ms": 81.31731944084167, + "p95_worker_service_time_ms": 80.70259928703308, + "p99_worker_service_time_ms": 84.26694274425506 + } + }, + "provenance": { + "base_config": "configs/experiments/flappy/realtime/starvla_openvla_fix_latency_0_30fps_measured_warmup200_3ep.yaml", + "checkpoint_kind": "best", + "model_artifact": { + "checkpoint": "checkpoints/steps_5000_pytorch_model.pt", + "model_config": "config.full.yaml", + "path_in_repo": ".", + "repo_id": "latency-sensitive-bench/openvla_flappy_fix_latency_0", + "source": "local" + }, + "session_ids": [ + 0, + 1, + 2, + 3, + 4 + ] + }, + "sample_model_type": "hidden_regime", + "source_run_id": "20260914T122201421825Z", + "summary": { + "frame_ms": 33.333333333333336, + "max_ms": 103.8806209564209, + "mean_effective_frames": 2.2762252352995147, + "mean_ms": 75.87417450998383, + "min_ms": 69.08178091049194, + "n_samples": 938, + "p50_frames": 2.2608331131935118, + "p50_ms": 75.36110377311707, + "p90_frames": 2.404656934261322, + "p90_ms": 80.15523114204407, + "p95_frames": 2.43951958322525, + "p95_ms": 81.31731944084167, + "p99_frames": 2.557028575229644, + "p99_ms": 85.23428584098815, + "prob_latency_gt_1_frame": 1.0, + "prob_latency_gt_2_frames": 1.0, + "prob_latency_gt_3_frames": 0.0021321961620469083, + "std_ms": 3.7074819521963067 + }, + "visualization_path": "latency_profile.png", + "workload_id": "flappy" +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/provenance.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..9dffc6d4e2d595269089223f89b654c4c1548406 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/provenance.json @@ -0,0 +1,97 @@ +{ + "task": "flappy", + "protocol": { + "gpu": 2, + "seed_start": 1000000, + "seed_end": 1000099, + "env_fps": 10, + "obs_fps": 10, + "max_raw_steps": 3600, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 75.87417450998383, + "profile": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/flappy/instance_a5037b165aa0cedc/profile.json", + "sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "checkpoint_weights": { + "bytes": 9785137569, + "sha256": "f784cb7d6c19872b4ac419f8a31459b395c12f23809f561ae134080a74c6e80a" + }, + "evaluation": { + "n_episodes": 100, + "mean_return": 384.8240045265853, + "std_return": 116.78777394316903, + "min_return": 36.60000045597553, + "max_return": 444.6000052243471, + "mean_length": 3119.31, + "std_length": 939.8693174585497, + "min_length": 314.0, + "max_length": 3600.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "flappy", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "flappy", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T122201421825Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "flappy", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42", + "profile_sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 64, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy.yaml", + "execution_audit": { + "issued_action_records": 311075, + "applied_action_records": 310911, + "dropped_action_records": 64, + "nonnoop_issued_records": 30817, + "finite_action_values": true, + "latency_sample_count": 311075, + "latency_mean_ms": 75.89784633675906, + "latency_std_ms": 3.799946378622932, + "latency_p95_ms": 81.3960393048375, + "latency_p99_ms": 87.23844517488543 + } + }, + "source_revision": { + "repo": "c3c6a39365a151e9b7a5e215452fd64e957c2b29", + "starvla": "ccca13c5177fe3d3c884b6e2de4965d916016649", + "runtime_fixes": [ + "mean-profile-preparation.patch", + "gym-language-contract.patch", + "loader-spawn-cache.patch", + "loader-spawn-test.patch", + "doom-mean-reset.patch" + ], + "pytorch3d": { + "revision": "33824be3cbc87a7dd1db0f6a9a9de9ac81b2d0ba", + "build": "transforms-only, no native render extension; QwenOFT uses transforms only" + }, + "decord": { + "version": "0.6.0", + "build": "official source CPU decoder CP310", + "wheel_sha256": "e193b356b1e984b4eff08d23b62e482c2c9e5037a6efdc0b1af47079ae2e4c47" + } + }, + "raw_records_format": "gzip(JSONL), lossless", + "startup_checks_included_in_score": false, + "results_status": "evaluation_complete; acceptance_not_inferred" +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json new file mode 100644 index 0000000000000000000000000000000000000000..57d8a8691bb9bd3e99061eb8e279b89268ee9383 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json @@ -0,0 +1,217 @@ +{ + "checkpoint_path": "/home/ubuntu/lzj/mean-profiling/flappy/vla-publication/checkpoints/model.pt", + "experiment_name": "flappy-mean5000-profile-simulation-100ep", + "latency": "profile_sample", + "latency_type": "profile_sample", + "lengths": [ + 3600, + 3600, + 3600, + 3600, + 3600, + 1861, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 2162, + 3600, + 3600, + 3600, + 3600, + 3600, + 955, + 3600, + 3600, + 3600, + 3600, + 2085, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 468, + 3600, + 3512, + 2237, + 2156, + 2161, + 3600, + 1180, + 3600, + 771, + 473, + 2159, + 3600, + 3600, + 3234, + 3600, + 2200, + 582, + 3600, + 3600, + 3600, + 3553, + 2834, + 3600, + 656, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 1109, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 3600, + 957, + 3600, + 3600, + 314, + 2125, + 3600, + 3600, + 3600, + 3600, + 1831, + 3600, + 1478, + 2478, + 3600, + 3600, + 3600, + 3600, + 3600 + ], + "mean_length": 3119.31, + "mean_return": 384.8240045265853, + "returns": [ + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 228.2000027000904, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 265.50000313669443, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 116.00000138580799, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 256.0000030249357, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 55.60000067949295, + 444.6000052243471, + 432.900005094707, + 274.8000032454729, + 264.90000312775373, + 265.4000031352043, + 444.6000052243471, + 143.90000171214342, + 444.6000052243471, + 93.1000011190772, + 56.10000068694353, + 265.2000031322241, + 444.6000052243471, + 444.6000052243471, + 398.80000469088554, + 444.6000052243471, + 270.2000031918287, + 69.70000084489584, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 437.90000515431166, + 348.9000041112304, + 444.6000052243471, + 78.9000009521842, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 135.0000016093254, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 116.20000138878822, + 444.6000052243471, + 444.6000052243471, + 36.60000045597553, + 260.90000308305025, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 225.20000265538692, + 444.6000052243471, + 180.9000021442771, + 305.2000035941601, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471, + 444.6000052243471 + ], + "seed": 1000000, + "source_profile_path": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/flappy/instance_a5037b165aa0cedc/profile.json", + "std_return": 116.78777394316901, + "suite_name": "profile_sample", + "timestamp_utc": "2026-10-01T06:42:51.932491+00:00" +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..a4b5e1527a34b6d2b179b3e421edcd577863dcf1 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:783f69617b6eeba129eadb2d698719018b185e13f9558e2fd644fd584ea5471e +size 23634236 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..859dcfcce636712935274362fb40228dd788d875 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c273135c50f8244c182e67fc63203c3bb9f95cc4d8654f26ea5a0008895c2350 +size 6608982 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..a2d6d2b94691817ae45a9b30f1ea71b48d210ae0 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ead85d36c65e4badc29e9e9d97f3a0b0899c7ad0116354adc91b364cfcf52a43 +size 6794 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..993dfad54ffa58b27f5448f1708f63e5ed5742c8 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6046f2601da69be9c8849b5ca4f69a89e3ad2354d17a8788708c0b2a20ddb47e +size 4678442 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..2fe21e1ec379bef81ffcc1b7d4b982025bb3329f --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b79be007401a136bb21ad8e25c4932b567c61dab18e3a052ae0f21ec790a15ea +size 4678436 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..a506d732f42c8d042187a8054fcb69ca5fda46bf --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8b9659fba7af43449720211464ae93a9fa76d044dcceae45b6c39d556ca9f404 +size 47 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..5441778983ca22a88fa86523831d9ea4f3339376 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8b7dcbafff1492142f70180b7c6c63cc53b8d866c0247569ebe7498200717e4e +size 794 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..557c8dba22329716082e7adaabf78d9579ab43db --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5bdf7b2ae8b367dd6412faaca477cdb70d3f6bdbc1142a1688d6f48f9801abd8 +size 7221452 diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6d136497e97788cc84a6fb209e4f5957fd1fc5d0 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml @@ -0,0 +1,85 @@ +experiment: + name: flappy-mean5000-profile-simulation-100ep + seed: 1000000 +backend: + run_mode: eval +executor: + mode: simulated + simulated_worker_capacity: 1 + simulated_inference_pool: true + inference_devices: + - cuda:0 + inference_batch_size: 32 +env: + name: flappy + gym_id: FlappyBird-v0 + env_fps: 10 + obs_fps: 10 + render_mode: rgb_array + observation_mode: image + use_lidar: false + normalize_obs: true + audio_on: false + frame_stack: 1 + obs_resize: + - 224 + - 224 + simulator: gpu + use_gpu_render: false + gpu_render_device: auto + gpu_render_profile: false + gpu_render_profile_interval: 200 + noop_action: noop + action_map: + noop: 0 + flap: 1 + oneshot_actions: + - flap + action_history_decisions: 8 +latency: + method: temporal + profile_path: /home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/flappy/instance_a5037b165aa0cedc/profile.json + profile_worker_slot: 0 + seed: 271828 + add_latency_info: false +scheduler: + hold_policy: one_frame_then_noop + ordering_policy: latest_ready +policy: + type: starvla + checkpoint_path: /home/ubuntu/lzj/mean-profiling/flappy/vla-publication/checkpoints/model.pt + model_config_path: /home/ubuntu/lzj/mean-profiling/flappy/vla-publication/config.full.yaml + device: cuda:0 + unnorm_key: new_embodiment + prompt_mode: latency_neutral + actions: + - noop + - flap + state_source: transport + backbone_path: /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct + worker_python_executable: /home/ubuntu/lzj/conda/envs/qwenoft/bin/python + action_prefix: + mode: none +evaluation: + eval_episodes: 100 + eval_parallel_envs: 32 + latency_bench_env_backend: flappy_gpu_batched + eval_latency_values: null + eval_max_steps: 3600 + eval_deterministic: true + eval_raw_reward: true + eval_suites: + fixed: [] + normal: [] + uniform: [] +logging: + output_dir: /home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy + save_step_records: true + save_action_records: true + save_latency_records: true + video: + enabled: false + wandb_project: null + wandb_group: null + wandb_job_type: null + simulated_pipeline_profile: false diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/statistics.json b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/statistics.json new file mode 100644 index 0000000000000000000000000000000000000000..15a41ce3a1890fc73dcdeee46f8bad4c59bc237d --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/statistics.json @@ -0,0 +1,36 @@ +{ + "n_episodes": 100, + "mean_return": 384.8240045265853, + "std_return": 116.78777394316903, + "min_return": 36.60000045597553, + "max_return": 444.6000052243471, + "mean_length": 3119.31, + "std_length": 939.8693174585497, + "min_length": 314.0, + "max_length": 3600.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "flappy", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "flappy", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T122201421825Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "flappy", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42", + "profile_sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 64, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy.yaml" +} diff --git a/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/stdout.log b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/stdout.log new file mode 100644 index 0000000000000000000000000000000000000000..d1310154151082d5f90c29adacecc0b1fd3ee446 --- /dev/null +++ b/latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42/evaluation/profile-simulation-100ep-20261001/stdout.log @@ -0,0 +1,397 @@ +[bench] run=flappy-mean5000-profile-simulation-100ep sweeps=1 episodes_per_sweep=100 total_episode_runs=100 +[bench] sweep 1/1: eval_latency=profile_sample +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/site-packages/pygame/pkgdata.py:25: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81. + from pkg_resources import resource_stream, resource_exists +10/01 [06:01:51] INFO | >> Loaded mixtures from Behavior registry.py:113 + (data_config): ['BEHAVIOR_challenge'] + INFO | >> Loaded data_config from DOMINO: registry.py:107 + ['robotwin'] + INFO | >> Loaded embodiment_tags from registry.py:110 + DOMINO (data_config): [] + INFO | >> Loaded mixtures from DOMINO registry.py:113 + (data_config): ['domino', + 'domino_clean', 'domino_random', + 'domino_cotrain'] + INFO | >> Loaded data_config from Franka: registry.py:107 + ['custom_robot_config', + 'demo_sim_franka_delta_joints', + 'SO101'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Franka (data_config): [] + INFO | >> Loaded mixtures from Franka registry.py:113 + (data_config): ['custom_dataset', + 'custom_dataset_2', + 'demo_sim_pick_place', 'SO101_pick'] + INFO | >> Loaded data_config from LIBERO: registry.py:107 + ['libero_franka'] + INFO | >> Loaded embodiment_tags from registry.py:110 + LIBERO (data_config): [] + INFO | >> Loaded mixtures from LIBERO registry.py:113 + (data_config): ['libero_all', + 'libero_goal', 'multi_robot'] + INFO | >> Loaded data_config from MIKASA: registry.py:107 + ['mikasa_franka_h1'] + INFO | >> Loaded embodiment_tags from registry.py:110 + MIKASA (data_config): + ['mikasa_franka_h1'] + INFO | >> Loaded mixtures from MIKASA registry.py:113 + (data_config): + ['local/intercept_grab_fast_vla_v0_h1_ + train'] + INFO | >> Loaded data_config from registry.py:107 + RoboChallenge_table30v2: + ['ur5_robochallenge', + 'arx5_robochallenge', + 'dosw1_robochallenge'] + INFO | >> Loaded embodiment_tags from registry.py:110 + RoboChallenge_table30v2 (data_config): + ['ur5_robochallenge', + 'arx5_robochallenge', + 'dosw1_robochallenge'] + INFO | >> Loaded mixtures from registry.py:113 + RoboChallenge_table30v2 (data_config): + ['robochallenge_table30v2_shred_paper' + , 'robochallenge_table30v2_ur5_all', + 'robochallenge_table30v2_arx5_all', + 'robochallenge_table30v2_dosw1_all'] + INFO | >> Loaded data_config from registry.py:107 + Robocasa_365: + ['panda_omron_robocasa365'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Robocasa_365 (data_config): [] + INFO | >> Loaded mixtures from registry.py:113 + Robocasa_365 (data_config): + ['robocasa365_open_drawer_target_human + ', + 'robocasa365_atomic_target_human_all', + 'robocasa365_composite_target_human_al + l', 'robocasa365_target_human_all'] + INFO | >> Loaded data_config from registry.py:107 + Robocasa_tabletop: + ['fourier_gr1_arms_waist'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Robocasa_tabletop (data_config): [] + INFO | >> Loaded mixtures from registry.py:113 + Robocasa_tabletop (data_config): + ['fourier_gr1_unified_1000'] + INFO | >> Loaded data_config from registry.py:107 + Robotwin: ['robotwin', 'robotwin50', + 'arx_x5'] + INFO | >> Loaded embodiment_tags from registry.py:110 + Robotwin (data_config): [] + INFO | >> Loaded mixtures from Robotwin registry.py:113 + (data_config): ['robotwin_all', + 'robotwin_all_50', 'robotwin', + 'robotwin_task1', 'robotwin_task2', + 'arx_x5'] + INFO | >> Loaded data_config from registry.py:107 + SimplerEnv: ['oxe_droid', + 'oxe_bridge', 'oxe_rt1'] + INFO | >> Loaded embodiment_tags from registry.py:110 + SimplerEnv (data_config): [] + INFO | >> Loaded mixtures from SimplerEnv registry.py:113 + (data_config): ['bridge', + 'bridge_rt_1'] + INFO | >> Loaded data_config from registry.py:107 + VLA-Arena: ['vla_arena_franka'] + INFO | >> Loaded embodiment_tags from registry.py:110 + VLA-Arena (data_config): [] + INFO | >> Loaded mixtures from VLA-Arena registry.py:113 + (data_config): ['vla_arena_L0_S', + 'vla_arena_L0_M', 'vla_arena_L0_L'] + INFO | >> Loaded data_config from registry.py:107 + rl_games: ['rl_games_flappy', + 'rl_games_demon_attack', + 'rl_games_defend_the_line', + 'rl_games_deadly_corridor', + 'rl_games_asterix', + 'rl_games_atlantis', + 'rl_games_gymnasium', + 'rl_games_gymnasium_discrete', + 'rl_games_gymnasium_native'] + INFO | >> Loaded embodiment_tags from registry.py:110 + rl_games (data_config): + ['rl_games_flappy', + 'rl_games_demon_attack', + 'rl_games_defend_the_line', + 'rl_games_deadly_corridor', + 'rl_games_asterix', + 'rl_games_atlantis', + 'rl_games_gymnasium', + 'rl_games_gymnasium_discrete', + 'rl_games_gymnasium_native'] + INFO | >> Loaded mixtures from rl_games registry.py:113 + (data_config): ['flappy_train', + 'flappy_train__bridge', + 'flappy_mixed_latency_train', + 'flappy_mixed_latency_train__bridge', + 'demon_attack_train', + 'demon_attack_train__bridge', + 'demon_attack_mixed_latency_train', + 'demon_attack_mixed_latency_train__bri + dge', 'defend_the_line_train', + 'defend_the_line_train__bridge', + 'defend_the_line_mixed_latency_train', + 'defend_the_line_mixed_latency_train__ + bridge', 'deadly_corridor_train', + 'deadly_corridor_train__bridge', + 'deadly_corridor_mixed_latency_train', + 'deadly_corridor_mixed_latency_train__ + bridge', 'asterix_train', + 'asterix_train__bridge', + 'asterix_mixed_latency_train', + 'asterix_mixed_latency_train__bridge', + 'atlantis_train', + 'atlantis_train__bridge', + 'atlantis_mixed_latency_train', + 'atlantis_mixed_latency_train__bridge' + , 'h1hand_balance_hard'] + INFO | >> PolicyServerWrapper: loading policy_wrapper.py:73 + framework from + /home/ubuntu/lzj/mean-profiling/f + lappy/vla-publication/checkpoints + /model.pt + INFO | >> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/fl + appy/vla-publication/checkpoints/mo + del.pt` +[QWen3] loading /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct with gradient_checkpointing=False + Loading checkpoint shards: 0%| | 0/2 [00:00> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/fl + appy/vla-publication/checkpoints/mo + del.pt` + INFO | >> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/fl + appy/vla-publication/checkpoints/mo + del.pt` + INFO | >> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/fl + appy/vla-publication/checkpoints/mo + del.pt` + INFO | >> PolicyNormProcessor policy_norm_processor.py:333 + ready: + robot_type=rl_games_flapp + y, + unnorm_key=new_embodiment + , + action_keys=['action.butt + on'] (dims=[2]), + state_keys=['state.game_s + tate'] + INFO | >> PolicyServerWrapper ready: policy_wrapper.py:126 + action_chunk_size=1, + default_unnorm_key=new_embodimen + t, + available_unnorm_keys=['new_embo + diment'], + action_keys=['action.button'], + state_keys=['state.game_state'] +[bench] sweep 1/1 episode 1/100 done +[bench] sweep 1/1 episode 2/100 done +[bench] sweep 1/1 episode 3/100 done +[bench] sweep 1/1 episode 4/100 done +[bench] sweep 1/1 episode 5/100 done +[bench] sweep 1/1 episode 6/100 done +[bench] sweep 1/1 episode 7/100 done +[bench] sweep 1/1 episode 8/100 done +[bench] sweep 1/1 episode 9/100 done +[bench] sweep 1/1 episode 10/100 done +[bench] sweep 1/1 episode 11/100 done +[bench] sweep 1/1 episode 12/100 done +[bench] sweep 1/1 episode 13/100 done +[bench] sweep 1/1 episode 14/100 done +[bench] sweep 1/1 episode 15/100 done +[bench] sweep 1/1 episode 16/100 done +[bench] sweep 1/1 episode 17/100 done +[bench] sweep 1/1 episode 18/100 done +[bench] sweep 1/1 episode 19/100 done +[bench] sweep 1/1 episode 20/100 done +[bench] sweep 1/1 episode 21/100 done +[bench] sweep 1/1 episode 22/100 done +[bench] sweep 1/1 episode 23/100 done +[bench] sweep 1/1 episode 24/100 done +[bench] sweep 1/1 episode 25/100 done +[bench] sweep 1/1 episode 26/100 done +[bench] sweep 1/1 episode 27/100 done +[bench] sweep 1/1 episode 28/100 done +[bench] sweep 1/1 episode 29/100 done +[bench] sweep 1/1 episode 30/100 done +[bench] sweep 1/1 episode 31/100 done +[bench] sweep 1/1 episode 32/100 done +[bench] sweep 1/1 episode 33/100 done +[bench] sweep 1/1 episode 34/100 done +[bench] sweep 1/1 episode 35/100 done +[bench] sweep 1/1 episode 36/100 done +[bench] sweep 1/1 episode 37/100 done +[bench] sweep 1/1 episode 38/100 done +[bench] sweep 1/1 episode 39/100 done +[bench] sweep 1/1 episode 40/100 done +[bench] sweep 1/1 episode 41/100 done +[bench] sweep 1/1 episode 42/100 done +[bench] sweep 1/1 episode 43/100 done +[bench] sweep 1/1 episode 44/100 done +[bench] sweep 1/1 episode 45/100 done +[bench] sweep 1/1 episode 46/100 done +[bench] sweep 1/1 episode 47/100 done +[bench] sweep 1/1 episode 48/100 done +[bench] sweep 1/1 episode 49/100 done +[bench] sweep 1/1 episode 50/100 done +[bench] sweep 1/1 episode 51/100 done +[bench] sweep 1/1 episode 52/100 done +[bench] sweep 1/1 episode 53/100 done +[bench] sweep 1/1 episode 54/100 done +[bench] sweep 1/1 episode 55/100 done +[bench] sweep 1/1 episode 56/100 done +[bench] sweep 1/1 episode 57/100 done +[bench] sweep 1/1 episode 58/100 done +[bench] sweep 1/1 episode 59/100 done +[bench] sweep 1/1 episode 60/100 done +[bench] sweep 1/1 episode 61/100 done +[bench] sweep 1/1 episode 62/100 done +[bench] sweep 1/1 episode 63/100 done +[bench] sweep 1/1 episode 64/100 done +[bench] sweep 1/1 episode 65/100 done +[bench] sweep 1/1 episode 66/100 done +[bench] sweep 1/1 episode 67/100 done +[bench] sweep 1/1 episode 68/100 done +[bench] sweep 1/1 episode 69/100 done +[bench] sweep 1/1 episode 70/100 done +[bench] sweep 1/1 episode 71/100 done +[bench] sweep 1/1 episode 72/100 done +[bench] sweep 1/1 episode 73/100 done +[bench] sweep 1/1 episode 74/100 done +[bench] sweep 1/1 episode 75/100 done +[bench] sweep 1/1 episode 76/100 done +[bench] sweep 1/1 episode 77/100 done +[bench] sweep 1/1 episode 78/100 done +[bench] sweep 1/1 episode 79/100 done +[bench] sweep 1/1 episode 80/100 done +[bench] sweep 1/1 episode 81/100 done +[bench] sweep 1/1 episode 82/100 done +[bench] sweep 1/1 episode 83/100 done +[bench] sweep 1/1 episode 84/100 done +[bench] sweep 1/1 episode 85/100 done +[bench] sweep 1/1 episode 86/100 done +[bench] sweep 1/1 episode 87/100 done +[bench] sweep 1/1 episode 88/100 done +[bench] sweep 1/1 episode 89/100 done +[bench] sweep 1/1 episode 90/100 done +[bench] sweep 1/1 episode 91/100 done +[bench] sweep 1/1 episode 92/100 done +[bench] sweep 1/1 episode 93/100 done +[bench] sweep 1/1 episode 94/100 done +[bench] sweep 1/1 episode 95/100 done +[bench] sweep 1/1 episode 96/100 done +[bench] sweep 1/1 episode 97/100 done +[bench] sweep 1/1 episode 98/100 done +[bench] sweep 1/1 episode 99/100 done +[bench] sweep 1/1 episode 100/100 done +[bench] sweep 1/1 complete elapsed=2511.2s +episode=0 return=444.600 steps=3600 mean_latency_ms=76.02271694866694 +episode=1 return=444.600 steps=3600 mean_latency_ms=76.14445348705047 +episode=2 return=444.600 steps=3600 mean_latency_ms=75.83047266244563 +episode=3 return=444.600 steps=3600 mean_latency_ms=76.04121221698036 +episode=4 return=444.600 steps=3600 mean_latency_ms=75.7789115791707 +episode=5 return=228.200 steps=1861 mean_latency_ms=76.22757676162651 +episode=6 return=444.600 steps=3600 mean_latency_ms=75.98373978309758 +episode=7 return=444.600 steps=3600 mean_latency_ms=75.85552109823348 +episode=8 return=444.600 steps=3600 mean_latency_ms=75.9782303085917 +episode=9 return=444.600 steps=3600 mean_latency_ms=75.94667987356688 +episode=10 return=444.600 steps=3600 mean_latency_ms=75.66396359484234 +episode=11 return=444.600 steps=3600 mean_latency_ms=75.7794525026407 +episode=12 return=444.600 steps=3600 mean_latency_ms=75.90110110734818 +episode=13 return=444.600 steps=3600 mean_latency_ms=76.01870178237883 +episode=14 return=444.600 steps=3600 mean_latency_ms=75.75567207010911 +episode=15 return=444.600 steps=3600 mean_latency_ms=75.83026036637241 +episode=16 return=444.600 steps=3600 mean_latency_ms=75.74502908171665 +episode=17 return=444.600 steps=3600 mean_latency_ms=75.84316844302293 +episode=18 return=444.600 steps=3600 mean_latency_ms=75.85876738771161 +episode=19 return=265.500 steps=2162 mean_latency_ms=75.89492798135642 +episode=20 return=444.600 steps=3600 mean_latency_ms=75.90859756288593 +episode=21 return=444.600 steps=3600 mean_latency_ms=75.93474621914784 +episode=22 return=444.600 steps=3600 mean_latency_ms=75.77022360156529 +episode=23 return=444.600 steps=3600 mean_latency_ms=75.8506098974935 +episode=24 return=444.600 steps=3600 mean_latency_ms=75.80511776716725 +episode=25 return=116.000 steps=955 mean_latency_ms=76.07937915327228 +episode=26 return=444.600 steps=3600 mean_latency_ms=75.77409482659607 +episode=27 return=444.600 steps=3600 mean_latency_ms=75.82354466933252 +episode=28 return=444.600 steps=3600 mean_latency_ms=75.92578714415393 +episode=29 return=444.600 steps=3600 mean_latency_ms=75.77326038618416 +episode=30 return=256.000 steps=2085 mean_latency_ms=75.8461606092662 +episode=31 return=444.600 steps=3600 mean_latency_ms=75.87053786258159 +episode=32 return=444.600 steps=3600 mean_latency_ms=75.90930861144982 +episode=33 return=444.600 steps=3600 mean_latency_ms=75.80530422686525 +episode=34 return=444.600 steps=3600 mean_latency_ms=76.05997569829616 +episode=35 return=444.600 steps=3600 mean_latency_ms=75.67579907153437 +episode=36 return=444.600 steps=3600 mean_latency_ms=76.07561842170198 +episode=37 return=444.600 steps=3600 mean_latency_ms=75.87459102177027 +episode=38 return=55.600 steps=468 mean_latency_ms=75.8887188983619 +episode=39 return=444.600 steps=3600 mean_latency_ms=75.86536772802552 +episode=40 return=432.900 steps=3512 mean_latency_ms=76.00355652525975 +episode=41 return=274.800 steps=2237 mean_latency_ms=75.7658282850597 +episode=42 return=264.900 steps=2156 mean_latency_ms=75.95264956954799 +episode=43 return=265.400 steps=2161 mean_latency_ms=75.82748305801191 +episode=44 return=444.600 steps=3600 mean_latency_ms=75.9295822845668 +episode=45 return=143.900 steps=1180 mean_latency_ms=75.94310218110371 +episode=46 return=444.600 steps=3600 mean_latency_ms=75.69568531179425 +episode=47 return=93.100 steps=771 mean_latency_ms=76.0527875505066 +episode=48 return=56.100 steps=473 mean_latency_ms=76.20882901957174 +episode=49 return=265.200 steps=2159 mean_latency_ms=76.05401077635972 +episode=50 return=444.600 steps=3600 mean_latency_ms=75.89333271844873 +episode=51 return=444.600 steps=3600 mean_latency_ms=75.89090159365671 +episode=52 return=398.800 steps=3234 mean_latency_ms=75.91218218803246 +episode=53 return=444.600 steps=3600 mean_latency_ms=75.86400590251726 +episode=54 return=270.200 steps=2200 mean_latency_ms=76.01590238337654 +episode=55 return=69.700 steps=582 mean_latency_ms=75.68422480575155 +episode=56 return=444.600 steps=3600 mean_latency_ms=75.87884524455251 +episode=57 return=444.600 steps=3600 mean_latency_ms=75.96981187494319 +episode=58 return=444.600 steps=3600 mean_latency_ms=76.03772455115222 +episode=59 return=437.900 steps=3553 mean_latency_ms=76.04088529786887 +episode=60 return=348.900 steps=2834 mean_latency_ms=75.79422825165413 +episode=61 return=444.600 steps=3600 mean_latency_ms=75.87728099437057 +episode=62 return=78.900 steps=656 mean_latency_ms=75.96352981662133 +episode=63 return=444.600 steps=3600 mean_latency_ms=75.80269270184165 +episode=64 return=444.600 steps=3600 mean_latency_ms=75.88518180564401 +episode=65 return=444.600 steps=3600 mean_latency_ms=75.87533034544981 +episode=66 return=444.600 steps=3600 mean_latency_ms=75.94241138050401 +episode=67 return=444.600 steps=3600 mean_latency_ms=75.95312277771471 +episode=68 return=444.600 steps=3600 mean_latency_ms=75.8998829764233 +episode=69 return=444.600 steps=3600 mean_latency_ms=75.98564617573034 +episode=70 return=444.600 steps=3600 mean_latency_ms=75.68328575087021 +episode=71 return=135.000 steps=1109 mean_latency_ms=75.99546963217229 +episode=72 return=444.600 steps=3600 mean_latency_ms=75.9923106611263 +episode=73 return=444.600 steps=3600 mean_latency_ms=75.80422251719546 +episode=74 return=444.600 steps=3600 mean_latency_ms=75.95469853250815 +episode=75 return=444.600 steps=3600 mean_latency_ms=75.74551875442629 +episode=76 return=444.600 steps=3600 mean_latency_ms=75.93301571087362 +episode=77 return=444.600 steps=3600 mean_latency_ms=75.98384926019328 +episode=78 return=444.600 steps=3600 mean_latency_ms=75.85055115368883 +episode=79 return=444.600 steps=3600 mean_latency_ms=75.97142616222317 +episode=80 return=444.600 steps=3600 mean_latency_ms=75.97039764106849 +episode=81 return=444.600 steps=3600 mean_latency_ms=75.74469321422862 +episode=82 return=116.200 steps=957 mean_latency_ms=76.0366526049804 +episode=83 return=444.600 steps=3600 mean_latency_ms=75.95924386190674 +episode=84 return=444.600 steps=3600 mean_latency_ms=76.0310580385874 +episode=85 return=36.600 steps=314 mean_latency_ms=75.8634823847272 +episode=86 return=260.900 steps=2125 mean_latency_ms=75.91783880059099 +episode=87 return=444.600 steps=3600 mean_latency_ms=76.16752514785735 +episode=88 return=444.600 steps=3600 mean_latency_ms=75.83331254385584 +episode=89 return=444.600 steps=3600 mean_latency_ms=76.2113387300584 +episode=90 return=444.600 steps=3600 mean_latency_ms=75.8334932097261 +episode=91 return=225.200 steps=1831 mean_latency_ms=75.75618859671614 +episode=92 return=444.600 steps=3600 mean_latency_ms=75.79683788505955 +episode=93 return=180.900 steps=1478 mean_latency_ms=75.96465307644473 +episode=94 return=305.200 steps=2478 mean_latency_ms=75.86565754734926 +episode=95 return=444.600 steps=3600 mean_latency_ms=75.74523644464854 +episode=96 return=444.600 steps=3600 mean_latency_ms=75.94353591524424 +episode=97 return=444.600 steps=3600 mean_latency_ms=75.81927739599219 +episode=98 return=444.600 steps=3600 mean_latency_ms=75.96229410618645 +episode=99 return=444.600 steps=3600 mean_latency_ms=75.94512877548694 +summary latency=profile_sample episodes=100 mean_return=384.824 std_return=116.788 mean_length=3119.3 +/home/ubuntu/lzj/conda/envs/qwenoft/lib/python3.10/multiprocessing/resource_tracker.py:224: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown + warnings.warn('resource_tracker: There appear to be %d ' diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/REPORT.md b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/REPORT.md new file mode 100644 index 0000000000000000000000000000000000000000..7f2af02a5f64a7ae2fdb82b2a93c6eaac45fb57d --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/REPORT.md @@ -0,0 +1,20 @@ +# QwenOFT mean-trained checkpoints under profile simulation + +Four final step-5000 H1 checkpoints; two rounds, one evaluation per physical GPU2/3,100 episodes each (400 total). + +The training latency was fixed mean; this evaluation samples the complete archived RTX3090 temporal hidden-regime profile. Simulator FPS, seeds, horizon, limits and model/profile identities are in evaluation-plan.json. Standard deviations below use ddof=0. Returns have task-specific scales. Startup checks are separate and excluded. + +| Task | Episodes | Return mean +/- SD | Length mean +/- SD | Success | Invalid | +|---|---:|---:|---:|---:|---:| +| flappy | 100 | 384.824005 +/- 116.787774 | 3119.31 +/- 939.87 | not provided by task | 0 | +| deadly_corridor | 100 | 1620.798776 +/- 913.624278 | 148.53 +/- 49.46 | not provided by task | 0 | +| ant | 100 | 1453.844064 +/- 693.727520 | 803.85 +/- 328.81 | not provided by task | 0 | +| intercept | 100 | 3.544349 +/- 7.071923 | 60.00 +/- 0.00 | 9/100 | 0 | + +No success metric is invented for Flappy/Deadly/Ant. Intercept reports the native accumulated success flag. No policy-quality acceptance gate is claimed. + +Compatibility repairs: portable robot_type copied from each actual training manifest (weights unchanged); official ViZDoom1.2.4 VizdoomCorridor-v0 uses the same deadly_corridor WAD as SF, preserves render contract and semantic seven-button ordering; public action space is equivalent MultiBinary7. Existing native render/button/history tests passed. Full eval source/patch and original profile assets are archived. + +Flappy/Deadly seeds1000000..1000099; Ant42..141; Intercept4242424242..4242424341. Latency seed271828. Flappy10/10Hz, Deadly35/8.75Hz, Ant10/10Hz, Intercept20/20Hz. Max raw frames3600/3600/1000/60; capacities1. MIKASA H1 holds last chunk action; no prefix, no DAgger. Ant keeps its training prompt label1 while execution latency is sampled. + +Raw JSONL logs are losslessly gzip-compressed for distribution; original uncompressed records remain on the experiment host. Empty observation_attempts files are retained; admission/drop evidence is in steps/actions. Per-task CSV and full400 episode CSV are provided. diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/all_episodes.csv b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/all_episodes.csv new file mode 100644 index 0000000000000000000000000000000000000000..be5e27101cb97f6a2dd0d85a0399d3fb8a5ea839 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/all_episodes.csv @@ -0,0 +1,401 @@ +task,episode_id,seed,return_env,length,mean_latency_ms,success +flappy,0,1000000,444.6000052243471,3600,76.02271694866694, +flappy,1,1000001,444.6000052243471,3600,76.14445348705047, +flappy,2,1000002,444.6000052243471,3600,75.83047266244563, +flappy,3,1000003,444.6000052243471,3600,76.04121221698036, +flappy,4,1000004,444.6000052243471,3600,75.7789115791707, +flappy,5,1000005,228.2000027000904,1861,76.22757676162651, +flappy,6,1000006,444.6000052243471,3600,75.98373978309758, +flappy,7,1000007,444.6000052243471,3600,75.85552109823348, +flappy,8,1000008,444.6000052243471,3600,75.9782303085917, +flappy,9,1000009,444.6000052243471,3600,75.94667987356688, +flappy,10,1000010,444.6000052243471,3600,75.66396359484234, +flappy,11,1000011,444.6000052243471,3600,75.7794525026407, +flappy,12,1000012,444.6000052243471,3600,75.90110110734818, +flappy,13,1000013,444.6000052243471,3600,76.01870178237883, +flappy,14,1000014,444.6000052243471,3600,75.75567207010911, +flappy,15,1000015,444.6000052243471,3600,75.83026036637241, +flappy,16,1000016,444.6000052243471,3600,75.74502908171665, +flappy,17,1000017,444.6000052243471,3600,75.84316844302293, +flappy,18,1000018,444.6000052243471,3600,75.85876738771161, +flappy,19,1000019,265.50000313669443,2162,75.89492798135642, +flappy,20,1000020,444.6000052243471,3600,75.90859756288593, +flappy,21,1000021,444.6000052243471,3600,75.93474621914784, +flappy,22,1000022,444.6000052243471,3600,75.77022360156529, +flappy,23,1000023,444.6000052243471,3600,75.8506098974935, +flappy,24,1000024,444.6000052243471,3600,75.80511776716725, +flappy,25,1000025,116.00000138580799,955,76.07937915327228, +flappy,26,1000026,444.6000052243471,3600,75.77409482659607, +flappy,27,1000027,444.6000052243471,3600,75.82354466933252, +flappy,28,1000028,444.6000052243471,3600,75.92578714415393, +flappy,29,1000029,444.6000052243471,3600,75.77326038618416, +flappy,30,1000030,256.0000030249357,2085,75.8461606092662, +flappy,31,1000031,444.6000052243471,3600,75.87053786258159, +flappy,32,1000032,444.6000052243471,3600,75.90930861144982, +flappy,33,1000033,444.6000052243471,3600,75.80530422686525, +flappy,34,1000034,444.6000052243471,3600,76.05997569829616, +flappy,35,1000035,444.6000052243471,3600,75.67579907153437, +flappy,36,1000036,444.6000052243471,3600,76.07561842170198, +flappy,37,1000037,444.6000052243471,3600,75.87459102177027, +flappy,38,1000038,55.60000067949295,468,75.8887188983619, +flappy,39,1000039,444.6000052243471,3600,75.86536772802552, +flappy,40,1000040,432.900005094707,3512,76.00355652525975, +flappy,41,1000041,274.8000032454729,2237,75.7658282850597, +flappy,42,1000042,264.90000312775373,2156,75.95264956954799, +flappy,43,1000043,265.4000031352043,2161,75.82748305801191, +flappy,44,1000044,444.6000052243471,3600,75.9295822845668, +flappy,45,1000045,143.90000171214342,1180,75.94310218110371, +flappy,46,1000046,444.6000052243471,3600,75.69568531179425, +flappy,47,1000047,93.1000011190772,771,76.0527875505066, +flappy,48,1000048,56.10000068694353,473,76.20882901957174, +flappy,49,1000049,265.2000031322241,2159,76.05401077635972, +flappy,50,1000050,444.6000052243471,3600,75.89333271844873, +flappy,51,1000051,444.6000052243471,3600,75.89090159365671, +flappy,52,1000052,398.80000469088554,3234,75.91218218803246, +flappy,53,1000053,444.6000052243471,3600,75.86400590251726, +flappy,54,1000054,270.2000031918287,2200,76.01590238337654, +flappy,55,1000055,69.70000084489584,582,75.68422480575155, +flappy,56,1000056,444.6000052243471,3600,75.87884524455251, +flappy,57,1000057,444.6000052243471,3600,75.96981187494319, +flappy,58,1000058,444.6000052243471,3600,76.03772455115222, +flappy,59,1000059,437.90000515431166,3553,76.04088529786887, +flappy,60,1000060,348.9000041112304,2834,75.79422825165413, +flappy,61,1000061,444.6000052243471,3600,75.87728099437057, +flappy,62,1000062,78.9000009521842,656,75.96352981662133, +flappy,63,1000063,444.6000052243471,3600,75.80269270184165, +flappy,64,1000064,444.6000052243471,3600,75.88518180564401, +flappy,65,1000065,444.6000052243471,3600,75.87533034544981, +flappy,66,1000066,444.6000052243471,3600,75.94241138050401, +flappy,67,1000067,444.6000052243471,3600,75.95312277771471, +flappy,68,1000068,444.6000052243471,3600,75.8998829764233, +flappy,69,1000069,444.6000052243471,3600,75.98564617573034, +flappy,70,1000070,444.6000052243471,3600,75.68328575087021, +flappy,71,1000071,135.0000016093254,1109,75.99546963217229, +flappy,72,1000072,444.6000052243471,3600,75.9923106611263, +flappy,73,1000073,444.6000052243471,3600,75.80422251719546, +flappy,74,1000074,444.6000052243471,3600,75.95469853250815, +flappy,75,1000075,444.6000052243471,3600,75.74551875442629, +flappy,76,1000076,444.6000052243471,3600,75.93301571087362, +flappy,77,1000077,444.6000052243471,3600,75.98384926019328, +flappy,78,1000078,444.6000052243471,3600,75.85055115368883, +flappy,79,1000079,444.6000052243471,3600,75.97142616222317, +flappy,80,1000080,444.6000052243471,3600,75.97039764106849, +flappy,81,1000081,444.6000052243471,3600,75.74469321422862, +flappy,82,1000082,116.20000138878822,957,76.0366526049804, +flappy,83,1000083,444.6000052243471,3600,75.95924386190674, +flappy,84,1000084,444.6000052243471,3600,76.0310580385874, +flappy,85,1000085,36.60000045597553,314,75.8634823847272, +flappy,86,1000086,260.90000308305025,2125,75.91783880059099, +flappy,87,1000087,444.6000052243471,3600,76.16752514785735, +flappy,88,1000088,444.6000052243471,3600,75.83331254385584, +flappy,89,1000089,444.6000052243471,3600,76.2113387300584, +flappy,90,1000090,444.6000052243471,3600,75.8334932097261, +flappy,91,1000091,225.20000265538692,1831,75.75618859671614, +flappy,92,1000092,444.6000052243471,3600,75.79683788505955, +flappy,93,1000093,180.9000021442771,1478,75.96465307644473, +flappy,94,1000094,305.2000035941601,2478,75.86565754734926, +flappy,95,1000095,444.6000052243471,3600,75.74523644464854, +flappy,96,1000096,444.6000052243471,3600,75.94353591524424, +flappy,97,1000097,444.6000052243471,3600,75.81927739599219, +flappy,98,1000098,444.6000052243471,3600,75.96229410618645, +flappy,99,1000099,444.6000052243471,3600,75.94512877548694, +deadly_corridor,0,1000000,337.47547912597656,72,71.90727374040254, +deadly_corridor,1,1000001,819.0284423828125,143,73.84762082340946, +deadly_corridor,2,1000002,2284.857650756836,182,72.79171012339609, +deadly_corridor,3,1000003,2276.2068634033203,189,76.345275285376, +deadly_corridor,4,1000004,805.2153015136719,150,73.86282581373551, +deadly_corridor,5,1000005,621.8231658935547,115,74.31105893586228, +deadly_corridor,6,1000006,2276.414749145508,176,74.2226331369995, +deadly_corridor,7,1000007,2284.310989379883,176,72.9072057957754, +deadly_corridor,8,1000008,81.07798767089844,49,73.20258272646697, +deadly_corridor,9,1000009,317.2351837158203,75,72.54774919154028, +deadly_corridor,10,1000010,2282.7608489990234,176,72.78293151689127, +deadly_corridor,11,1000011,88.11907958984375,45,72.60486105128022, +deadly_corridor,12,1000012,2281.468536376953,176,72.29193331603048, +deadly_corridor,13,1000013,2276.6868591308594,178,72.73330265771509, +deadly_corridor,14,1000014,2276.1705932617188,178,73.30067987408609, +deadly_corridor,15,1000015,2282.6631622314453,177,72.49405489224537, +deadly_corridor,16,1000016,2280.300033569336,172,72.80884970803692, +deadly_corridor,17,1000017,2280.4182891845703,182,73.03539182090206, +deadly_corridor,18,1000018,2281.2594451904297,177,72.50972089313564, +deadly_corridor,19,1000019,479.8523712158203,99,72.58046231642126, +deadly_corridor,20,1000020,2279.7379455566406,181,72.47468246266928, +deadly_corridor,21,1000021,2284.9097442626953,197,83.18983231769475, +deadly_corridor,22,1000022,2286.2730407714844,172,72.84281562147524, +deadly_corridor,23,1000023,244.51919555664062,74,76.23461799191558, +deadly_corridor,24,1000024,2279.957275390625,195,72.94927214021655, +deadly_corridor,25,1000025,2283.952178955078,179,73.18968843008061, +deadly_corridor,26,1000026,2276.701370239258,178,72.87702909462648, +deadly_corridor,27,1000027,2277.142562866211,190,72.45412386128042, +deadly_corridor,28,1000028,2279.025634765625,177,74.11102172804317, +deadly_corridor,29,1000029,2285.7152099609375,177,71.63189230597281, +deadly_corridor,30,1000030,53.374298095703125,44,72.51418721312025, +deadly_corridor,31,1000031,2279.8080444335938,183,72.72702656843174, +deadly_corridor,32,1000032,2282.307357788086,178,74.33584751930213, +deadly_corridor,33,1000033,2282.834014892578,192,73.95005063555192, +deadly_corridor,34,1000034,2284.200241088867,188,76.29368894499888, +deadly_corridor,35,1000035,2287.2159118652344,179,72.81890806090988, +deadly_corridor,36,1000036,2284.693832397461,183,76.28284599973325, +deadly_corridor,37,1000037,2283.2066650390625,178,72.1797344044525, +deadly_corridor,38,1000038,2281.032196044922,178,73.74343783824916, +deadly_corridor,39,1000039,2282.960678100586,190,73.24816830891406, +deadly_corridor,40,1000040,2287.094253540039,185,72.35711232966574, +deadly_corridor,41,1000041,2279.3030853271484,179,72.42125368367608, +deadly_corridor,42,1000042,440.0892791748047,104,73.92064892672727, +deadly_corridor,43,1000043,2280.8592529296875,177,72.36020918178356, +deadly_corridor,44,1000044,2283.4308471679688,189,75.93658060557208, +deadly_corridor,45,1000045,2282.324264526367,181,73.54224681770178, +deadly_corridor,46,1000046,326.0184631347656,74,73.1983876441008, +deadly_corridor,47,1000047,2279.086135864258,182,73.00958120503027, +deadly_corridor,48,1000048,2280.3804626464844,179,73.17268244992928, +deadly_corridor,49,1000049,2276.215301513672,189,75.47590644230628, +deadly_corridor,50,1000050,2278.132034301758,182,74.50495464842548, +deadly_corridor,51,1000051,2285.6056518554688,181,73.41699294418743, +deadly_corridor,52,1000052,2287.240921020508,173,73.22110809114655, +deadly_corridor,53,1000053,310.81517028808594,73,74.09003681120738, +deadly_corridor,54,1000054,2276.6219787597656,175,72.98605010243534, +deadly_corridor,55,1000055,2276.2769470214844,194,75.17704077845171, +deadly_corridor,56,1000056,2278.861602783203,178,72.97353037051572, +deadly_corridor,57,1000057,2279.728561401367,181,73.96913002154926, +deadly_corridor,58,1000058,2280.544464111328,176,73.02432805290651, +deadly_corridor,59,1000059,487.829833984375,108,78.89398217393664, +deadly_corridor,60,1000060,567.0655517578125,113,72.64874721482185, +deadly_corridor,61,1000061,2278.210220336914,177,72.96068484971086, +deadly_corridor,62,1000062,2281.436721801758,186,75.46710866924751, +deadly_corridor,63,1000063,382.2119903564453,89,81.21157315209366, +deadly_corridor,64,1000064,246.2946014404297,70,73.9736408486285, +deadly_corridor,65,1000065,285.21240234375,76,73.13661133681993, +deadly_corridor,66,1000066,310.6737365722656,75,73.40468658737086, +deadly_corridor,67,1000067,346.1162872314453,75,72.1929723632303, +deadly_corridor,68,1000068,804.7056121826172,150,73.76397959753224, +deadly_corridor,69,1000069,2285.6442108154297,184,75.13255757158333, +deadly_corridor,70,1000070,730.5995788574219,132,73.25446825350764, +deadly_corridor,71,1000071,86.91796875,47,76.28335745963689, +deadly_corridor,72,1000072,60.30122375488281,44,76.83513093208644, +deadly_corridor,73,1000073,768.6264343261719,141,77.27057350071598, +deadly_corridor,74,1000074,2280.1071166992188,172,74.16699734355548, +deadly_corridor,75,1000075,860.9334106445312,151,73.15118478347584, +deadly_corridor,76,1000076,722.9459228515625,143,75.5655785931314, +deadly_corridor,77,1000077,2276.8687438964844,182,72.95102474014934, +deadly_corridor,78,1000078,368.3357238769531,79,71.51096709276341, +deadly_corridor,79,1000079,-76.45918273925781,17,72.24888432102617, +deadly_corridor,80,1000080,2281.5543823242188,183,73.32589540463356, +deadly_corridor,81,1000081,2281.6688842773438,171,73.10600900440717, +deadly_corridor,82,1000082,2277.5223083496094,178,73.55648700566698, +deadly_corridor,83,1000083,42.30937194824219,41,73.52700344736942, +deadly_corridor,84,1000084,2285.8980407714844,176,71.98655161011203, +deadly_corridor,85,1000085,68.90191650390625,45,72.84773487604696, +deadly_corridor,86,1000086,2286.2190551757812,171,72.82303966497733, +deadly_corridor,87,1000087,281.1173553466797,76,72.26983276661764, +deadly_corridor,88,1000088,2283.1607971191406,175,73.49638264342678, +deadly_corridor,89,1000089,2277.888946533203,177,73.44736473371472, +deadly_corridor,90,1000090,429.36326599121094,93,71.86172378947977, +deadly_corridor,91,1000091,252.0751953125,70,72.26459581736903, +deadly_corridor,92,1000092,2278.306442260742,192,80.97328482778371, +deadly_corridor,93,1000093,2285.236801147461,175,74.02717585214627, +deadly_corridor,94,1000094,857.2727355957031,152,85.59110000526613, +deadly_corridor,95,1000095,2275.9288024902344,199,73.62958803645523, +deadly_corridor,96,1000096,2286.8704833984375,179,72.31519682456816, +deadly_corridor,97,1000097,2278.048355102539,181,73.50330330803081, +deadly_corridor,98,1000098,2277.4480743408203,178,76.78472725777, +deadly_corridor,99,1000099,2276.9671478271484,178,77.9678189026336, +ant,0,42,1846.1103431567394,1000,89.89614608291177, +ant,1,43,2415.720790707953,1000,90.00308114332259, +ant,2,44,457.34421085068755,177,89.83716885697598, +ant,3,45,1421.7952163289683,1000,89.87909631338808, +ant,4,46,2037.7234409469488,937,89.82685347370092, +ant,5,47,2330.630175869275,1000,90.47193606091501, +ant,6,48,1161.643572255748,429,89.84194070141322, +ant,7,49,2351.1524624990343,1000,89.92640891799017, +ant,8,50,513.2964809479813,210,89.88895656571908, +ant,9,51,1126.8652528911032,660,89.91361550654544, +ant,10,52,1693.436933192597,1000,89.84960962337662, +ant,11,53,948.3780972955639,1000,89.94678527711802, +ant,12,54,2322.052445211472,1000,90.11873818885832, +ant,13,55,960.4026770814776,1000,90.93377411320307, +ant,14,56,1464.564005196777,1000,89.80893705661644, +ant,15,57,1110.548792782156,1000,89.99466844889166, +ant,16,58,2246.207900740156,1000,90.1624262080728, +ant,17,59,85.64836938561511,60,89.87005518664785, +ant,18,60,340.54799067574436,143,89.94240076131771, +ant,19,61,2457.088748930458,1000,89.92054036086635, +ant,20,62,2166.2512677098603,1000,89.95452553058773, +ant,21,63,2357.957592244385,1000,89.86780458600198, +ant,22,64,1654.8780938737275,871,90.08433827425095, +ant,23,65,1499.367100151414,1000,89.89663615668341, +ant,24,66,2297.4032619179525,1000,90.09818426014289, +ant,25,67,1253.360764666355,543,89.9390124443734, +ant,26,68,1221.270312709775,1000,89.84986177450952, +ant,27,69,2389.2476464763376,1000,89.95772586857817, +ant,28,70,1682.5290233886233,707,89.76145439054764, +ant,29,71,2474.676425615127,1000,89.82093759631324, +ant,30,72,382.9231146443659,256,90.69916524888657, +ant,31,73,1837.8126619276347,1000,90.03642087221974, +ant,32,74,227.19436616673684,101,89.8553742761573, +ant,33,75,1700.6312067622644,1000,89.80097198453268, +ant,34,76,960.9452812639541,372,89.8420903148968, +ant,35,77,2290.6720141359438,1000,89.91770573449698, +ant,36,78,328.5729178056416,162,90.01187187392946, +ant,37,79,1180.073938772476,1000,89.81938304804656, +ant,38,80,817.4190215442345,363,89.85140773938038, +ant,39,81,1651.2255208727013,1000,91.17610023451576, +ant,40,82,1428.174672693164,1000,89.8551155619885, +ant,41,83,1627.3838925098842,1000,90.55986754698809, +ant,42,84,1079.756369746183,680,90.17098553312343, +ant,43,85,2173.9447393037276,1000,89.84319301261918, +ant,44,86,409.90633829945847,160,89.66802828269809, +ant,45,87,2467.2636019929073,1000,89.90844708827387, +ant,46,88,657.4084558813478,248,89.86487149424892, +ant,47,89,974.7436031610902,1000,89.76305094278182, +ant,48,90,1510.5184342975385,1000,90.24355118464125, +ant,49,91,602.2339441184535,260,89.7103209703719, +ant,50,92,760.9784375126189,316,89.8206829517188, +ant,51,93,1941.172113330597,1000,90.30785204408768, +ant,52,94,624.3590446196446,281,89.94582387208622, +ant,53,95,2163.4347041279893,1000,89.84848132390947, +ant,54,96,1126.9957963444238,1000,89.84637728060243, +ant,55,97,1405.131632695366,1000,90.18855922596491, +ant,56,98,1206.2916757636292,1000,89.78065539051504, +ant,57,99,2392.7980761515178,1000,89.76388668266138, +ant,58,100,964.0216541467705,1000,89.82618651237911, +ant,59,101,2252.192880003706,1000,89.8197082349776, +ant,60,102,2471.9158497657563,1000,89.96642568195992, +ant,61,103,1902.8491241623092,1000,89.87542708971246, +ant,62,104,1435.6661382989703,1000,90.28644124851098, +ant,63,105,1668.3237703695809,1000,89.86433221097877, +ant,64,106,1813.291243529155,1000,89.85118001877315, +ant,65,107,446.72353548541076,189,89.8309544306309, +ant,66,108,130.84194814079504,74,89.82416773165995, +ant,67,109,2315.857153770824,1000,90.25959750757508, +ant,68,110,288.3915792961347,116,90.09271984792927, +ant,69,111,894.0228631227924,1000,89.89663691508213, +ant,70,112,2030.322535823717,1000,89.84028619017428, +ant,71,113,507.9449555916754,215,90.57267432538549, +ant,72,114,2377.7373967468293,1000,89.84919425782105, +ant,73,115,897.3077114027096,1000,89.90431472264346, +ant,74,116,1454.612590266188,1000,91.19515970740413, +ant,75,117,2292.457960175467,1000,89.8333901030839, +ant,76,118,1424.378337790017,1000,89.88029014661089, +ant,77,119,1441.1111023164538,1000,89.79844243631413, +ant,78,120,1265.4771503717611,1000,89.86009503143968, +ant,79,121,1662.8808067819505,1000,90.5661722205243, +ant,80,122,2508.917122342891,1000,89.87403626041336, +ant,81,123,1655.3510139158748,1000,90.05039760075688, +ant,82,124,1387.3843721247736,821,90.10235730111886, +ant,83,125,646.4356689469432,271,89.7559653760994, +ant,84,126,2172.801064037805,1000,89.84602989356796, +ant,85,127,165.9213897970373,72,89.66756877688618, +ant,86,128,1063.1483912161111,1000,89.77409215132576, +ant,87,129,1000.135342286622,1000,89.81900933661238, +ant,88,130,1977.2359176146426,1000,89.77653862908736, +ant,89,131,1937.1674235355138,1000,90.12773943823525, +ant,90,132,1344.7729257831547,1000,90.39620143170467, +ant,91,133,786.3379828975102,441,89.82893206036925, +ant,92,134,1391.060299752017,1000,89.86676880070257, +ant,93,135,503.300235688713,250,89.94819176115624, +ant,94,136,2446.7482357041768,1000,89.84848658183878, +ant,95,137,1171.9102336514923,1000,89.92785850220504, +ant,96,138,2356.7311711183065,1000,90.42933754946152, +ant,97,139,2356.12199478712,1000,89.82049779117614, +ant,98,140,1389.2987977192308,1000,90.82209581044775, +ant,99,141,967.2335383727841,1000,89.80016695371027, +intercept,0,4242424242,0.7267571190313902,60,99.89614420497905,0.0 +intercept,1,4242424243,2.9096362272975966,60,99.91707940536706,0.0 +intercept,2,4242424244,3.2060351513209753,60,97.78809018716221,0.0 +intercept,3,4242424245,0.7574528902187012,60,98.54894447730877,0.0 +intercept,4,4242424246,0.6827895979695313,60,99.11745353519741,0.0 +intercept,5,4242424247,29.923812823486514,60,99.04064156549293,1.0 +intercept,6,4242424248,0.7661087726592086,60,98.25342313549518,0.0 +intercept,7,4242424249,0.8284444468154106,60,97.98431264506286,0.0 +intercept,8,4242424250,0.9707721562881488,60,99.10671115977826,0.0 +intercept,9,4242424251,1.0944434545235708,60,99.10142489904808,0.0 +intercept,10,4242424252,0.7526731102407211,60,98.24263629181895,0.0 +intercept,11,4242424253,1.0327306617691647,60,99.90610126116793,0.0 +intercept,12,4242424254,24.08529434411321,60,99.04964452767656,1.0 +intercept,13,4242424255,0.8258126199943945,60,99.06333184347895,0.0 +intercept,14,4242424256,0.6465023508935701,60,99.96017435988418,0.0 +intercept,15,4242424257,1.2059930491086561,60,99.07293754243183,0.0 +intercept,16,4242424258,0.8975468523567542,60,99.10975490804557,0.0 +intercept,17,4242424259,0.638558203499997,60,99.9008234011206,0.0 +intercept,18,4242424260,2.473904824233614,60,99.1007534285042,0.0 +intercept,19,4242424261,0.8594156300532632,60,98.2804424689215,0.0 +intercept,20,4242424262,0.7127419076277874,60,97.42140552034121,0.0 +intercept,21,4242424263,1.1195833964738995,60,99.05356389575846,0.0 +intercept,22,4242424264,1.4589147588121705,60,99.10476263429966,0.0 +intercept,23,4242424265,22.348254217096837,60,98.14243140713285,1.0 +intercept,24,4242424266,27.43761277961312,60,99.03063235183511,1.0 +intercept,25,4242424267,0.797963114338927,60,98.95851806063928,0.0 +intercept,26,4242424268,0.6415987604705151,60,99.1202532952496,0.0 +intercept,27,4242424269,1.502438226743834,60,99.8369766656745,0.0 +intercept,28,4242424270,1.277322537265718,60,99.13638822823135,0.0 +intercept,29,4242424271,0.6413188653605175,60,99.8165233572777,0.0 +intercept,30,4242424272,26.015227647672873,60,99.93359984997578,1.0 +intercept,31,4242424273,0.7568511647114065,60,98.29798580223347,0.0 +intercept,32,4242424274,0.7758818510046694,60,95.9119617819155,0.0 +intercept,33,4242424275,0.743274000211386,60,99.16579733811342,0.0 +intercept,34,4242424276,0.9812663898337632,60,99.96913332715677,0.0 +intercept,35,4242424277,0.7364500367548317,60,98.4144170848438,0.0 +intercept,36,4242424278,0.7676261149172205,60,99.87913624991887,0.0 +intercept,37,4242424279,2.6105462690466084,60,99.0124647390605,0.0 +intercept,38,4242424280,0.8922563010128215,60,99.49582641131909,0.0 +intercept,39,4242424281,0.7909053032053635,60,99.95776157301488,0.0 +intercept,40,4242424282,27.747763212013524,60,99.89232705853966,1.0 +intercept,41,4242424283,2.830903574009426,60,99.11182141335861,0.0 +intercept,42,4242424284,3.749473527306691,60,99.89401411987875,0.0 +intercept,43,4242424285,3.2371535471174866,60,98.58941395009701,0.0 +intercept,44,4242424286,1.141169616690604,60,98.95023432158384,0.0 +intercept,45,4242424287,1.2504711685760412,60,99.8783128676535,0.0 +intercept,46,4242424288,1.1401455145678483,60,99.09364640302553,0.0 +intercept,47,4242424289,1.1743367564631626,60,98.24703755640672,0.0 +intercept,48,4242424290,0.6911400489043444,60,98.98847807253395,0.0 +intercept,49,4242424291,0.966755291854497,60,98.31297463384391,0.0 +intercept,50,4242424292,3.7725237559643574,60,98.1307167401627,0.0 +intercept,51,4242424293,0.7292428385990206,60,99.08520847604322,0.0 +intercept,52,4242424294,2.733719722367823,60,99.8949988335446,0.0 +intercept,53,4242424295,2.7277548569836654,60,99.16542541107671,0.0 +intercept,54,4242424296,0.8013565168366767,60,98.2801475641182,0.0 +intercept,55,4242424297,0.9918300381395966,60,98.74883429246843,0.0 +intercept,56,4242424298,3.8384227409260347,60,98.30485570834159,0.0 +intercept,57,4242424299,2.525593837024644,60,99.08211861473346,0.0 +intercept,58,4242424300,1.1939986812940333,60,99.95562586586023,0.0 +intercept,59,4242424301,1.1946645161951892,60,99.11887142756973,0.0 +intercept,60,4242424302,0.6632764584392135,60,99.92733404817194,0.0 +intercept,61,4242424303,0.7345126099826302,60,99.61093201950378,0.0 +intercept,62,4242424304,1.1547945403144695,60,98.88382479344455,0.0 +intercept,63,4242424305,1.0395031699445099,60,99.10455669644611,0.0 +intercept,64,4242424306,2.7713681719324086,60,99.99607387713222,0.0 +intercept,65,4242424307,3.8083399715833366,60,99.90908449191997,0.0 +intercept,66,4242424308,3.1245881704380736,60,99.86388390473627,0.0 +intercept,67,4242424309,0.9936205917911138,60,99.06530098425861,0.0 +intercept,68,4242424310,0.6479002644773573,60,97.31661851374615,0.0 +intercept,69,4242424311,1.09404552471824,60,99.06482130667098,0.0 +intercept,70,4242424312,0.725047086874838,60,99.39714496924636,0.0 +intercept,71,4242424313,2.085218493710272,60,99.91926924929075,0.0 +intercept,72,4242424314,25.112157980707707,60,99.89495984140663,1.0 +intercept,73,4242424315,0.7960666966973804,60,99.91315720008677,0.0 +intercept,74,4242424316,1.8899870013119653,60,99.8635479883608,0.0 +intercept,75,4242424317,24.77215793245705,60,99.08978442272605,1.0 +intercept,76,4242424318,0.763190906640375,60,99.1181927131107,0.0 +intercept,77,4242424319,0.8356004936795216,60,98.87472332915829,0.0 +intercept,78,4242424320,24.543561146681895,60,98.85109478338812,1.0 +intercept,79,4242424321,0.7962639288743958,60,96.64267992061197,0.0 +intercept,80,4242424322,0.6807828926102957,60,98.70551839611774,0.0 +intercept,81,4242424323,1.1704122956143692,60,99.1162675413269,0.0 +intercept,82,4242424324,0.8024117537715938,60,99.10984056283594,0.0 +intercept,83,4242424325,1.0154686415335163,60,99.90653765962175,0.0 +intercept,84,4242424326,0.6267238368745893,60,99.14313411902761,0.0 +intercept,85,4242424327,1.1180786813492887,60,99.87378109642233,0.0 +intercept,86,4242424328,1.0531825890648179,60,99.90759275984404,0.0 +intercept,87,4242424329,0.7319892354425974,60,99.07934787032669,0.0 +intercept,88,4242424330,1.1460731038823724,60,98.32584786308246,0.0 +intercept,89,4242424331,1.145515855285339,60,99.853580446333,0.0 +intercept,90,4242424332,3.1438898412743583,60,98.23875463665809,0.0 +intercept,91,4242424333,1.1678254807484336,60,99.94300368083988,0.0 +intercept,92,4242424334,1.1468605129048228,60,98.28856657896678,0.0 +intercept,93,4242424335,2.816772125195712,60,99.10609232867152,0.0 +intercept,94,4242424336,1.1577836629003286,60,99.93574594730708,0.0 +intercept,95,4242424337,1.0533778404060286,60,99.92123883389186,0.0 +intercept,96,4242424338,0.8533297177054919,60,99.06174133027585,0.0 +intercept,97,4242424339,0.7617567333800253,60,99.95316521359209,0.0 +intercept,98,4242424340,0.9895546428160742,60,99.11385213912092,0.0 +intercept,99,4242424341,0.770722996792756,60,99.44169788411487,0.0 diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/comparison.csv b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/comparison.csv new file mode 100644 index 0000000000000000000000000000000000000000..1b381df6789eea28ea56149c4b780cf0cade01c0 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/comparison.csv @@ -0,0 +1,5 @@ +task,episodes,return_mean,return_sd,length_mean,length_sd,success_count,success_rate,invalid_actions,dropped_actions +flappy,100,384.8240045265853,116.78777394316903,3119.31,939.8693174585497,,,0,64 +deadly_corridor,100,1620.7987757873534,913.6242782186637,148.53,49.455930888013825,,,0,0 +ant,100,1453.844063807972,693.7275200567642,803.85,328.8088312378486,,,0,0 +intercept,100,3.5443485127069287,7.07192296411853,60.0,0.0,9,0.09,0,10 diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/comparison.json b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/comparison.json new file mode 100644 index 0000000000000000000000000000000000000000..8486070c582599f0cb0c336c70e6569823f4990f --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/comparison.json @@ -0,0 +1,205 @@ +{ + "condition": "profile-latency", + "executor_mode": "simulated", + "latency_method": "temporal/profile_sample", + "episodes_per_checkpoint": 100, + "total_episodes": 400, + "checkpoints_metadata_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "results": { + "flappy": { + "n_episodes": 100, + "mean_return": 384.8240045265853, + "std_return": 116.78777394316903, + "min_return": 36.60000045597553, + "max_return": 444.6000052243471, + "mean_length": 3119.31, + "std_length": 939.8693174585497, + "min_length": 314.0, + "max_length": 3600.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "flappy", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "flappy", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T122201421825Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "flappy", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42", + "profile_sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 64, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy.yaml", + "execution_audit": { + "issued_action_records": 311075, + "applied_action_records": 310911, + "dropped_action_records": 64, + "nonnoop_issued_records": 30817, + "finite_action_values": true, + "latency_sample_count": 311075, + "latency_mean_ms": 75.89784633675906, + "latency_std_ms": 3.799946378622932, + "latency_p95_ms": 81.3960393048375, + "latency_p99_ms": 87.23844517488543 + } + }, + "deadly_corridor": { + "n_episodes": 100, + "mean_return": 1620.7987757873534, + "std_return": 913.6242782186637, + "min_return": -76.45918273925781, + "max_return": 2287.240921020508, + "mean_length": 148.53, + "std_length": 49.455930888013825, + "min_length": 17.0, + "max_length": 199.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "doom_deadly_corridor", + "model_id": "openvla", + "gpu_class": "1x-rtx3090", + "workload_id": "deadly_corridor", + "instance_id": "instance_a5037b165aa0cedc", + "source_run_id": "20260914T171446047509Z", + "profile_ref": null, + "env_fps": 35.0, + "obs_fps": 8.75, + "frame_ms": 28.571428571428573, + "latency_type": "profile_sample", + "task": "deadly_corridor", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42", + "profile_sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor.yaml", + "execution_audit": { + "issued_action_records": 3753, + "applied_action_records": 3673, + "dropped_action_records": 0, + "nonnoop_issued_records": 3753, + "finite_action_values": true, + "latency_sample_count": 3753, + "latency_mean_ms": 74.01999621872471, + "latency_std_ms": 5.5537519652567635, + "latency_p95_ms": 89.54825982614612, + "latency_p99_ms": 95.97310052501227 + } + }, + "ant": { + "n_episodes": 100, + "mean_return": 1453.844063807972, + "std_return": 693.7275200567642, + "min_return": 85.64836938561511, + "max_return": 2508.917122342891, + "mean_length": 803.85, + "std_length": 328.8088312378486, + "min_length": 60.0, + "max_length": 1000.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "LatencyBench/AntContinuous-v0", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "ant", + "instance_id": "instance_859cf1e47bca6046", + "source_run_id": "20260911T033037730561Z", + "profile_ref": null, + "env_fps": 10.0, + "obs_fps": 10.0, + "frame_ms": 100.0, + "latency_type": "profile_sample", + "task": "ant", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42", + "profile_sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 0, + "unique_seeds": 100, + "physical_gpu": 2, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant.yaml", + "execution_audit": { + "issued_action_records": 79573, + "applied_action_records": 79465, + "dropped_action_records": 0, + "nonnoop_issued_records": 79573, + "finite_action_values": true, + "latency_sample_count": 79573, + "latency_mean_ms": 90.00919554158884, + "latency_std_ms": 2.514492574433973, + "latency_p95_ms": 91.11971585797141, + "latency_p99_ms": 102.67108120995428 + } + }, + "intercept": { + "n_episodes": 100, + "mean_return": 3.5443485127069287, + "std_return": 7.07192296411853, + "min_return": 0.6267238368745893, + "max_return": 29.923812823486514, + "mean_length": 60.0, + "std_length": 0.0, + "min_length": 60.0, + "max_length": 60.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "mikasa_intercept_grab_fast", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "mikasa_intercept_grab_fast", + "instance_id": "instance_3a0d42681a03715c", + "source_run_id": "20260909T044501695676Z", + "profile_ref": null, + "env_fps": 20.0, + "obs_fps": 20.0, + "frame_ms": 50.0, + "latency_type": "profile_sample", + "task": "intercept", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0", + "profile_sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 10, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept.yaml", + "success_count": 9, + "success_rate": 0.09, + "execution_audit": { + "issued_action_records": 2974, + "applied_action_records": 2864, + "dropped_action_records": 10, + "nonnoop_issued_records": 2974, + "finite_action_values": true, + "latency_sample_count": 2974, + "latency_mean_ms": 99.11060319379854, + "latency_std_ms": 4.301543980874005, + "latency_p95_ms": 100.2889407458356, + "latency_p99_ms": 100.64616770379737 + } + } + }, + "quality_acceptance": "not inferred; observed statistics only" +} diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/episodes.csv b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/episodes.csv new file mode 100644 index 0000000000000000000000000000000000000000..f926cea1fa69e001ccb73fd99b25d7efc63f99ba --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/episodes.csv @@ -0,0 +1,101 @@ +episode_id,seed,return_env,length,mean_latency_ms,invalid_actions,dropped_actions +0,4242424242,0.7267571190313902,60,99.89614420497905,0,0 +1,4242424243,2.9096362272975966,60,99.91707940536706,0,0 +2,4242424244,3.2060351513209753,60,97.78809018716221,0,0 +3,4242424245,0.7574528902187012,60,98.54894447730877,0,0 +4,4242424246,0.6827895979695313,60,99.11745353519741,0,0 +5,4242424247,29.923812823486514,60,99.04064156549293,0,0 +6,4242424248,0.7661087726592086,60,98.25342313549518,0,1 +7,4242424249,0.8284444468154106,60,97.98431264506286,0,0 +8,4242424250,0.9707721562881488,60,99.10671115977826,0,0 +9,4242424251,1.0944434545235708,60,99.10142489904808,0,0 +10,4242424252,0.7526731102407211,60,98.24263629181895,0,0 +11,4242424253,1.0327306617691647,60,99.90610126116793,0,0 +12,4242424254,24.08529434411321,60,99.04964452767656,0,0 +13,4242424255,0.8258126199943945,60,99.06333184347895,0,0 +14,4242424256,0.6465023508935701,60,99.96017435988418,0,0 +15,4242424257,1.2059930491086561,60,99.07293754243183,0,0 +16,4242424258,0.8975468523567542,60,99.10975490804557,0,1 +17,4242424259,0.638558203499997,60,99.9008234011206,0,0 +18,4242424260,2.473904824233614,60,99.1007534285042,0,0 +19,4242424261,0.8594156300532632,60,98.2804424689215,0,0 +20,4242424262,0.7127419076277874,60,97.42140552034121,0,0 +21,4242424263,1.1195833964738995,60,99.05356389575846,0,0 +22,4242424264,1.4589147588121705,60,99.10476263429966,0,0 +23,4242424265,22.348254217096837,60,98.14243140713285,0,0 +24,4242424266,27.43761277961312,60,99.03063235183511,0,0 +25,4242424267,0.797963114338927,60,98.95851806063928,0,0 +26,4242424268,0.6415987604705151,60,99.1202532952496,0,0 +27,4242424269,1.502438226743834,60,99.8369766656745,0,0 +28,4242424270,1.277322537265718,60,99.13638822823135,0,1 +29,4242424271,0.6413188653605175,60,99.8165233572777,0,0 +30,4242424272,26.015227647672873,60,99.93359984997578,0,0 +31,4242424273,0.7568511647114065,60,98.29798580223347,0,0 +32,4242424274,0.7758818510046694,60,95.9119617819155,0,0 +33,4242424275,0.743274000211386,60,99.16579733811342,0,0 +34,4242424276,0.9812663898337632,60,99.96913332715677,0,0 +35,4242424277,0.7364500367548317,60,98.4144170848438,0,0 +36,4242424278,0.7676261149172205,60,99.87913624991887,0,0 +37,4242424279,2.6105462690466084,60,99.0124647390605,0,0 +38,4242424280,0.8922563010128215,60,99.49582641131909,0,0 +39,4242424281,0.7909053032053635,60,99.95776157301488,0,0 +40,4242424282,27.747763212013524,60,99.89232705853966,0,0 +41,4242424283,2.830903574009426,60,99.11182141335861,0,0 +42,4242424284,3.749473527306691,60,99.89401411987875,0,1 +43,4242424285,3.2371535471174866,60,98.58941395009701,0,0 +44,4242424286,1.141169616690604,60,98.95023432158384,0,0 +45,4242424287,1.2504711685760412,60,99.8783128676535,0,0 +46,4242424288,1.1401455145678483,60,99.09364640302553,0,0 +47,4242424289,1.1743367564631626,60,98.24703755640672,0,0 +48,4242424290,0.6911400489043444,60,98.98847807253395,0,0 +49,4242424291,0.966755291854497,60,98.31297463384391,0,0 +50,4242424292,3.7725237559643574,60,98.1307167401627,0,0 +51,4242424293,0.7292428385990206,60,99.08520847604322,0,0 +52,4242424294,2.733719722367823,60,99.8949988335446,0,0 +53,4242424295,2.7277548569836654,60,99.16542541107671,0,0 +54,4242424296,0.8013565168366767,60,98.2801475641182,0,0 +55,4242424297,0.9918300381395966,60,98.74883429246843,0,0 +56,4242424298,3.8384227409260347,60,98.30485570834159,0,1 +57,4242424299,2.525593837024644,60,99.08211861473346,0,0 +58,4242424300,1.1939986812940333,60,99.95562586586023,0,1 +59,4242424301,1.1946645161951892,60,99.11887142756973,0,1 +60,4242424302,0.6632764584392135,60,99.92733404817194,0,0 +61,4242424303,0.7345126099826302,60,99.61093201950378,0,0 +62,4242424304,1.1547945403144695,60,98.88382479344455,0,0 +63,4242424305,1.0395031699445099,60,99.10455669644611,0,0 +64,4242424306,2.7713681719324086,60,99.99607387713222,0,0 +65,4242424307,3.8083399715833366,60,99.90908449191997,0,0 +66,4242424308,3.1245881704380736,60,99.86388390473627,0,0 +67,4242424309,0.9936205917911138,60,99.06530098425861,0,1 +68,4242424310,0.6479002644773573,60,97.31661851374615,0,0 +69,4242424311,1.09404552471824,60,99.06482130667098,0,0 +70,4242424312,0.725047086874838,60,99.39714496924636,0,0 +71,4242424313,2.085218493710272,60,99.91926924929075,0,1 +72,4242424314,25.112157980707707,60,99.89495984140663,0,0 +73,4242424315,0.7960666966973804,60,99.91315720008677,0,0 +74,4242424316,1.8899870013119653,60,99.8635479883608,0,0 +75,4242424317,24.77215793245705,60,99.08978442272605,0,0 +76,4242424318,0.763190906640375,60,99.1181927131107,0,0 +77,4242424319,0.8356004936795216,60,98.87472332915829,0,0 +78,4242424320,24.543561146681895,60,98.85109478338812,0,0 +79,4242424321,0.7962639288743958,60,96.64267992061197,0,0 +80,4242424322,0.6807828926102957,60,98.70551839611774,0,0 +81,4242424323,1.1704122956143692,60,99.1162675413269,0,0 +82,4242424324,0.8024117537715938,60,99.10984056283594,0,0 +83,4242424325,1.0154686415335163,60,99.90653765962175,0,0 +84,4242424326,0.6267238368745893,60,99.14313411902761,0,0 +85,4242424327,1.1180786813492887,60,99.87378109642233,0,0 +86,4242424328,1.0531825890648179,60,99.90759275984404,0,0 +87,4242424329,0.7319892354425974,60,99.07934787032669,0,0 +88,4242424330,1.1460731038823724,60,98.32584786308246,0,0 +89,4242424331,1.145515855285339,60,99.853580446333,0,0 +90,4242424332,3.1438898412743583,60,98.23875463665809,0,0 +91,4242424333,1.1678254807484336,60,99.94300368083988,0,0 +92,4242424334,1.1468605129048228,60,98.28856657896678,0,0 +93,4242424335,2.816772125195712,60,99.10609232867152,0,0 +94,4242424336,1.1577836629003286,60,99.93574594730708,0,0 +95,4242424337,1.0533778404060286,60,99.92123883389186,0,0 +96,4242424338,0.8533297177054919,60,99.06174133027585,0,0 +97,4242424339,0.7617567333800253,60,99.95316521359209,0,0 +98,4242424340,0.9895546428160742,60,99.11385213912092,0,1 +99,4242424341,0.770722996792756,60,99.44169788411487,0,0 diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/eval_config.yaml b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/eval_config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b0741bdff2b0925d3c5fa842ffd3e72f6f2750f --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/eval_config.yaml @@ -0,0 +1,68 @@ +experiment: + name: intercept-mean5000-profile-simulation-100ep + seed: 4242424242 +executor: + mode: simulated + simulated_worker_capacity: 1 + simulated_inference_pool: true + inference_devices: + - cuda:0 + inference_batch_size: 32 +env: + name: mikasa_intercept_grab_fast + env_fps: 20 + obs_fps: 20 + frame_stack: 1 + noop_action: + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + base_prompt: Intercept the rolling ball and grasp it to stop it. + simulator_device: gpu +latency: + method: temporal + profile_path: /home/ubuntu/lzj/profiles/intercept-published/profiles/qwenoft/1x-rtx3090/mikasa_intercept_grab_fast/instance_3a0d42681a03715c/profile.json + profile_worker_slot: 0 + seed: 271828 + add_latency_info: false +scheduler: + hold_policy: hold + ordering_policy: latest_ready + hold_last_chunk_action: true +policy: + action_prefix: + mode: none + type: starvla + checkpoint_path: /home/ubuntu/lzj/mean-profiling/intercept/vla-publication/checkpoints/model.pt + model_config_path: /home/ubuntu/lzj/mean-profiling/intercept/vla-publication/config.full.yaml + task_contract_path: /home/ubuntu/lzj/mean-profiling/intercept/vla-publication/task_contract.json + device: cuda:0 + state_info_key: mikasa_proprio + image_views_info_key: mikasa_image_views + backbone_path: /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct + worker_python_executable: /home/ubuntu/lzj/conda/envs/qwenoft/bin/python +evaluation: + eval_episodes: 100 + eval_parallel_envs: 32 + eval_max_steps: 60 + eval_deterministic: true + eval_latency_values: null + eval_raw_reward: true + eval_suites: + fixed: [] + normal: [] + uniform: [] +logging: + output_dir: /home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept + video: + enabled: false + save_step_records: true + save_action_records: true + save_latency_records: true + wandb_project: null + wandb_group: null + wandb_job_type: null diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py new file mode 100644 index 0000000000000000000000000000000000000000..d5edbe901e40e8e9cbb2e0763281fdfcd77dbfcc --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/batched_simulated.py @@ -0,0 +1,702 @@ +from __future__ import annotations + +import time +from collections.abc import Callable, Mapping, Sequence +from dataclasses import dataclass, field +from pathlib import Path + +import numpy as np + +from latency_bench.core.clock import EnvClock +from latency_bench.core.decision_action_history import DecisionActionHistory +from latency_bench.core.timing import StageProfiler, profiler_scope +from latency_bench.core.types import ActionEvent, EpisodeMetrics, LatencyRecord, Observation, StepRecord +from latency_bench.envs.atari import TRUE_EPISODE_END_INFO_KEY +from latency_bench.envs.base import EnvAdapter +from latency_bench.executors._simulated_timeline import ( + SimulatedResultTimeline, + SimulatedWorkerCapacity, + build_simulated_action_event, +) +from latency_bench.executors.base import BatchedExecutor +from latency_bench.executors.env_step_backend import EnvStepBackend, env_action_space +from latency_bench.latency.sample import LatencySample +from latency_bench.latency.samplers import LatencySampler +from latency_bench.logging.metrics import ( + compute_episode_metrics, + compute_episode_metrics_from_aggregates, + episode_raw_fact_metadata, + latency_type_from_source, + profile_metadata_from_source, +) +from latency_bench.logging.records import build_step_record +from latency_bench.logging.trajectory_logger import TrajectoryLogger +from latency_bench.policy.action_prefix import with_action_prefix +from latency_bench.policy.base import PolicyRunner +from latency_bench.scheduler.action_queue import ActionScheduler +from latency_bench.scheduler.decision import DecisionScheduler +from latency_bench.utils.io import write_json +from latency_bench.utils.stats import series_stats + + +@dataclass +class _EpisodeBuffers: + step_records: list[StepRecord] | None = None + action_events: list[ActionEvent] | None = None + latency_records: list[LatencyRecord] | None = None + latency_values_ms: list[float] = field(default_factory=list) + episode_return_env: float = 0.0 + survival_steps: int = 0 + game_score: float | None = None + return_raw: float | None = None + num_actions: int = 0 + num_dropped_actions: int = 0 + num_invalid_actions: int = 0 + submitted_observation_frames: int = 0 + dropped_observation_count: int = 0 + soft_reset_count: int = 0 + final_lives: int | None = None + final_is_true_episode_end: bool | None = None + task_metrics: dict | None = None + task_metric_moments: dict | None = None + + def record_step(self, *, reward: float, info: dict) -> None: + self.episode_return_env += float(reward) + self.survival_steps += 1 + if "invalid_action" in info and info["invalid_action"]: + self.num_invalid_actions += 1 + if "soft_reset" in info and info["soft_reset"]: + self.soft_reset_count += 1 + if "lives" in info: + self.final_lives = info["lives"] + if TRUE_EPISODE_END_INFO_KEY in info: + self.final_is_true_episode_end = info[TRUE_EPISODE_END_INFO_KEY] + if "game_score" in info: + self.game_score = float(info["game_score"]) + if "score" in info: + self.game_score = float(info["score"]) + if "task_metrics" in info: + self.task_metrics = info["task_metrics"] + if "task_metric_moments" in info: + self.task_metric_moments = info["task_metric_moments"] + self._update_return_raw(info) + extra_stats = info["episode_extra_stats"] if "episode_extra_stats" in info else None + if isinstance(extra_stats, dict): + self._update_return_raw(extra_stats) + + def _update_return_raw(self, stats: dict) -> None: + for key in ("return_raw", "raw_return", "episodic_raw_return", "episode/raw_return"): + if key in stats and stats[key] is not None: + self.return_raw = float(stats[key]) + + +@dataclass +class _SlotState: + slot_id: int + env: EnvAdapter + latency_source: LatencySampler + action_scheduler: ActionScheduler + result_timeline: SimulatedResultTimeline + active: bool = False + episode_id: int | None = None + episode_seed: int | None = None + env_step: int = 0 + recent_drop_count: int = 0 + decision_action_history: DecisionActionHistory | None = None + decision_admitted: bool = False + decision_issued_action: object = None + buffers: _EpisodeBuffers = field(default_factory=_EpisodeBuffers) + worker_capacity: SimulatedWorkerCapacity = field( + default_factory=lambda: SimulatedWorkerCapacity(capacity=None, busy_until_by_worker={}) + ) + + +@dataclass +class _PendingPolicyObservation: + slot: _SlotState + observation: Observation + obs_id: int + latency_sample: LatencySample + worker_slot: int + + +class BatchedSimulatedLatencyExecutor(BatchedExecutor): + """Run multiple simulated episodes concurrently with independent slot state. + + The main process owns policy inference, latency scheduling, episode accounting, + and logging. Env stepping can be serial in-process or delegated to worker + subprocesses through env_backend. + """ + + def __init__( + self, + *, + env_backend: EnvStepBackend, + policy: PolicyRunner, + decision_scheduler: DecisionScheduler, + latency_sources: Sequence[LatencySampler], + action_schedulers: Sequence[ActionScheduler], + clock: EnvClock, + logger: TrajectoryLogger | None = None, + episode_latency_source_factory: Callable[[int], LatencySampler] | None = None, + simulated_worker_capacity: int | None = None, + profile_pipeline: bool = False, + inference_pool=None, + action_prefix=None, + action_history_decisions: int | None = None, + ): + slot_count = env_backend.num_slots + self.env_backend = env_backend + self.envs = list(env_backend.slot_handles) + self.policy = policy + self.decision_scheduler = decision_scheduler + self.clock = clock + self.logger = logger + self.profile_pipeline = bool(profile_pipeline) + self.inference_pool = inference_pool + self.action_prefix = action_prefix + self._pipeline_profile_rows: list[dict[str, float]] = [] + self.simulated_worker_capacity = simulated_worker_capacity + self._collect_step_records = bool(logger is not None and logger.save_step_records) + self._collect_action_records = bool(logger is not None and logger.save_action_records) + self._collect_latency_records = bool(logger is not None and logger.save_latency_records) + self.episode_latency_source_factory = episode_latency_source_factory + self.slots = [ + _SlotState( + slot_id=slot_id, + env=self.envs[slot_id], + latency_source=latency_sources[slot_id], + action_scheduler=action_schedulers[slot_id], + result_timeline=SimulatedResultTimeline( + ordering_policy=action_schedulers[slot_id].ordering_policy + ), + decision_action_history=( + DecisionActionHistory( + env_action_space(self.envs[slot_id]), num_envs=1, decisions=action_history_decisions + ) if action_history_decisions is not None else None + ), + buffers=self._new_episode_buffers(), + worker_capacity=SimulatedWorkerCapacity( + capacity=simulated_worker_capacity, + busy_until_by_worker={}, + ), + ) + for slot_id in range(slot_count) + ] + self._next_obs_id = 0 + self._next_action_id = 0 + self.started_episodes = 0 + self.completed_episodes = 0 + self._completed_metrics: dict[int, EpisodeMetrics] = {} + self._completed_buffers: dict[int, _EpisodeBuffers] = {} + self._episode_log_order: list[int] = [] + self._next_episode_log_index = 0 + + @property + def num_slots(self) -> int: + return len(self.slots) + + def close(self) -> None: + if self.inference_pool is not None: + self.inference_pool.close() + self.env_backend.close() + + def run_episodes( + self, + *, + episode_ids: Sequence[int], + seeds: Sequence[int | None], + eval_max_steps: int = 10000, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + ) -> list[EpisodeMetrics]: + if eval_max_steps < 0: + raise ValueError("eval_max_steps must be non-negative") + episode_ids = [int(episode_id) for episode_id in episode_ids] + if len(seeds) != len(episode_ids): + raise ValueError("seeds length must match episode_ids length") + + self._reset_run_state(episode_ids) + if not episode_ids: + return [] + + next_episode_index = 0 + initial_slots = min(self.num_slots, len(episode_ids)) + for slot in self.slots[:initial_slots]: + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + + while self.completed_episodes < len(episode_ids): + active_slots = self._active_slots() + if eval_max_steps == 0: + for slot in active_slots: + self._complete_slot(slot, on_episode_complete=on_episode_complete) + if next_episode_index < len(episode_ids): + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + continue + + observations = [] + observation_slots: list[_SlotState] = [] + step_capacity_info: dict[int, dict[str, int | bool | None]] = {} + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + slot.worker_capacity.release_ready(slot.env_step) + self._deliver_arrived_results(slot, raw_frame=slot.env_step) + observation_submitted = False + observation_dropped = False + if self.decision_scheduler.should_observe(slot.env_step, current_time_ms): + prefix_request_pending = ( + self.action_prefix is not None + and self.action_prefix["mode"] != "none" + and slot.result_timeline.pending_observation_count > 0 + ) + if slot.worker_capacity.can_submit() and not prefix_request_pending: + observation_slots.append(slot) + observation_submitted = True + else: + slot.buffers.dropped_observation_count += 1 + observation_dropped = True + slot.recent_drop_count += 1 + if self.simulated_worker_capacity is not None: + step_capacity_info[slot.slot_id] = { + "observation_submitted": observation_submitted, + "observation_dropped": observation_dropped, + } + if slot.decision_action_history is not None and slot.env_step % self.clock.obs_stride_raw_frames == 0: + slot.decision_admitted = observation_submitted + slot.decision_issued_action = slot.action_scheduler.noop_action.value + + observe_ms = 0.0 + if observation_slots: + observe_start = time.perf_counter() + observations_by_slot = self.env_backend.observe_slots([slot.slot_id for slot in observation_slots]) + observe_ms = (time.perf_counter() - observe_start) * 1000.0 + pending_observations = [ + self._sample_policy_observation( + slot, + self._policy_observation( + slot, + observations_by_slot[slot.slot_id], + transport=( + slot.decision_action_history.observation()[0] + if slot.decision_action_history is not None else None + ), + ), + ) + for slot in observation_slots + ] + observations = [pending.observation for pending in pending_observations] + + profile_row = None + if observations: + profiler = StageProfiler(enabled=self.profile_pipeline) + with profiler_scope(profiler): + policy_outputs = ( + self.inference_pool.predict_batch(observations) + if self.inference_pool is not None + else self.policy.predict_batch(observations) + ) + if len(policy_outputs) != len(observations): + raise RuntimeError("policy.predict_batch returned the wrong number of outputs") + if self.profile_pipeline: + profile_row = { + "active_slots": float(len(active_slots)), + "batch_size": float(len(observations)), + "observe_slots_ms": observe_ms, + **{key: float(value) for key, value in profiler.timings.items()}, + } + for pending, policy_output in zip(pending_observations, policy_outputs): + if pending.slot.decision_action_history is not None: + pending.slot.decision_issued_action = policy_output.action.value + self._enqueue_policy_output( + pending.slot, + pending.observation, + policy_output, + obs_id=pending.obs_id, + latency_sample=pending.latency_sample, + worker_slot=pending.worker_slot, + ) + + actions_by_slot = {} + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + self._deliver_arrived_results(slot, raw_frame=slot.env_step) + active_action = slot.action_scheduler.update(slot.env_step, current_time_ms) + actions_by_slot[slot.slot_id] = active_action + + env_step_start = time.perf_counter() + step_responses = self.env_backend.step_slots(actions_by_slot) + if profile_row is not None: + profile_row["env_step_ms"] = (time.perf_counter() - env_step_start) * 1000.0 + self._pipeline_profile_rows.append(profile_row) + for slot in active_slots: + current_time_ms = self.clock.step_to_time_ms(slot.env_step) + active_action = actions_by_slot[slot.slot_id] + if ( + slot.decision_action_history is not None + and (slot.env_step + 1) % self.clock.obs_stride_raw_frames == 0 + ): + slot.decision_action_history.append( + [0], [slot.decision_admitted], + [slot.decision_issued_action], [active_action.value], + ) + result = step_responses[slot.slot_id].result + soft_reset = bool(result.info.get("soft_reset")) if isinstance(result.info, dict) else False + episode_done = bool(result.done or result.truncated) and not soft_reset + if slot.buffers.step_records is not None: + record = build_step_record( + episode_id=int(slot.episode_id), + env_step=slot.env_step, + scheduled_time_ms=current_time_ms, + active_action=active_action, + reward=result.reward, + done=episode_done, + info=result.info, + active_event=slot.action_scheduler.latest_applied_event, + frame_ms=self.clock.frame_ms, + latency_type=latency_type_from_source(slot.latency_source), + ) + slot.buffers.step_records.append(record) + slot.buffers.record_step(reward=float(result.reward), info=record.info) + else: + slot.buffers.record_step(reward=float(result.reward), info=result.info) + if self.simulated_worker_capacity is not None and slot.buffers.step_records is not None: + slot.buffers.step_records[-1].info.update( + { + **step_capacity_info[slot.slot_id], + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + ) + if soft_reset: + slot.buffers.num_dropped_actions += slot.action_scheduler.dropped_events_count + slot.action_scheduler.reset() + slot.result_timeline.reset() + self._reset_policy_state(slot.slot_id) + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + slot.recent_drop_count = 0 + + slot.env_step += 1 + if episode_done or slot.env_step >= eval_max_steps: + self._complete_slot(slot, on_episode_complete=on_episode_complete) + if next_episode_index < len(episode_ids): + self._start_slot( + slot, + episode_id=episode_ids[next_episode_index], + seed=seeds[next_episode_index], + ) + next_episode_index += 1 + + self._write_pipeline_profile_summary() + return self._ordered_metrics(episode_ids) + + def _policy_observation( + self, + slot: _SlotState, + observation: Observation, + transport: np.ndarray | None = None, + ) -> Observation: + observation = with_action_prefix(observation, slot.action_scheduler, self.action_prefix) + metadata = dict(observation.metadata) + metadata["slot_id"] = slot.slot_id + metadata["episode_id"] = int(slot.episode_id) + metadata["action_noise_seed"] = slot.episode_seed + data = observation.data + if transport is not None: + data = {**data, "transport": transport} if isinstance(data, Mapping) else {"obs": data, "transport": transport} + return Observation( + data=data, + env_step=observation.env_step, + sim_time_ms=observation.sim_time_ms, + metadata=metadata, + ) + + def _sample_policy_observation( + self, + slot: _SlotState, + observation: Observation, + ) -> _PendingPolicyObservation: + obs_id = self._next_obs_id + self._next_obs_id += 1 + raw_frame = int(slot.env_step) + current_time_ms = self.clock.step_to_time_ms(raw_frame) + worker_slot = slot.worker_capacity.assign_worker() + latency_context = { + "observation": observation, + "obs_id": obs_id, + "env_step": raw_frame, + "raw_frame": raw_frame, + "sim_time_ms": current_time_ms, + "episode_id": slot.episode_id, + "slot_id": slot.slot_id, + "worker_slot": worker_slot, + "recent_drop_count": slot.recent_drop_count, + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + latency_sample = slot.latency_source.sample(latency_context) + metadata = dict(observation.metadata) + metadata["obs_id"] = obs_id + policy_observation = Observation( + data=observation.data, + env_step=observation.env_step, + sim_time_ms=observation.sim_time_ms, + metadata=metadata, + ) + slot.worker_capacity.submit( + worker_slot, raw_frame + latency_sample.worker_service_raw_frames + ) + return _PendingPolicyObservation( + slot=slot, + obs_id=obs_id, + latency_sample=latency_sample, + worker_slot=worker_slot, + observation=policy_observation, + ) + + def _reset_run_state(self, episode_ids: Sequence[int]) -> None: + self.started_episodes = 0 + self.completed_episodes = 0 + self._pipeline_profile_rows.clear() + self._completed_metrics.clear() + self._completed_buffers.clear() + self._episode_log_order = [int(episode_id) for episode_id in episode_ids] + self._next_episode_log_index = 0 + for slot in self.slots: + slot.active = False + slot.episode_id = None + slot.episode_seed = None + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + + def _active_slots(self) -> list[_SlotState]: + return [slot for slot in self.slots if slot.active] + + def _deliver_arrived_results(self, slot: _SlotState, *, raw_frame: int | None) -> None: + released, dropped = slot.result_timeline.release_arrived(raw_frame) + slot.buffers.num_dropped_actions += len(dropped) + for event in released: + slot.action_scheduler.enqueue(event) + + def _start_slot(self, slot: _SlotState, *, episode_id: int, seed: int | None) -> None: + if self.episode_latency_source_factory is not None: + slot.latency_source = self.episode_latency_source_factory(episode_id) + slot.active = True + slot.episode_id = int(episode_id) + slot.episode_seed = None if seed is None else int(seed) + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + self._reset_policy_state(slot.slot_id) + self.env_backend.reset_slot(slot.slot_id, episode_id=episode_id, seed=seed) + self.started_episodes += 1 + + def _reset_policy_state(self, slot_id: int) -> None: + if self.inference_pool is not None: + self.inference_pool.reset_state(slot_id) + else: + self.policy.reset_state(slot_id=slot_id) + + def _complete_slot( + self, + slot: _SlotState, + *, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + ) -> None: + if not slot.active or slot.episode_id is None: + return + episode_id = int(slot.episode_id) + self._deliver_arrived_results(slot, raw_frame=None) + slot.buffers.num_dropped_actions += slot.action_scheduler.dropped_events_count + metrics = self._compute_episode_metrics( + episode_id=episode_id, + buffers=slot.buffers, + metadata=episode_raw_fact_metadata( + mode="simulated", + episode_seed=slot.episode_seed, + env_fps=self.clock.env_fps, + obs_fps=self.clock.obs_fps, + frame_ms=self.clock.frame_ms, + latency_type=latency_type_from_source(slot.latency_source), + latency_source=slot.latency_source, + ) + | slot.action_scheduler.chunk_metrics() + | ( + { + "submitted_observation_frames": slot.buffers.submitted_observation_frames, + "dropped_observation_count": slot.buffers.dropped_observation_count, + "simulated_worker_capacity": self.simulated_worker_capacity, + "inference_worker_count": self.simulated_worker_capacity, + "in_flight_count": slot.worker_capacity.in_flight_count, + "idle_worker_count": slot.worker_capacity.idle_worker_count, + } + if self.simulated_worker_capacity is not None + else {} + ), + ) + self._completed_metrics[episode_id] = metrics + self._completed_buffers[episode_id] = slot.buffers + self.completed_episodes += 1 + slot.active = False + slot.episode_id = None + slot.episode_seed = None + slot.env_step = 0 + slot.recent_drop_count = 0 + slot.buffers = self._new_episode_buffers() + slot.action_scheduler.reset() + slot.result_timeline.reset() + slot.worker_capacity.reset() + if slot.decision_action_history is not None: + slot.decision_action_history.reset() + self._flush_completed_in_episode_order() + if on_episode_complete is not None: + on_episode_complete(metrics) + + def _enqueue_policy_output( + self, + slot: _SlotState, + observation, + policy_output, + *, + obs_id: int, + latency_sample: LatencySample, + worker_slot: int, + ) -> None: + raw_frame = int(slot.env_step) + latency_ms = latency_sample.latency_ms + ready_raw_frame = raw_frame + latency_sample.action_ready_raw_frames + ready_time_ms = self.clock.step_to_time_ms(ready_raw_frame) + latency_type = latency_type_from_source(slot.latency_source) + profile_metadata = profile_metadata_from_source(slot.latency_source) + slot_metadata = { + "episode_id": int(slot.episode_id), + "slot_id": int(slot.slot_id), + "worker_id": int(worker_slot), + } + latency_record, event = build_simulated_action_event( + action_id=self._next_action_id, + obs_id=obs_id, + policy_output=policy_output, + raw_frame=raw_frame, + ready_raw_frame=ready_raw_frame, + ready_time_ms=ready_time_ms, + latency_sample=latency_sample, + frame_ms=self.clock.frame_ms, + latency_type=latency_type, + profile_metadata=profile_metadata, + latency_record_metadata=slot_metadata, + extra_event_metadata=slot_metadata, + ) + self._next_action_id += 1 + slot.result_timeline.submit(obs_id=obs_id, ready_raw_frame=ready_raw_frame, event=event) + slot.buffers.submitted_observation_frames += 1 + slot.recent_drop_count = 0 + slot.buffers.num_actions += 1 + if slot.buffers.action_events is not None: + slot.buffers.action_events.append(event) + slot.buffers.latency_values_ms.append(latency_ms) + if slot.buffers.latency_records is not None: + slot.buffers.latency_records.append(latency_record) + + def _flush_completed_in_episode_order(self) -> None: + if self.logger is None: + return + while self._next_episode_log_index < len(self._episode_log_order): + episode_id = self._episode_log_order[self._next_episode_log_index] + if episode_id not in self._completed_metrics: + break + buffers = self._completed_buffers[episode_id] + metrics = self._completed_metrics[episode_id] + if buffers.step_records is not None: + for record in buffers.step_records: + self.logger.log_step(record) + if buffers.action_events is not None: + for event in buffers.action_events: + self.logger.log_action_event(event) + if buffers.latency_records is not None: + for latency_record in buffers.latency_records: + self.logger.log_latency(latency_record) + self.logger.log_episode_metrics(metrics) + self._next_episode_log_index += 1 + + def _ordered_metrics(self, episode_ids: Sequence[int]) -> list[EpisodeMetrics]: + return [self._completed_metrics[int(episode_id)] for episode_id in episode_ids] + + def _new_episode_buffers(self) -> _EpisodeBuffers: + return _EpisodeBuffers( + step_records=[] if self._collect_step_records else None, + action_events=[] if self._collect_action_records else None, + latency_records=[] if self._collect_latency_records else None, + ) + + def _write_pipeline_profile_summary(self) -> None: + if not self.profile_pipeline or self.logger is None or not self._pipeline_profile_rows: + return + keys = sorted({key for row in self._pipeline_profile_rows for key in row}) + summary = { + "num_profiled_batches": len(self._pipeline_profile_rows), + **{ + key: series_stats([float(row[key]) for row in self._pipeline_profile_rows if key in row]) + for key in keys + }, + } + write_json(Path(self.logger.output_dir) / "simulated_pipeline_summary.json", summary) + + def _compute_episode_metrics( + self, + *, + episode_id: int, + buffers: _EpisodeBuffers, + metadata: dict, + ) -> EpisodeMetrics: + if buffers.task_metrics is not None: + metadata["task_metrics"] = buffers.task_metrics + if buffers.task_metric_moments is not None: + metadata["task_metric_moments"] = buffers.task_metric_moments + if buffers.final_lives is not None: + metadata["final_lives"] = buffers.final_lives + if buffers.final_is_true_episode_end is not None: + metadata["final_is_true_episode_end"] = buffers.final_is_true_episode_end + metadata["soft_reset_count"] = buffers.soft_reset_count + if buffers.step_records is not None and buffers.action_events is not None: + return compute_episode_metrics( + episode_id=episode_id, + step_records=buffers.step_records, + action_events=buffers.action_events, + latency_values_ms=buffers.latency_values_ms, + metadata=metadata, + frame_ms=self.clock.frame_ms, + ) + return compute_episode_metrics_from_aggregates( + episode_id=episode_id, + episode_return_env=buffers.episode_return_env, + survival_steps=buffers.survival_steps, + return_raw=buffers.return_raw, + game_score=buffers.game_score, + latency_values_ms=buffers.latency_values_ms, + num_actions=buffers.num_actions, + num_dropped_actions=buffers.num_dropped_actions, + num_invalid_actions=buffers.num_invalid_actions, + metadata=metadata, + ) diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch new file mode 100644 index 0000000000000000000000000000000000000000..bdeca64d184242eedaa57463d2e0a242f43e19ba --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly-compatibility.patch @@ -0,0 +1,99 @@ +diff --git a/latency_bench/envs/deadly_corridor.py b/latency_bench/envs/deadly_corridor.py +index 4dcaa48c..dc4d1186 100644 +--- a/latency_bench/envs/deadly_corridor.py ++++ b/latency_bench/envs/deadly_corridor.py +@@ -5,7 +5,7 @@ from collections import deque + from typing import Any + + import numpy as np +-from gymnasium.spaces import Box, Tuple ++from gymnasium.spaces import Box, MultiBinary, Tuple + + from latency_bench.core.types import Action, Observation, StepResult + from latency_bench.envs.base import EnvAdapter +@@ -346,6 +346,7 @@ class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + export_env_raw_rgb_frames: bool = True, + ): + import gymnasium as gym ++ import vizdoom + import vizdoom.gymnasium_wrapper # noqa: F401 (registers the env ids) + + env_cfg = config["env"] +@@ -360,30 +361,27 @@ class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + ) + if key in env_cfg + } +- attempts = [ +- ("VizdoomDeadlyCorridor-MultiBinary-v1", {}), +- ("VizdoomDeadlyCorridor-MultiBinary-v0", {}), +- ("VizdoomDeadlyCorridor-v1", {"max_buttons_pressed": 0}), +- ("VizdoomDeadlyCorridor-v0", {"max_buttons_pressed": 0}), +- ] +- last_exc: Exception | None = None +- self.gym_env = None +- for env_id, kwargs in attempts: +- try: +- # frame_skip=1: the latency_bench scheduler advances obs_stride raw +- # frames per decision and holds the action between observations. +- self.gym_env = gym.make( +- env_id, render_mode="rgb_array", frame_skip=1, **render_options, **kwargs +- ) +- self.env_id = env_id +- break +- except (gym.error.NameNotFound, gym.error.VersionNotFound, gym.error.NamespaceNotFound) as exc: +- last_exc = exc +- if self.gym_env is None: +- raise RuntimeError(f"Failed to create Deadly Corridor MultiBinary env: {last_exc}") ++ # ViZDoom registers deadly_corridor.cfg under this official Gym ID. ++ self.env_id = "VizdoomCorridor-v0" ++ self.gym_env = gym.make( ++ self.env_id, render_mode="rgb_array", frame_skip=1, max_buttons_pressed=0, ++ ) ++ game = self.gym_env.unwrapped.game ++ game.close() ++ for key, value in render_options.items(): ++ if key == "screen_resolution": ++ value = getattr(vizdoom.ScreenResolution, value) ++ getattr(game, f"set_{key}")(value) ++ game.init() ++ self.gym_env.unwrapped.observation_space.spaces["screen"] = Box( ++ 0, 255, ++ shape=(game.get_screen_height(), game.get_screen_width(), game.get_screen_channels()), ++ dtype=np.uint8, ++ ) + + self._runtime_button_order = _deadly_runtime_button_names(self.gym_env) + self._num_buttons = len(self._runtime_button_order) ++ self.gym_env.unwrapped.action_space = MultiBinary(self._num_buttons) + self.noop_action = noop_action or Action( + value=[0] * self._num_buttons, name="NOOP", is_noop=True + ) +diff --git a/tests/integration/test_deadly_render_contract.py b/tests/integration/test_deadly_render_contract.py +index 535db22a..09894b1b 100644 +--- a/tests/integration/test_deadly_render_contract.py ++++ b/tests/integration/test_deadly_render_contract.py +@@ -5,11 +5,13 @@ import json + import numpy as np + import pytest + +-pytest.importorskip("vizdoom", minversion="1.3.0") ++pytest.importorskip("vizdoom", minversion="1.2.4") + pytest.importorskip("sample_factory") + + from latency_bench.envs.deadly_corridor import DeadlyCorridorEnvAdapter, DeadlyCorridorVlaEnvAdapter + from latency_bench.envs.raw_rgb import ENV_RAW_RGB_FRAME_STACK_INFO_KEY ++from latency_bench.core.types import Action ++from gymnasium.spaces import MultiBinary + from scripts.tasks.decision_history.eval_vla_hist8 import evaluation_config + + +@@ -33,6 +35,9 @@ def test_hist8_deadly_vla_uses_the_teacher_resolution_and_hud(tmp_path): + # The health/ammo panel is stable across the two engine reset paths; + # the animated face and enemies can differ with their RNG streams. + np.testing.assert_array_equal(teacher_frame[-20:, :64], student_frame[-20:, :64]) ++ assert isinstance(student.gym_env.action_space, MultiBinary) ++ step = student.step(Action(value=[1, 0, 0, 0, 0, 0, 1], name="forward_attack")) ++ assert np.isfinite(step.reward) + finally: + teacher.close() + student.close() diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py new file mode 100644 index 0000000000000000000000000000000000000000..1016be20dca9e949c752192edb407b9a84e30e35 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/deadly_corridor.py @@ -0,0 +1,455 @@ +from __future__ import annotations + +import copy +from collections import deque +from typing import Any + +import numpy as np +from gymnasium.spaces import Box, MultiBinary, Tuple + +from latency_bench.core.types import Action, Observation, StepResult +from latency_bench.envs.base import EnvAdapter +from latency_bench.utils.array import looks_chw +from latency_bench.envs.raw_rgb import RawRgbFrameStackBuffer + + +def _noop_action_from_space(space) -> Any: + n = getattr(space, "n", None) + if n is not None: + return 0 + if isinstance(space, Tuple): + return tuple(_noop_action_from_space(subspace) for subspace in space.spaces) + if isinstance(space, Box): + import numpy as np + + return np.zeros(space.shape, dtype=space.dtype) + raise TypeError(f"Unsupported action space for Deadly Corridor no-op action: {space}") + + +def _coerce_noop_action_for_space(value: Any, space) -> Any: + if isinstance(space, Tuple): + if isinstance(value, (list, tuple)): + if len(value) != len(space.spaces): + raise ValueError( + f"Deadly Corridor no-op action length {len(value)} does not match action space {space}" + ) + return tuple( + _coerce_noop_action_for_space(item, subspace) + for item, subspace in zip(value, space.spaces) + ) + if value == 0: + return _noop_action_from_space(space) + return value + + +def _spec_with_reward_scaling(spec: Any, disable_reward_scaling: bool) -> Any: + if not disable_reward_scaling: + return spec + spec_to_use = copy.copy(spec) + spec_to_use.reward_scaling = 1.0 + return spec_to_use + + +def _synchronous_eval_fps_from_config(config: dict[str, Any], default: int = 35) -> int: + env_cfg = config.get("env", {}) + try: + fps = int(float(env_cfg.get("env_fps", default))) + except (TypeError, ValueError) as exc: + raise ValueError("env_fps must be positive") from exc + if fps <= 0: + raise ValueError("env_fps must be positive") + return fps + + +def _build_sample_factory_eval_cfg(config: dict[str, Any]) -> Any: + from training.deadly_corridor_sf import integration + from training.common.utils import maybe_set_cli_override + + integration.register_deadly_corridor_components() + base_cfg = integration.SAMPLE_FACTORY_CONFIG_PARSER.parse_eval( + integration.build_cli_args_from_config(config) + ) + eval_fps = _synchronous_eval_fps_from_config(config) + cfg = copy.deepcopy(base_cfg) + if _requires_sample_factory_checkpoint_config(config): + from sample_factory.cfg.arguments import load_from_checkpoint + + cfg = load_from_checkpoint(cfg) + + for key in ( + "seed", + "res_w", + "res_h", + "wide_aspect_ratio", + ): + if hasattr(base_cfg, key): + maybe_set_cli_override(cfg, key, getattr(base_cfg, key)) + maybe_set_cli_override(cfg, "frame_stack", 1) + explicit_max_episode_steps = int(getattr(base_cfg, "max_episode_steps", 0) or 0) + if explicit_max_episode_steps > 0: + maybe_set_cli_override(cfg, "max_episode_steps", explicit_max_episode_steps) + else: + eval_max_steps = int(getattr(base_cfg, "eval_max_steps", 0) or 0) + if eval_max_steps > 0: + maybe_set_cli_override(cfg, "max_episode_steps", eval_max_steps) + + maybe_set_cli_override(cfg, "mode", "eval") + maybe_set_cli_override(cfg, "latency_type", "zero") + maybe_set_cli_override(cfg, "fixed_latency_ms", 0.0) + maybe_set_cli_override(cfg, "env_frameskip", 1) + maybe_set_cli_override(cfg, "eval_env_frameskip", 1) + maybe_set_cli_override(cfg, "num_envs", 1) + maybe_set_cli_override(cfg, "no_render", True) + maybe_set_cli_override(cfg, "save_video", False) + maybe_set_cli_override(cfg, "fps", eval_fps) + maybe_set_cli_override(cfg, "eval_deterministic", bool(getattr(base_cfg, "eval_deterministic", True))) + maybe_set_cli_override(cfg, "disable_reward_scaling", bool(getattr(base_cfg, "eval_raw_reward", False))) + return cfg + + +def _requires_sample_factory_checkpoint_config(config: dict[str, Any]) -> bool: + policy_type = str(config.get("policy", {}).get("type", "")).strip().lower() + return policy_type == "deadly_corridor_sf" + + +def _seed_initialized_vizdoom_game(env: Any, seed: int) -> bool: + unwrapped = getattr(env, "unwrapped", env) + game = getattr(unwrapped, "game", None) + if game is None: + return False + unwrapped.seed(int(seed)) + game.set_seed(int(unwrapped.curr_seed)) + return True + + +class DeadlyCorridorEnvAdapter(EnvAdapter): + """Latency-bench adapter for ViZDoom Deadly Corridor using the SF Doom env stack.""" + OBSERVATION_TYPE = "vizdoom_frame_v1" + + def __init__( + self, + *, + config: dict[str, Any], + noop_action: Action | None = None, + export_env_raw_rgb_frames: bool = False, + ): + env_cfg = config["env"] + env_id = str(env_cfg.get("env_id", "doom_deadly_corridor")) + env_fps = float(env_cfg.get("env_fps", 35)) + self.frame_stack = int(env_cfg.get("frame_stack", 1) or 1) + + from sample_factory.utils.attr_dict import AttrDict + from sf_examples.vizdoom.doom.doom_utils import DOOM_ENVS, make_doom_env_from_spec + + cfg = _build_sample_factory_eval_cfg(config) + spec = next((item for item in DOOM_ENVS if item.name == str(env_id)), None) + if spec is None: + raise ValueError(f"Unknown ViZDoom env spec: {env_id}") + spec_to_use = _spec_with_reward_scaling( + spec, + disable_reward_scaling=bool(getattr(cfg, "disable_reward_scaling", False)), + ) + self.gym_env = make_doom_env_from_spec( + spec_to_use, + str(env_id), + cfg, + AttrDict(worker_index=0, vector_index=0, env_id=0), + render_mode=None, + ) + self.cfg = cfg + self.env_id = env_id + self.env_fps = float(env_fps) + action_space = self.gym_env.action_space + noop_value = _noop_action_from_space(action_space) + if noop_action is None: + self.noop_action = Action(value=noop_value, name=str(noop_value), is_noop=True) + else: + coerced_noop_value = _coerce_noop_action_for_space(noop_action.value, action_space) + self.noop_action = Action( + value=coerced_noop_value, + name=str(coerced_noop_value), + is_noop=True, + is_oneshot=noop_action.is_oneshot, + ) + self.env_step = 0 + self.export_env_raw_rgb_frames = bool(export_env_raw_rgb_frames) + self._last_info: dict[str, Any] = {} + self._last_frame: Any = None + self._observed_frames: deque[np.ndarray] = deque(maxlen=self.frame_stack) + self._raw_rgb_frames = RawRgbFrameStackBuffer(num_frames=self.frame_stack) + + def reset(self, seed: int | None = None) -> Observation: + self.env_step = 0 + self._observed_frames.clear() + if seed is not None: + if _seed_initialized_vizdoom_game(self.gym_env, int(seed)): + obs, info = self.gym_env.reset() + else: + try: + obs, info = self.gym_env.reset(seed=seed) + except TypeError: + obs, info = self.gym_env.reset() + else: + obs, info = self.gym_env.reset() + self._last_frame = obs + self._last_info = dict(info or {}) + self._reset_frame_stack(obs) + if self.export_env_raw_rgb_frames: + self._reset_raw_rgb_frame_stack() + return self._make_observation(info=self._last_info) + + def step(self, action: Action) -> StepResult: + gym_action = action.value + obs, reward, terminated, truncated, info = self.gym_env.step(gym_action) + self.env_step += 1 + self._last_frame = obs + self._last_info = dict(info or {}) + self._append_frame(obs) + if self.export_env_raw_rgb_frames and not bool(terminated or truncated): + self._append_raw_rgb_frame() + observation = self._make_observation(info=self._last_info) + step_info = dict(self._last_info) + step_info.update( + { + "env_step": self.env_step, + "sim_time_ms": self.env_step * self.frame_ms, + "applied_action": gym_action, + "applied_action_name": action.name, + "observation": "vizdoom_frame_v1", + } + ) + return StepResult( + observation=observation, + reward=float(reward), + done=bool(terminated), + truncated=bool(truncated), + info=step_info, + ) + + def observe(self) -> Observation: + if self._last_frame is None: + raise RuntimeError("DeadlyCorridorEnvAdapter has no current observation; call reset() first") + metadata = self._metadata(self._last_info) + return Observation( + data=self._policy_frame_stack(), + env_step=self.env_step, + sim_time_ms=self.env_step * self.frame_ms, + metadata=metadata, + ) + + def render_game_frame(self) -> np.ndarray: + return np.transpose(self.gym_env.unwrapped.game.get_state().screen_buffer, (1, 2, 0)) + + def close(self) -> None: + self.gym_env.close() + + def _reset_frame_stack(self, frame: Any) -> None: + self._observed_frames.clear() + self._append_frame(frame) + + def _append_frame(self, frame: Any) -> None: + self._observed_frames.append(_single_frame_data(frame)) + + def _policy_frame_stack(self) -> np.ndarray: + frames = list(self._observed_frames) + if not frames: + raise RuntimeError("Deadly Corridor observe() has no current frame; call reset() first") + if len(frames) < self.frame_stack: + frames = [frames[0]] * (self.frame_stack - len(frames)) + frames + frames = [np.asarray(frame, dtype=np.uint8) for frame in frames[-self.frame_stack :]] + if self.frame_stack == 1: + return frames[-1] + axis = 0 if looks_chw(frames[0]) else -1 + return np.concatenate(frames, axis=axis) + + +def _single_frame_data(frame: Any) -> np.ndarray: + value = frame.get("obs") if isinstance(frame, dict) else frame + arr = np.asarray(value, dtype=np.uint8) + if arr.ndim == 2: + return arr[..., None] + if arr.ndim != 3: + raise ValueError(f"Expected Deadly Corridor image frame with 2 or 3 dims, got {arr.shape!r}") + return arr + + +# Fixed semantic button order the StarVLA multibinary head is trained against. +# Mirrors starVLA.training.rl_games.eval_core._semantic_to_runtime_multibinary. +DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER = ( + "MOVE_FORWARD", + "MOVE_BACKWARD", + "MOVE_LEFT", + "MOVE_RIGHT", + "TURN_LEFT", + "TURN_RIGHT", + "ATTACK", +) + + +def _deadly_runtime_button_names(gym_env: Any) -> list[str]: + """Return the live ViZDoom action-button order (ports eval_core helper). + + The MultiBinary action vector is indexed by the game's available-button + order, which is not guaranteed to equal the semantic order the head emits. + """ + + def _button_name(button: Any) -> str: + name = getattr(button, "name", None) + if name is not None: + return str(name) + text = str(button) + return text.split(".")[-1] if "." in text else text + + for candidate in (gym_env, getattr(gym_env, "unwrapped", None)): + if candidate is None: + continue + for attr_name in ("game", "_game"): + game = getattr(candidate, attr_name, None) + if game is None: + continue + getter = getattr(game, "get_available_buttons", None) + if getter is None: + continue + names = [_button_name(button) for button in getter()] + if names: + return names + return list(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + + +def _semantic_to_runtime_multibinary(semantic_values: list[int], runtime_order: list[str]) -> list[int]: + semantic_map = { + name: int(semantic_values[idx]) + for idx, name in enumerate(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + if idx < len(semantic_values) + } + return [semantic_map.get(name, 0) for name in runtime_order] + + +class DeadlyCorridorVlaEnvAdapter(EnvAdapter): + """Deadly Corridor adapter for StarVLA eval, matching eval_core's env. + + Unlike :class:`DeadlyCorridorEnvAdapter` (sample_factory, factorised action + tuple), this uses the gymnasium ``VizdoomDeadlyCorridor-MultiBinary`` env so + the model's multibinary head can fire arbitrary button subsets, exactly like + ``starVLA.training.rl_games.eval_core``. Native ``frame_skip=1`` is used so + latency_bench's observation-cadence scheduler owns the obs_stride stepping + (see ObservationCadenceDecisionScheduler); setting a native skip would + double-count it. + """ + + OBSERVATION_TYPE = "vizdoom_frame_v1" + + def __init__( + self, + *, + config: dict[str, Any], + noop_action: Action | None = None, + export_env_raw_rgb_frames: bool = True, + ): + import gymnasium as gym + import vizdoom + import vizdoom.gymnasium_wrapper # noqa: F401 (registers the env ids) + + env_cfg = config["env"] + self.env_fps = float(env_cfg.get("env_fps", 35)) + self.frame_stack = int(env_cfg.get("frame_stack", 1) or 1) + # The raw teacher view is part of the policy's observation contract. + render_options = { + key: env_cfg[key] + for key in ( + "screen_resolution", "render_hud", "render_crosshair", + "render_weapon", "render_decals", "render_particles", + ) + if key in env_cfg + } + # ViZDoom registers deadly_corridor.cfg under this official Gym ID. + self.env_id = "VizdoomCorridor-v0" + self.gym_env = gym.make( + self.env_id, render_mode="rgb_array", frame_skip=1, max_buttons_pressed=0, + ) + game = self.gym_env.unwrapped.game + game.close() + for key, value in render_options.items(): + if key == "screen_resolution": + value = getattr(vizdoom.ScreenResolution, value) + getattr(game, f"set_{key}")(value) + game.init() + self.gym_env.unwrapped.observation_space.spaces["screen"] = Box( + 0, 255, + shape=(game.get_screen_height(), game.get_screen_width(), game.get_screen_channels()), + dtype=np.uint8, + ) + + self._runtime_button_order = _deadly_runtime_button_names(self.gym_env) + self._num_buttons = len(self._runtime_button_order) + self.gym_env.unwrapped.action_space = MultiBinary(self._num_buttons) + self.noop_action = noop_action or Action( + value=[0] * self._num_buttons, name="NOOP", is_noop=True + ) + self.export_env_raw_rgb_frames = bool(export_env_raw_rgb_frames) + self.env_step = 0 + self._last_info: dict[str, Any] = {} + self._last_frame: Any = None + self._raw_rgb_frames = RawRgbFrameStackBuffer(num_frames=self.frame_stack) + + def reset(self, seed: int | None = None) -> Observation: + self.env_step = 0 + try: + obs, info = self.gym_env.reset(seed=seed) + except TypeError: + obs, info = self.gym_env.reset() + self._last_frame = obs + self._last_info = dict(info or {}) + if self.export_env_raw_rgb_frames: + self._reset_raw_rgb_frame_stack() + return self._make_observation(info=self._last_info) + + def step(self, action: Action) -> StepResult: + # action.value is a 7-dim multibinary vector in semantic order; re-order + # to the live game's button layout before stepping the MultiBinary env. + semantic = [int(v) for v in np.asarray(action.value).reshape(-1).tolist()] + expected = len(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) + if len(semantic) != expected: + raise ValueError( + "DeadlyCorridorVlaEnvAdapter expects a " + f"{expected}-dim multibinary action in semantic order, got " + f"{len(semantic)} values ({action.value!r}). This usually means the " + "policy decoded a non-multibinary layout; ensure the deadly head is " + "action_layout=multibinary_7 and reached the multibinary decode path." + ) + runtime_buttons = _semantic_to_runtime_multibinary(semantic, self._runtime_button_order) + gym_action = np.asarray(runtime_buttons, dtype=np.int8) + obs, reward, terminated, truncated, info = self.gym_env.step(gym_action) + self.env_step += 1 + self._last_frame = obs + self._last_info = dict(info or {}) + if self.export_env_raw_rgb_frames and not bool(terminated or truncated): + self._append_raw_rgb_frame() + observation = self._make_observation(info=self._last_info) + step_info = dict(self._last_info) + step_info.update( + { + "env_step": self.env_step, + "sim_time_ms": self.env_step * self.frame_ms, + "applied_action": runtime_buttons, + "applied_action_name": action.name, + "observation": self.OBSERVATION_TYPE, + } + ) + return StepResult( + observation=observation, + reward=float(reward), + done=bool(terminated), + truncated=bool(truncated), + info=step_info, + ) + + def observe(self) -> Observation: + return self._make_observation(info=self._last_info) + + def render_game_frame(self) -> np.ndarray: + frame = self.gym_env.render() + return np.asarray(frame, dtype=np.uint8) + + def close(self) -> None: + self.gym_env.close() diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py new file mode 100644 index 0000000000000000000000000000000000000000..13f64ef81329e0b3c9296a066e17196a7e6c1d56 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/decision_action_history.py @@ -0,0 +1,60 @@ +"""Causal action history sampled at completed decision boundaries.""" + +from __future__ import annotations + +import numpy as np +from gymnasium.spaces import Discrete, MultiBinary, Tuple + + +class DecisionActionHistory: + """Encode admission, admitted command, and last applied action for each decision.""" + + def __init__(self, action_space, *, num_envs: int, decisions: int): + self._multibinary = isinstance(action_space, MultiBinary) + if isinstance(action_space, Discrete): + self.action_sizes = (action_space.n,) + elif isinstance(action_space, Tuple) and all(isinstance(space, Discrete) for space in action_space.spaces): + self.action_sizes = tuple(space.n for space in action_space.spaces) + elif self._multibinary and action_space.shape == (7,): + self.action_sizes = (3, 3, 3, 2) + else: + raise NotImplementedError(f"Decision action history does not support {action_space!r}") + self.decisions = decisions + self.action_dim = sum(size - 1 for size in self.action_sizes) + self.step_dim = 1 + 2 * self.action_dim + self.data = np.zeros((num_envs, decisions, self.step_dim), dtype=np.float32) + self._basis = tuple(np.eye(size, dtype=np.float32)[:, 1:] for size in self.action_sizes) + + @property + def observation_dim(self) -> int: + return self.decisions * self.step_dim + + def reset(self, indices=None) -> None: + if indices is None: + self.data.fill(0) + else: + self.data[indices] = 0 + + def append(self, indices, admitted, issued_actions, applied_actions) -> None: + admitted = np.asarray(admitted, dtype=np.float32).reshape(-1) + issued = self._encode(issued_actions) * admitted[:, None] + applied = self._encode(applied_actions) + rows = self.data[indices].copy() + rows[:, :-1] = rows[:, 1:] + rows[:, -1, 0] = admitted + rows[:, -1, 1 : 1 + self.action_dim] = issued + rows[:, -1, 1 + self.action_dim :] = applied + self.data[indices] = rows + + def observation(self) -> np.ndarray: + return self.data.reshape(self.data.shape[0], self.observation_dim).copy() + + def _encode(self, actions) -> np.ndarray: + if self._multibinary: + # The VLA button order is move, strafe, turn, attack; teacher history + # encodes turn, move, strafe, attack. Keep both opposing bits if issued. + return np.asarray(actions, dtype=np.float32).reshape(-1, 7)[:, [4, 5, 0, 1, 2, 3, 6]] + values = np.asarray(actions, dtype=np.int64).reshape(-1, len(self.action_sizes)) + return np.concatenate( + [basis[values[:, index]] for index, basis in enumerate(self._basis)], axis=1 + ) diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py new file mode 100644 index 0000000000000000000000000000000000000000..fca10fa388e20c9df4abb2f235ac4ea284a3144c --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/eval_driver.py @@ -0,0 +1,216 @@ +"""Single evaluation driver: run one config's episodes and attach metadata. + +This is the core ``run_from_config`` and its episode-side helpers. Sweep/suite +orchestration lives in :mod:`latency_bench.eval.sweeps`; the CLI in +:mod:`latency_bench.run`. +""" +from __future__ import annotations + +from collections.abc import Callable, Sequence +from pathlib import Path +from typing import Any + +import yaml + +from training.common.utils import seed_everything +from latency_bench.core.types import EpisodeMetrics, ExecutorMode +from latency_bench.eval.config import ( + _episode_seed, + _eval_episodes, + _eval_max_steps, + _evaluation_seed, + resolve_evaluation_config, +) +from latency_bench.eval.reporting import _write_non_sweep_summary +from latency_bench.envs.base import EnvAdapter +from latency_bench.executors.base import BatchedExecutor +from latency_bench.executors.factory import build_executor +from latency_bench.executors.realtime_warmup import plot_realtime_eval_latency +from latency_bench.latency.config import latency_type_from_config +from latency_bench.logging.action_trace_replay import record_videos_from_action_trace +from latency_bench.logging.video import select_episode_return_stratified + + +def run_from_config( + config: dict[str, Any], + extra_metadata: dict[str, Any] | None = None, + *, + write_summary: bool = True, + on_episode_complete: Callable[[EpisodeMetrics], None] | None = None, + episode_ids: Sequence[int] | None = None, + policy: Any | None = None, + env: EnvAdapter | None = None, + env_backend: Any | None = None, + inference_devices: list[str] | None = None, +) -> list[EpisodeMetrics]: + eval_max_steps = _eval_max_steps(config) + resolve_evaluation_config(config) + if ( + policy is None + and env is None + and env_backend is None + and config["policy"]["type"] == "starvla" + ): + from latency_bench.policy.starvla import prepare_starvla_checkpoint_input_config + + prepare_starvla_checkpoint_input_config(config) + + experiment_cfg = config["experiment"] + policy_cfg = config["policy"] + logging_cfg = config["logging"] + seed = _evaluation_seed(config) + configured_num_episodes = _eval_episodes(config) + selected_episode_ids = list(range(configured_num_episodes)) if episode_ids is None else list(episode_ids) + seed_everything(seed) + + executor_kwargs = {} + if policy is not None: + executor_kwargs["policy"] = policy + if env is not None: + executor_kwargs["env"] = env + if env_backend is not None: + executor_kwargs["env_backend"] = env_backend + if inference_devices is not None: + executor_kwargs["inference_devices"] = inference_devices + executor = build_executor(config, **executor_kwargs) + metrics = [] + warmup_metadata_by_episode: dict[int, dict[str, Any]] = {} + try: + output_dir = Path(logging_cfg["output_dir"]) + output_dir.mkdir(parents=True, exist_ok=True) + (output_dir / "resolved_config.yaml").write_text( + yaml.safe_dump(config, sort_keys=False), encoding="utf-8" + ) + if isinstance(executor, BatchedExecutor): + warmup_metadata = executor.run_warmup() + run_episodes_kwargs: dict[str, Any] = { + "episode_ids": selected_episode_ids, + "seeds": [_episode_seed(config, episode_id) for episode_id in selected_episode_ids], + "eval_max_steps": eval_max_steps, + } + if on_episode_complete is not None: + run_episodes_kwargs["on_episode_complete"] = on_episode_complete + metrics = list(executor.run_episodes(**run_episodes_kwargs)) + warmup_metadata_by_episode.update( + (episode_id, warmup_metadata) for episode_id in selected_episode_ids + ) + else: + warmup_metadata = executor.run_warmup() + for episode_id in selected_episode_ids: + warmup_metadata_by_episode[episode_id] = warmup_metadata + episode_metrics = executor.run_episode( + episode_id=episode_id, + seed=_episode_seed(config, episode_id), + eval_max_steps=eval_max_steps, + ) + metrics.append(episode_metrics) + if on_episode_complete is not None: + on_episode_complete(episode_metrics) + metrics.sort(key=lambda item: int(item.episode_id)) + for episode_metrics in metrics: + for key, value in _evaluation_raw_fact_metadata(config, int(episode_metrics.episode_id)).items(): + if episode_metrics.metadata.get(key) is None: + episode_metrics.metadata[key] = value + episode_metrics.metadata.update(warmup_metadata_by_episode[int(episode_metrics.episode_id)]) + if "measurement" in config: + episode_metrics.metadata["measurement"] = config["measurement"] + episode_metrics.metadata["config_name"] = experiment_cfg.get("name") + episode_metrics.metadata["run_name"] = experiment_cfg.get("name") + if "checkpoint_path" in policy_cfg: + episode_metrics.metadata["checkpoint_path"] = policy_cfg["checkpoint_path"] + if "profile_path" in config["latency"]: + episode_metrics.metadata["source_profile_path"] = config["latency"]["profile_path"] + if "checkpoint_kind" in policy_cfg: + episode_metrics.metadata["checkpoint_kind"] = str(policy_cfg["checkpoint_kind"]) + episode_metrics.metadata["output_dir"] = str(logging_cfg["output_dir"]) + if "action_prefix" in policy_cfg: + episode_metrics.metadata["action_prefix"] = policy_cfg["action_prefix"] + if extra_metadata: + episode_metrics.metadata.update(extra_metadata) + if executor.logger is not None: + executor.logger.flush() + _record_realtime_eval_latency_plot(config, executor) + if write_summary: + _write_non_sweep_summary(config, metrics) + _record_stratified_replay_videos(config, metrics, seed=seed) + finally: + executor.close() + return metrics + + +def _record_realtime_eval_latency_plot(config: dict[str, Any], executor: Any) -> None: + if ExecutorMode(config["executor"]["mode"]) != ExecutorMode.REALTIME: + return + if not config["logging"]["save_latency_records"]: + return + + latency_values = list(executor.logger.latency_ms_values) + plot_realtime_eval_latency( + latency_values, + Path(config["logging"]["output_dir"]) / "eval_latency_trace.png", + ) + + +def _record_stratified_replay_videos( + config: dict[str, Any], + metrics: list[EpisodeMetrics], + *, + seed: int, +) -> None: + if "video" not in config["logging"]: + return + video_cfg = config["logging"]["video"] + if not video_cfg["enabled"]: + return + if not config["logging"]["save_step_records"]: + # Replay reads steps.jsonl, which is only written when save_step_records is on. + # Without it (e.g. factor-sweep evals) skip video instead of crashing on a missing file. + return + if ExecutorMode(config["executor"]["mode"]) == ExecutorMode.REALTIME: + return + selections = select_episode_return_stratified( + metrics, + num_bins=video_cfg["num_bins"], + seed=seed, + ) + record_videos_from_action_trace(config, selections=selections, metrics=metrics) + + +def _evaluation_raw_fact_metadata(config: dict[str, Any], episode_id: int) -> dict[str, Any]: + env_cfg = config.get("env", {}) + policy_cfg = config.get("policy", {}) + latency_cfg = config.get("latency", {}) + executor_cfg = config.get("executor", {}) + env_fps = float(env_cfg["env_fps"]) if "env_fps" in env_cfg else None + obs_fps = float(env_cfg["obs_fps"]) if "obs_fps" in env_cfg else None + frame_ms = None if env_fps is None or env_fps <= 0 else 1000.0 / env_fps + executor_mode = str(executor_cfg.get("mode", "")).strip().lower() + latency_type = latency_type_from_config(latency_cfg) + if executor_mode == "paused": + latency_type = "zero" + elif executor_mode == "realtime": + latency_type = "measured" + return { + "mode": executor_cfg.get("mode"), + "episode_seed": _episode_seed(config, episode_id), + "policy_id": _metadata_id(policy_cfg, "policy_id", "id", "type"), + "env_id": _metadata_id(env_cfg, "env_id", "id", "name"), + "model_id": latency_cfg.get("model_id"), + "gpu_class": latency_cfg.get("gpu_class"), + "workload_id": latency_cfg.get("workload_id"), + "instance_id": latency_cfg.get("instance_id"), + "source_run_id": latency_cfg.get("source_run_id"), + "profile_ref": latency_cfg.get("profile_ref"), + "env_fps": env_fps, + "obs_fps": obs_fps, + "frame_ms": frame_ms, + "latency_type": latency_type, + } + + +def _metadata_id(config: dict[str, Any], *keys: str) -> str | None: + for key in keys: + value = config.get(key) + if value is not None: + return str(value) + return None diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py new file mode 100644 index 0000000000000000000000000000000000000000..19206ec9dfd77ce730fbaee77fe932956dac0b4d --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/mikasa_evaluate.py @@ -0,0 +1,341 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path +from typing import Any + +import gymnasium as gym +import numpy as np +import torch +import yaml + + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT)) +sys.path.insert(0, str(ROOT / "third_party" / "MIKASA-Robo")) + +from latency_bench.core.types import Action, Observation # noqa: E402 +from latency_bench.executors.gpu_batched_env_step_backend import ( # noqa: E402 + GpuBatchedEnvStepBackendBase, + SlotStepOutcome, +) +from mikasa_robo_suite.seed_reset import ( # noqa: E402 + reset_seeded_slot as _reset_seeded_slot, + reset_seeded_slots as _reset_seeded_slots, +) + + +ENV_ID = "InterceptGrabFast-VLA-v0" +INSTRUCTION = "Intercept the rolling ball and grasp it to stop it." +START_SEED = 4242424242 +MIKASA_IMAGE_VIEWS_INFO_KEY = "mikasa_image_views" +MIKASA_STATE_INFO_KEY = "mikasa_proprio" + + +def _scalar(value: Any) -> Any: + if torch.is_tensor(value): + return value.detach().reshape(-1)[0].cpu().item() + return np.asarray(value).reshape(-1)[0].item() + + +def _make_raw_env( + obs_mode: str, + num_envs: int = 1, + simulator_device: str = "gpu", +): + import mikasa_robo_suite.vla.memory_envs # noqa: F401 + + return gym.make( + ENV_ID, + num_envs=num_envs, + obs_mode=obs_mode, + control_mode="pd_ee_delta_pose", + render_mode="all", + sim_backend=simulator_device, + render_backend=simulator_device, + reward_mode="normalized_dense", + ) + + +def _make_ppo_env(num_envs: int = 1, simulator_device: str = "gpu"): + from baselines.ppo.ppo_memtasks import FlattenRGBDObservationWrapper + from mani_skill.vector.wrappers.gymnasium import ManiSkillVectorEnv + from mikasa_robo_suite.vla.dataset_collectors.get_mikasa_robo_datasets import ( + env_info, + ) + + env = _make_raw_env( + "state", + num_envs=num_envs, + simulator_device=simulator_device, + ) + wrappers, _ = env_info(ENV_ID) + for wrapper, kwargs in wrappers: + env = wrapper(env, **kwargs) + env = FlattenRGBDObservationWrapper(env, rgb=False, depth=False, state=True) + return ManiSkillVectorEnv( + env, + num_envs, + ignore_terminations=True, + record_metrics=True, + ) + + +def _make_vla_env(num_envs: int = 1, simulator_device: str = "gpu"): + from mikasa_robo_suite.vla.utils.apply_wrappers import apply_mikasa_vla_wrappers + + return apply_mikasa_vla_wrappers( + _make_raw_env( + "rgb", + num_envs=num_envs, + simulator_device=simulator_device, + ), + include_overlays=False, + ) + + +class _PpoPolicy: + def __init__(self, env, checkpoint: Path): + from baselines.ppo.ppo_memtasks import AgentStateOnly + + self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") + self.agent = AgentStateOnly(env).to(self.device) + self.agent.load_state_dict(torch.load(checkpoint, map_location=self.device)) + self.agent.eval() + + def forward(self, observation): + with torch.no_grad(): + return self.agent.get_action( + {key: value.to(self.device) for key, value in observation.items()}, + deterministic=True, + ) + + +class MikasaEnvStepBackend(GpuBatchedEnvStepBackendBase): + """Own the native MIKASA simulator and its 7D action contract.""" + + backend_name = "mikasa_gpu_batched" + + def __init__(self, *, config: dict[str, Any], num_slots: int, env=None): + noop_action = Action( + value=np.asarray(config["env"]["noop_action"], dtype=np.float32), + name="noop", + is_noop=True, + ) + super().__init__( + config=config, + noop_action=noop_action, + num_slots=num_slots, + action_space=gym.spaces.Box(-1.0, 1.0, shape=(7,), dtype=np.float32), + ) + self.env = ( + _make_vla_env( + num_envs=num_slots, + simulator_device=config["env"]["simulator_device"], + ) + if env is None + else env + ) + self._episode_seeds = [0] * num_slots + self._success = np.zeros(num_slots, dtype=np.bool_) + self._observation, _ = self.env.reset(seed=self._episode_seeds) + + def _reset_slot_observation(self, slot_id: int, *, seed: int | None) -> Observation: + if seed is not None: + self._episode_seeds[slot_id] = int(seed) + self._observation, _ = _reset_seeded_slot( + self.env, + slot_id=slot_id, + seed=self._episode_seeds[slot_id], + ) + self._env_steps[slot_id] = 0 + self._success[slot_id] = False + return self._observation_for_slot(slot_id) + + def _observe_slot_observations( + self, + slot_ids: list[int], + ) -> dict[int, Observation]: + return {slot_id: self._observation_for_slot(slot_id) for slot_id in slot_ids} + + def _step_cores( + self, + slot_ids: list[int], + *, + actions: np.ndarray, + active_mask: np.ndarray, + ) -> Any: + del slot_ids, active_mask + tensor_actions = torch.as_tensor( + actions, + dtype=torch.float32, + device=self.env.unwrapped.device, + ) + self._observation, reward, terminated, truncated, info = self.env.step( + tensor_actions + ) + return reward, terminated, truncated, info + + def _slot_step_outcome(self, state: Any, slot_id: int) -> SlotStepOutcome: + reward, terminated, truncated, info = state + success = bool(_slot_value(info["success"], slot_id)) + self._success[slot_id] |= success + return SlotStepOutcome( + reward=float(_slot_value(reward, slot_id)), + done=bool(_slot_value(terminated, slot_id)), + truncated=bool(_slot_value(truncated, slot_id)), + info={ + "success": success, + "task_metrics": {"success": float(self._success[slot_id])}, + }, + ) + + def _observation_for_slot(self, slot_id: int) -> Observation: + rgb = self._observation["rgb"] + if torch.is_tensor(rgb): + rgb = rgb.detach().cpu().numpy() + rgb = np.asarray(rgb) + views = np.stack( + [ + np.asarray(rgb[slot_id, :, :, :3], dtype=np.uint8), + np.asarray(rgb[slot_id, :, :, 3:6], dtype=np.uint8), + ] + ) + metadata = { + MIKASA_IMAGE_VIEWS_INFO_KEY: views, + MIKASA_STATE_INFO_KEY: self._observation["proprio"][slot_id].detach().cpu().numpy(), + "slot_id": slot_id, + } + if "action_prefix_state_key" in self.config["env"]: + metadata["action_prefix_state_key"] = self.config["env"]["action_prefix_state_key"] + if "returned_action_context" in self.config["env"]: + context = self.config["env"]["returned_action_context"] + metadata["returned_action_context"] = { + **context, + "order": np.asarray(context["order"]), + "low": np.asarray(context["low"], dtype=np.float32), + "high": np.asarray(context["high"], dtype=np.float32), + } + return Observation( + data=None, + env_step=int(self._env_steps[slot_id]), + sim_time_ms=float(self._env_steps[slot_id]) * self._frame_ms, + metadata=metadata, + ) + + def close(self) -> None: + if not self.closed: + self.env.close() + super().close() + + +def _slot_value(value: Any, slot_id: int) -> Any: + if torch.is_tensor(value): + return value.detach().reshape(-1)[slot_id].cpu().item() + return np.asarray(value).reshape(-1)[slot_id].item() + + +def _evaluate(args: argparse.Namespace) -> dict[str, Any]: + env = _make_ppo_env() + policy = _PpoPolicy(env, args.checkpoint) + seeds = [] + successes = [] + returns = [] + lengths = [] + try: + for episode_index in range(args.episodes): + seed = START_SEED + episode_index + observation, _ = env.reset(seed=seed) + success_once = False + episode_return = 0.0 + for step in range(60): + action = policy.forward(observation) + observation, reward, terminated, truncated, info = env.step(action) + success_once = success_once or bool(_scalar(info["success"])) + episode_return += float(_scalar(reward)) + if bool(_scalar(terminated)) or bool(_scalar(truncated)): + break + seeds.append(seed) + successes.append(success_once) + returns.append(episode_return) + lengths.append(step + 1) + finally: + env.close() + summary = { + "seeds": seeds, + "successes": successes, + "success_rate": float(np.mean(successes)), + "returns": returns, + "lengths": lengths, + } + (args.output_dir / "summary.json").write_text( + json.dumps(summary, indent=2) + "\n", encoding="utf-8" + ) + return summary + + +def _latency_eval(argv: list[str]) -> None: + from latency_bench.core.config import load_config + from latency_bench.eval.config import apply_runtime_overrides + from latency_bench.eval.driver import run_from_config + + parser = argparse.ArgumentParser() + parser.add_argument("--eval-config", type=Path, required=True) + parser.add_argument("--checkpoint-path", type=Path) + parser.add_argument("--model-config-path", type=Path) + parser.add_argument("--task-contract-path", type=Path) + parser.add_argument("--run-name") + parser.add_argument("--output-dir", type=Path) + parser.add_argument("--latency-method", choices=("zero", "temporal")) + parser.add_argument("--profile-path", type=Path) + parser.add_argument("--latency-seed", type=int) + args = parser.parse_args(argv) + config = load_config(args.eval_config) + apply_runtime_overrides( + config, + checkpoint_path=args.checkpoint_path, + model_config_path=args.model_config_path, + task_contract_path=args.task_contract_path, + run_name=args.run_name, + output_dir=args.output_dir, + latency_method=args.latency_method, + latency_profile_path=args.profile_path, + latency_seed=args.latency_seed, + ) + output_dir = Path(config["logging"]["output_dir"]) + output_dir.mkdir(parents=True, exist_ok=True) + (output_dir / "eval_config.yaml").write_text( + yaml.safe_dump(config, sort_keys=False), encoding="utf-8" + ) + backend = MikasaEnvStepBackend( + config=config, + num_slots=int(config["evaluation"]["eval_parallel_envs"]), + ) + run_from_config( + config, + env_backend=backend, + inference_devices=config["executor"]["inference_devices"], + ) + + +def main() -> None: + if sys.argv[1:2] == ["latency-eval"]: + _latency_eval(sys.argv[2:]) + return + + parser = argparse.ArgumentParser() + parser.add_argument("--policy", choices=("ppo",), required=True) + parser.add_argument("--checkpoint", type=Path) + parser.add_argument("--episodes", type=int, default=50) + parser.add_argument("--output-dir", type=Path, required=True) + args = parser.parse_args() + args.output_dir.mkdir(parents=True, exist_ok=True) + + _evaluate(args) + + +if __name__ == "__main__": + main() diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py new file mode 100644 index 0000000000000000000000000000000000000000..7b9cab3b3fad77d6f6a3b7919ff4a8124c98d011 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla.py @@ -0,0 +1,1207 @@ +from __future__ import annotations + +import json +import sys +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any + +import numpy as np +import numpy.typing as npt +from PIL import Image + +from latency_bench.core.actions import ActionResolver +from latency_bench.core.clock import EnvClock +from latency_bench.core.timing import current_profiler +from latency_bench.core.types import Action, Observation, PolicyOutput +from latency_bench.data.ghost_trail import GhostTrailConfig, build_flappy_ghost_trail_window +from latency_bench.data.state_normalization import min_max_normalize_state +from latency_bench.envs.raw_rgb import ENV_RAW_RGB_FRAME_STACK_INFO_KEY +from latency_bench.envs.gymnasium_task import ( + gymnasium_action_space_contract, + gymnasium_task_contract, +) +from latency_bench.policy.base import PolicyRunner +from latency_bench.policy.starvla_prompts import load_latency_prompt_map, resolve_starvla_prompt + +from latency_bench.utils.paths import REPO_ROOT + + +STARVLA_ROOT = REPO_ROOT / "third_party" / "starVLA" +STATEFUL_STARVLA_MODEL_IDS: tuple[str, ...] = ( + "pi0", + "pi-0", + "pi05", + "pi-0.5", + "gr00t", + "qwenpi", + "qwenpi_v3", + "qwengr00t", +) +STATELESS_STARVLA_MODEL_IDS: tuple[str, ...] = ( + "openvla", + "qwenoft", +) + +DEMON_ATTACK_ACTION_LABELS: tuple[str, ...] = ( + "NOOP", + "FIRE", + "RIGHT", + "LEFT", + "RIGHTFIRE", + "LEFTFIRE", +) +DEADLY_CORRIDOR_TURN_LABELS: tuple[str, ...] = ( + "TURN_NOOP", + "TURN_LEFT", + "TURN_RIGHT", +) +DEADLY_CORRIDOR_MOVE_LABELS: tuple[str, ...] = ( + "MOVE_NOOP", + "MOVE_FORWARD", + "MOVE_BACKWARD", +) +DEADLY_CORRIDOR_STRAFE_LABELS: tuple[str, ...] = ( + "STRAFE_NOOP", + "MOVE_LEFT", + "MOVE_RIGHT", +) +DEADLY_CORRIDOR_ATTACK_LABELS: tuple[str, ...] = ( + "ATTACK_NOOP", + "ATTACK", +) + + +class StarVlaPolicyRunner(PolicyRunner): + """Translate observations and model outputs using the task action contract.""" + + def __init__( + self, + *, + wrapper: Any, + checkpoint_path: str, + device: str, + unnorm_key: str | None, + env_name: str, + action_resolver: ActionResolver, + action_refs: Sequence[Any], + latency_prompt_map: dict[str, Any] | None = None, + base_prompt: str | None = None, + latency_prompt_key: int | str | None = None, + prompt_mode: str | None = None, + obs_resize: tuple[int, int] | None = None, + image_transform_config: Mapping[str, Any] | None = None, + observation_stride_raw_frames: int, + model_cfg: Mapping[str, Any] | None = None, + state_normalization: Mapping[str, Any] | None = None, + state_source: str | None = None, + image_views_info_key: str | None = None, + action_output_type: str | None = None, + ) -> None: + self._wrapper = wrapper + self._obs_resize = tuple(obs_resize) if obs_resize else None + self._checkpoint_path = checkpoint_path + self._device = device + self._unnorm_key = unnorm_key + self._env_name = env_name + self._action_by_raw_id = { + raw_action_id: action_resolver.resolve(action_ref) + for raw_action_id, action_ref in enumerate(action_refs) + } + self._latency_prompt_map = latency_prompt_map + self._base_prompt = base_prompt + self._latency_prompt_key = latency_prompt_key + self._prompt_mode = str(prompt_mode or "default").strip().lower() + self._image_transform_config = dict(image_transform_config or {"image_transform": "raw_rgb"}) + self._image_transform = str( + self._image_transform_config.get("image_transform", "raw_rgb") or "raw_rgb" + ).strip().lower() + model_cfg = ( + _normalized_model_cfg_from_wrapper(wrapper) + if model_cfg is None + else _normalized_model_cfg(model_cfg) + ) + self._include_state = _include_state_from_model_cfg(model_cfg) + self._state_dim = _state_dim_from_model_cfg(model_cfg) if self._include_state else None + self._state_normalization = dict(state_normalization or {}) + self._state_source = state_source + self._image_views_info_key = image_views_info_key + self._action_output_type = action_output_type + vla_data = (model_cfg.get("datasets", {}) or {}).get("vla_data", {}) or {} + self._pack_image_sequence = ( + bool(vla_data["pack_image_sequence"]) + if "pack_image_sequence" in vla_data + else False + ) + self._image_sequence_length = ( + int(vla_data["image_sequence_length"]) + if self._pack_image_sequence + else 1 + ) + self._observation_stride_raw_frames = int(observation_stride_raw_frames) + self._image_sequence_raw_span = ( + 1 + + (self._image_sequence_length - 1) + * self._observation_stride_raw_frames + ) + self._num_obs_frames = int(vla_data.get("num_obs_frames", 1) or 1) + self._image_mode = str(vla_data.get("image_mode", "single")) + self._stitch_grid = tuple(vla_data.get("stitch_grid", [2, 2])) + framework_cfg = model_cfg["framework"] + kv_cfg = framework_cfg["kv_memory"] if "kv_memory" in framework_cfg else {} + self._kv_memory_enabled = bool(kv_cfg["enabled"]) if "enabled" in kv_cfg else False + + def reset_state(self, slot_id: int | None = None) -> None: + # Clears the model's per-slot KV memory at episode boundaries (req3). + # No-op unless the framework maintains KV memory. + reset = getattr(self._wrapper, "reset_memory", None) + if callable(reset): + reset(slot_id) + + def predict(self, observation: Observation) -> PolicyOutput: + return self.predict_batch([observation])[0] + + def predict_batch(self, observations: Sequence[Observation]) -> list[PolicyOutput]: + profiler = current_profiler() + with profiler.time("policy_build_example_ms"): + examples = [self._build_example(observation) for observation in observations] + with profiler.time("policy_wrapper_predict_action_ms"): + prediction = self._wrapper.predict_action( + examples=examples, unnorm_key=self._unnorm_key, profiler=profiler + ) + with profiler.time("policy_decode_ms"): + outputs = [ + self._decode_prediction( + prediction=prediction, + index=index, + observation=observation, + example=example, + ) + for index, (observation, example) in enumerate(zip(observations, examples)) + ] + return outputs + + def _decode_prediction( + self, + *, + prediction: dict[str, Any], + index: int, + observation: Observation, + example: dict[str, Any], + ) -> PolicyOutput: + actions = np.asarray(prediction["actions"]) + raw_action_scores = ( + np.asarray(prediction["raw_action_scores"]) + if "raw_action_scores" in prediction + else None + ) + return self._policy_output( + observation=observation, + example=example, + action_payload=actions[index, 0], + action_output_type=( + prediction["action_output_type"] + if self._action_output_type is None + else self._action_output_type + ), + raw_action_scores=None if raw_action_scores is None else raw_action_scores[index, 0], + ) + + def _build_example(self, observation: Observation) -> dict[str, Any]: + frame_source = observation.metadata[ + ENV_RAW_RGB_FRAME_STACK_INFO_KEY + if self._image_views_info_key is None + else self._image_views_info_key + ] + frames = observation_data_to_hwc_uint8_frames(frame_source) # oldest .. newest + transformed = self._transformed_frame(frames=frames, observation=observation) + + if self._image_views_info_key is not None: + pass + elif self._pack_image_sequence: + if transformed is not None: + raise ValueError( + "WanOFT packed image sequences require image_transform=raw_rgb" + ) + if len(frames) < self._image_sequence_raw_span: + raise ValueError( + "WanOFT packed image sequence requires " + f"{self._image_sequence_raw_span} raw frames for " + f"{self._image_sequence_length} decision observations at stride " + f"{self._observation_stride_raw_frames}, got {len(frames)}" + ) + frames = frames[ + -self._image_sequence_raw_span + :: self._observation_stride_raw_frames + ] + elif transformed is not None: + frames = [transformed] + elif self._image_mode == "single" or self._kv_memory_enabled: + frames = frames[-1:] + else: + # Select the temporal observation window to match training (_pack_sample). + raw_span = 1 + (self._num_obs_frames - 1) * self._observation_stride_raw_frames + frames = frames[-raw_span :: self._observation_stride_raw_frames] + + prompt = resolve_starvla_prompt( + env_name=self._env_name, + observation_metadata=observation.metadata, + latency_prompt_map=self._latency_prompt_map, + base_prompt=self._base_prompt, + latency_prompt_key=self._latency_prompt_key, + prompt_mode=self._prompt_mode, + ) + + if self._image_mode == "stitch": + if transformed is not None: + raise ValueError("image_transform is not compatible with image_mode=stitch") + # Tile the window into one image; matches _pack_sample's stitch branch + # (raw frames passed to stitch_frames, which resizes each cell to 224). + images = [_get_stitch_frames()(frames, grid=self._stitch_grid, size=(224, 224))] + else: + if self._obs_resize is not None: + height, width = self._obs_resize + # match training preprocessing exactly: gr00t LeRobotSingleDataset._pack_sample + # does `Image.fromarray(image).resize((224, 224))` (PIL default resample = BICUBIC). + frames = [ + np.asarray(Image.fromarray(frame).resize((width, height)), dtype=np.uint8) + for frame in frames + ] + images = [Image.fromarray(frame) for frame in frames] + + example = { + "image": images, + "lang": prompt, + } + if self._kv_memory_enabled: + example["slot_id"] = observation.metadata["slot_id"] + elif "slot_id" in observation.metadata: + example["slot_id"] = observation.metadata["slot_id"] + if self._include_state: + if self._state_source == "transport": + state = np.asarray(observation.data["transport"], dtype=np.float32) + example["state"] = state.reshape(1, self._state_dim) + elif self._state_normalization: + state = np.asarray( + observation.metadata["gymnasium_state"], dtype=np.float32 + ) + state_min = np.asarray(self._state_normalization["min"], dtype=np.float32) + state_max = np.asarray(self._state_normalization["max"], dtype=np.float32) + state = min_max_normalize_state(state, state_min, state_max) + example["state"] = state.reshape(1, self._state_dim) + else: + example["state"] = np.zeros((1, self._state_dim), dtype=np.float32) + return example + + def _transformed_frame( + self, + *, + frames: Sequence[npt.NDArray[np.uint8]], + observation: Observation, + ) -> npt.NDArray[np.uint8] | None: + if self._image_transform in {"", "none", "raw", "raw_rgb"}: + return None + if self._image_transform not in {"flappy_ghost_trail", "demon_attack_ghost_trail"}: + raise ValueError(f"Unsupported StarVLA image_transform={self._image_transform!r}") + if self._image_transform == "flappy_ghost_trail" and self._env_name != "flappy": + raise ValueError("image_transform=flappy_ghost_trail is only supported for env_name=flappy") + if self._image_transform == "demon_attack_ghost_trail" and self._env_name != "demon_attack": + raise ValueError("image_transform=demon_attack_ghost_trail is only supported for env_name=demon_attack") + + config = GhostTrailConfig( + image_transform=self._image_transform, + history_frames=int(self._image_transform_config.get("history_frames", 5)), + gamma=float(self._image_transform_config.get("gamma", 1.3)), + min_alpha=int(self._image_transform_config.get("min_alpha", 35)), + ground_fraction=float(self._image_transform_config.get("ground_fraction", 0.22)), + scroll_px_per_step=float(self._image_transform_config.get("scroll_px_per_step", 4.0)), + ) + if self._image_transform == "demon_attack_ghost_trail": + # env_step counts raw ALE frames (buffer updated 4× per decision step). + # frames[-0:] == frames, so env_step=0 falls back to the full reset-fill buffer. + valid_count = min(len(frames), int(observation.env_step)) + else: + max_frames = max(1, int(config.history_frames) + 1) + valid_count = min(len(frames), max(1, int(observation.env_step) + 1), max_frames) + window = [np.asarray(frame, dtype=np.uint8) for frame in frames[-valid_count:]] + + if self._image_transform == "demon_attack_ghost_trail": + from latency_bench.data.ghost_trail_demon import build_demon_attack_ghost_trail_window + steps_arg = list(range(len(window))) + return build_demon_attack_ghost_trail_window(window, steps_arg, config=config) + + current_step = int(observation.env_step) + start_step = current_step - valid_count + 1 + steps = list(range(start_step, current_step + 1)) + return build_flappy_ghost_trail_window(window, steps, config=config) + + def _policy_output( + self, + *, + observation: Observation, + example: dict[str, Any], + action_payload: npt.NDArray[Any], + action_output_type: str, + raw_action_scores: npt.NDArray[Any] | None, + ) -> PolicyOutput: + payload = np.asarray(action_payload) + action, action_metadata = action_from_starvla_payload( + payload=payload, + env_name=self._env_name, + action_by_raw_id=self._action_by_raw_id, + action_output_type=action_output_type, + ) + metadata = { + "policy_type": "starvla", + "prompt_source": "latency_prompt_map" if self._latency_prompt_map is not None else "base", + "checkpoint_path": self._checkpoint_path, + "unnorm_key": self._unnorm_key, + "device": self._device, + "input_frame_count": len(example["image"]), + "image_transform": self._image_transform, + "action_output_type": action_output_type, + "action_payload": to_jsonable_action_payload(payload), + "kv_memory_enabled": self._kv_memory_enabled, + **action_metadata, + } + if self._pack_image_sequence: + metadata["image_sequence_length"] = self._image_sequence_length + metadata["input_frame_raw_stride"] = self._observation_stride_raw_frames + metadata["input_frame_raw_span"] = self._image_sequence_raw_span + if "slot_id" in example: + metadata["slot_id"] = example["slot_id"] + if raw_action_scores is not None: + metadata["raw_action_scores"] = [ + float(item) for item in np.asarray(raw_action_scores, dtype=np.float32).tolist() + ] + if "latency_raw_frames" in observation.metadata: + metadata["latency_raw_frames"] = observation.metadata["latency_raw_frames"] + if "latency_ms" in observation.metadata: + metadata["latency_ms"] = observation.metadata["latency_ms"] + if self._latency_prompt_key is not None: + metadata["latency_prompt_key"] = self._latency_prompt_key + return PolicyOutput( + action=action, + raw_output=metadata["action_payload"], + metadata=metadata, + ) + + +def observation_data_to_hwc_uint8_frames(data: Any) -> list[npt.NDArray[np.uint8]]: + frame = _extract_observation_array(data) + if frame.ndim == 4 and frame.shape[-1] == 3: + return [_as_uint8_image(item) for item in frame] + if frame.ndim == 4 and frame.shape[1] == 3: + return [_as_uint8_image(np.transpose(item, (1, 2, 0))) for item in frame] + if frame.ndim == 3 and frame.shape[-1] == 3: + return [_as_uint8_image(frame)] + if frame.ndim == 3 and frame.shape[0] == 3: + return [_as_uint8_image(np.transpose(frame, (1, 2, 0)))] + if ( + frame.ndim == 3 + and frame.shape[0] % 3 == 0 + and frame.shape[0] < frame.shape[1] + and frame.shape[0] < frame.shape[2] + ): + return [ + _as_uint8_image(np.transpose(frame[start : start + 3, :, :], (1, 2, 0))) + for start in range(0, frame.shape[0], 3) + ] + if frame.ndim == 3 and frame.shape[-1] % 3 == 0: + return [ + _as_uint8_image(frame[:, :, start : start + 3]) + for start in range(0, frame.shape[-1], 3) + ] + if frame.ndim == 3 and frame.shape[0] % 3 == 0: + return [ + _as_uint8_image(np.transpose(frame[start : start + 3, :, :], (1, 2, 0))) + for start in range(0, frame.shape[0], 3) + ] + return [_as_uint8_image(frame)] + + +def decode_starvla_action( + *, + vector: npt.NDArray[Any], + env_name: str, + action_by_raw_id: Mapping[int, Action], + action_layout: str | None = None, +) -> tuple[Action, dict[str, Any]]: + deadly_layout = None + if str(env_name) == "deadly_corridor": + action_dim = int(np.asarray(vector).shape[-1]) + deadly_layouts = { + 7: "deadly_corridor_semantic_7", + 11: "deadly_corridor_factorized_11", + 54: "deadly_corridor_joint_54", + } + if action_dim not in deadly_layouts: + raise ValueError( + "Deadly Corridor StarVLA action vector expected 7, 11, or 54 " + f"values, got {action_dim}" + ) + deadly_layout = deadly_layouts[action_dim] + asterix_layout = None + if str(env_name) == "asterix": + action_dim = int(np.asarray(vector).shape[-1]) + if action_layout is not None: + asterix_layout = str(action_layout).strip().lower() + else: + asterix_layout = "factorized_6" if action_dim < 9 else "discrete_9" + + decode_rl_games_actions, _, _ = _load_rl_games_action_decode() + prediction = decode_rl_games_actions( + normalized_actions=np.asarray(vector), + env_name=str(env_name), + deadly_action_layout=(deadly_layout.removeprefix("deadly_corridor_") if deadly_layout is not None else None), + asterix_action_layout=asterix_layout, + ) + action, metadata = action_from_starvla_payload( + payload=np.asarray(prediction["actions"]), + env_name=env_name, + action_by_raw_id=action_by_raw_id, + action_output_type=prediction["action_output_type"], + ) + if deadly_layout is not None: + metadata["action_layout"] = deadly_layout + if deadly_layout == "deadly_corridor_joint_54": + turn, move, strafe, attack = action.value + metadata["raw_action_id"] = turn * 18 + move * 6 + strafe * 2 + attack + elif deadly_layout == "deadly_corridor_semantic_7": + semantic_actions = ( + [0, 1, 0, 0], + [0, 2, 0, 0], + [0, 0, 1, 0], + [0, 0, 2, 0], + [1, 0, 0, 0], + [2, 0, 0, 0], + [0, 0, 0, 1], + ) + metadata["raw_action_id"] = semantic_actions.index(action.value) + if asterix_layout is not None: + metadata["action_layout"] = asterix_layout + return action, metadata + + +# Fixed semantic button order the StarVLA multibinary head is trained against. +# Mirrors starVLA.training.rl_games.eval_core._semantic_to_runtime_multibinary; +# the env adapter re-orders this to the live ViZDoom button layout. +DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER = ( + "MOVE_FORWARD", + "MOVE_BACKWARD", + "MOVE_LEFT", + "MOVE_RIGHT", + "TURN_LEFT", + "TURN_RIGHT", + "ATTACK", +) + + +def action_from_starvla_payload( + *, + payload: npt.NDArray[Any], + env_name: str, + action_by_raw_id: Mapping[int, Action], + action_output_type: str = "", +) -> tuple[Action, dict[str, Any]]: + if str(action_output_type) == "rl_games_continuous": + values = [float(item) for item in np.asarray(payload).reshape(-1).tolist()] + return Action( + value=values, + name="continuous_torque", + is_noop=all(value == 0.0 for value in values), + is_oneshot=False, + ), {"continuous_action": values} + if str(env_name) == "demon_attack": + return demon_attack_action_from_id(int(np.asarray(payload).reshape(-1)[0])) + if str(env_name) == "deadly_corridor": + # Multibinary heads emit an already-thresholded 7-dim button vector in + # fixed semantic order; the env adapter re-orders it to the live ViZDoom + # button layout. Keep it as-is rather than reinterpreting it as a + # [turn, move, strafe, attack] categorical tuple. + if str(action_output_type) == "rl_games_deadly_corridor_multibinary": + buttons = [int(item) for item in np.asarray(payload).reshape(-1).tolist()] + active = [ + DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER[idx] + for idx, pressed in enumerate(buttons) + if idx < len(DEADLY_CORRIDOR_SEMANTIC_BUTTON_ORDER) and pressed + ] + action_name = "+".join(active) if active else "NOOP" + return Action( + value=buttons, + name=action_name, + is_noop=not any(buttons), + is_oneshot=False, + ), { + "decoded_multibinary_buttons": buttons, + "action_label": action_name, + "action_layout": "deadly_corridor_multibinary_7", + } + return deadly_corridor_action_from_tuple( + action_value=[int(item) for item in np.asarray(payload).reshape(-1).tolist()], + metadata={"action_layout": "deadly_corridor_tuple"}, + ) + raw_action_id = int(np.asarray(payload).reshape(-1)[0]) + return action_by_raw_id[raw_action_id], {"raw_action_id": raw_action_id} + + +def to_jsonable_action_payload(payload: npt.NDArray[Any]) -> Any: + value = np.asarray(payload).tolist() + if isinstance(value, list) and len(value) == 1: + return value[0] + return value + + +def demon_attack_action_from_id(action_id: int) -> tuple[Action, dict[str, Any]]: + action = Action( + value=action_id, + name=DEMON_ATTACK_ACTION_LABELS[action_id], + is_noop=action_id == 0, + is_oneshot=False, + ) + return action, {"raw_action_id": action_id, "action_label": action.name} + + +def deadly_corridor_action_from_tuple( + *, + action_value: list[int], + metadata: dict[str, Any], +) -> tuple[Action, dict[str, Any]]: + turn, move, strafe, attack = action_value + action_value = [turn, move, strafe, attack] + turn_label = DEADLY_CORRIDOR_TURN_LABELS[turn] + move_label = DEADLY_CORRIDOR_MOVE_LABELS[move] + strafe_label = DEADLY_CORRIDOR_STRAFE_LABELS[strafe] + attack_label = DEADLY_CORRIDOR_ATTACK_LABELS[attack] + active_labels = [ + label + for label in (turn_label, move_label, strafe_label, attack_label) + if not label.endswith("_NOOP") + ] + action_name = "+".join(active_labels) if active_labels else "NOOP" + return Action( + value=action_value, + name=action_name, + is_noop=action_value == [0, 0, 0, 0], + is_oneshot=False, + ), { + "decoded_action_tuple": action_value, + "turn_label": turn_label, + "move_label": move_label, + "strafe_label": strafe_label, + "attack_label": attack_label, + "action_label": action_name, + **metadata, + } + + +def _extract_observation_array(data: Any) -> npt.NDArray[Any]: + if isinstance(data, Mapping): + return np.asarray(data["observation"]) + return np.asarray(data) + + +def _as_uint8_image(frame: npt.NDArray[Any]) -> npt.NDArray[np.uint8]: + return np.ascontiguousarray(frame, dtype=np.uint8) + + +def _normalized_model_cfg(model_cfg: Mapping[str, Any]) -> dict[str, Any]: + _ensure_starvla_path() + from omegaconf import OmegaConf + from starVLA.model.framework.share_tools import apply_config_compat + + cfg = OmegaConf.create(model_cfg) + apply_config_compat(cfg) + _apply_model_family_include_state_compat(cfg) + return OmegaConf.to_container(cfg, resolve=True) + + +def _normalized_model_cfg_from_wrapper(wrapper: Any) -> dict[str, Any]: + return _normalized_model_cfg(wrapper._model_cfg) + + +def _load_starvla_model_config(path: str | Path) -> dict[str, Any]: + from omegaconf import OmegaConf + + return _normalized_model_cfg(OmegaConf.load(path)) + + +def _apply_model_family_include_state_compat(cfg: Any) -> None: + from omegaconf import OmegaConf + + if OmegaConf.select(cfg, "datasets.vla_data.include_state") is not None: + return + + model_ids = ( + _normalized_optional_config_string(cfg, ("model",)), + _normalized_optional_config_string(cfg, ("rl_games", "model_alias")), + _normalized_optional_config_string(cfg, ("framework", "name")), + ) + if any(model_id in STATEFUL_STARVLA_MODEL_IDS for model_id in model_ids if model_id is not None): + OmegaConf.update(cfg, "datasets.vla_data.include_state", True, force_add=True) + return + if any(model_id in STATELESS_STARVLA_MODEL_IDS for model_id in model_ids if model_id is not None): + OmegaConf.update(cfg, "datasets.vla_data.include_state", False, force_add=True) + + +def _normalized_optional_config_string(cfg: Any, path: tuple[str, ...]) -> str | None: + from omegaconf import OmegaConf + + value = OmegaConf.select(cfg, ".".join(path)) + if value is None: + return None + return str(value).strip().lower() + + +def _state_dim_from_model_cfg(model_cfg: dict[str, Any]) -> int: + return model_cfg["framework"]["action_model"]["state_dim"] + + +def _include_state_from_model_cfg(model_cfg: dict[str, Any]) -> bool: + return model_cfg["datasets"]["vla_data"]["include_state"] + + +_STITCH_FRAMES = None + + +def _get_stitch_frames(): + """Lazily import starVLA's stitch_frames (starVLA path is added at runtime).""" + global _STITCH_FRAMES + if _STITCH_FRAMES is None: + _ensure_starvla_path() + from starVLA.training.trainer_utils.trainer_tools import stitch_frames + + _STITCH_FRAMES = stitch_frames + return _STITCH_FRAMES + + +def _ensure_starvla_path() -> None: + starvla_root = str(STARVLA_ROOT) + if starvla_root not in sys.path: + sys.path.insert(0, starvla_root) + + +def _observation_stride_raw_frames(config: Mapping[str, Any]) -> int: + env_cfg = config["env"] + return EnvClock( + env_fps=float(env_cfg["env_fps"]), + obs_fps=float(env_cfg["obs_fps"]), + ).obs_stride_raw_frames + + +def apply_starvla_model_input_config( + config: dict[str, Any], + *, + model_cfg: Mapping[str, Any], + image_transform: str = "raw_rgb", +) -> None: + """Match latency_bench's raw frame stack to a saved StarVLA input contract.""" + vla_data = model_cfg["datasets"]["vla_data"] + pack_image_sequence = ( + bool(vla_data["pack_image_sequence"]) + if "pack_image_sequence" in vla_data + else False + ) + normalized_transform = str(image_transform).strip().lower() + raw_image_transform = normalized_transform in {"", "none", "raw", "raw_rgb"} + if pack_image_sequence: + if not raw_image_transform: + raise ValueError( + "WanOFT packed image sequences require image_transform=raw_rgb" + ) + input_frame_count = int(vla_data["image_sequence_length"]) + else: + if not raw_image_transform: + return + framework_cfg = model_cfg["framework"] + kv_cfg = framework_cfg["kv_memory"] if "kv_memory" in framework_cfg else {} + kv_memory_enabled = bool(kv_cfg["enabled"]) if "enabled" in kv_cfg else False + if kv_memory_enabled: + return + image_mode = str(vla_data["image_mode"]) if "image_mode" in vla_data else "single" + if image_mode == "single": + return + input_frame_count = int(vla_data["num_obs_frames"]) + + observation_stride = _observation_stride_raw_frames(config) + required_raw_frames = 1 + (input_frame_count - 1) * observation_stride + config["env"]["frame_stack"] = max( + int(config["env"]["frame_stack"]), + required_raw_frames, + ) + + +def prepare_starvla_checkpoint_input_config(config: dict[str, Any]) -> None: + """Apply the saved checkpoint input contract before env construction.""" + if config["policy"]["type"] != "starvla": + return + + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + if config["env"]["name"] == "gymnasium": + contract = json.loads( + Path(policy_cfg["task_contract_path"]).read_text(encoding="utf-8") + ) + config["env"]["state_space"] = {"labels": contract["state_labels"]} + if contract["robot_type"] in ("latency_balance_profile_h8", "latency_balance_profile_h16"): + config["env"]["name"] = "balance_profile" + config["env"]["action_context_horizon"] = contract["action_horizon"] + config["env"]["frame_stack"] = 1 + return + if "model_config_path" in policy_cfg: + model_cfg = _load_starvla_model_config(policy_cfg["model_config_path"]) + else: + _ensure_starvla_path() + from starVLA.model.framework.share_tools import read_mode_config + + saved_model_cfg, _norm_stats = read_mode_config(policy_cfg["checkpoint_path"]) + model_cfg = _normalized_model_cfg(saved_model_cfg) + if config["env"]["name"] == "gymnasium": + image_size = model_cfg["rl_games"]["env_eval"]["image_size"] + config["env"]["obs_resize"] = [image_size, image_size] + image_transform_cfg = ( + policy_cfg["image_transform_config"] + if "image_transform_config" in policy_cfg + else {} + ) + image_transform = ( + image_transform_cfg["image_transform"] + if "image_transform" in image_transform_cfg + else "raw_rgb" + ) + apply_starvla_model_input_config( + config, + model_cfg=model_cfg, + image_transform=image_transform, + ) + + +def _load_policy_wrapper_class() -> Any: + _ensure_starvla_path() + from deployment.model_server.policy_wrapper import PolicyServerWrapper + + return PolicyServerWrapper + + +def _profiler_stage(profiler: Any, name: str) -> Any: + from contextlib import nullcontext + + return profiler.time(name) if profiler is not None else nullcontext() + + +def _load_rl_games_action_decode() -> tuple[Any, Any, Any]: + _ensure_starvla_path() + from deployment.model_server.rl_games_action_decode import ( + decode_rl_games_actions, + resolve_asterix_action_decode_spec, + resolve_deadly_action_decode_spec, + ) + + return decode_rl_games_actions, resolve_deadly_action_decode_spec, resolve_asterix_action_decode_spec + + +class LiveStarVlaWrapper: + """In-process stand-in for ``PolicyServerWrapper`` over a *live* framework. + + During training the trainer already holds the model in memory + (``accelerator.unwrap_model(self.model)`` — the same object eval_core calls). + This wrapper exposes only the rl_games-mode surface ``StarVlaPolicyRunner`` + uses — ``predict_action`` (framework forward + rl_games decode), + ``reset_memory`` passthrough, and the ``_model_cfg`` attribute — so no + checkpoint reload is needed. The disk-backed ``PolicyNormProcessor`` is never + built because rl_games decoding ignores un-normalization stats. + """ + + def __init__( + self, + *, + framework: Any, + model_cfg: dict[str, Any], + env_name: str, + rl_games_action_env_dim: int | None = None, + gymnasium_action_space_type: str = "discrete", + action_layout: str | None = None, + multibinary_threshold: float | None = None, + ) -> None: + self._framework = framework + self._model_cfg = model_cfg + self._rl_games_env_name = str(env_name) + self._rl_games_action_env_dim = rl_games_action_env_dim + self._gymnasium_action_space_type = gymnasium_action_space_type + ( + self._decode_rl_games_actions, + resolve_deadly_action_decode_spec, + resolve_asterix_action_decode_spec, + ) = _load_rl_games_action_decode() + self._action_layout = action_layout + self._multibinary_threshold = multibinary_threshold + if self._rl_games_env_name == "deadly_corridor": + self._action_layout, self._multibinary_threshold = resolve_deadly_action_decode_spec( + model_cfg, + action_layout=action_layout, + multibinary_threshold=multibinary_threshold, + ) + elif self._rl_games_env_name == "asterix": + self._action_layout = resolve_asterix_action_decode_spec( + model_cfg, + action_layout=action_layout, + ) + + def reset_memory(self, slot_id: int | None = None) -> None: + reset = getattr(self._framework, "reset_memory", None) + if callable(reset): + reset(slot_id) + + def predict_action( + self, + examples: list[dict[str, Any]], + unnorm_key: str | None = None, + **kwargs: Any, + ) -> dict[str, Any]: + # unnorm_key is unused in rl_games mode; kept for interface parity. + del unnorm_key + profiler = kwargs["profiler"] if "profiler" in kwargs else None + out = self._framework.predict_action(examples=examples, **kwargs) + normalized = np.asarray(out["normalized_actions"]) # (B, T, D) + decode_kwargs: dict[str, Any] = {} + if self._rl_games_env_name == "gymnasium": + decode_kwargs["action_env_dim"] = self._rl_games_action_env_dim + if self._gymnasium_action_space_type == "box": + decode_kwargs["gymnasium_action_space_type"] = "box" + with _profiler_stage(profiler, "starvla_wrapper_rl_games_decode_ms"): + return self._decode_rl_games_actions( + normalized_actions=normalized, + env_name=self._rl_games_env_name, + deadly_action_layout=( + self._action_layout + if self._rl_games_env_name == "deadly_corridor" + else None + ), + deadly_multibinary_threshold=( + self._multibinary_threshold + if self._rl_games_env_name == "deadly_corridor" + else None + ), + asterix_action_layout=( + self._action_layout + if self._rl_games_env_name == "asterix" + else None + ), + **decode_kwargs, + ) + + +_LEGACY_GYMNASIUM_TASK_NAMES = { + "ant_rgb_state": "ant", + "half_cheetah_rgb_state": "half_cheetah", + "hopper_rgb_state": "hopper", + "humanoid_rgb_state": "humanoid", + "inverted_pendulum_rgb_state": "inverted_pendulum", + "swimmer_rgb_state": "swimmer", + "walker2d_rgb_state": "walker2d", +} + + +def _canonical_gymnasium_contract_namespace( + contract: Mapping[str, Any], +) -> dict[str, Any]: + canonical = dict(contract) + task_name = canonical["task_name"] + if task_name in _LEGACY_GYMNASIUM_TASK_NAMES: + canonical["task_name"] = _LEGACY_GYMNASIUM_TASK_NAMES[task_name] + if canonical["env_id"] == "LatencyBench/HopperRgbState-v0": + canonical["env_id"] = "LatencyBench/Hopper-v0" + canonical["registration_imports"] = [ + "latency_bench.envs.gymnasium_hopper" + if module == "latency_bench.envs.gymnasium_hopper_rgb_state" + else module + for module in canonical["registration_imports"] + ] + return canonical + + +def _validate_gymnasium_starvla_contract( + *, + env_cfg: Mapping[str, Any], + policy_cfg: Mapping[str, Any], + model_cfg: Mapping[str, Any], + manifest: Mapping[str, Any], +) -> None: + eval_contract = gymnasium_task_contract(env_cfg) + manifest_task = manifest.get("gymnasium_task") + expected = policy_cfg.get( + "gymnasium_training_task_contract", manifest_task or eval_contract + ) + comparable_eval_contract = {**eval_contract, "make_kwargs": expected["make_kwargs"]} + if _canonical_gymnasium_contract_namespace( + comparable_eval_contract + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA training contract or dataset manifest" + ) + if manifest.get("integration_name", "gymnasium") != "gymnasium": + raise ValueError("StarVLA task manifest is not a Gymnasium handoff") + if manifest_task is not None: + if _canonical_gymnasium_contract_namespace( + manifest_task + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA dataset manifest" + ) + model_contract = model_cfg["datasets"]["vla_data"].get("gymnasium_task_contract") + if model_contract is not None: + if _canonical_gymnasium_contract_namespace( + model_contract + ) != _canonical_gymnasium_contract_namespace(expected): + raise ValueError( + "Evaluation Gymnasium task contract does not match the StarVLA model config" + ) + action_space = gymnasium_action_space_contract(env_cfg) + action_layout = str(policy_cfg.get("action_layout", "") or "").strip().lower() + is_asterix_factorized = ( + str(env_cfg.get("task_name", "")) == "asterix" + and action_layout in {"factorized_6", "factorized6", "asterix_factorized_6", "asterix_factorized6"} + ) + if not is_asterix_factorized and manifest["active_action_dim"] != len(action_space["labels"]): + raise ValueError( + "StarVLA dataset active_action_dim does not match its Gymnasium action catalog" + ) + if ( + model_cfg["framework"]["action_model"]["action_env_dim"] + != manifest["active_action_dim"] + ): + raise ValueError( + "StarVLA model action_env_dim does not match the dataset manifest" + ) + model_uses_state = bool(model_cfg["datasets"]["vla_data"]["include_state"]) + manifest_has_state_metadata = ( + "uses_state" in manifest or "state_labels" in manifest + ) + manifest_uses_state = bool(manifest.get("uses_state", model_uses_state)) + if manifest_has_state_metadata: + if policy_cfg.get("state_source") != "transport" and manifest_uses_state != ("state_space" in expected): + raise ValueError( + "StarVLA dataset uses_state does not match the Gymnasium state space" + ) + if manifest_uses_state != model_uses_state: + raise ValueError( + "StarVLA dataset uses_state does not match the model include_state" + ) + if manifest_has_state_metadata and manifest_uses_state: + state_labels = manifest["state_labels"] + expected_state_labels = expected["state_space"]["labels"] if policy_cfg.get("state_source") != "transport" else state_labels + if state_labels != expected_state_labels: + raise ValueError( + "StarVLA dataset state_labels do not match the Gymnasium state space" + ) + if manifest["state_dim"] != len(state_labels): + raise ValueError( + "StarVLA dataset state_dim does not match its state_labels" + ) + if ( + model_cfg["framework"]["action_model"]["state_dim"] + != manifest["state_dim"] + ): + raise ValueError( + "StarVLA model state_dim does not match the dataset manifest" + ) + if not manifest["state_normalization"]: + raise ValueError( + "StarVLA state-enabled dataset manifest is missing state_normalization" + ) + + +def _starvla_runner_kwargs( + config: dict[str, Any], + action_resolver: ActionResolver, + model_cfg: Mapping[str, Any] | None, + *, + base_prompt: str | None, +) -> dict[str, Any]: + """Resolve task and input settings shared by checkpoint and resident models.""" + env_cfg = config["env"] + policy_cfg = config["policy"] + if env_cfg["name"] == "gymnasium": + task_manifest = json.loads( + Path(policy_cfg["task_manifest_path"]).read_text(encoding="utf-8") + ) + _validate_gymnasium_starvla_contract( + env_cfg=env_cfg, + policy_cfg=policy_cfg, + model_cfg=model_cfg, + manifest=task_manifest, + ) + semantic_env_name = env_cfg["task_name"] + action_refs = env_cfg.get("action_order", []) + base_prompt = env_cfg["base_prompt"] + state_normalization = task_manifest.get("state_normalization") + else: + semantic_env_name = env_cfg["name"] + action_refs = policy_cfg.get("actions", action_resolver.default_action_refs()) + state_normalization = policy_cfg["state_normalization"] if "state_normalization" in policy_cfg else None + return dict( + unnorm_key=policy_cfg.get("unnorm_key"), + env_name=semantic_env_name, + action_resolver=action_resolver, + action_refs=action_refs, + latency_prompt_map=( + load_latency_prompt_map(policy_cfg["latency_prompt_map_path"]) + if "latency_prompt_map_path" in policy_cfg + else None + ), + base_prompt=base_prompt, + latency_prompt_key=policy_cfg.get("latency_prompt_key"), + prompt_mode=policy_cfg.get("prompt_mode"), + obs_resize=tuple(env_cfg["obs_resize"]) if env_cfg.get("obs_resize") else None, + image_transform_config=policy_cfg.get("image_transform_config"), + observation_stride_raw_frames=_observation_stride_raw_frames(config), + model_cfg=model_cfg, + state_normalization=state_normalization, + state_source=policy_cfg["state_source"] if "state_source" in policy_cfg else None, + ) + + +def build_starvla_policy( + config: dict[str, Any], + action_resolver: ActionResolver, +) -> PolicyRunner: + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + _ensure_starvla_path() + from latency_bench.policy.starvla_tasks import build_task_starvla_policy + + return build_task_starvla_policy(config) + env_cfg = config["env"] + integration_env_name = env_cfg["name"] + model_cfg = ( + _load_starvla_model_config(policy_cfg["model_config_path"]) + if integration_env_name == "gymnasium" or "model_config_path" in policy_cfg + else None + ) + runner_kwargs = _starvla_runner_kwargs( + config, action_resolver, model_cfg, base_prompt=env_cfg.get("base_prompt") + ) + wrapper_cls = _load_policy_wrapper_class() + wrapper_kwargs: dict[str, Any] = dict( + ckpt_path=policy_cfg["checkpoint_path"], + device=policy_cfg["device"], + use_bf16=True, + unnorm_key=runner_kwargs["unnorm_key"], + action_output_mode=( + policy_cfg["action_output_mode"] + if "action_output_mode" in policy_cfg + else "rl_games" + ), + rl_games_env_name=integration_env_name, + rl_games_action_layout=( + policy_cfg["action_layout"] if "action_layout" in policy_cfg else None + ), + rl_games_multibinary_threshold=( + policy_cfg["multibinary_threshold"] + if "multibinary_threshold" in policy_cfg + else None + ), + ) + if "backbone_path" in policy_cfg: + wrapper_kwargs["backbone_path"] = policy_cfg["backbone_path"] + if integration_env_name == "gymnasium": + action_space = gymnasium_action_space_contract(env_cfg) + wrapper_kwargs["rl_games_action_env_dim"] = len(action_space["labels"]) + if action_space["type"] == "box": + wrapper_kwargs["rl_games_gymnasium_action_space_type"] = "box" + wrapper_kwargs["rl_games_env_name"] = integration_env_name + wrapper = wrapper_cls(**wrapper_kwargs) + return StarVlaPolicyRunner( + wrapper=wrapper, + checkpoint_path=policy_cfg["checkpoint_path"], + device=policy_cfg["device"], + **runner_kwargs, + image_views_info_key=( + policy_cfg["image_views_info_key"] + if "image_views_info_key" in policy_cfg + else None + ), + action_output_type=( + policy_cfg["action_output_type"] + if "action_output_type" in policy_cfg + else None + ), + ) + + +def build_live_starvla_policy( + *, + framework: Any, + model_cfg: dict[str, Any], + config: dict[str, Any], + action_resolver: ActionResolver | None = None, +) -> PolicyRunner: + """Build a StarVLA policy around a *live* in-memory framework (no reload). + + Mirrors ``build_starvla_policy`` but swaps the ckpt-loading + ``PolicyServerWrapper`` for :class:`LiveStarVlaWrapper`, so the trainer's + resident model is evaluated directly. ``model_cfg`` is the in-memory model + config (e.g. ``read_mode_config`` output) the wrapper would otherwise read + from disk. + """ + policy_cfg = config["policy"] + if "task_contract_path" in policy_cfg: + from latency_bench.policy.starvla_tasks import TaskStarVlaPolicyRunner + + contract = json.loads( + Path(policy_cfg["task_contract_path"]).read_text(encoding="utf-8") + ) + return TaskStarVlaPolicyRunner( + framework, + policy_config=policy_cfg, + model_config=model_cfg, + contract=contract, + ) + + env_cfg = config["env"] + integration_env_name = env_cfg["name"] + normalized_model_cfg = ( + _normalized_model_cfg(model_cfg) + if integration_env_name == "gymnasium" + else None + ) + # Resident evaluation historically takes non-Gymnasium prompts from the map. + runner_kwargs = _starvla_runner_kwargs( + config, action_resolver, normalized_model_cfg, base_prompt=None + ) + wrapper_kwargs: dict[str, Any] = dict( + framework=framework, + model_cfg=model_cfg, + env_name=integration_env_name, + action_layout=policy_cfg["action_layout"] if "action_layout" in policy_cfg else None, + multibinary_threshold=( + policy_cfg["multibinary_threshold"] + if "multibinary_threshold" in policy_cfg + else None + ), + ) + if integration_env_name == "gymnasium": + action_space = gymnasium_action_space_contract(env_cfg) + wrapper_kwargs["rl_games_action_env_dim"] = len(action_space["labels"]) + if action_space["type"] == "box": + wrapper_kwargs["gymnasium_action_space_type"] = "box" + wrapper_kwargs["env_name"] = integration_env_name + wrapper = LiveStarVlaWrapper(**wrapper_kwargs) + return StarVlaPolicyRunner( + wrapper=wrapper, + checkpoint_path=policy_cfg.get("checkpoint_path", ""), + device=policy_cfg.get("device", "cuda"), + **runner_kwargs, + ) + + +__all__ = [ + "LiveStarVlaWrapper", + "StarVlaPolicyRunner", + "apply_starvla_model_input_config", + "build_live_starvla_policy", + "build_starvla_policy", + "decode_starvla_action", + "observation_data_to_hwc_uint8_frames", + "prepare_starvla_checkpoint_input_config", +] diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py new file mode 100644 index 0000000000000000000000000000000000000000..4a5cc03813f1bc524a30a7b6b7292a4f4fca85d4 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-code/starvla_tasks.py @@ -0,0 +1,92 @@ +"""StarVLA inference using the task's training observation/action contract.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import numpy as np +from PIL import Image + +from latency_bench.core.types import Action, Observation, PolicyOutput +from latency_bench.data.starvla_tasks import denormalize, normalize +from latency_bench.policy.base import PolicyRunner + + +class TaskStarVlaPolicyRunner(PolicyRunner): + """Map task RGB/state into a StarVLA model and decode its action chunk.""" + + def __init__(self, framework, *, policy_config: dict, model_config: dict, contract: dict): + self.framework = framework + self.policy_config = policy_config + self.model_config = model_config + self.contract = contract + + def _example(self, observation: Observation) -> dict: + cfg = self.policy_config + state = normalize( + observation.metadata[cfg["state_info_key"]], + self.contract["normalization"]["state"], + ).reshape(1, self.contract["state_dim"]) + data_cfg = self.model_config["datasets"]["vla_data"] + height, width = data_cfg["obs_image_size"] + images = [ + Image.fromarray(frame).resize((width, height)) + for frame in observation.metadata[cfg["image_views_info_key"]] + ] + if data_cfg["image_mode"] == "stitch_views": + from starVLA.training.trainer_utils.trainer_tools import stitch_frames + + # MIKASA's two simultaneous views form one Wan observation, not a video. + images = [stitch_frames(images, grid=data_cfg["stitch_grid"], size=(width, height))] + example = {"image": images, "state": state, "lang": self.contract["prompt"]} + if "action_prefix" in observation.metadata: + example["action_prefix"] = normalize( + observation.metadata["action_prefix"], + self.contract["normalization"]["action"], + ) + example["action_prefix_mask"] = observation.metadata["action_prefix_mask"] + return example + + def predict(self, observation: Observation) -> PolicyOutput: + return self.predict_batch([observation])[0] + + def predict_batch(self, observations: list[Observation]) -> list[PolicyOutput]: + prediction = self.framework.predict_action( + examples=[self._example(observation) for observation in observations] + ) + actions = denormalize( + prediction["normalized_actions"], self.contract["normalization"]["action"] + ) + # Prefix heads were excluded from the loss; retain the frozen controller plan. + for chunk, observation in zip(actions, observations): + if "action_prefix" in observation.metadata: + mask = observation.metadata["action_prefix_mask"] + chunk[mask] = observation.metadata["action_prefix"][mask] + return [ + PolicyOutput( + action=Action(value=chunk[0].tolist(), name="task_command"), + action_chunk=chunk, + raw_output=chunk.tolist(), + metadata={"policy_type": "starvla", "task": self.contract["task"]}, + ) + for chunk in actions + ] + + +def build_task_starvla_policy(config: dict) -> TaskStarVlaPolicyRunner: + # StarVLA and torch are optional in the simulator process; workers own them. + import torch + from starVLA.model.framework.base_framework import baseframework + from starVLA.model.framework.share_tools import read_mode_config + + cfg = config["policy"] + model_config, _ = read_mode_config(cfg["checkpoint_path"]) + framework = baseframework.from_pretrained( + cfg["checkpoint_path"], backbone_path=cfg["backbone_path"] + ) + framework = framework.to(device=cfg["device"], dtype=torch.bfloat16).eval() + contract = json.loads(Path(cfg["task_contract_path"]).read_text(encoding="utf-8")) + return TaskStarVlaPolicyRunner( + framework, policy_config=cfg, model_config=model_config, contract=contract + ) diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json new file mode 100644 index 0000000000000000000000000000000000000000..371542eee194a26ff888c211317abe89050b38e0 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/evaluation-plan.json @@ -0,0 +1,105 @@ +{ + "condition": "profile-latency", + "executor_mode": "simulated", + "latency_method": "temporal", + "profile_source": "originalRTX3090immutableprofiles", + "episodes_per_checkpoint": 100, + "total_episodes": 400, + "rounds": [ + [ + "flappy", + "deadly_corridor" + ], + [ + "ant", + "intercept" + ] + ], + "physical_gpu_assignments": { + "flappy": 2, + "deadly_corridor": 3, + "ant": 2, + "intercept": 3 + }, + "single_gpu_per_job": true, + "round2_requires_both_round1_complete": true, + "latency_seed": 271828, + "tasks": { + "flappy": { + "gpu": 2, + "seed_start": 1000000, + "seed_end": 1000099, + "env_fps": 10, + "obs_fps": 10, + "max_raw_steps": 3600, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/flappy/vla/starvla-qwenoft-h1/flappy-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 75.87417450998383, + "profile": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/flappy/instance_a5037b165aa0cedc/profile.json", + "sha256": "c266cba7ebac05c7e37bc83f1f16fe4b27dab97942ff43290e6c41514f6e39fc" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/flappy", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "deadly_corridor": { + "gpu": 3, + "seed_start": 1000000, + "seed_end": 1000099, + "env_fps": 35, + "obs_fps": 8.75, + "max_raw_steps": 3600, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/deadly-corridor/vla/starvla-qwenoft-h1/deadly_corridor-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 73.69250777493353, + "profile": "/home/ubuntu/lzj/profiles/published/profiles/openvla/1x-rtx3090/deadly_corridor/instance_a5037b165aa0cedc/profile.json", + "sha256": "bbfb93cef9f63750a9a159ec10a93a9fc6c25e4cb0cac3b39532663b4dc11eba" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/deadly_corridor", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "ant": { + "gpu": 2, + "seed_start": 42, + "seed_end": 141, + "env_fps": 10, + "obs_fps": 10, + "max_raw_steps": 1000, + "parallel_envs": 16, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/ant/vla/starvla-qwenoft-h1/ant-qwenoft-mean-3090-s42", + "profile": { + "mean_ms": 90.56460638563993, + "profile": "/home/ubuntu/lzj/mean-profiling/reference/checkpoint-configs/latency-aware/ant/small-policy/sample-factory-qwenoft-rtx3090-v1/profile/profile.json", + "sha256": "0e49f72ec05ac600a54e07bbca5af3143708444d606167f33fa21788a9fefe50" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/ant", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "intercept": { + "gpu": 3, + "seed_start": 4242424242, + "seed_end": 4242424341, + "env_fps": 20, + "obs_fps": 20, + "max_raw_steps": 60, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0", + "profile": { + "mean_ms": 99.05021289731565, + "profile": "/home/ubuntu/lzj/profiles/intercept-published/profiles/qwenoft/1x-rtx3090/mikasa_intercept_grab_fast/instance_3a0d42681a03715c/profile.json", + "sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + } + } +} diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/execution_audit.json b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/execution_audit.json new file mode 100644 index 0000000000000000000000000000000000000000..652ac4be43a3b43a9a6288d5395b396cc61ff136 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/execution_audit.json @@ -0,0 +1,12 @@ +{ + "issued_action_records": 2974, + "applied_action_records": 2864, + "dropped_action_records": 10, + "nonnoop_issued_records": 2974, + "finite_action_values": true, + "latency_sample_count": 2974, + "latency_mean_ms": 99.11060319379854, + "latency_std_ms": 4.301543980874005, + "latency_p95_ms": 100.2889407458356, + "latency_p99_ms": 100.64616770379737 +} diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/files.sha256.json b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/files.sha256.json new file mode 100644 index 0000000000000000000000000000000000000000..593cb1f913541b8e8425253afb17af9b3ca27b63 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/files.sha256.json @@ -0,0 +1,130 @@ +{ + "REPORT.md": { + "bytes": 2163, + "sha256": "b2fd63b1cde2a415b2d77daf0184ce5ec3b6ada1aa199c21941fb04031f77db6" + }, + "all_episodes.csv": { + "bytes": 25198, + "sha256": "bd41f35a464250ed6f9bc16e466aa9f1b55a48ac29072bb0ec3559a24129edac" + }, + "comparison.csv": { + "bytes": 438, + "sha256": "a0c7cf191395dd851bd6222bdf61427f15782fea2db4416ddc072a5f5dc8a861" + }, + "comparison.json": { + "bytes": 7826, + "sha256": "0d5dd042439466aee84cd0d96c31a27a951e57965a7e468cb73ec07883f1f751" + }, + "episodes.csv": { + "bytes": 5932, + "sha256": "c93e1707d186193962d91e177db677ac28d6f59cec2c630723b4b354b6002a84" + }, + "eval_config.yaml": { + "bytes": 1905, + "sha256": "9b832e26a5e70c5262b4c9e14430c2d7cb06a45a2f1f9c24332bfdbeccc467e5" + }, + "evaluation-code/batched_simulated.py": { + "bytes": 31282, + "sha256": "b901f966d911feab7962a32f21095cb90f7880121811f2b4eab2193afe1381db" + }, + "evaluation-code/deadly-compatibility.patch": { + "bytes": 4570, + "sha256": "623676cc4542b1eab6c9395b163b369ddc605353c1de02d17d8f713167ee07fa" + }, + "evaluation-code/deadly_corridor.py": { + "bytes": 17902, + "sha256": "47f7bc65cba9853e66d79ed2a28f844bd2a094f1285458be166045f2db1690dc" + }, + "evaluation-code/decision_action_history.py": { + "bytes": 2746, + "sha256": "14a9d223e775745b6c402dbce9e2a50a1c3f7b5b9fe528150ef8689126fe97cf" + }, + "evaluation-code/eval_driver.py": { + "bytes": 9133, + "sha256": "330030270fbb695bc5f14037ef7349650bd20c53c881c1159ee55ea066408d9e" + }, + "evaluation-code/mikasa_evaluate.py": { + "bytes": 11466, + "sha256": "6cf9ffee25fcfd6f3255c520fc544c48ff2c8f8912e5369c2410a709820c4ffd" + }, + "evaluation-code/starvla.py": { + "bytes": 48378, + "sha256": "6d9988f3a28d39e46c2f6e80da85edebc42cafa629a2b9f75000414324c1065a" + }, + "evaluation-code/starvla_tasks.py": { + "bytes": 4029, + "sha256": "3fc74169d1554d9dc3358ed85e450cca75eb69bc1fff85284c1054a605633a52" + }, + "evaluation-plan.json": { + "bytes": 4698, + "sha256": "b758a5fb72dcdef49d025e2fd168d024ebd8b18b2b00125145b3cde38b16a318" + }, + "execution_audit.json": { + "bytes": 358, + "sha256": "909d6ce6238528d1275c27832b495e541285ca0ca7e80d58c6ae73c346fc20c3" + }, + "profile/latency_burst_model.json": { + "bytes": 625, + "sha256": "83d0fae3530cf06ec49399c9911ffe54ad8018e060d80f1dc2026ec6f0022d6c" + }, + "profile/latency_distribution.json": { + "bytes": 25146, + "sha256": "4ac20f4eafb253a7bf88cf21e1729e5025ca313529e2856d008d5a85a9319c0e" + }, + "profile/profile.json": { + "bytes": 2190, + "sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984" + }, + "provenance.json": { + "bytes": 3900, + "sha256": "bae581409194ed5891c11b8c37e9529e54d438fda4764402689769fd1fffc78c" + }, + "queue_eval_latency_profile_sample.json": { + "bytes": 5257, + "sha256": "7c1c74a77301bc73a7e6430b61ee1731fba764d0fb1dd76382114da59b8d0d88" + }, + "raw-records/actions.jsonl.gz": { + "bytes": 435518, + "sha256": "a1c41d3d54f5c29128cbe5cab91f58ec42d3301e9cb9ea068d9a1c48b3d797e8" + }, + "raw-records/e2e_latencies.jsonl.gz": { + "bytes": 51906, + "sha256": "214f7c734a889ed4022203073fc42e83a931ec006f4d3e9f0a7d354b87cb50fb" + }, + "raw-records/episode_metrics.jsonl.gz": { + "bytes": 7477, + "sha256": "db379aff0be63814fed317db79b3834ee9cec8c6896313a6759b539b44a67c7b" + }, + "raw-records/infer_latencies.jsonl.gz": { + "bytes": 47253, + "sha256": "5560ddac213898b5f54f3a333884bd531f54b7ab3fcb4c670157521835c4a227" + }, + "raw-records/latencies.jsonl.gz": { + "bytes": 47247, + "sha256": "c81eb0dca742565bdda3f5e6d605a4b19cd92b43717a28f20be90c08df78ac91" + }, + "raw-records/observation_attempts.jsonl.gz": { + "bytes": 47, + "sha256": "b5696823564889c3075fe8b31161c33543789d91483e71611bd82d67839f1014" + }, + "raw-records/queue_eval_results.jsonl.gz": { + "bytes": 1513, + "sha256": "d086f85d515bcdfc2faf7d0263c6aceaf22d767f279fa4441efcca66e0fd5ae4" + }, + "raw-records/steps.jsonl.gz": { + "bytes": 524803, + "sha256": "b360ae43443a17a01ccef3e9b2c97f15e0c852604d59e4d2ed06d239ab0e1db9" + }, + "resolved_config.yaml": { + "bytes": 1941, + "sha256": "df81c68443f4a48d3ad80dcb8bdf24ce4b5aca41567385c85ad8dcbe228351ea" + }, + "statistics.json": { + "bytes": 1315, + "sha256": "00817a8463d75230a38c206c3f8c3d36b2d0a7f4482dffa776063decb7af4ee4" + }, + "stdout.log": { + "bytes": 1758, + "sha256": "0b77f0966739ee831f98e1bf7f596cf5a353b39861d9f87b90c446c35b7e6bf9" + } +} diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json new file mode 100644 index 0000000000000000000000000000000000000000..d857fba8c8a47ba29911a86a20816145514a1de5 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/profile/latency_burst_model.json @@ -0,0 +1,30 @@ +{ + "burst_dwell_distribution": null, + "burst_dwell_lengths": [], + "burst_merge_gap_records": 30, + "burst_rank_processes": [], + "burst_slot_rank_templates": [], + "model_type": "hidden_regime", + "pre_worker_time_ms": null, + "regime_step_counts": { + "burst": 0, + "calm": 7505 + }, + "regime_transition_counts": { + "burst": { + "burst": 0, + "calm": 0 + }, + "calm": { + "burst": 0, + "calm": 7500 + } + }, + "reset_scope": "session", + "schema_version": 12, + "spike_median_multiplier": 1.25, + "spike_threshold_ms_by_worker_slot": { + "0": 124.09547124058008 + }, + "worker_count": 1 +} diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json new file mode 100644 index 0000000000000000000000000000000000000000..a2efa2f55bfcd49c31eb66275e7e16ab4c122638 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/profile/latency_distribution.json @@ -0,0 +1,1027 @@ +{ + "schema_version": 3, + "worker_slots": { + "0": { + "all": { + "count": 7505, + "observation_to_action_latency_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 57.612886011600494, + 57.62146864297986, + 58.7545331081748, + 66.09737751752138, + 74.82502317726612, + 74.91858472645282, + 75.01819090604782, + 75.05339576900005, + 75.06971918046474, + 75.10700300306081, + 75.12112951159477, + 75.16202610909939, + 75.19575421214104, + 75.42616620063782, + 75.65311696827412, + 99.37247110009193, + 99.43262921273708, + 99.47681441307068, + 99.52412345409394, + 99.54871280789375, + 99.56331015527249, + 99.57241803407669, + 99.5812383234501, + 99.58828883171081, + 99.59576745927333, + 99.60240037441254, + 99.61199377477169, + 99.629284709692, + 99.67789617180824, + 99.7064346075058, + 99.72905080020428, + 99.75321701169014, + 99.77889660596847, + 99.80081658363342, + 99.81962535083294, + 99.83213859796524, + 99.84125447273254, + 99.84778462648391, + 99.85274956524373, + 99.85727146267891, + 99.86183978319168, + 99.86456000804901, + 99.86759921610356, + 99.87019740343094, + 99.87176775038242, + 99.87412478923798, + 99.87562672793865, + 99.8770145714283, + 99.87828935980797, + 99.87934100627899, + 99.8807034611702, + 99.88206803798676, + 99.88330855071544, + 99.88427200317383, + 99.88555290102958, + 99.88647729158401, + 99.88741657137871, + 99.88831980228424, + 99.88920986056328, + 99.89009560346604, + 99.89088672697544, + 99.89175403118134, + 99.89239759445191, + 99.89311588406562, + 99.89379814565181, + 99.89464781284332, + 99.89529819786549, + 99.89636748433114, + 99.89710994660854, + 99.8981822013855, + 99.89941228628159, + 99.90051797032356, + 99.90158712565899, + 99.9030942440033, + 99.90450049042701, + 99.9058184504509, + 99.90708529949188, + 99.90844941139221, + 99.9102594166994, + 99.9125352203846, + 99.91486594080925, + 99.91745400428772, + 99.92060535252094, + 99.92482746839524, + 99.9294671267271, + 99.93411362171173, + 99.94073301553726, + 99.95224863290787, + 99.96378195285797, + 99.98486397266387, + 100.01060213148594, + 100.03342998027802, + 100.06439234018326, + 100.08884423971176, + 100.11190702319145, + 100.1371660888195, + 100.16162499785423, + 100.1739651799202, + 100.18246213495732, + 100.1873430609703, + 100.19385551512241, + 100.20146602392197, + 100.2111359834671, + 100.22382822036744, + 100.24639928638935, + 100.28458214998246, + 100.32741177082062, + 100.3517196893692, + 100.36628523170948, + 100.4074354171753, + 100.64497736990452, + 100.64893850594758, + 100.65099319577217, + 100.65379960268736, + 100.65773456037044, + 100.66102886348963, + 100.66966495037079, + 100.68931616842747, + 100.7418359965086, + 100.78067521154881, + 100.8345368938148, + 101.12626532873509, + 101.19136601686478 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "spearman_rho": 0.9243654356338395, + "worker_service_time_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 57.08379399776459, + 57.09226791679859, + 58.26461273595691, + 65.8206928896904, + 74.3083177959919, + 74.40505877763033, + 74.52166484594345, + 74.54348520338536, + 74.58248136281968, + 74.61044075489045, + 74.62466361284255, + 74.65019178837538, + 74.67433359622956, + 74.91301276683808, + 75.17338513433933, + 98.8286515533924, + 98.93079552054405, + 98.96998397111892, + 99.01225567758084, + 99.04239894151688, + 99.06330541372299, + 99.07603198289871, + 99.08675174713134, + 99.09452852606773, + 99.10335409939289, + 99.11095643043518, + 99.12102273106575, + 99.13388040661812, + 99.15576733052731, + 99.17665178775788, + 99.20076440274715, + 99.22319501638412, + 99.25007324516773, + 99.26879998445511, + 99.28192260563374, + 99.2940465927124, + 99.30185323953629, + 99.30911481380463, + 99.31377121210099, + 99.31942344307899, + 99.32239929437637, + 99.32588696479797, + 99.32903581559658, + 99.33187481760979, + 99.33426531255245, + 99.33669203519821, + 99.33975677192211, + 99.34182575941085, + 99.344867131114, + 99.3473904132843, + 99.35013988018036, + 99.35334751009941, + 99.35609721541405, + 99.3590588092804, + 99.36210224628448, + 99.3658879995346, + 99.36862123012543, + 99.37164657115936, + 99.37495642602444, + 99.37807421088219, + 99.3803928911686, + 99.38310897350311, + 99.385760602355, + 99.38778692483902, + 99.39003224372864, + 99.39227157831192, + 99.39429570734501, + 99.39590540528297, + 99.39792068302631, + 99.39988957643509, + 99.40171358287334, + 99.40340998768806, + 99.40481108427048, + 99.4066559791565, + 99.40821809470654, + 99.40977981686592, + 99.41181422770023, + 99.41371536254883, + 99.41585898399353, + 99.4180584013462, + 99.420166644454, + 99.42292004823685, + 99.42697884738445, + 99.43130620121956, + 99.43675144016743, + 99.44651156663895, + 99.45808677375317, + 99.47644877433777, + 99.49587154388428, + 99.52255980968475, + 99.54652854800224, + 99.57171154022217, + 99.60106147825718, + 99.62191702127457, + 99.63395856618881, + 99.64684218764305, + 99.65841352939606, + 99.66941176652908, + 99.67911484539509, + 99.68980298638344, + 99.69720814526082, + 99.70772802829742, + 99.72056367397309, + 99.73819247484207, + 99.75959515571594, + 99.7885009765625, + 99.84809502959251, + 99.98601661920547, + 100.07802231311798, + 100.10256896018981, + 100.37615032196045, + 100.3818331310153, + 100.38439734458923, + 100.38643792003393, + 100.38879033625126, + 100.39409460574389, + 100.40106156349182, + 100.40533482283354, + 100.4536669152975, + 100.5038772636652, + 100.55455599203708, + 100.83995364856716, + 100.92318505048752 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + } + }, + "steady": { + "count": 7505, + "observation_to_action_latency_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 57.612886011600494, + 57.62146864297986, + 58.7545331081748, + 66.09737751752138, + 74.82502317726612, + 74.91858472645282, + 75.01819090604782, + 75.05339576900005, + 75.06971918046474, + 75.10700300306081, + 75.12112951159477, + 75.16202610909939, + 75.19575421214104, + 75.42616620063782, + 75.65311696827412, + 99.37247110009193, + 99.43262921273708, + 99.47681441307068, + 99.52412345409394, + 99.54871280789375, + 99.56331015527249, + 99.57241803407669, + 99.5812383234501, + 99.58828883171081, + 99.59576745927333, + 99.60240037441254, + 99.61199377477169, + 99.629284709692, + 99.67789617180824, + 99.7064346075058, + 99.72905080020428, + 99.75321701169014, + 99.77889660596847, + 99.80081658363342, + 99.81962535083294, + 99.83213859796524, + 99.84125447273254, + 99.84778462648391, + 99.85274956524373, + 99.85727146267891, + 99.86183978319168, + 99.86456000804901, + 99.86759921610356, + 99.87019740343094, + 99.87176775038242, + 99.87412478923798, + 99.87562672793865, + 99.8770145714283, + 99.87828935980797, + 99.87934100627899, + 99.8807034611702, + 99.88206803798676, + 99.88330855071544, + 99.88427200317383, + 99.88555290102958, + 99.88647729158401, + 99.88741657137871, + 99.88831980228424, + 99.88920986056328, + 99.89009560346604, + 99.89088672697544, + 99.89175403118134, + 99.89239759445191, + 99.89311588406562, + 99.89379814565181, + 99.89464781284332, + 99.89529819786549, + 99.89636748433114, + 99.89710994660854, + 99.8981822013855, + 99.89941228628159, + 99.90051797032356, + 99.90158712565899, + 99.9030942440033, + 99.90450049042701, + 99.9058184504509, + 99.90708529949188, + 99.90844941139221, + 99.9102594166994, + 99.9125352203846, + 99.91486594080925, + 99.91745400428772, + 99.92060535252094, + 99.92482746839524, + 99.9294671267271, + 99.93411362171173, + 99.94073301553726, + 99.95224863290787, + 99.96378195285797, + 99.98486397266387, + 100.01060213148594, + 100.03342998027802, + 100.06439234018326, + 100.08884423971176, + 100.11190702319145, + 100.1371660888195, + 100.16162499785423, + 100.1739651799202, + 100.18246213495732, + 100.1873430609703, + 100.19385551512241, + 100.20146602392197, + 100.2111359834671, + 100.22382822036744, + 100.24639928638935, + 100.28458214998246, + 100.32741177082062, + 100.3517196893692, + 100.36628523170948, + 100.4074354171753, + 100.64497736990452, + 100.64893850594758, + 100.65099319577217, + 100.65379960268736, + 100.65773456037044, + 100.66102886348963, + 100.66966495037079, + 100.68931616842747, + 100.7418359965086, + 100.78067521154881, + 100.8345368938148, + 101.12626532873509, + 101.19136601686478 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + }, + "spearman_rho": 0.9243654356338395, + "worker_service_time_ms": { + "distribution_type": "inverse_cdf", + "latency_ms": [ + 57.08379399776459, + 57.09226791679859, + 58.26461273595691, + 65.8206928896904, + 74.3083177959919, + 74.40505877763033, + 74.52166484594345, + 74.54348520338536, + 74.58248136281968, + 74.61044075489045, + 74.62466361284255, + 74.65019178837538, + 74.67433359622956, + 74.91301276683808, + 75.17338513433933, + 98.8286515533924, + 98.93079552054405, + 98.96998397111892, + 99.01225567758084, + 99.04239894151688, + 99.06330541372299, + 99.07603198289871, + 99.08675174713134, + 99.09452852606773, + 99.10335409939289, + 99.11095643043518, + 99.12102273106575, + 99.13388040661812, + 99.15576733052731, + 99.17665178775788, + 99.20076440274715, + 99.22319501638412, + 99.25007324516773, + 99.26879998445511, + 99.28192260563374, + 99.2940465927124, + 99.30185323953629, + 99.30911481380463, + 99.31377121210099, + 99.31942344307899, + 99.32239929437637, + 99.32588696479797, + 99.32903581559658, + 99.33187481760979, + 99.33426531255245, + 99.33669203519821, + 99.33975677192211, + 99.34182575941085, + 99.344867131114, + 99.3473904132843, + 99.35013988018036, + 99.35334751009941, + 99.35609721541405, + 99.3590588092804, + 99.36210224628448, + 99.3658879995346, + 99.36862123012543, + 99.37164657115936, + 99.37495642602444, + 99.37807421088219, + 99.3803928911686, + 99.38310897350311, + 99.385760602355, + 99.38778692483902, + 99.39003224372864, + 99.39227157831192, + 99.39429570734501, + 99.39590540528297, + 99.39792068302631, + 99.39988957643509, + 99.40171358287334, + 99.40340998768806, + 99.40481108427048, + 99.4066559791565, + 99.40821809470654, + 99.40977981686592, + 99.41181422770023, + 99.41371536254883, + 99.41585898399353, + 99.4180584013462, + 99.420166644454, + 99.42292004823685, + 99.42697884738445, + 99.43130620121956, + 99.43675144016743, + 99.44651156663895, + 99.45808677375317, + 99.47644877433777, + 99.49587154388428, + 99.52255980968475, + 99.54652854800224, + 99.57171154022217, + 99.60106147825718, + 99.62191702127457, + 99.63395856618881, + 99.64684218764305, + 99.65841352939606, + 99.66941176652908, + 99.67911484539509, + 99.68980298638344, + 99.69720814526082, + 99.70772802829742, + 99.72056367397309, + 99.73819247484207, + 99.75959515571594, + 99.7885009765625, + 99.84809502959251, + 99.98601661920547, + 100.07802231311798, + 100.10256896018981, + 100.37615032196045, + 100.3818331310153, + 100.38439734458923, + 100.38643792003393, + 100.38879033625126, + 100.39409460574389, + 100.40106156349182, + 100.40533482283354, + 100.4536669152975, + 100.5038772636652, + 100.55455599203708, + 100.83995364856716, + 100.92318505048752 + ], + "quantile_levels": [ + 0.0, + 0.0001, + 0.0005, + 0.001, + 0.002, + 0.003, + 0.004, + 0.005, + 0.006, + 0.007, + 0.008, + 0.009, + 0.01, + 0.02, + 0.03, + 0.04, + 0.05, + 0.06, + 0.07, + 0.08, + 0.09, + 0.1, + 0.11, + 0.12, + 0.13, + 0.14, + 0.15, + 0.16, + 0.17, + 0.18, + 0.19, + 0.2, + 0.21, + 0.22, + 0.23, + 0.24, + 0.25, + 0.26, + 0.27, + 0.28, + 0.29, + 0.3, + 0.31, + 0.32, + 0.33, + 0.34, + 0.35, + 0.36, + 0.37, + 0.38, + 0.39, + 0.4, + 0.41, + 0.42, + 0.43, + 0.44, + 0.45, + 0.46, + 0.47, + 0.48, + 0.49, + 0.5, + 0.51, + 0.52, + 0.53, + 0.54, + 0.55, + 0.56, + 0.57, + 0.58, + 0.59, + 0.6, + 0.61, + 0.62, + 0.63, + 0.64, + 0.65, + 0.66, + 0.67, + 0.68, + 0.69, + 0.7, + 0.71, + 0.72, + 0.73, + 0.74, + 0.75, + 0.76, + 0.77, + 0.78, + 0.79, + 0.8, + 0.81, + 0.82, + 0.83, + 0.84, + 0.85, + 0.86, + 0.87, + 0.88, + 0.89, + 0.9, + 0.91, + 0.92, + 0.93, + 0.94, + 0.95, + 0.96, + 0.97, + 0.98, + 0.99, + 0.991, + 0.992, + 0.993, + 0.994, + 0.995, + 0.996, + 0.997, + 0.998, + 0.999, + 0.9995, + 0.9999, + 1.0 + ] + } + } + } + } +} diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/profile/profile.json b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/profile/profile.json new file mode 100644 index 0000000000000000000000000000000000000000..22b4d0dd275f2dfde5c7c996197cac820bb8b5f0 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/profile/profile.json @@ -0,0 +1,66 @@ +{ + "burst_model_path": "latency_burst_model.json", + "distribution_path": "latency_distribution.json", + "env_fps": 20, + "frame_ms": 50.0, + "gpu_class": "1x-rtx3090", + "instance_id": "instance_3a0d42681a03715c", + "latency_kind": "observation_to_action_latency", + "latency_method": "temporal", + "model_id": "qwenoft", + "n_admitted_observations": 7505, + "n_capacity_drops": 7495, + "n_observation_attempts": 15000, + "per_slot_summary": { + "0": { + "admitted_count": 7505, + "mean_observation_to_action_latency_ms": 99.05021289731565, + "mean_worker_service_time_ms": 98.55270653801072, + "p95_observation_to_action_latency_ms": 100.32732703685761, + "p95_worker_service_time_ms": 99.84762226343155, + "p99_worker_service_time_ms": 100.37557450771331 + } + }, + "provenance": { + "base_config": "configs/examples/mikasa/official_starvla_profile.yaml", + "checkpoint_kind": "best", + "model_artifact": { + "checkpoint": "checkpoints/model.pt", + "model_config": "config.yaml", + "path_in_repo": "zero-latency/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/mikasa_intercept_grab_fast_qwenoft_formal_latest_r8gpu8_20260907T082154Z", + "repo_id": "latency-sensitive-bench/benchmark-models", + "source": "hf" + }, + "session_ids": [ + 0, + 1, + 2, + 3, + 4 + ] + }, + "sample_model_type": "hidden_regime", + "source_run_id": "20260909T044501695676Z", + "summary": { + "frame_ms": 50.0, + "max_ms": 101.19136601686478, + "mean_effective_frames": 1.981004257946313, + "mean_ms": 99.05021289731565, + "min_ms": 57.612886011600494, + "n_samples": 7505, + "p50_frames": 1.9978350806236267, + "p50_ms": 99.89175403118134, + "p90_frames": 2.0040290887355803, + "p90_ms": 100.20145443677902, + "p95_frames": 2.006546540737152, + "p95_ms": 100.32732703685761, + "p99_frames": 2.0128874400138854, + "p99_ms": 100.64437200069428, + "prob_latency_gt_1_frame": 1.0, + "prob_latency_gt_2_frames": 0.2139906728847435, + "prob_latency_gt_3_frames": 0.0, + "std_ms": 4.583189247461202 + }, + "visualization_path": "latency_profile.png", + "workload_id": "mikasa_intercept_grab_fast" +} diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/provenance.json b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..ce4f48fbf72176b91809abe7b0ea58b401d13835 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/provenance.json @@ -0,0 +1,99 @@ +{ + "task": "intercept", + "protocol": { + "gpu": 3, + "seed_start": 4242424242, + "seed_end": 4242424341, + "env_fps": 20, + "obs_fps": 20, + "max_raw_steps": 60, + "parallel_envs": 32, + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0", + "profile": { + "mean_ms": 99.05021289731565, + "profile": "/home/ubuntu/lzj/profiles/intercept-published/profiles/qwenoft/1x-rtx3090/mikasa_intercept_grab_fast/instance_3a0d42681a03715c/profile.json", + "sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984" + }, + "config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept.yaml", + "output": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept", + "metrics": "return mean/std/min/max;length mean/std;success where nativeprovided;latency/drops/invalid/action audits;100 episode vectors" + }, + "checkpoint_weights": { + "bytes": 9785101729, + "sha256": "7bea6746bf7973ff7d4e4e8a457a3bfbc0ce981c80fe5f7afadca651ec5ca85e" + }, + "evaluation": { + "n_episodes": 100, + "mean_return": 3.5443485127069287, + "std_return": 7.07192296411853, + "min_return": 0.6267238368745893, + "max_return": 29.923812823486514, + "mean_length": 60.0, + "std_length": 0.0, + "min_length": 60.0, + "max_length": 60.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "mikasa_intercept_grab_fast", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "mikasa_intercept_grab_fast", + "instance_id": "instance_3a0d42681a03715c", + "source_run_id": "20260909T044501695676Z", + "profile_ref": null, + "env_fps": 20.0, + "obs_fps": 20.0, + "frame_ms": 50.0, + "latency_type": "profile_sample", + "task": "intercept", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0", + "profile_sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 10, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept.yaml", + "success_count": 9, + "success_rate": 0.09, + "execution_audit": { + "issued_action_records": 2974, + "applied_action_records": 2864, + "dropped_action_records": 10, + "nonnoop_issued_records": 2974, + "finite_action_values": true, + "latency_sample_count": 2974, + "latency_mean_ms": 99.11060319379854, + "latency_std_ms": 4.301543980874005, + "latency_p95_ms": 100.2889407458356, + "latency_p99_ms": 100.64616770379737 + } + }, + "source_revision": { + "repo": "c3c6a39365a151e9b7a5e215452fd64e957c2b29", + "starvla": "ccca13c5177fe3d3c884b6e2de4965d916016649", + "runtime_fixes": [ + "mean-profile-preparation.patch", + "gym-language-contract.patch", + "loader-spawn-cache.patch", + "loader-spawn-test.patch", + "doom-mean-reset.patch" + ], + "pytorch3d": { + "revision": "33824be3cbc87a7dd1db0f6a9a9de9ac81b2d0ba", + "build": "transforms-only, no native render extension; QwenOFT uses transforms only" + }, + "decord": { + "version": "0.6.0", + "build": "official source CPU decoder CP310", + "wheel_sha256": "e193b356b1e984b4eff08d23b62e482c2c9e5037a6efdc0b1af47079ae2e4c47" + } + }, + "raw_records_format": "gzip(JSONL), lossless", + "startup_checks_included_in_score": false, + "results_status": "evaluation_complete; acceptance_not_inferred" +} diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json new file mode 100644 index 0000000000000000000000000000000000000000..5c4e51ef9d931330002647f1d4da8e4bfdea1bf9 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/queue_eval_latency_profile_sample.json @@ -0,0 +1,325 @@ +{ + "checkpoint_path": "/home/ubuntu/lzj/mean-profiling/intercept/vla-publication/checkpoints/model.pt", + "experiment_name": "intercept-mean5000-profile-simulation-100ep", + "latency": "profile_sample", + "latency_type": "profile_sample", + "lengths": [ + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60, + 60 + ], + "mean_length": 60.0, + "mean_return": 3.5443485127069287, + "returns": [ + 0.7267571190313902, + 2.9096362272975966, + 3.2060351513209753, + 0.7574528902187012, + 0.6827895979695313, + 29.923812823486514, + 0.7661087726592086, + 0.8284444468154106, + 0.9707721562881488, + 1.0944434545235708, + 0.7526731102407211, + 1.0327306617691647, + 24.08529434411321, + 0.8258126199943945, + 0.6465023508935701, + 1.2059930491086561, + 0.8975468523567542, + 0.638558203499997, + 2.473904824233614, + 0.8594156300532632, + 0.7127419076277874, + 1.1195833964738995, + 1.4589147588121705, + 22.348254217096837, + 27.43761277961312, + 0.797963114338927, + 0.6415987604705151, + 1.502438226743834, + 1.277322537265718, + 0.6413188653605175, + 26.015227647672873, + 0.7568511647114065, + 0.7758818510046694, + 0.743274000211386, + 0.9812663898337632, + 0.7364500367548317, + 0.7676261149172205, + 2.6105462690466084, + 0.8922563010128215, + 0.7909053032053635, + 27.747763212013524, + 2.830903574009426, + 3.749473527306691, + 3.2371535471174866, + 1.141169616690604, + 1.2504711685760412, + 1.1401455145678483, + 1.1743367564631626, + 0.6911400489043444, + 0.966755291854497, + 3.7725237559643574, + 0.7292428385990206, + 2.733719722367823, + 2.7277548569836654, + 0.8013565168366767, + 0.9918300381395966, + 3.8384227409260347, + 2.525593837024644, + 1.1939986812940333, + 1.1946645161951892, + 0.6632764584392135, + 0.7345126099826302, + 1.1547945403144695, + 1.0395031699445099, + 2.7713681719324086, + 3.8083399715833366, + 3.1245881704380736, + 0.9936205917911138, + 0.6479002644773573, + 1.09404552471824, + 0.725047086874838, + 2.085218493710272, + 25.112157980707707, + 0.7960666966973804, + 1.8899870013119653, + 24.77215793245705, + 0.763190906640375, + 0.8356004936795216, + 24.543561146681895, + 0.7962639288743958, + 0.6807828926102957, + 1.1704122956143692, + 0.8024117537715938, + 1.0154686415335163, + 0.6267238368745893, + 1.1180786813492887, + 1.0531825890648179, + 0.7319892354425974, + 1.1460731038823724, + 1.145515855285339, + 3.1438898412743583, + 1.1678254807484336, + 1.1468605129048228, + 2.816772125195712, + 1.1577836629003286, + 1.0533778404060286, + 0.8533297177054919, + 0.7617567333800253, + 0.9895546428160742, + 0.770722996792756 + ], + "seed": 4242424242, + "source_profile_path": "/home/ubuntu/lzj/profiles/intercept-published/profiles/qwenoft/1x-rtx3090/mikasa_intercept_grab_fast/instance_3a0d42681a03715c/profile.json", + "std_return": 7.071922964118529, + "suite_name": "profile_sample", + "task_metrics": { + "success": { + "mean": 0.09, + "std": 0.28618176042508375, + "values": [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 1.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ] + } + }, + "timestamp_utc": "2026-10-01T06:51:53.252454+00:00" +} diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..4f6e5270fe827359afe21172c72b9994106dd4e3 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/actions.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a1c41d3d54f5c29128cbe5cab91f58ec42d3301e9cb9ea068d9a1c48b3d797e8 +size 435518 diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..86c85879e06bb1bdebd86b0cad9b0485d7feb569 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/e2e_latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:214f7c734a889ed4022203073fc42e83a931ec006f4d3e9f0a7d354b87cb50fb +size 51906 diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..82118b45eb67e5dd4323b9cbf81a13e1aadd2fe1 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/episode_metrics.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:db379aff0be63814fed317db79b3834ee9cec8c6896313a6759b539b44a67c7b +size 7477 diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..b97793721b148e93974008f37f74407b1fee8b1c --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/infer_latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5560ddac213898b5f54f3a333884bd531f54b7ab3fcb4c670157521835c4a227 +size 47253 diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..457b574981a529003954dd14d7176d458bd597ef --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/latencies.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c81eb0dca742565bdda3f5e6d605a4b19cd92b43717a28f20be90c08df78ac91 +size 47247 diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..6d4cb769c7b8b799d336079f11abbffdee22a96d --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/observation_attempts.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b5696823564889c3075fe8b31161c33543789d91483e71611bd82d67839f1014 +size 47 diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..fccbc96b1db0d17a593f35d2eac82c9610d81205 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/queue_eval_results.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d086f85d515bcdfc2faf7d0263c6aceaf22d767f279fa4441efcca66e0fd5ae4 +size 1513 diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..2cd4bf8cb7c9879a2059a3ca940ef10758bcb20e --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/raw-records/steps.jsonl.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b360ae43443a17a01ccef3e9b2c97f15e0c852604d59e4d2ed06d239ab0e1db9 +size 524803 diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e40398ddfa01330923024c37b644dfdd3e8083a4 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/resolved_config.yaml @@ -0,0 +1,69 @@ +experiment: + name: intercept-mean5000-profile-simulation-100ep + seed: 4242424242 +executor: + mode: simulated + simulated_worker_capacity: 1 + simulated_inference_pool: true + inference_devices: + - cuda:0 + inference_batch_size: 32 +env: + name: mikasa_intercept_grab_fast + env_fps: 20 + obs_fps: 20 + frame_stack: 1 + noop_action: + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + - 0.0 + base_prompt: Intercept the rolling ball and grasp it to stop it. + simulator_device: gpu +latency: + method: temporal + profile_path: /home/ubuntu/lzj/profiles/intercept-published/profiles/qwenoft/1x-rtx3090/mikasa_intercept_grab_fast/instance_3a0d42681a03715c/profile.json + profile_worker_slot: 0 + seed: 271828 + add_latency_info: false +scheduler: + hold_policy: hold + ordering_policy: latest_ready + hold_last_chunk_action: true +policy: + action_prefix: + mode: none + type: starvla + checkpoint_path: /home/ubuntu/lzj/mean-profiling/intercept/vla-publication/checkpoints/model.pt + model_config_path: /home/ubuntu/lzj/mean-profiling/intercept/vla-publication/config.full.yaml + task_contract_path: /home/ubuntu/lzj/mean-profiling/intercept/vla-publication/task_contract.json + device: cuda:0 + state_info_key: mikasa_proprio + image_views_info_key: mikasa_image_views + backbone_path: /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct + worker_python_executable: /home/ubuntu/lzj/conda/envs/qwenoft/bin/python +evaluation: + eval_episodes: 100 + eval_parallel_envs: 32 + eval_max_steps: 60 + eval_deterministic: true + eval_latency_values: null + eval_raw_reward: true + eval_suites: + fixed: [] + normal: [] + uniform: [] +logging: + output_dir: /home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept + video: + enabled: false + save_step_records: true + save_action_records: true + save_latency_records: true + wandb_project: null + wandb_group: null + wandb_job_type: null + simulated_pipeline_profile: false diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/statistics.json b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/statistics.json new file mode 100644 index 0000000000000000000000000000000000000000..6c6b7b6e18ace118870b85bb2a02055bad2f4da8 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/statistics.json @@ -0,0 +1,38 @@ +{ + "n_episodes": 100, + "mean_return": 3.5443485127069287, + "std_return": 7.07192296411853, + "min_return": 0.6267238368745893, + "max_return": 29.923812823486514, + "mean_length": 60.0, + "std_length": 0.0, + "min_length": 60.0, + "max_length": 60.0, + "return_field": "episode_return_env", + "length_field": "survival_steps", + "mode": "simulated", + "policy_id": "starvla", + "env_id": "mikasa_intercept_grab_fast", + "model_id": "qwenoft", + "gpu_class": "1x-rtx3090", + "workload_id": "mikasa_intercept_grab_fast", + "instance_id": "instance_3a0d42681a03715c", + "source_run_id": "20260909T044501695676Z", + "profile_ref": null, + "env_fps": 20.0, + "obs_fps": 20.0, + "frame_ms": 50.0, + "latency_type": "profile_sample", + "task": "intercept", + "checkpoint_revision": "d4825bbad8d15bc50b6b8481eb2abc84846a54fb", + "checkpoint_path": "latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0", + "profile_sha256": "31a9b15d6b04182439934f0b377cb3d09be16ce21f3cf95c1049918f8a5d4984", + "condition": "profile-latency", + "invalid_actions": 0, + "dropped_actions": 10, + "unique_seeds": 100, + "physical_gpu": 3, + "eval_config": "/home/ubuntu/lzj/mean-profiling/evaluation/profile-latency-100ep-20261001/intercept.yaml", + "success_count": 9, + "success_rate": 0.09 +} diff --git a/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/stdout.log b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/stdout.log new file mode 100644 index 0000000000000000000000000000000000000000..ab90a3c3493aa6d19f16c0a5fe5d5c86b9e10af9 --- /dev/null +++ b/latency-aware/mikasa-intercept-grab-fast/vla/starvla-qwenoft-h1/intercept-qwenoft-mean-3090-h1-s0/evaluation/profile-simulation-100ep-20261001/stdout.log @@ -0,0 +1,17 @@ +/home/ubuntu/lzj/conda/envs/mikasa/lib/python3.10/site-packages/torch/cuda/__init__.py:61: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you. + import pynvml # type: ignore[import] +/home/ubuntu/lzj/conda/envs/mikasa/lib/python3.10/site-packages/sapien/_vulkan_tricks.py:21: UserWarning: Failed to find system libvulkan. Fallback to SAPIEN builtin libvulkan. + warn("Failed to find system libvulkan. Fallback to SAPIEN builtin libvulkan.") +10/01 [06:51:01] INFO | >> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/in + tercept/vla-publication/checkpoints + /model.pt` + INFO | >> [*] Loading from local share_tools.py:418 + checkpoint path + `/home/ubuntu/lzj/mean-profiling/in + tercept/vla-publication/checkpoints + /model.pt` +[QWen3] loading /home/ubuntu/lzj/models/Qwen3-VL-4B-Instruct with gradient_checkpointing=True + Loading checkpoint shards: 0%| | 0/2 [00:00