File size: 58,457 Bytes
5170d86
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>OmniAgent — ICML 2026 Booth Talk</title>
<link rel="stylesheet" href="deck.css?v=2">
<style>deck-stage:not(:defined){visibility:hidden}</style>
</head>
<body class="no-foot">

<deck-stage width="1920" height="1080" no-rail>

<!-- ============ 1 · HOOK / TITLE ============ -->
<section data-label="Hook" data-speaker-notes="(~60s) Hi everyone, thanks for stopping by. I'm Zhenghao from CUHK, and this is joint work with the Qwen team. I'd like to share our ICML paper, OmniAgent.

Here's the problem we started from. When you ask a video model about one specific moment in a long video, it still watches the entire video, and it often misses the moment you actually care about.

What we did is teach the model to search instead of watch. OmniAgent is a 7B model built on Qwen2.5-Omni, and it decides for itself what to look at, what to listen to, and when to stop. That's enough to beat a model ten times larger on long video, while reading about 73 percent fewer frames. Let me show you how it works.">
  <div class="slide" data-screen-label="01 · Hook">
    <div class="title-logos rise" style="position: absolute; top: 84px; right: 110px; display: flex; align-items: center;">
      <img src="assets/qwen-logo.jpeg" alt="Qwen" style="height: 60px; mix-blend-mode: multiply;">
    </div>
    <div class="title-wrap">
      <div class="eyebrow rise">ICML 2026 · Seoul · Qwen Booth #B400 · Jul 8, 13:20 KST</div>
      <p class="title-brand rise d1" style="font-size: 118px; margin-bottom: 16px;">OmniAgent</p>
      <p class="title-main title-main-only rise d2" style="font-size: 42px; margin-top: 0; margin-bottom: 64px; color: var(--ink-soft); font-weight: 600;">Native Active Perception as Reasoning for Omni-Modal Understanding</p>
      <div class="title-proof rise d2">
        <div class="title-proof-card">
          <p class="title-proof-n">7B &gt; 72B</p>
          <p class="title-proof-l">beats a <strong>10× larger</strong> passive model on LVBench</p>
        </div>
        <div class="title-proof-card">
          <p class="title-proof-n">−73%</p>
          <p class="title-proof-l">frames ingested per video — <strong>203 vs 768</strong></p>
        </div>
        <div class="title-proof-card">
          <p class="title-proof-n">10/10</p>
          <p class="title-proof-l">all 10 benchmarks improve — up to <strong>7× on temporal grounding</strong></p>
        </div>
      </div>
      <p class="title-speaker rise d3">Zhenghao Xing · CUHK Ph.D. Candidate · Multimodal Agent &amp; Audio-Visual Reasoning</p>
      <p class="title-authors rise d3">Zhenghao Xing* · Ruiyang Xu* · Yuxuan Wang* · Jinzheng He · Ziyang Ma · Qize Yang · Yunfei Chu · Jin Xu† · Junyang Lin · Chi-Wing Fu · Pheng-Ann Heng†</p>
      <p class="title-affil rise d3">Qwen · CUHK · SJTU · NTU &nbsp;&nbsp;·&nbsp;&nbsp; *equal contribution &nbsp;&nbsp;·&nbsp;&nbsp; †corresponding author &nbsp;&nbsp;·&nbsp;&nbsp; built on Qwen2.5-Omni-7B</p>
      <div class="title-strip rise d3" aria-hidden="true">
        <span class="title-strip-track"></span>
        <i class="tp" style="left: 12%; animation-delay: 0.9s;"></i>
        <i class="tp" style="left: 46%; animation-delay: 1.05s;"></i>
        <i class="tp audio" style="left: 51%; animation-delay: 1.2s;"></i>
        <i class="tp" style="left: 55%; animation-delay: 1.35s;"></i>
        <i class="tp" style="left: 84%; animation-delay: 1.5s;"></i>
      </div>
    </div>
  </div>
</section>

<!-- ============ 2 · PROBLEM ============ -->
<section data-label="Problem" data-speaker-notes="(~60s) A passive model samples a fixed set of frames and then answers. Even if your question is about one single moment in a forty-minute vlog, most of the frame budget lands nowhere near that moment.

That causes two problems at once. The compute grows with the video length, and the real evidence gets drowned in irrelevant context.

What we actually want is for perception to scale with the question instead of the video.">
  <div class="slide" data-screen-label="02 · Problem">
    <div class="eyebrow">Motivation</div>
    <h2 class="title">Passive models watch everything. Cost grows with the video.</h2>
    <div class="vpack">
    <div class="query-card rise">
      <p class="query-q">&ldquo;Why are the vlogger family members so happy at 22:03?&rdquo;</p>
      <p class="query-meta">2,448 s vlog · audio: yes</p>
    </div>
    <div class="strip-row rise d1">
      <p class="strip-name">Passive<span>watch it all</span></p>
      <div class="stripcol">
        <div class="strip passive">
          <div class="probe passive-probe" style="left: 1.5%;"></div>
          <div class="probe passive-probe" style="left: 4.0%;"></div>
          <div class="probe passive-probe" style="left: 6.5%;"></div>
          <div class="probe passive-probe" style="left: 9.0%;"></div>
          <div class="probe passive-probe" style="left: 11.5%;"></div>
          <div class="probe passive-probe" style="left: 14.0%;"></div>
          <div class="probe passive-probe" style="left: 16.5%;"></div>
          <div class="probe passive-probe" style="left: 19.0%;"></div>
          <div class="probe passive-probe" style="left: 21.5%;"></div>
          <div class="probe passive-probe" style="left: 24.0%;"></div>
          <div class="probe passive-probe" style="left: 26.5%;"></div>
          <div class="probe passive-probe" style="left: 29.0%;"></div>
          <div class="probe passive-probe" style="left: 31.5%;"></div>
          <div class="probe passive-probe" style="left: 34.0%;"></div>
          <div class="probe passive-probe" style="left: 36.5%;"></div>
          <div class="probe passive-probe" style="left: 39.0%;"></div>
          <div class="probe passive-probe" style="left: 41.5%;"></div>
          <div class="probe passive-probe" style="left: 44.0%;"></div>
          <div class="probe passive-probe" style="left: 46.5%;"></div>
          <div class="probe passive-probe" style="left: 49.0%;"></div>
          <div class="probe passive-probe" style="left: 51.5%;"></div>
          <div class="probe passive-probe" style="left: 54.0%;"></div>
          <div class="probe passive-probe" style="left: 56.5%;"></div>
          <div class="probe passive-probe" style="left: 59.0%;"></div>
          <div class="probe passive-probe" style="left: 61.5%;"></div>
          <div class="probe passive-probe" style="left: 64.0%;"></div>
          <div class="probe passive-probe" style="left: 66.5%;"></div>
          <div class="probe passive-probe" style="left: 69.0%;"></div>
          <div class="probe passive-probe" style="left: 71.5%;"></div>
          <div class="probe passive-probe" style="left: 74.0%;"></div>
          <div class="probe passive-probe" style="left: 76.5%;"></div>
          <div class="probe passive-probe" style="left: 79.0%;"></div>
          <div class="probe passive-probe" style="left: 81.5%;"></div>
          <div class="probe passive-probe" style="left: 84.0%;"></div>
          <div class="probe passive-probe" style="left: 86.5%;"></div>
          <div class="probe passive-probe" style="left: 89.0%;"></div>
          <div class="probe passive-probe" style="left: 91.5%;"></div>
          <div class="probe passive-probe" style="left: 94.0%;"></div>
          <div class="probe passive-probe" style="left: 96.5%;"></div>
          <div class="probe passive-probe" style="left: 99.0%;"></div>
        </div>
        <div class="strip-scale"><span>0:00</span><span class="end">40:48</span></div>
      </div>
      <p class="strip-cap"><strong>768 uniform frame probes</strong> — dense across the whole video (illustrative)</p>
    </div>
    <div class="strip-row rise d2">
      <p class="strip-name">What the query needs<span>query-driven</span></p>
      <div class="stripcol">
        <div class="strip active-strip">
          <div class="probe" style="left: 50%;"></div>
          <div class="probe" style="left: 53%;"></div>
          <div class="probe audio" style="left: 55.5%;"></div>
          <div class="probe" style="left: 60%;"></div>
        </div>
        <div class="strip-scale"><span>0:00</span><span class="mid" style="left: 55.5%;">▲ 22:03</span><span class="end">40:48</span></div>
      </div>
      <p class="strip-cap"><strong>4 targeted observations</strong> — frames + audio around 22:03</p>
    </div>
    <div class="cost-line rise d3">
      <span class="bad">today: pay for video length</span>
      <span class="good">wanted: pay for question difficulty</span>
    </div>
    </div>
    <div class="foot">OmniAgent · ICML 2026</div>
  </div>
</section>

<!-- ============ 3 · CONTRIBUTIONS ============ -->
<section data-label="Contributions" data-speaker-notes="(~40s) This is our core idea. OmniAgent is a native omni-modal agent for active perception. It is one model that sees, hears, reasons, and acts, with no external tools and no pipeline.

The four cards show what it does. It requests frames on demand. It switches to audio when frames alone can't settle the question. It keeps short text notes instead of raw pixels. And on every turn it decides what to do next, and stopping is a decision too.

The bottom row shows the contrast. There is no external captioner, no ASR toolchain, and no caption-everything pre-scan. All of the search behavior lives inside the one model, and the next slides show how we build and train it.">
  <div class="slide" data-screen-label="03 · Contributions">
    <div class="eyebrow">Our idea</div>
    <h2 class="title">See. Hear. Reason. Act. Inside one native model.</h2>
    <p class="subtitle rise"><strong>OmniAgent</strong> — a <strong>native omni-modal agent for active perception</strong>, built on Qwen2.5-Omni-7B. One model that sees, hears, and decides what to look at next — no external tools, no pipeline.</p>
    <div class="idea-grid rise d1">
      <div class="idea-card">
        <span class="icon-badge" aria-hidden="true"><svg class="icon" viewBox="0 0 24 24"><path d="M2 12s3.5-6 10-6 10 6 10 6-3.5 6-10 6-10-6-10-6z"></path><circle cx="12" cy="12" r="3"></circle></svg></span>
        <p class="idea-k">see</p>
        <h3 class="idea-h">Request frames on demand</h3>
        <p class="idea-p">Broad scans locate candidates; dense scans refine boundaries.</p>
      </div>
      <div class="idea-card">
        <span class="icon-badge" aria-hidden="true"><svg class="icon" viewBox="0 0 24 24"><path d="M12 3v18"></path><path d="M8 7v10"></path><path d="M16 7v10"></path><path d="M4 10v4"></path><path d="M20 10v4"></path></svg></span>
        <p class="idea-k">hear</p>
        <h3 class="idea-h">Switch to audio when needed</h3>
        <p class="idea-p">Audio confirms what frames alone can't.</p>
      </div>
      <div class="idea-card">
        <span class="icon-badge" aria-hidden="true"><svg class="icon" viewBox="0 0 24 24"><path d="M9 18h6"></path><path d="M10 22h4"></path><path d="M8.5 14.5A6 6 0 1 1 15.5 14.5c-.8.6-1.5 1.5-1.5 2.5h-4c0-1-.7-1.9-1.5-2.5z"></path></svg></span>
        <p class="idea-k">reason</p>
        <h3 class="idea-h">Remember notes, not pixels</h3>
        <p class="idea-p">Raw media is deleted each turn; short notes persist.</p>
      </div>
      <div class="idea-card">
        <span class="icon-badge" aria-hidden="true"><svg class="icon" viewBox="0 0 24 24"><path d="M5 12h14"></path><path d="m13 6 6 6-6 6"></path></svg></span>
        <p class="idea-k">act</p>
        <h3 class="idea-h">Decide what to do next</h3>
        <p class="idea-p">Frames, audio, clip, or answer — stopping is a decision too.</p>
      </div>
    </div>
    <div class="principle-strip rise d2">
      <div>
        <p class="principle-k">not this</p>
        <p class="principle-p">external captioner / ASR toolchain · exhaustive caption-everything pre-scan</p>
      </div>
      <div>
        <p class="principle-k accent">this</p>
        <p class="principle-p"><strong>native active perception</strong> — train the looking, not just the answering</p>
      </div>
      <div>
        <p class="principle-k accent">result</p>
        <p class="principle-p"><strong>small + active &gt; big + passive</strong> — exact margins in the results</p>
      </div>
    </div>
    <div class="foot">OmniAgent · ICML 2026</div>
  </div>
</section>

<!-- ============ 4 · METHOD: OTA LOOP ============ -->
<section data-label="Method — OTA loop" data-speaker-notes="(~60s) Here's how the loop works. The model starts with just the query and some basic video metadata. On every turn it writes an observation, a thought, and an action.

The observation is a short text summary of what it just saw or heard. The thought says what is still missing. The action asks for the next piece of input, which can be frames, audio, a clip, or the final answer.

The key design choice is that raw media gets deleted after every turn. Only the text trace stays, so the context grows with the reasoning instead of the video length.

The environment itself is deliberately simple. It only cuts out the requested segment and returns it.">
  <div class="slide" style="--pad-top: 76px; --pad-bottom: 96px; --gap-title: 26px;" data-screen-label="04 · Method: OTA loop">
    <div class="eyebrow">Method — active perception as a POMDP</div>
    <h2 class="title" style="margin-bottom: 28px;">Look, Think, Act. Repeat.</h2>
    <div class="ota grow rise d1">
      <div class="ota-box">
        <p class="ota-k">Persistent memory</p>
        <h3 class="ota-h">Query + short notes only</h3>
        <p class="ota-p">Starts as the question plus video metadata. Grows one note per turn — never raw media.</p>
      </div>
      <div class="ota-link">
        <div class="ota-lane"><p class="lane-lab">write a note</p><p class="lane-arr"></p></div>
        <div class="ota-lane"><p class="lane-lab">read every turn</p><p class="lane-arr rev"></p></div>
      </div>
      <div class="ota-box ota-model">
        <p class="ota-k">OmniAgent — one native model</p>
        <div class="ota-steps">
          <p class="ota-step"><svg class="icon" viewBox="0 0 24 24" aria-hidden="true"><path d="M2 12s3.5-6 10-6 10 6 10 6-3.5 6-10 6-10-6-10-6z"></path><circle cx="12" cy="12" r="3"></circle></svg><b>Observation</b><span>what did I just see / hear?</span></p>
          <p class="ota-step"><svg class="icon" viewBox="0 0 24 24" aria-hidden="true"><path d="M9 18h6"></path><path d="M10 22h4"></path><path d="M8.5 14.5A6 6 0 1 1 15.5 14.5c-.8.6-1.5 1.5-1.5 2.5h-4c0-1-.7-1.9-1.5-2.5z"></path></svg><b>Thought</b><span>what is still missing?</span></p>
          <p class="ota-step"><svg class="icon" viewBox="0 0 24 24" aria-hidden="true"><path d="M5 12h14"></path><path d="m13 6 6 6-6 6"></path></svg><b>Action</b><span>get_frames · get_audio · get_clip · answer</span></p>
        </div>
      </div>
      <div class="ota-link">
        <div class="ota-lane"><p class="lane-lab">action</p><p class="lane-arr"></p></div>
        <div class="ota-lane"><p class="lane-lab">transient percept — purged after the turn</p><p class="lane-arr rev"></p></div>
      </div>
      <div class="ota-box">
        <p class="ota-k">Video environment Ω</p>
        <h3 class="ota-h">A dumb media extractor</h3>
        <p class="ota-p">Returns the requested frames, audio, or A/V clip. No captioner, no ASR, no retrieval — all perception happens inside the model.</p>
      </div>
    </div>
    <div class="chip-row rise d2">
      <p class="chip"><strong>purge</strong> — raw media deleted each turn; only notes persist</p>
      <p class="chip"><strong>cost</strong> — grows with the reasoning, not the video length</p>
      <p class="chip"><strong>stop</strong> — the answer action ends the loop once evidence suffices</p>
    </div>
  </div>
</section>

<!-- ============ 5 · WORKED EXAMPLE: SEARCH ============ -->
<section data-label="Worked example — search" data-speaker-notes="(~85s) Now let's see how it actually runs on a real example. The query asks for every time range where Linda Grimbeck discusses tourism security, and the correct answer is two intervals, one from 22 to 76 seconds and one from about 175 to 226 seconds.

The agent starts with a broad scan, 50 frames across the full video. This scan is not for answering, it just builds a map. Most of the frames turn out to be irrelevant, but two windows look promising, one near 21 seconds and one near 175.

Then it switches to audio and listens to those two candidate windows. The audio carries the real evidence here. We hear about smart cameras, an intelligence network, a tourism victim support helpline, and tourism monitors. So both windows really are about tourism security.

After that it goes back to frames, this time dense ones right at the edges, checking before, inside, and after each span. So the full pattern is a broad visual search, then audio verification, then boundary refinement, and every action is chosen to resolve a specific doubt from the previous turn.">
  <div class="slide" data-screen-label="05 · Worked example: search">
    <div class="eyebrow">Worked example · temporal grounding</div>
    <h2 class="title" style="margin-bottom: 28px;">Scan. Verify. Refine.</h2>
    <div class="example-grid">
      <div class="example-query rise">
        <p class="example-k">real demo trajectory</p>
        <p class="example-q">Find all time ranges where &ldquo;Linda Grimbeck discusses tourism security measures in an interview.&rdquo;</p>
        <div class="example-meta">
          <p><strong>Video:</strong> 260.10 s · audio + visual news clip</p>
          <p><strong>Task:</strong> temporal grounding, no explanation in final answer</p>
          <p><strong>First action:</strong> get_frames(0.0, 260.1, 50)</p>
        </div>
      </div>
      <div class="search-board rise d1">
        <div class="scan-strip">
          <figure class="scan-card">
            <img src="assets/demo-linda/scan-000.jpg" alt="Opening frame from the full-video scan">
            <span class="scan-time">0.00s</span>
            <figcaption class="scan-cap">not the evidence</figcaption>
          </figure>
          <figure class="scan-card hit">
            <img src="assets/demo-linda/scan-021.jpg" alt="First candidate interview segment with Linda Grimbeck">
            <span class="scan-time">21.23s</span>
            <figcaption class="scan-cap">candidate A</figcaption>
          </figure>
          <figure class="scan-card hit">
            <img src="assets/demo-linda/scan-175.jpg" alt="Second candidate interview segment with Linda Grimbeck">
            <span class="scan-time">175s</span>
            <figcaption class="scan-cap">candidate B</figcaption>
          </figure>
          <figure class="scan-card">
            <img src="assets/demo-linda/scan-260.jpg" alt="End frame from the full-video scan">
            <span class="scan-time">260s</span>
            <figcaption class="scan-cap">end card</figcaption>
          </figure>
        </div>
        <div class="candidate-map">
          <p class="candidate-map-k">1 · quick glance — 50 frames, the model's own choice (not a 768-frame pre-scan)</p>
          <div class="candidate-track">
            <span class="track-start">0s</span>
            <span class="track-end">260s</span>
            <span class="track-window a">A · 22-76s</span>
            <span class="track-window b">B · 175-226s</span>
          </div>
          <p class="candidate-map-p">Two promising islands. Both need verifying — and their edges are still fuzzy.</p>
        </div>
        <div class="evidence-grid">
          <div class="audio-evidence">
            <p class="block-k">2 · switch to audio</p>
            <ul class="audio-list">
              <li>smart cameras</li>
              <li>intelligence network</li>
              <li>tourism victim support helpline</li>
              <li>tourism monitors</li>
            </ul>
            <p class="audio-note">Frames found the candidates. Audio confirms the topic: tourism security.</p>
          </div>
          <div class="boundary-evidence">
            <p class="block-k">3 · refine cuts</p>
            <div class="boundary-rows">
              <div class="boundary-row">
                <p class="boundary-label">range A</p>
                <figure class="boundary-thumb"><img src="assets/demo-linda/boundary-first-before.jpg" alt="Frame before the first Linda Grimbeck segment"><span>before</span></figure>
                <figure class="boundary-thumb inside"><img src="assets/demo-linda/boundary-first-inside.jpg" alt="Frame inside the first Linda Grimbeck segment"><span>inside</span></figure>
                <figure class="boundary-thumb"><img src="assets/demo-linda/boundary-first-after.jpg" alt="Frame after the first Linda Grimbeck segment"><span>after</span></figure>
              </div>
              <div class="boundary-row">
                <p class="boundary-label">range B</p>
                <figure class="boundary-thumb"><img src="assets/demo-linda/boundary-second-before.jpg" alt="Frame before the second Linda Grimbeck segment"><span>before</span></figure>
                <figure class="boundary-thumb inside"><img src="assets/demo-linda/boundary-second-inside.jpg" alt="Frame inside the second Linda Grimbeck segment"><span>inside</span></figure>
                <figure class="boundary-thumb"><img src="assets/demo-linda/boundary-second-after.jpg" alt="Frame after the second Linda Grimbeck segment"><span>after</span></figure>
              </div>
            </div>
          </div>
        </div>
      </div>
    </div>
  </div>
</section>

<!-- ============ 6 · WORKED EXAMPLE: TRACE ============ -->
<section data-label="Worked example — OTA trace" data-speaker-notes="(~55s) When the search finishes, what stays in memory is a compact text state. It records candidate A, candidate B, the audio confirmation of the topic, and the fact that the boundaries still need tightening. That is enough to answer, and the full trace is nine turns.

If you read that trace, it looks less like a video summary and more like a lab notebook. What did I inspect, what did I learn, what is still uncertain, and what should I look at next.">
  <div class="slide" data-screen-label="06 · Worked example: OTA trace">
    <div class="eyebrow">Worked example · OTA trace</div>
    <h2 class="title" style="margin-bottom: 28px;">Only the text memory persists</h2>
    <div class="memory-grid">
      <div class="memory-panel media-panel rise">
        <p class="memory-k">Transient percepts</p>
        <h3 class="memory-h">Used once, then deleted</h3>
        <div class="percept-stack">
          <div class="percept-card">
            <span class="percept-ico"><svg class="icon" viewBox="0 0 24 24"><rect x="3" y="5" width="18" height="14" rx="2"></rect><path d="M8 13h4"></path><path d="M15 9h2"></path></svg></span>
            <div><p class="percept-t">50-frame global scan</p><p class="percept-p">finds candidates, not the answer</p></div>
            <p class="purge">purge</p>
          </div>
          <div class="percept-card">
            <span class="percept-ico audio"><svg class="icon" viewBox="0 0 24 24"><path d="M12 3v18"></path><path d="M8 7v10"></path><path d="M16 7v10"></path><path d="M4 10v4"></path><path d="M20 10v4"></path></svg></span>
            <div><p class="percept-t">Audio on candidate windows</p><p class="percept-p">verifies the security topic</p></div>
            <p class="purge">purge</p>
          </div>
          <div class="percept-card">
            <span class="percept-ico"><svg class="icon" viewBox="0 0 24 24"><path d="M4 5h16"></path><path d="M4 19h16"></path><rect x="6" y="8" width="12" height="8" rx="1.5"></rect></svg></span>
            <div><p class="percept-t">Dense boundary frames</p><p class="percept-p">before / inside / after the cuts</p></div>
            <p class="purge">purge</p>
          </div>
        </div>
      </div>
      <div class="memory-panel text-panel rise d1">
        <p class="memory-k">Persistent text memory</p>
        <h3 class="memory-h">What remains reads like a lab notebook</h3>
        <div class="memory-trace">
          <p><span>01</span> Candidate A around 22-76 s; candidate B around 175-226 s.</p>
          <p><span>02</span> Audio says smart cameras, intelligence network, victim support helpline, tourism monitors.</p>
          <p><span>03</span> Both candidates are Linda Grimbeck discussing tourism security.</p>
          <p><span>04</span> Remaining work: refine start/end boundaries, then answer.</p>
        </div>
        <div class="answer-strip">
          <p class="answer-strip-k">answer action · turn 9</p>
          <div class="answer-strip-ranges">
            <p>[22.00, 76.00]</p>
            <p>[175.04, 225.96]</p>
          </div>
        </div>
      </div>
    </div>
    <div class="foot">OmniAgent · ICML 2026</div>
  </div>
</section>

<!-- ============ 7 · METHOD: AGENTIC SFT ============ -->
<section data-label="Method — Agentic SFT" data-speaker-notes="(~65s) Now, how do we train this behavior? The first thing we tried was ordinary video QA fine-tuning, and it actually made the base model worse on LVBench. The score dropped from 43.0 to 41.6, because more answer labels only reinforce the passive habit.

So Agentic SFT supervises the trajectory instead of the answer. A teacher explores multiple action paths, and we only keep a path when the final answer is correct and the reasoning is supported by what was actually observed.

That gives us 58K clean trajectories, and LVBench goes from 43.0 to 48.7. In short, we train the looking, not just the answering.">
  <div class="slide" data-screen-label="07 · Method: Agentic SFT">
    <div class="eyebrow">Method — trajectory supervision</div>
    <h2 class="title" style="margin-bottom: 26px;">Train the looking, not the answering</h2>
    <div class="sft-grid">
      <div class="sft-panel passive rise">
        <p class="sft-k">ordinary video-QA SFT</p>
        <h3 class="sft-h">Supervise the answer</h3>
        <div class="sft-flow">
          <p>sampled video tokens</p>
          <span></span>
          <p>answer label</p>
        </div>
        <p class="sft-p">More answer labels just reinforced the passive habit — the score dropped.</p>
        <div class="sft-score down">
          <p>LVBench</p>
          <strong>43.0 → 41.6</strong>
        </div>
      </div>
      <div class="sft-panel active rise d1">
        <p class="sft-k">agentic SFT</p>
        <h3 class="sft-h">Supervise the trajectory</h3>
        <div class="sft-flow">
          <p>Observation</p>
          <span></span>
          <p>Thought</p>
          <span></span>
          <p>Action</p>
        </div>
        <p class="sft-p">Best-of-N exploration with self-correction; a path is kept only if the answer is right and the reasoning matches what was seen.</p>
        <div class="sft-score up">
          <p>LVBench</p>
          <strong>43.0 → 48.7</strong>
        </div>
      </div>
    </div>
    <div class="sft-bottom rise d2">
      <div><p class="sft-bottom-n">58K</p><p class="sft-bottom-l">accepted trajectories</p></div>
      <div><p class="sft-bottom-n">2 filters</p><p class="sft-bottom-l">right answer + supported reasoning</p></div>
      <div><p class="sft-bottom-n">learns when</p><p class="sft-bottom-l">to scan, listen, zoom in, and stop</p></div>
    </div>
    <div class="foot">OmniAgent · ICML 2026</div>
  </div>
</section>

<!-- ============ 8 · METHOD: TAURA ============ -->
<section data-label="Method — TAURA" data-speaker-notes="(~65s) After SFT we run reinforcement learning, and we make one change to the usual recipe. Plain GRPO gives every turn the same credit, so the turn that found the evidence earns exactly as much as a routine follow-up turn.

TAURA fixes this with token entropy. We observed that about 79.2 percent of the critical branching turns have above-average entropy, which means high uncertainty tends to mark the decision points.

So TAURA rescales each turn's advantage by its normalized entropy. Discovery turns earn more credit, and confident wrong turns get punished harder. On LVBench, Agentic SFT reaches 48.7, plain GRPO reaches 49.8, and TAURA reaches 50.5.">
  <div class="slide" data-screen-label="08 · Method: TAURA">
    <div class="eyebrow">Method — turn-level credit</div>
    <h2 class="title" style="margin-bottom: 20px;">Not every turn deserves the same credit</h2>
    <div class="duo">
      <div class="panel problem rise d1">
        <p class="panel-k">Problem — advantage homogenization</p>
        <h3 class="panel-h">GRPO pays every turn the same</h3>
        <p class="panel-p">Every turn gets the same credit — the turn that found the evidence earns no more than filler.</p>
        <div class="bigstat">
          <p class="bigstat-n">79.2%</p>
          <p class="bigstat-l">of decision-point turns show above-average entropy — uncertainty marks the forks</p>
        </div>
      </div>
      <div class="panel fix rise d2">
        <p class="panel-k">Fix — entropy-rescaled turn credit</p>
        <h3 class="panel-h">Weight each turn by its uncertainty</h3>
        <p class="panel-p">Pay each turn by its uncertainty. Discovery earns more; routine earns less; confident mistakes are punished harder.</p>
        <div class="tw-row">
          <div class="tw-bar" style="--h: 26%"></div>
          <div class="tw-bar" style="--h: 20%"></div>
          <div class="tw-bar" style="--h: 32%"></div>
          <div class="tw-bar hi" style="--h: 100%"><span class="tw-tag">discovery turn</span></div>
          <div class="tw-bar" style="--h: 18%"></div>
          <div class="tw-bar" style="--h: 28%"></div>
          <div class="tw-bar" style="--h: 24%"></div>
        </div>
        <p class="tw-cap">Illustrative — routine turns get down-weighted; the turn that finds the evidence earns amplified credit.</p>
        <div class="panel-formula formula-card">
          <p class="formula-k">TAURA — Turn-aware Adaptive Uncertainty Rescaled Advantage</p>
          <div class="formula-line">
            <span class="m-term"><span class="m-var">Â</span><sub>i,k</sub></span>
            <span class="m-eq">=</span>
            <span class="m-labeled"><span class="m-term"><span class="m-var">A</span><sub>i</sub></span><em>trajectory advantage</em></span>
            <span class="formula-dot">·</span>
            <span class="m-labeled"><span class="m-term"><span class="m-var">w</span><sub>i,k</sub></span><em>turn-aware weight</em></span>
          </div>
          <div class="formula-line secondary">
            <span class="m-term"><span class="m-var">w</span><sub>i,k</sub></span>
            <span class="m-eq">=</span>
            <span class="m-frac">
              <span class="m-num"><span class="m-var">H</span><sub>i,k</sub></span>
              <span class="m-den">
                <span class="m-mini"><span>1</span><span><span class="m-var">N</span><sub>G</sub></span></span>
                <span class="m-sum"><sup>G</sup><span class="m-sigma">Σ</span><sub>j=1</sub></span>
                <span class="m-sum"><sup>K<sub>j</sub></sup><span class="m-sigma">Σ</span><sub>m=1</sub></span>
                <span class="m-term"><span class="m-var">H</span><sub>j,m</sub></span>
              </span>
            </span>
            <span class="m-gloss">← this turn's entropy<br>← mean entropy over the group</span>
          </div>
          <p class="formula-note">Normalized so E[w<sub>i,k</sub>] = 1; credit shifts to high-entropy discovery turns without changing gradient scale.</p>
        </div>
      </div>
    </div>
    <p class="flow-note rise d3" style="margin-top: 26px;"><strong>Every stage earns its keep</strong> — LVBench: SFT 48.7 → GRPO 49.8 → <strong>TAURA 50.5</strong> · DailyOmni: GRPO dips to 62.2, <strong>TAURA 64.8</strong></p>
  </div>
</section>

<!-- ============ 9 · HEADLINE RESULT (dark) ============ -->
<section data-label="Headline result" data-speaker-notes="(~55s) This is the headline result in full. On LVBench, OmniAgent-7B gets 50.5 and Qwen2.5-VL-72B gets 47.3, and we use about 203 frames per video compared with their 768.

Passive systems mostly buy accuracy by ingesting more frames, and that is the dashed trend line in the chart. We sit above that line and to the left, which means higher accuracy with fewer frames. The model is not seeing more, it is choosing better.">
  <div class="slide" data-screen-label="09 · Headline result">
    <div class="eyebrow">Headline result — LVBench · hour-plus videos</div>
    <h2 class="title" style="margin-bottom: 30px;">Small model. Sharper eyes.</h2>
    <div class="headline-grid">
      <div class="vs-block">
        <div class="vs-row rise">
          <p class="vs-num win">50.5%</p>
          <p class="vs-label"><strong>OmniAgent-7B</strong>active perception</p>
        </div>
        <div class="vs-row rise d1">
          <p class="vs-num lose">47.3%</p>
          <p class="vs-label"><strong>Qwen2.5-VL-72B</strong>passive, 10× the parameters</p>
        </div>
        <hr class="vs-divider">
        <p class="vs-frames rise d2">frames per video: <strong>203</strong> vs 768 — <strong>~73% fewer</strong></p>
      </div>
      <div class="fig-card rise d2" style="height: 660px; flex-direction: column; gap: 16px; padding: 28px 36px 20px;">
        <div class="leg">
          <span class="leg-i"><i class="dm"></i>OmniAgent (active)</span>
          <span class="leg-i"><i class="dm ag"></i>agentic baselines</span>
          <span class="leg-i"><i class="dot"></i>passive models</span>
        </div>
        <svg class="svfig" viewBox="-14 70 1014 590" role="img" aria-label="LVBench accuracy versus frames ingested: OmniAgent-7B sits top-left, above the passive trend">
          <text class="sv-axis-title" transform="translate(8 350) rotate(-90)" text-anchor="middle">LVBench accuracy (%)</text>

          <g stroke="#ece7de" stroke-width="2">
            <line x1="70" y1="536" x2="950" y2="536"></line>
            <line x1="70" y1="448" x2="950" y2="448"></line>
            <line x1="70" y1="360" x2="950" y2="360"></line>
            <line x1="70" y1="272" x2="950" y2="272"></line>
            <line x1="70" y1="184" x2="950" y2="184"></line>
          </g>
          <line x1="70" y1="120" x2="70" y2="580" stroke="#d8d3c8" stroke-width="2"></line>
          <g class="sv-tick y" text-anchor="end">
            <text x="54" y="543">36</text>
            <text x="54" y="455">40</text>
            <text x="54" y="367">44</text>
            <text x="54" y="279">48</text>
            <text x="54" y="191">52</text>
          </g>
          <line x1="80" y1="470" x2="950" y2="314" stroke="#b9b2a4" stroke-width="3" stroke-dasharray="10 10"></line>
          <text class="sv-cap" x="862" y="371" text-anchor="middle" font-style="italic">passive trend</text>
          <g fill="#93a7cd">
            <circle cx="127" cy="538" r="13"></circle>
            <circle cx="127" cy="470" r="13"></circle>
            <circle cx="203" cy="461" r="13"></circle>
            <circle cx="175" cy="382" r="15"></circle>
            <circle cx="691" cy="333" r="15"></circle>
            <circle cx="898" cy="402" r="17"></circle>
            <circle cx="691" cy="287" r="30"></circle>
          </g>
          <g fill="#d99a3d">
            <rect x="-13" y="-13" width="26" height="26" transform="translate(410,320) rotate(45)"></rect>
            <rect x="-13" y="-13" width="26" height="26" transform="translate(725,419) rotate(45)"></rect>
          </g>
          <rect x="-18" y="-18" width="36" height="36" fill="var(--accent)" transform="translate(234,217) rotate(45)"></rect>
          <text class="sv-hero" x="272" y="210">OmniAgent-7B</text>
          <text class="sv-sub" x="272" y="244">50.5 @ ~203 frames</text>
          <text class="sv-lab" x="736" y="283">Qwen2.5-VL-72B</text>
          <text class="sv-min" x="736" y="311">47.3 @ 768</text>
          <text class="sv-min" x="150" y="546">LongVA-7B</text>
          <text class="sv-min" x="127" y="446" text-anchor="middle">VISTA-7B</text>
          <text class="sv-min" x="224" y="469">Kangaroo-7B</text>
          <text class="sv-min" x="204" y="390">Qwen2.5-Omni-7B (base)</text>
          <text class="sv-min" x="658" y="341" text-anchor="end">Qwen2.5-VL-7B</text>
          <text class="sv-min" x="898" y="448" text-anchor="middle">Vamba-10B</text>
          <text class="sv-min" x="410" y="288" text-anchor="middle">Zoom-Zero-7B</text>
          <text class="sv-min" x="725" y="460" text-anchor="middle">LongVT-7B</text>
          <line x1="70" y1="580" x2="960" y2="580" stroke="#d8d3c8" stroke-width="2"></line>
          <g class="sv-tick" text-anchor="middle">
            <text x="70" y="616">0</text>
            <text x="277" y="616">256</text>
            <text x="484" y="616">512</text>
            <text x="691" y="616">768</text>
            <text x="898" y="616">1024</text>
          </g>
          <text class="sv-cap" x="515" y="652" text-anchor="middle">frames ingested per video →</text>
        </svg>
      </div>
    </div>
  </div>
</section>

<!-- ============ 10 · BENCHMARK SWEEP ============ -->
<section data-label="Benchmark sweep" data-speaker-notes="(~45s) The sweep checks whether the LVBench result generalizes, and it does. OmniAgent improves the base model on all ten benchmarks, and the biggest gains appear exactly where evidence search is hardest.

On VSI-Bench the gain comes from spatial reasoning, on OmniVideoBench it comes from audio-visual evidence, and on LVBench and VideoMME-Long it comes from long-video reasoning. The smaller QA sets move less, which is what you would expect if the looking is doing the real work.">
  <div class="slide" data-screen-label="10 · Benchmark sweep">
    <div class="eyebrow">Results — benchmark sweep</div>
    <h2 class="title">Broad gains. Strongest on search-heavy tasks.</h2>
    <div class="sweep-layout">
      <div class="sweep-claim rise">
        <p class="sweep-n">10/10</p>
        <p class="sweep-p">all paper benchmarks improve over Qwen2.5-Omni-7B</p>
        <p class="sweep-note">Paper Tables 1-3: video understanding, audio-visual understanding, and temporal grounding. VUE-TR reports Vision+Audio / Vision.</p>
      </div>
      <div class="sweep-metrics detailed rise d1">
        <div class="sweep-group">Video understanding <span>Table 1</span></div>
        <div class="sweep-row"><p class="label">VideoMME <span>overall / long</span></p><p class="score">64.8 → 67.8<br><em>54.8 → 59.6</em></p><div class="sweep-track"><i style="--w: 9%;"></i></div><strong>+3.0 / +4.8</strong></div>
        <div class="sweep-row hl"><p class="label">VSI-Bench <span>reasoning · avg 1.6 min</span></p><p class="score">35.5 → 48.4</p><div class="sweep-track"><i style="--w: 38%;"></i></div><strong>+12.9</strong></div>
        <div class="sweep-row"><p class="label">MLVU <span>long · 3-120 min</span></p><p class="score">65.2 → 71.1</p><div class="sweep-track"><i style="--w: 18%;"></i></div><strong>+5.9</strong></div>
        <div class="sweep-row"><p class="label">Minerva <span>reasoning · 2-90 min</span></p><p class="score">33.4 → 41.4</p><div class="sweep-track"><i style="--w: 24%;"></i></div><strong>+8.0</strong></div>
        <div class="sweep-row hl"><p class="label">LVBench <span>long · avg 68 min</span></p><p class="score">43.0 → 50.5</p><div class="sweep-track"><i style="--w: 22%;"></i></div><strong>+7.5</strong></div>
        <div class="sweep-group">Audio-visual understanding <span>Table 2</span></div>
        <div class="sweep-row"><p class="label">DailyOmni <span>generic · avg 0.7 min</span></p><p class="score">60.1 → 64.8</p><div class="sweep-track"><i style="--w: 14%;"></i></div><strong>+4.7</strong></div>
        <div class="sweep-row"><p class="label">WorldSense <span>generic · avg 2.4 min</span></p><p class="score">45.4 → 47.2</p><div class="sweep-track"><i style="--w: 5%;"></i></div><strong>+1.8</strong></div>
        <div class="sweep-row hl"><p class="label">OmniVideoBench <span>reasoning · avg 6.4 min</span></p><p class="score">29.3 → 37.1</p><div class="sweep-track"><i style="--w: 23%;"></i></div><strong>+7.8</strong></div>
        <div class="sweep-group">Temporal grounding <span>Table 3 · IoU</span></div>
        <div class="sweep-row"><p class="label">LongVALE <span>avg 3.9 min</span></p><p class="score">5.7 → 39.1</p><div class="sweep-track"><i style="--w: 98%;"></i></div><strong>+33.4</strong></div>
        <div class="sweep-row"><p class="label">VUE-TR <span>V+A / vision</span></p><p class="score">3.5 → 36.5<br><em>8.0 → 46.1</em></p><div class="sweep-track"><i style="--w: 100%;"></i></div><strong>+33.0 / +38.1</strong></div>
      </div>
    </div>
    <div class="foot">OmniAgent · ICML 2026</div>
  </div>
</section>

<!-- ============ 11 · TEMPORAL GROUNDING ============ -->
<section data-label="Temporal grounding" data-speaker-notes="(~60s) Temporal grounding is the strictest test here, because the model has to output actual time spans, and there is no way to bluff that.

A passive model has to compress the whole video and then guess the boundaries. OmniAgent searches instead, exactly like the worked example. It scans broadly, verifies in whichever modality carries the evidence, and then refines the cut points.

The gains are large. On LongVALE the base model sits at 5.7 IoU, and OmniAgent reaches 39.1. On VUE-TR, where both vision and audio matter, we reach 36.5, which is above GPT-4o and Gemini-2.5-Pro in this evaluation.">
  <div class="slide" data-screen-label="11 · Temporal grounding">
    <div class="eyebrow">Results — temporal grounding</div>
    <h2 class="title">Finding the moment is a search problem.</h2>
    <div class="temporal-layout">
      <div class="temporal-left rise">
        <div class="ground-card temporal-main">
          <p class="ground-bench">LongVALE · IoU — vs. base model</p>
          <div class="ground-move">
            <p class="ground-from">5.7</p>
            <p class="ground-arrow"></p>
            <p class="ground-to">39.1</p>
          </div>
          <p class="ground-delta">+33.4 absolute — a near-7× jump</p>
        </div>
        <div class="search-steps">
          <div><p class="search-step-k">scan</p><p>find candidate windows</p></div>
          <div><p class="search-step-k">verify</p><p>use audio or vision evidence</p></div>
          <div><p class="search-step-k">refine</p><p>tighten start and end cuts</p></div>
        </div>
      </div>
      <div class="ground-card temporal-frontier rise d1">
        <p class="ground-bench">VUE-TR (vision + audio) · IoU — frontier comparison</p>
        <div class="cmp">
          <div class="cmp-top"><span class="compare-label">OmniAgent-7B</span><span class="compare-val win">36.5</span></div>
          <div class="cbar win"><i style="--w: 100%;"></i></div>
        </div>
        <div class="cmp">
          <div class="cmp-top"><span class="compare-label">GPT-4o</span><span class="compare-val">11.1</span></div>
          <div class="cbar"><i style="--w: 30.4%;"></i></div>
        </div>
        <div class="cmp">
          <div class="cmp-top"><span class="compare-label">Gemini-2.5-Pro</span><span class="compare-val">12.1</span></div>
          <div class="cbar"><i style="--w: 33.2%;"></i></div>
        </div>
        <p class="frontier-note">Closed-source generalists see the video; OmniAgent also learns where to look next.</p>
      </div>
    </div>
    <div class="foot">OmniAgent · ICML 2026</div>
  </div>
</section>

<!-- ============ 12 · TEST-TIME SCALING ============ -->
<section data-label="Test-time scaling" data-speaker-notes="(~50s) One result we are particularly excited about is test-time scaling. If we simply raise the maximum turn budget from 6 to 52, VideoMME-Long accuracy keeps climbing, from 53.4 all the way to 59.6. That is a six-point gain with no retraining at all, you just let the model think longer.

And the model does not blindly spend all 52 turns. Even with the cap that high, it only uses about 11.7 on average, because answering is an action too and it stops as soon as the evidence is enough. In other words, the extra compute goes exactly where it should, to the hard questions.">
  <div class="slide" data-screen-label="12 · Test-time scaling">
    <div class="eyebrow">Analysis — test-time scaling</div>
    <h2 class="title" style="margin-bottom: 28px;">It thinks more only when it has to</h2>
    <div class="split">
      <div class="split-stats">
        <div class="rise d1">
          <p class="stat-h">+6.2%</p>
          <p class="stat-l">VideoMME-Long accuracy, as the max turn budget K grows 6 → 52</p>
        </div>
        <div class="rise d2">
          <p class="stat-h">~11.7</p>
          <p class="stat-l">turns actually used even at K = 52 — it answers once the evidence is enough</p>
        </div>
        <div class="rise d3">
          <p class="stat-h">16.9 → 5.7</p>
          <p class="stat-l">turns per video-hour as videos grow 20–40 → 120–140 min (8.5 → 12.5 turns) — compute follows the question, not the clock</p>
        </div>
      </div>
      <div class="fig-card grow rise d2" style="flex-direction: column; gap: 16px; padding: 28px 36px 20px;">
        <div class="leg">
          <span class="leg-i"><i class="dm"></i>accuracy — VideoMME-Long</span>
          <span class="leg-i"><i class="sq"></i>turns actually used</span>
        </div>
        <svg class="svfig" viewBox="0 0 1000 584" role="img" aria-label="Accuracy rises from 53.44 to 59.56 as the max turn budget grows from 6 to 52, while actual turns rise from 4.99 to 11.66">
          <text class="sv-cap" x="58" y="42">accuracy score ↑</text>
          <text class="sv-cap" x="58" y="337">actual turns ↑</text>

          <g stroke="#e6e1d8" stroke-width="2">
            <line x1="86" y1="260" x2="940" y2="260"></line>
            <line x1="86" y1="210" x2="940" y2="210"></line>
            <line x1="86" y1="160" x2="940" y2="160"></line>
            <line x1="86" y1="110" x2="940" y2="110"></line>
            <line x1="86" y1="502" x2="940" y2="502"></line>
          </g>
          <g class="sv-tick" text-anchor="end">
            <text x="74" y="268">54</text>
            <text x="74" y="218">56</text>
            <text x="74" y="168">58</text>
            <text x="74" y="118">60</text>
          </g>
          <g class="sv-tick" text-anchor="middle">
            <text x="100" y="536">6</text>
            <text x="260" y="536">12</text>
            <text x="420" y="536">22</text>
            <text x="580" y="536">32</text>
            <text x="740" y="536">42</text>
            <text x="900" y="536">52</text>
          </g>

          <g fill="#cdd7ea">
            <rect x="62" y="450" width="76" height="52" rx="6"></rect>
            <rect x="222" y="424" width="76" height="78" rx="6"></rect>
            <rect x="382" y="402" width="76" height="100" rx="6"></rect>
            <rect x="542" y="395" width="76" height="107" rx="6"></rect>
            <rect x="702" y="381" width="76" height="121" rx="6"></rect>
            <rect x="862" y="380" width="76" height="122" rx="6"></rect>
          </g>
          <g class="sv-bar" text-anchor="middle">
            <text x="100" y="438">4.99</text>
            <text x="260" y="412">7.43</text>
            <text x="420" y="390">9.53</text>
            <text x="580" y="383">10.18</text>
            <text x="740" y="369">11.51</text>
            <text x="900" y="368">11.66</text>
          </g>

          <polyline points="100,274 260,237 420,204 580,162 740,165 900,121" fill="none" stroke="var(--accent)" stroke-width="6" stroke-linejoin="round"></polyline>
          <g fill="var(--accent)">
            <rect x="-11" y="-11" width="22" height="22" transform="translate(100,274) rotate(45)"></rect>
            <rect x="-11" y="-11" width="22" height="22" transform="translate(260,237) rotate(45)"></rect>
            <rect x="-11" y="-11" width="22" height="22" transform="translate(420,204) rotate(45)"></rect>
            <rect x="-11" y="-11" width="22" height="22" transform="translate(580,162) rotate(45)"></rect>
            <rect x="-11" y="-11" width="22" height="22" transform="translate(740,165) rotate(45)"></rect>
            <rect x="-11" y="-11" width="22" height="22" transform="translate(900,121) rotate(45)"></rect>
          </g>
          <g class="sv-acc-lab" text-anchor="middle">
            <text x="118" y="246">53.44</text>
            <text x="260" y="210">54.89</text>
            <text x="420" y="177">56.22</text>
            <text x="580" y="132">57.89</text>
            <text x="740" y="139">57.78</text>
            <text x="900" y="94">59.56</text>
          </g>
          <text class="sv-cap" x="500" y="580" text-anchor="middle">max turn budget K →</text>
        </svg>
      </div>
    </div>
    <div class="foot">OmniAgent · ICML 2026</div>
  </div>
</section>

<!-- ============ 13 · WEB DEMO ============ -->
<section data-label="Web demo" data-speaker-notes="(~25s) We also release a web demo, which is the same loop exposed in a browser. You can pick a built-in example or upload your own video, watch every observation, thought, and action step, and export the full trajectory as JSON.

We are not running it live at the booth, but everything is open source. One launch script and a single A100 gets you this exact interface, so you can try it on your own videos.">
  <div class="slide demo-slide" style="--pad-bottom: 140px;" data-screen-label="13 · Web demo">
    <div class="eyebrow">Release — web demo</div>
    <h2 class="title" style="margin-bottom: 30px;">The same loop, visible in the browser</h2>
    <div class="demo-grid">
      <div class="demo-panel rise">
        <p class="demo-k">inspectable interface</p>
        <ul class="demo-list">
          <li><strong>Choose or upload</strong>Built-in examples or your own video.</li>
          <li><strong>Watch the loop</strong>Observation → Thought → Action, every turn.</li>
          <li><strong>Inspect evidence</strong>Frames, audio, exported trajectory JSON.</li>
        </ul>
        <p class="demo-note">Fully open-source — one launch script and a single A100 80GB gets you this exact interface.</p>
      </div>
      <div class="demo-stage rise d1">
        <video class="demo-video" src="webdemo.mp4" poster="assets/webdemo-poster.jpg" muted playsinline controls preload="metadata" data-autoplay-on-slide></video>
        <p class="demo-caption"><strong>Recorded walkthrough:</strong> the active-perception loop exposed as a web interface for real inspection.</p>
      </div>
    </div>
  </div>
</section>

<!-- ============ 14 · TAKEAWAYS + CLOSING ============ -->
<section data-label="Takeaways + closing" data-speaker-notes="(~55s) Let me close with three takeaways. First, perception can be reasoning. Watching less but in the right places beats watching everything. Second, you have to train the search itself, because training on answers alone made the model worse. Third, active beats bigger, since our 7B outperforms a 72B while using far fewer frames.

One honest limitation is latency. The loop is sequential, so it is slower than a single forward pass, and parallel exploration is our next step.

Everything is released, including the code, the environment, both checkpoints, the SFT recipe, and the web demo, and inference runs on a single A100. Feel free to scan the QR codes. Our poster is this afternoon from 2:30 to 4:15 in Hall A, board 1200. Thank you.">
  <div class="slide no-slide-foot" data-screen-label="14 · Takeaways + closing">
    <div class="eyebrow">Takeaways</div>
    <h2 class="title" style="margin-bottom: 30px;">Three things to remember</h2>
    <div class="take-grid" style="flex: 0 0 auto;">
      <div class="take-card rise">
        <div class="take-head"><p class="take-num">1</p><span class="icon-badge" aria-hidden="true"><svg class="icon" viewBox="0 0 24 24"><path d="M2 12s3.5-6 10-6 10 6 10 6-3.5 6-10 6-10-6-10-6z"></path><circle cx="12" cy="12" r="3"></circle></svg></span></div>
        <h3 class="take-h">Perception can be reasoning</h3>
        <p class="take-p">It turns hours of video into a few notes — <strong>cost tracks the question, not the video</strong>.</p>
      </div>
      <div class="take-card rise d1">
        <div class="take-head"><p class="take-num">2</p><span class="icon-badge" aria-hidden="true"><svg class="icon" viewBox="0 0 24 24"><path d="M12 3v18"></path><path d="M6 7h12"></path><path d="M7 12h10"></path><path d="M8 17h8"></path></svg></span></div>
        <h3 class="take-h">Train the looking, not just the answering</h3>
        <p class="take-p">Passive SFT made things worse (43.0 → 41.6). <strong>Agentic SFT + TAURA</strong> fix that.</p>
      </div>
      <div class="take-card rise d2">
        <div class="take-head"><p class="take-num">3</p><span class="icon-badge" aria-hidden="true"><svg class="icon" viewBox="0 0 24 24"><path d="m3 17 6-6 4 4 8-8"></path><path d="M14 7h7v7"></path></svg></span></div>
        <h3 class="take-h">Active beats bigger</h3>
        <p class="take-p"><strong>7B &gt; 72B on LVBench</strong>, ~73% fewer frames — and it keeps improving the more it thinks (+6.2%).</p>
      </div>
    </div>
    <p class="limit-line rise d3"><strong>One honest caveat</strong> — the sequential loop adds latency vs. a single forward pass; parallelized exploration is next.</p>
    <div class="finale rise d3">
      <div>
        <p class="finale-url">github.com/HarryHsing/OmniAgent</p>
        <p class="finale-meta">
          <strong>Released</strong> — code · env · <strong>RL-7B / SFT-7B</strong> · demo · <span style="white-space: nowrap;">Apache-2.0</span><br>
          <strong>Paper</strong> — arXiv:2606.19341 · <strong>Booth</strong> — Jul 8 · 13:20 · <span style="white-space: nowrap;">Qwen #B400</span><br>
          <strong>Poster</strong> — Wed Jul 8 · <span style="white-space: nowrap;">2:30–4:15 PM KST</span> · <span style="white-space: nowrap;">Hall A #1200</span><br><span style="font-size: 21px;">“Native Active Perception as Reasoning for Omni-Modal Understanding”</span>
        </p>
      </div>
      <div class="finale-qrs">
        <div class="finale-qr">
          <img src="assets/qr-github.png" alt="QR code linking to the OmniAgent GitHub repository">
          <p class="finale-qr-label"><svg viewBox="0 0 16 16" aria-hidden="true"><path d="M8 0C3.58 0 0 3.58 0 8c0 3.54 2.29 6.53 5.47 7.59.4.07.55-.17.55-.38 0-.19-.01-.82-.01-1.49-2.01.37-2.53-.49-2.69-.94-.09-.23-.48-.94-.82-1.13-.28-.15-.68-.52-.01-.53.63-.01 1.08.58 1.23.82.72 1.21 1.87.87 2.33.66.07-.52.28-.87.51-1.07-1.78-.2-3.64-.89-3.64-3.95 0-.87.31-1.59.82-2.15-.08-.2-.36-1.02.08-2.12 0 0 .67-.21 2.2.82.64-.18 1.32-.27 2-.27.68 0 1.36.09 2 .27 1.53-1.04 2.2-.82 2.2-.82.44 1.1.16 1.92.08 2.12.51.56.82 1.27.82 2.15 0 3.07-1.87 3.75-3.65 3.95.29.25.54.73.54 1.48 0 1.07-.01 1.93-.01 2.2 0 .21.15.46.55.38A8.01 8.01 0 0 0 16 8c0-4.42-3.58-8-8-8z"></path></svg>GitHub</p>
        </div>
        <div class="finale-qr">
          <img src="assets/qr-x-thread.png" alt="QR code linking to the OmniAgent thread on X">
          <p class="finale-qr-label"><svg viewBox="0 0 24 24" aria-hidden="true"><path d="M18.24 2.25h3.31l-7.23 8.26 8.5 11.24h-6.66l-5.21-6.82-5.97 6.82H1.67l7.73-8.84L1.25 2.25h6.83l4.71 6.23 5.45-6.23z"></path></svg>Thread</p>
        </div>
      </div>
    </div>
  </div>
</section>

</deck-stage>

<template id="__bundler_thumbnail">
  <svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 100 100">
    <rect width="100" height="100" fill="#1a1d2e"/>
    <circle cx="50" cy="50" r="20" fill="none" stroke="#e8734a" stroke-width="4"/>
    <circle cx="50" cy="50" r="7" fill="#e8734a"/>
  </svg>
</template>

<script src="deck-stage.js"></script>
<script src="deck-sync.js"></script>
<script>
customElements.whenDefined('deck-stage').then(() => {
  const stage = document.querySelector('deck-stage');
  if (!stage) return;

  const syncDemoPlayback = (slide) => {
    document.querySelectorAll('video').forEach((video) => {
      if (!slide || !slide.contains(video) || !video.hasAttribute('data-autoplay-on-slide')) {
        video.pause();
        return;
      }
      video.play().catch(() => {});
    });
  };

  stage.addEventListener('slidechange', (event) => {
    syncDemoPlayback(event.detail && event.detail.slide);
  });

  syncDemoPlayback(document.querySelector('[data-deck-active]'));
});
</script>
<script>
// True fullscreen (hides browser tabs & address bar): press F or click the ⛶ button.
(function () {
  function toggleFullscreen() {
    if (document.fullscreenElement) {
      document.exitFullscreen();
    } else {
      document.documentElement.requestFullscreen().catch(() => {
        alert('无法进入全屏:请先用"在新标签页打开"在浏览器单独打开本文件,再点击 ⛶ 或按 F。');
      });
    }
  }

  document.addEventListener('keydown', (e) => {
    if ((e.key === 'f' || e.key === 'F') && !e.metaKey && !e.ctrlKey && !e.altKey) {
      toggleFullscreen();
    }
  });

  const btn = document.createElement('button');
  btn.id = 'fullscreen-btn';
  btn.title = '全屏 (F)';
  btn.textContent = '⛶';
  btn.setAttribute('data-omelette-chrome', '');
  btn.style.cssText = 'position:fixed;right:14px;bottom:14px;z-index:9999;width:40px;height:40px;border:none;border-radius:8px;background:rgba(0,0,0,0.55);color:#fff;font-size:20px;line-height:1;cursor:pointer;opacity:0.35;transition:opacity .2s;';
  btn.addEventListener('mouseenter', () => { btn.style.opacity = '1'; });
  btn.addEventListener('mouseleave', () => { btn.style.opacity = '0.35'; });
  btn.addEventListener('click', toggleFullscreen);
  document.addEventListener('fullscreenchange', () => {
    btn.textContent = document.fullscreenElement ? '⛶' : '⛶';
    btn.title = document.fullscreenElement ? '退出全屏 (F / Esc)' : '全屏 (F)';
  });
  document.body.appendChild(btn);
})();
</script>

</body>
</html>