File size: 3,158 Bytes
0185029
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
# Copy this file, then replace every ../local_data placeholder.
#
# OraRL never downloads data or media. You must obtain each dataset under its
# license, keep the required attribution, and point input/media_root at your
# own local, licensed copy. The optional license and url values are audit
# metadata only and are never opened by the builder.
#
# The seven quotas reproduce the final paper mixture: 100,032 train prompts
# (1,563 batches of 64), with 62,656 structured prompts and 37,376 answer-only
# prompts. The canary is
# additional held-out data and is not subtracted from the train target.
version: 1
seed: 42
target: 100032
canary_size: 512
max_prompts_per_media: 2
require_media: true
allow_shortfall: false

# Supply JSON/JSONL benchmark records before a release build. Any candidate
# sharing a normalized prompt identity or media identity is removed.
benchmark_excludes:
  - ../local_data/exclusions/public_benchmarks.jsonl

sources:
  - name: temporal_grounding_train
    input: ../local_data/annotations/temporal_grounding.jsonl
    task: temporal grounding
    family: temporal
    quota: 20096
    media_root: ../local_data/media/temporal
    license: REPLACE_WITH_DATASET_LICENSE
    url: REPLACE_WITH_DATASET_HOME_PAGE

  - name: tracking_train
    input: ../local_data/annotations/tracking.jsonl
    task: tracking
    family: tracking
    quota: 13952
    media_root: ../local_data/media/tracking
    license: REPLACE_WITH_DATASET_LICENSE
    url: REPLACE_WITH_DATASET_HOME_PAGE

  - name: segmentation_train
    input: ../local_data/annotations/segmentation.jsonl
    task: segmentation
    family: segmentation
    quota: 12032
    media_root: ../local_data/media/segmentation
    license: REPLACE_WITH_DATASET_LICENSE
    url: REPLACE_WITH_DATASET_HOME_PAGE

  - name: spatial_grounding_train
    input: ../local_data/annotations/spatial_grounding.jsonl
    task: spatial grounding
    family: spatial
    quota: 7040
    media_root: ../local_data/media/spatial
    license: REPLACE_WITH_DATASET_LICENSE
    url: REPLACE_WITH_DATASET_HOME_PAGE

  - name: spatial_temporal_grounding_train
    input: ../local_data/annotations/spatial_temporal_grounding.jsonl
    task: spatial-temporal grounding
    family: stvg
    quota: 9536
    media_root: ../local_data/media/spatial_temporal
    license: REPLACE_WITH_DATASET_LICENSE
    url: REPLACE_WITH_DATASET_HOME_PAGE

  - name: video_qa_train
    input: ../local_data/annotations/video_qa.jsonl
    task: video_qa_mc
    family: video_qa
    quota: 20288
    media_root: ../local_data/media/video_qa
    license: REPLACE_WITH_DATASET_LICENSE
    url: REPLACE_WITH_DATASET_HOME_PAGE

  - name: spatial_intelligence_train
    input: ../local_data/annotations/spatial_intelligence.jsonl
    task: spatial intelligence
    family: spatial_intelligence
    # Keep subtype labels such as object_counting and object_rel_direction;
    # the reward adapter uses them to select the official scoring rule.
    preserve_problem_type: true
    quota: 17088
    media_root: ../local_data/media/spatial_intelligence
    license: REPLACE_WITH_DATASET_LICENSE
    url: REPLACE_WITH_DATASET_HOME_PAGE