futurefantasy commited on
Commit
4bde30c
·
verified ·
1 Parent(s): c52a7a7

Rename repo to VLAC-Cut and fix parameter metadata

Browse files
README.md CHANGED
@@ -9,9 +9,9 @@ tags:
9
  - progress-estimation
10
  ---
11
 
12
- # VLAC-Cut-Qwen3VL-30B-A3B-Progress
13
 
14
- VLAC-Cut-Qwen3VL-30B-A3B-Progress is a video-language progress estimation model for robotic manipulation. Given a task description and a video, it predicts time-progress keypoints that can be aligned into a progress curve.
15
 
16
  This release contains the inference-ready model weights, tokenizer and processor files, three bundled demo episodes, reference prediction outputs, and minimal local inference utilities.
17
 
@@ -33,7 +33,7 @@ This release contains the inference-ready model weights, tokenizer and processor
33
  ```python
34
  from transformers import AutoModelForImageTextToText, AutoProcessor
35
 
36
- model_id = "InternRobotics/VLAC-Cut-Qwen3VL-30B-A3B-Progress"
37
 
38
  processor = AutoProcessor.from_pretrained(model_id)
39
  model = AutoModelForImageTextToText.from_pretrained(
@@ -49,7 +49,7 @@ Run progress inference on an arbitrary video:
49
 
50
  ```bash
51
  python quick_start/run_example.py \
52
- --model-path /path/to/VLAC-Cut-Qwen3VL-30B-A3B-Progress \
53
  --video-path /path/to/video.mp4 \
54
  --task-instruction "<natural-language task instruction>" \
55
  --task-plan $'<optional step-by-step task plan>' \
 
9
  - progress-estimation
10
  ---
11
 
12
+ # VLAC-Cut
13
 
14
+ VLAC-Cut is a video-language progress estimation model for robotic manipulation. Given a task description and a video, it predicts time-progress keypoints that can be aligned into a progress curve.
15
 
16
  This release contains the inference-ready model weights, tokenizer and processor files, three bundled demo episodes, reference prediction outputs, and minimal local inference utilities.
17
 
 
33
  ```python
34
  from transformers import AutoModelForImageTextToText, AutoProcessor
35
 
36
+ model_id = "InternRobotics/VLAC-Cut"
37
 
38
  processor = AutoProcessor.from_pretrained(model_id)
39
  model = AutoModelForImageTextToText.from_pretrained(
 
49
 
50
  ```bash
51
  python quick_start/run_example.py \
52
+ --model-path /path/to/VLAC-Cut \
53
  --video-path /path/to/video.mp4 \
54
  --task-instruction "<natural-language task instruction>" \
55
  --task-plan $'<optional step-by-step task plan>' \
model.safetensors.index.json CHANGED
@@ -1,6 +1,6 @@
1
  {
2
  "metadata": {
3
- "total_parameters": 664816,
4
  "total_size": 62141508064
5
  },
6
  "weight_map": {
 
1
  {
2
  "metadata": {
3
+ "total_parameters": 31070754032,
4
  "total_size": 62141508064
5
  },
6
  "weight_map": {
quick_start/README.md CHANGED
@@ -23,7 +23,7 @@ Run inference:
23
 
24
  ```bash
25
  python quick_start/run_example.py \
26
- --model-path /path/to/VLAC-Cut-Qwen3VL-30B-A3B-Progress \
27
  --video-path /path/to/video.mp4 \
28
  --task-instruction "<natural-language task instruction>" \
29
  --task-plan $'<optional step-by-step task plan>' \
 
23
 
24
  ```bash
25
  python quick_start/run_example.py \
26
+ --model-path /path/to/VLAC-Cut \
27
  --video-path /path/to/video.mp4 \
28
  --task-instruction "<natural-language task instruction>" \
29
  --task-plan $'<optional step-by-step task plan>' \
training_summary.json CHANGED
@@ -1,5 +1,5 @@
1
  {
2
- "model_name": "VLAC2-Qwen3VL-30B-A3B-Progress",
3
  "base_model": "Qwen/Qwen3-VL-30B-A3B-Instruct",
4
  "exported_architecture": "Qwen3VLMoeForConditionalGeneration",
5
  "config_model_type": "qwen3_vl_moe",
@@ -8,7 +8,7 @@
8
  "template": "qwen3_vl",
9
  "transformers_version": "4.57.6",
10
  "notes": [
11
- "Released VLAC2 model for video-language progress estimation in robotic manipulation.",
12
  "This summary records the base model and runtime loading information for the released checkpoint."
13
  ]
14
  }
 
1
  {
2
+ "model_name": "VLAC-Cut",
3
  "base_model": "Qwen/Qwen3-VL-30B-A3B-Instruct",
4
  "exported_architecture": "Qwen3VLMoeForConditionalGeneration",
5
  "config_model_type": "qwen3_vl_moe",
 
8
  "template": "qwen3_vl",
9
  "transformers_version": "4.57.6",
10
  "notes": [
11
+ "Released VLAC-Cut model for video-language progress estimation in robotic manipulation.",
12
  "This summary records the base model and runtime loading information for the released checkpoint."
13
  ]
14
  }