Image-Text-to-Text
Transformers
Safetensors
qwen3_5
vllm
video
multimodal
reinforcement-learning
temporal-grounding
object-tracking
video-segmentation
visual-question-answering
spatial-reasoning
qwen3.5
conversational
Instructions to use OraRL/Video-ORA-4B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use OraRL/Video-ORA-4B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-text-to-text", model="OraRL/Video-ORA-4B") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("OraRL/Video-ORA-4B") model = AutoModelForMultimodalLM.from_pretrained("OraRL/Video-ORA-4B", device_map="auto") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use OraRL/Video-ORA-4B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "OraRL/Video-ORA-4B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "OraRL/Video-ORA-4B", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker
docker model run hf.co/OraRL/Video-ORA-4B
- SGLang
How to use OraRL/Video-ORA-4B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "OraRL/Video-ORA-4B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "OraRL/Video-ORA-4B", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "OraRL/Video-ORA-4B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "OraRL/Video-ORA-4B", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }' - Docker Model Runner
How to use OraRL/Video-ORA-4B with Docker Model Runner:
docker model run hf.co/OraRL/Video-ORA-4B
| # Copyright 2024 Bytedance Ltd. and/or its affiliates | |
| # | |
| # Licensed under the Apache License, Version 2.0 (the "License"); | |
| # you may not use this file except in compliance with the License. | |
| # You may obtain a copy of the License at | |
| # | |
| # http://www.apache.org/licenses/LICENSE-2.0 | |
| # | |
| # Unless required by applicable law or agreed to in writing, software | |
| # distributed under the License is distributed on an "AS IS" BASIS, | |
| # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | |
| # See the License for the specific language governing permissions and | |
| # limitations under the License. | |
| import json | |
| import os | |
| import ray | |
| from omegaconf import OmegaConf | |
| from ..single_controller.ray import RayWorkerGroup | |
| from ..utils.tokenizer import get_processor, get_tokenizer | |
| from ..workers.fsdp_workers import FSDPWorker | |
| from ..workers.reward import AutoRewardManager | |
| from .config import PPOConfig | |
| from .data_loader import create_dataloader | |
| from .ray_trainer import RayPPOTrainer, ResourcePoolManager, Role | |
| # please make sure main_task is not scheduled on head | |
| class Runner: | |
| """A runner for RL training.""" | |
| def run(self, config: PPOConfig): | |
| # print config | |
| print(json.dumps(config.to_dict(), indent=2)) | |
| # instantiate tokenizer | |
| tokenizer_path = config.worker.actor.model.tokenizer_path | |
| tokenizer = get_tokenizer( | |
| tokenizer_path, | |
| override_chat_template=config.data.override_chat_template, | |
| trust_remote_code=config.worker.actor.model.trust_remote_code, | |
| use_fast=True, | |
| ) | |
| processor = get_processor( | |
| tokenizer_path, | |
| override_chat_template=config.data.override_chat_template, | |
| trust_remote_code=config.worker.actor.model.trust_remote_code, | |
| use_fast=True, | |
| ) | |
| # define worker classes | |
| ray_worker_group_cls = RayWorkerGroup | |
| role_worker_mapping = { | |
| Role.ActorRolloutRef: ray.remote(FSDPWorker), | |
| Role.Critic: ray.remote(FSDPWorker), | |
| } | |
| global_pool_id = "global_pool" | |
| resource_pool_spec = { | |
| global_pool_id: [config.trainer.n_gpus_per_node] * config.trainer.nnodes, | |
| } | |
| mapping = { | |
| Role.ActorRolloutRef: global_pool_id, | |
| Role.Critic: global_pool_id, | |
| } | |
| resource_pool_manager = ResourcePoolManager(resource_pool_spec=resource_pool_spec, mapping=mapping) | |
| RemoteRewardManager = ray.remote(AutoRewardManager).options(num_cpus=config.worker.reward.num_cpus) | |
| reward_fn = RemoteRewardManager.remote(config.worker.reward, tokenizer) | |
| val_reward_fn = RemoteRewardManager.remote(config.worker.reward, tokenizer) | |
| train_dataloader, val_dataloader = create_dataloader( | |
| config.data, tokenizer, processor, | |
| model_path=config.worker.actor.model.model_path, | |
| ) | |
| trainer = RayPPOTrainer( | |
| config=config, | |
| tokenizer=tokenizer, | |
| processor=processor, | |
| train_dataloader=train_dataloader, | |
| val_dataloader=val_dataloader, | |
| role_worker_mapping=role_worker_mapping, | |
| resource_pool_manager=resource_pool_manager, | |
| ray_worker_group_cls=ray_worker_group_cls, | |
| reward_fn=reward_fn, | |
| val_reward_fn=val_reward_fn, | |
| ) | |
| trainer.init_workers() | |
| trainer.fit() | |
| def main(): | |
| cli_args = OmegaConf.from_cli() | |
| default_config = OmegaConf.structured(PPOConfig()) | |
| if hasattr(cli_args, "config"): | |
| config_path = cli_args.pop("config", None) | |
| file_config = OmegaConf.load(config_path) | |
| default_config = OmegaConf.merge(default_config, file_config) | |
| ppo_config = OmegaConf.merge(default_config, cli_args) | |
| ppo_config: PPOConfig = OmegaConf.to_object(ppo_config) | |
| ppo_config.deep_post_init() | |
| if not ray.is_initialized(): | |
| runtime_env = { | |
| "env_vars": { | |
| "TOKENIZERS_PARALLELISM": "true", | |
| "NCCL_DEBUG": "WARN", | |
| "VLLM_LOGGING_LEVEL": "WARN", | |
| "TORCH_NCCL_AVOID_RECORD_STREAMS": "1", | |
| "PYTORCH_CUDA_ALLOC_CONF": "expandable_segments:False", | |
| "CUDA_DEVICE_MAX_CONNECTIONS": "1", | |
| "VLLM_ALLREDUCE_USE_SYMM_MEM": "0", | |
| } | |
| } | |
| # Forward diagnostic/guard toggles (VERL_*) from the driver to every Ray | |
| # worker so the no-cache video token/feature audit + guard behave | |
| # consistently across nodes (e.g. VERL_DIAGNOSE_VIDEO_MISMATCH_LOG). | |
| for _key, _val in os.environ.items(): | |
| if _key.startswith("VERL_"): | |
| runtime_env["env_vars"].setdefault(_key, _val) | |
| ray.init(runtime_env=runtime_env) | |
| runner = Runner.remote() | |
| ray.get(runner.run.remote(ppo_config)) | |
| if ppo_config.trainer.ray_timeline is not None: | |
| # use `export RAY_PROFILING=1` to record the ray timeline | |
| ray.timeline(filename=ppo_config.trainer.ray_timeline) | |
| if __name__ == "__main__": | |
| main() | |