Image-Text-to-Text
Transformers
Safetensors
qwen3_5
vllm
video
multimodal
reinforcement-learning
temporal-grounding
object-tracking
video-segmentation
visual-question-answering
spatial-reasoning
qwen3.5
conversational
Instructions to use OraRL/Video-ORA-9B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use OraRL/Video-ORA-9B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-text-to-text", model="OraRL/Video-ORA-9B") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("OraRL/Video-ORA-9B") model = AutoModelForMultimodalLM.from_pretrained("OraRL/Video-ORA-9B", device_map="auto") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use OraRL/Video-ORA-9B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "OraRL/Video-ORA-9B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "OraRL/Video-ORA-9B", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker
docker model run hf.co/OraRL/Video-ORA-9B
- SGLang
How to use OraRL/Video-ORA-9B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "OraRL/Video-ORA-9B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "OraRL/Video-ORA-9B", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "OraRL/Video-ORA-9B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "OraRL/Video-ORA-9B", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }' - Docker Model Runner
How to use OraRL/Video-ORA-9B with Docker Model Runner:
docker model run hf.co/OraRL/Video-ORA-9B
| # Copyright 2024 Bytedance Ltd. and/or its affiliates | |
| # | |
| # Licensed under the Apache License, Version 2.0 (the "License"); | |
| # you may not use this file except in compliance with the License. | |
| # You may obtain a copy of the License at | |
| # | |
| # http://www.apache.org/licenses/LICENSE-2.0 | |
| # | |
| # Unless required by applicable law or agreed to in writing, software | |
| # distributed under the License is distributed on an "AS IS" BASIS, | |
| # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | |
| # See the License for the specific language governing permissions and | |
| # limitations under the License. | |
| """ | |
| Implement Actor | |
| """ | |
| import os | |
| from collections import defaultdict | |
| from typing import Any, Optional | |
| import numpy as np | |
| import torch | |
| import torch.distributed as dist | |
| from einops import rearrange | |
| from ray.experimental.tqdm_ray import tqdm | |
| from torch import nn | |
| from torch.distributed.fsdp import FullyShardedDataParallel as FSDP | |
| from ...protocol import DataProto, batch_collate | |
| from ...trainer.core_algos import average_loss, compute_kl, compute_policy_loss | |
| from ...utils import torch_functional as VF | |
| from ...utils.py_functional import append_to_dict | |
| from ...utils.seqlen_balancing import prepare_dynamic_batch, restore_dynamic_batch | |
| from ...utils.ulysses import gather_outputs_and_unpad, ulysses_pad_and_slice_inputs | |
| from .base import BasePPOActor | |
| from .config import ActorConfig | |
| try: | |
| from flash_attn.bert_padding import index_first_axis, pad_input, rearrange, unpad_input | |
| except ImportError: | |
| pass | |
| __all__ = ["DataParallelPPOActor"] | |
| def _disable_tqdm() -> bool: | |
| return os.getenv("VERL_DISABLE_TQDM", "0") == "1" | |
| def _loss_units(response_mask: torch.Tensor, loss_avg_mode: str) -> torch.Tensor: | |
| """Number of local averaging units represented by a (micro-)batch.""" | |
| if loss_avg_mode == "token": | |
| return response_mask.to(dtype=torch.float32).sum() | |
| if loss_avg_mode == "seq": | |
| return response_mask.new_tensor(float(response_mask.shape[0]), dtype=torch.float32) | |
| raise ValueError(f"Unsupported loss_avg_mode={loss_avg_mode!r}; expected 'token' or 'seq'.") | |
| def _micro_batch_loss_weight( | |
| response_mask: torch.Tensor, | |
| *, | |
| global_loss_units: torch.Tensor, | |
| world_size: int, | |
| loss_avg_mode: str, | |
| ) -> torch.Tensor: | |
| """Undo micro-batch averaging before FSDP's cross-rank gradient average.""" | |
| if float(global_loss_units.item()) <= 0: | |
| raise ValueError("global_loss_units must be positive.") | |
| return _loss_units(response_mask, loss_avg_mode) * world_size / global_loss_units | |
| class DataParallelPPOActor(BasePPOActor): | |
| def __init__( | |
| self, | |
| config: ActorConfig, | |
| actor_module: nn.Module, | |
| actor_optimizer: Optional[torch.optim.Optimizer] = None, | |
| ): | |
| """ | |
| When optimizer is None, it is Reference Policy | |
| """ | |
| super().__init__(config) | |
| self.rank = int(os.getenv("RANK", "0")) | |
| self.world_size = int(os.getenv("WORLD_SIZE", "1")) | |
| self.actor_module = actor_module | |
| self.actor_optimizer = actor_optimizer | |
| if config.use_torch_compile: | |
| self.log_probs_from_logits = torch.compile(VF.log_probs_from_logits, dynamic=True) | |
| else: | |
| self.log_probs_from_logits = VF.log_probs_from_logits | |
| def _forward_micro_batch(self, micro_batch: dict[str, torch.Tensor], temperature: float) -> torch.Tensor: | |
| """ | |
| Returns: | |
| log_probs: # (bs, response_len) | |
| """ | |
| input_ids = micro_batch["input_ids"] | |
| batch_size, seqlen = input_ids.shape | |
| attention_mask = micro_batch["attention_mask"] | |
| position_ids = micro_batch["position_ids"] | |
| responses = micro_batch["responses"] | |
| response_length = responses.size(-1) | |
| if position_ids.dim() == 3: # qwen2vl mrope | |
| position_ids = position_ids.transpose(0, 1) # (bsz, 4, seqlen) -> (4, bsz, seqlen) | |
| multi_modal_inputs = defaultdict(list) | |
| if "multi_modal_inputs" in micro_batch: | |
| multi_modal_inputs = batch_collate(micro_batch["multi_modal_inputs"]) | |
| merged = {} | |
| for key, value in multi_modal_inputs.items(): | |
| tensors = [v for v in value if v is not None] | |
| if tensors: | |
| merged[key] = torch.cat(tensors, dim=0) | |
| multi_modal_inputs = merged | |
| else: | |
| multi_modal_inputs = {} | |
| if self.config.padding_free: | |
| input_ids_rmpad, indices, *_ = unpad_input(input_ids.unsqueeze(-1), attention_mask) # (total_nnz, 1) | |
| input_ids_rmpad = input_ids_rmpad.transpose(0, 1) # (1, total_nnz) | |
| # unpad the position_ids to align the rotary | |
| if position_ids.dim() == 3: | |
| position_ids_rmpad = ( | |
| index_first_axis(rearrange(position_ids, "c b s ... -> (b s) c ..."), indices) | |
| .transpose(0, 1) | |
| .unsqueeze(1) | |
| ) # (4, bsz, seqlen) -> (4, 1, bsz * seqlen) | |
| else: | |
| position_ids_rmpad = index_first_axis( | |
| rearrange(position_ids.unsqueeze(-1), "b s ... -> (b s) ..."), indices | |
| ).transpose(0, 1) | |
| # for compute the log_prob | |
| input_ids_rmpad_rolled = torch.roll(input_ids_rmpad, shifts=-1, dims=1) # (1, total_nnz) | |
| # pad and slice the inputs if sp > 1 | |
| if self.config.ulysses_size > 1: | |
| input_ids_rmpad, position_ids_rmpad, pad_size = ulysses_pad_and_slice_inputs( | |
| input_ids_rmpad, position_ids_rmpad, sp_size=self.config.ulysses_size | |
| ) | |
| input_ids_rmpad_rolled, _, _ = ulysses_pad_and_slice_inputs( | |
| input_ids_rmpad_rolled, None, self.config.ulysses_size | |
| ) | |
| input_ids_rmpad_rolled = input_ids_rmpad_rolled.squeeze(0) # ((total_nnz / sp) + pad) | |
| # only pass input_ids and position_ids to enable flash_attn_varlen | |
| output = self.actor_module( | |
| input_ids=input_ids_rmpad, | |
| attention_mask=None, | |
| position_ids=position_ids_rmpad, | |
| **multi_modal_inputs, | |
| use_cache=False, | |
| ) # prevent model thinks we are generating | |
| logits_rmpad = output.logits.squeeze(0) # (total_nnz, vocab_size) | |
| logits_rmpad.div_(temperature) | |
| # ((total_nnz / sp) + pad) | |
| log_probs = self.log_probs_from_logits(logits=logits_rmpad, labels=input_ids_rmpad_rolled) | |
| # gather log_prob if sp > 1 | |
| if self.config.ulysses_size > 1: | |
| # gather and unpad for the ulysses sp | |
| log_probs = gather_outputs_and_unpad(log_probs, gather_dim=0, unpad_dim=0, padding_size=pad_size) | |
| # pad back to (bsz, seqlen) | |
| full_log_probs = pad_input( | |
| hidden_states=log_probs.unsqueeze(-1), indices=indices, batch=batch_size, seqlen=seqlen | |
| ) | |
| log_probs = full_log_probs.squeeze(-1)[:, -response_length - 1 : -1] # (bsz, response_length) | |
| else: | |
| output = self.actor_module( | |
| input_ids=input_ids, | |
| attention_mask=attention_mask, | |
| position_ids=position_ids, | |
| **multi_modal_inputs, | |
| use_cache=False, | |
| ) | |
| logits: torch.Tensor = output.logits | |
| logits.div_(temperature) | |
| logits = logits[:, -response_length - 1 : -1, :] # (bsz, response_length, vocab_size) | |
| log_probs = self.log_probs_from_logits(logits, responses) # (bsz, response_length) | |
| return log_probs | |
| def _optimizer_step(self) -> torch.Tensor: | |
| if isinstance(self.actor_module, FSDP): | |
| grad_norm = self.actor_module.clip_grad_norm_(self.config.max_grad_norm) | |
| else: | |
| grad_norm = nn.utils.clip_grad_norm_(self.actor_module.parameters(), max_norm=self.config.max_grad_norm) | |
| if not torch.isfinite(grad_norm): | |
| print("Gradient norm is not finite. Skip update.") | |
| else: | |
| self.actor_optimizer.step() | |
| self.actor_optimizer.zero_grad() | |
| return grad_norm | |
| def compute_log_prob(self, data: DataProto) -> torch.Tensor: | |
| """Compute the log probability of the responses given input_ids, attention_mask and position_ids | |
| Args: | |
| data (DataProto): a DataProto containing keys | |
| ``input_ids``: tensor of shape [batch_size, sequence_length]. torch.int64. Note that input_ids is the | |
| concatenation of prompt and response. Note that ``sequence_length = prompt_length + response_length``. | |
| ``attention_mask``: tensor of shape [batch_size, sequence_length]. torch.int64. | |
| ``position_ids``: tensor of shape [batch_size, sequence_length]. torch.int64. | |
| ``responses``: tensor of shape [batch_size, response_length]. torch.int64. | |
| Returns: | |
| torch.Tensor: the log_prob tensor | |
| """ | |
| self.actor_module.eval() | |
| temperature = data.meta_info["temperature"] | |
| select_keys = ["input_ids", "attention_mask", "position_ids", "responses"] | |
| non_tensor_select_keys = ["multi_modal_inputs"] | |
| data = data.select(select_keys, non_tensor_select_keys) | |
| if self.config.dynamic_batching: | |
| if self.config.max_token_len_per_gpu is not None: | |
| max_token_len = self.config.max_token_len_per_gpu | |
| else: | |
| max_token_len = self.config.micro_batch_size_per_device_for_experience * data.batch["input_ids"].size( | |
| -1 | |
| ) | |
| micro_batches, batch_idx_list = prepare_dynamic_batch(data, max_token_len=max_token_len) | |
| else: | |
| micro_batches = data.split(self.config.micro_batch_size_per_device_for_experience) | |
| log_probs_lst = [] | |
| if self.rank == 0 and not _disable_tqdm(): | |
| micro_batches = tqdm(micro_batches, desc="Compute log probs", position=1) | |
| for micro_batch in micro_batches: | |
| model_inputs = {**micro_batch.batch, **micro_batch.non_tensor_batch} | |
| log_probs = self._forward_micro_batch(model_inputs, temperature=temperature) | |
| log_probs_lst.append(log_probs) | |
| log_probs = torch.concat(log_probs_lst, dim=0) | |
| if self.config.dynamic_batching: | |
| log_probs = restore_dynamic_batch(log_probs, batch_idx_list) | |
| return log_probs | |
| def update_policy(self, data: DataProto) -> dict[str, Any]: | |
| self.actor_module.train() | |
| temperature = data.meta_info["temperature"] # temperature must be in the data.meta_info to avoid slient error | |
| select_keys = ["input_ids", "attention_mask", "position_ids", "responses", "response_mask"] | |
| select_keys.extend(["old_log_probs", "ref_log_probs", "advantages"]) | |
| non_tensor_select_keys = ["multi_modal_inputs"] | |
| skip_old_log_probs = bool(data.meta_info.get("skip_old_log_probs", False)) | |
| # Split to make minibatch iterator for updating the actor | |
| # See PPO paper for details. https://arxiv.org/abs/1707.06347 | |
| mini_batches = data.select(select_keys, non_tensor_select_keys).split(self.config.global_batch_size_per_device) | |
| metrics = defaultdict(list) | |
| for _ in range(self.config.ppo_epochs): | |
| if self.rank == 0 and not _disable_tqdm(): | |
| mini_batches = tqdm(mini_batches, desc="Train mini-batches", position=1) | |
| for mini_batch in mini_batches: | |
| # ``average_loss`` already reduces each micro-batch by either | |
| # response token or sequence. Scale it back by the same unit | |
| # before FSDP averages gradients across ranks. Using token count | |
| # here unconditionally silently reintroduced length weighting | |
| # even when loss_avg_mode="seq". | |
| global_loss_units = _loss_units( | |
| mini_batch.batch["response_mask"], | |
| self.config.loss_avg_mode, | |
| ) | |
| dist.all_reduce(global_loss_units, op=dist.ReduceOp.SUM) | |
| if self.config.dynamic_batching: | |
| if self.config.max_token_len_per_gpu is not None: | |
| max_token_len = self.config.max_token_len_per_gpu | |
| else: | |
| max_input_len = mini_batch.batch["input_ids"].size(-1) | |
| max_token_len = self.config.micro_batch_size_per_device_for_update * max_input_len | |
| micro_batches, _ = prepare_dynamic_batch(mini_batch, max_token_len=max_token_len) | |
| else: | |
| micro_batches = mini_batch.split(self.config.micro_batch_size_per_device_for_update) | |
| if self.rank == 0 and not _disable_tqdm(): | |
| micro_batches = tqdm(micro_batches, desc="Update policy", position=2) | |
| for micro_batch in micro_batches: | |
| model_inputs = {**micro_batch.batch, **micro_batch.non_tensor_batch} | |
| response_mask = model_inputs["response_mask"] | |
| advantages = model_inputs["advantages"] | |
| # all return: (bsz, response_length) | |
| log_probs = self._forward_micro_batch(model_inputs, temperature=temperature) | |
| old_log_probs = model_inputs.get("old_log_probs") | |
| if old_log_probs is None: | |
| if not skip_old_log_probs: | |
| raise KeyError("old_log_probs is required unless skip_old_log_probs=True.") | |
| old_log_probs = log_probs.detach() | |
| pg_loss, pg_metrics = compute_policy_loss( | |
| old_log_probs=old_log_probs, | |
| log_probs=log_probs, | |
| advantages=advantages, | |
| response_mask=response_mask, | |
| clip_ratio_low=self.config.clip_ratio_low, | |
| clip_ratio_high=self.config.clip_ratio_high, | |
| clip_ratio_dual=self.config.clip_ratio_dual, | |
| loss_type=self.config.loss_type, | |
| loss_avg_mode=self.config.loss_avg_mode, | |
| ) | |
| if self.config.use_kl_loss and "ref_log_probs" in model_inputs: | |
| ref_log_probs = model_inputs["ref_log_probs"] | |
| # compute kl loss | |
| kld = compute_kl( | |
| log_probs=log_probs, | |
| ref_log_probs=ref_log_probs, | |
| kl_penalty=self.config.kl_penalty, | |
| ) | |
| kl_loss = average_loss(kld, response_mask, mode=self.config.loss_avg_mode) | |
| loss = pg_loss + kl_loss * self.config.kl_coef | |
| metrics["actor/kl_loss"] = kl_loss.detach().item() | |
| metrics["actor/kl_coef"] = self.config.kl_coef | |
| else: | |
| loss = pg_loss | |
| loss = loss * _micro_batch_loss_weight( | |
| response_mask, | |
| global_loss_units=global_loss_units, | |
| world_size=self.world_size, | |
| loss_avg_mode=self.config.loss_avg_mode, | |
| ) | |
| loss.backward() | |
| batch_metrics = {f"actor/{k}": v for k, v in pg_metrics.items()} | |
| batch_metrics["actor/pg_loss"] = pg_loss.detach().item() | |
| append_to_dict(metrics, batch_metrics) | |
| grad_norm = self._optimizer_step() | |
| append_to_dict(metrics, {"actor/grad_norm": grad_norm.detach().item()}) | |
| return metrics | |