Image-Text-to-Text
Transformers
Safetensors
qwen3_5
vllm
video
multimodal
reinforcement-learning
temporal-grounding
object-tracking
video-segmentation
visual-question-answering
spatial-reasoning
qwen3.5
conversational
Instructions to use OraRL/Video-ORA-9B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use OraRL/Video-ORA-9B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-text-to-text", model="OraRL/Video-ORA-9B") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("OraRL/Video-ORA-9B") model = AutoModelForMultimodalLM.from_pretrained("OraRL/Video-ORA-9B", device_map="auto") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use OraRL/Video-ORA-9B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "OraRL/Video-ORA-9B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "OraRL/Video-ORA-9B", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker
docker model run hf.co/OraRL/Video-ORA-9B
- SGLang
How to use OraRL/Video-ORA-9B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "OraRL/Video-ORA-9B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "OraRL/Video-ORA-9B", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "OraRL/Video-ORA-9B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "OraRL/Video-ORA-9B", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }' - Docker Model Runner
How to use OraRL/Video-ORA-9B with Docker Model Runner:
docker model run hf.co/OraRL/Video-ORA-9B
| # Copyright 2024 Bytedance Ltd. and/or its affiliates | |
| # | |
| # Licensed under the Apache License, Version 2.0 (the "License"); | |
| # you may not use this file except in compliance with the License. | |
| # You may obtain a copy of the License at | |
| # | |
| # http://www.apache.org/licenses/LICENSE-2.0 | |
| # | |
| # Unless required by applicable law or agreed to in writing, software | |
| # distributed under the License is distributed on an "AS IS" BASIS, | |
| # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | |
| # See the License for the specific language governing permissions and | |
| # limitations under the License. | |
| """ | |
| Contain small python utility functions | |
| """ | |
| import importlib.metadata | |
| import importlib.util | |
| import os | |
| import re | |
| from contextlib import contextmanager | |
| from functools import lru_cache | |
| from typing import Any, Optional, Union | |
| import numpy as np | |
| import yaml | |
| from codetiming import Timer | |
| from packaging import version | |
| from yaml import Dumper | |
| def is_sci_notation(number: float) -> bool: | |
| pattern = re.compile(r"^[+-]?\d+(\.\d*)?[eE][+-]?\d+$") | |
| return bool(pattern.match(str(number))) | |
| def float_representer(dumper: Dumper, number: Union[float, np.float32, np.float64]): | |
| if is_sci_notation(number): | |
| value = str(number) | |
| if "." not in value and "e" in value: | |
| value = value.replace("e", ".0e", 1) | |
| else: | |
| value = str(round(number, 3)) | |
| return dumper.represent_scalar("tag:yaml.org,2002:float", value) | |
| yaml.add_representer(float, float_representer) | |
| yaml.add_representer(np.float32, float_representer) | |
| yaml.add_representer(np.float64, float_representer) | |
| def is_package_available(name: str) -> bool: | |
| return importlib.util.find_spec(name) is not None | |
| def get_package_version(name: str) -> "version.Version": | |
| try: | |
| return version.parse(importlib.metadata.version(name)) | |
| except Exception: | |
| return version.parse("0.0.0") | |
| def is_transformers_version_greater_than(content: str): | |
| return get_package_version("transformers") >= version.parse(content) | |
| def union_two_dict(dict1: dict[str, Any], dict2: dict[str, Any]) -> dict[str, Any]: | |
| """Union two dict. Will throw an error if there is an item not the same object with the same key.""" | |
| for key in dict2.keys(): | |
| if key in dict1: | |
| assert dict1[key] == dict2[key], f"{key} in dict1 and dict2 are not the same object" | |
| dict1[key] = dict2[key] | |
| return dict1 | |
| def append_to_dict(data: dict[str, list[Any]], new_data: dict[str, Any]) -> None: | |
| """Append dict to a dict of list.""" | |
| for key, val in new_data.items(): | |
| if key not in data: | |
| data[key] = [] | |
| data[key].append(val) | |
| def unflatten_dict(data: dict[str, Any], sep: str = "/") -> dict[str, Any]: | |
| unflattened = {} | |
| for key, value in data.items(): | |
| pieces = key.split(sep) | |
| pointer = unflattened | |
| for piece in pieces[:-1]: | |
| if piece not in pointer: | |
| pointer[piece] = {} | |
| pointer = pointer[piece] | |
| pointer[pieces[-1]] = value | |
| return unflattened | |
| def flatten_dict(data: dict[str, Any], parent_key: str = "", sep: str = "/") -> dict[str, Any]: | |
| flattened = {} | |
| for key, value in data.items(): | |
| new_key = parent_key + sep + key if parent_key else key | |
| if isinstance(value, dict): | |
| flattened.update(flatten_dict(value, new_key, sep=sep)) | |
| else: | |
| flattened[new_key] = value | |
| return flattened | |
| def convert_dict_to_str(data: dict[str, Any]) -> str: | |
| return yaml.dump(data, indent=2) | |
| def get_abs_path(path: str, prompt: str = "File") -> Optional[str]: | |
| if path is not None: | |
| if os.path.exists(path): # ray job uses absolute path | |
| return os.path.abspath(path) | |
| else: | |
| print(f"{prompt} {path} not found.") | |
| def timer(name: str, timing_raw: dict[str, float]): | |
| with Timer(name=name, logger=None) as timer: | |
| yield | |
| timing_raw[name] = timer.last | |