Image-Text-to-Text
Transformers
Safetensors
kimi_k3
feature-extraction
compressed-tensors
conversational
custom_code
Instructions to use SinterForge/Kimi-K3 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use SinterForge/Kimi-K3 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-text-to-text", model="SinterForge/Kimi-K3", trust_remote_code=True) messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("SinterForge/Kimi-K3", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use SinterForge/Kimi-K3 with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "SinterForge/Kimi-K3" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "SinterForge/Kimi-K3", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker
docker model run hf.co/SinterForge/Kimi-K3
- SGLang
How to use SinterForge/Kimi-K3 with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "SinterForge/Kimi-K3" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "SinterForge/Kimi-K3", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "SinterForge/Kimi-K3" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "SinterForge/Kimi-K3", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }' - Docker Model Runner
How to use SinterForge/Kimi-K3 with Docker Model Runner:
docker model run hf.co/SinterForge/Kimi-K3
File size: 7,660 Bytes
c866901 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 | """Kimi-K3 processor: wraps vision processor + tokenizer into a single interface.
Chat rendering (including XTML tool-result ordering) is handled by the
tokenizer's Python encoder; this processor adds multimodal media preprocessing.
"""
from transformers.feature_extraction_utils import BatchFeature
from transformers.processing_utils import ProcessorMixin
from transformers.utils import logging
from .media_utils import ensure_media_type
logger = logging.get_logger(__name__)
# ββ KimiK3Processor βββββββββββββββββββββββββββββββββββββββββββββββββββ
class KimiK3Processor(ProcessorMixin):
r"""
Constructs a KimiK3 processor which wraps a KimiK3 image processor
and a tokenizer into a single processor.
[`KimiK3Processor`] offers all the functionalities of
[`KimiK3VisionProcessor`] and [`TikTokenTokenizer`].
Args:
image_processor ([`KimiK3VisionProcessor`], *optional*):
The image processor is a required input.
tokenizer ([`TikTokenTokenizer`], *optional*):
The tokenizer is a required input.
chat_template (`str`, *optional*): Kept for ProcessorMixin
compatibility. Kimi K3 chat encoding is implemented in Python by
the tokenizer.
"""
attributes = ["image_processor", "tokenizer"]
valid_kwargs = ["chat_template"]
image_processor_class = "AutoImageProcessor"
tokenizer_class = "AutoTokenizer"
def __init__(
self,
image_processor=None,
tokenizer=None,
chat_template=None,
**kwargs,
):
super().__init__(image_processor,
tokenizer,
chat_template=chat_template)
self.media_processor = image_processor
self.image_placeholder = "<|kimi_image_placeholder|>"
# ββ Media preprocessing ββββββββββββββββββββββββββββββββββββββββββββ
def update_raw_text(self, text: str, image_prompts: list[str]) -> str:
# Replace image placeholders
image_count = text.count(self.image_placeholder)
if image_count > 0:
assert image_count == len(image_prompts), (
f"image placeholder count {image_count} != "
f"image_prompts count {len(image_prompts)}")
text_parts = text.split(self.image_placeholder)
assert len(text_parts) == len(image_prompts) + 1
text = "".join([
text_parts[i] + image_prompts[i]
for i in range(len(image_prompts))
])
text += text_parts[-1]
return text
def preprocess_medias(self,
medias: list[dict]) -> tuple[list[dict], list[str]]:
"""Process media items and generate corresponding prompts.
Returns:
A tuple of (updated_medias, image_prompts).
"""
updated_medias = []
image_prompts = []
for media in medias:
if media['type'] == 'image':
updated_medias.append(media)
img = ensure_media_type(
media,
transparent_bg_config=self.media_processor.
_transparent_bg_config,
transparent_bg_fill_stage=self.media_processor.
_transparent_bg_fill_stage,
)['image']
w, h = img.size
image_prompts.append(
self.media_processor.make_image_prompt(w, h))
else:
raise ValueError(f"unsupported media type: {media['type']}")
return updated_medias, image_prompts
# ββ Main entry points ββββββββββββββββββββββββββββββββββββββββββββββ
def __call__(self,
messages: list[dict] = None,
medias: list[dict] = None,
text: str = None,
return_tensors: str = "pt",
**kwargs) -> BatchFeature:
"""
Process multimodal inputs for Kimi-K3 model.
Args:
messages: List of message dicts with 'role' and 'content' fields.
If provided, medias and text will be extracted automatically.
medias: Pre-extracted list of media dicts.
text: Pre-formatted text string.
return_tensors: Format of returned tensors. Default: 'pt'.
**kwargs: Additional arguments passed to apply_chat_template.
Returns:
BatchFeature with fields: input_ids, attention_mask,
pixel_values, grid_thws.
"""
if messages is None and (medias is None or text is None):
raise ValueError(
"Provide either 'messages' or both 'medias' and 'text'")
if medias is not None and text is not None:
updated_medias, image_prompts = (self.preprocess_medias(medias))
preprocessed = self.media_processor.preprocess(
updated_medias, return_tensors=return_tensors)
text = self.update_raw_text(text, image_prompts)
text_inputs = self.tokenizer(text, return_tensors=return_tensors)
return BatchFeature(data={**text_inputs, **preprocessed.data})
if medias is None:
medias = self._extract_medias_from_messages(messages)
updated_medias, image_prompts = (self.preprocess_medias(medias))
preprocessed = self.media_processor.preprocess(
updated_medias, return_tensors=return_tensors)
if text is None:
text_inputs = self.tokenizer.apply_chat_template(
messages,
tokenize=True,
return_tensors=return_tensors,
return_dict=True,
image_prompts=image_prompts,
**kwargs)
return BatchFeature(data={**text_inputs, **preprocessed.data})
text = self.update_raw_text(text, image_prompts)
text_inputs = self.tokenizer(text, return_tensors=return_tensors)
return BatchFeature(data={**text_inputs, **preprocessed.data})
@staticmethod
def _extract_medias_from_messages(messages: list[dict]) -> list[dict]:
"""Extract media items from messages in a single pass."""
medias = []
for msg in messages:
if msg['role'] != 'user' or not msg.get('content'):
continue
for content_part in msg['content']:
if not isinstance(content_part, dict):
continue
content_type = content_part.get('type')
if content_type in ['image_url', 'image']:
image_data = content_part.get(content_type)
assert image_data is not None, f"image data is missing for content part: {content_part}"
medias.append({
'type': 'image',
'image': image_data,
})
return medias
def apply_chat_template(self, messages, **kwargs):
return self.tokenizer.apply_chat_template(messages, **kwargs)
def batch_decode(self, *args, **kwargs):
return self.tokenizer.batch_decode(*args, **kwargs)
def decode(self, *args, **kwargs):
return self.tokenizer.decode(*args, **kwargs)
@property
def model_input_names(self):
return ['input_ids', 'attention_mask', 'pixel_values', 'grid_thws']
|