Text Generation
Transformers
Safetensors
llada2_moe
dllm
diffusion
llm
text_generation
conversational
custom_code
Instructions to use inclusionAI/LLaDA2.2-flash with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use inclusionAI/LLaDA2.2-flash with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="inclusionAI/LLaDA2.2-flash", trust_remote_code=True) messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("inclusionAI/LLaDA2.2-flash", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use inclusionAI/LLaDA2.2-flash with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "inclusionAI/LLaDA2.2-flash" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "inclusionAI/LLaDA2.2-flash", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/inclusionAI/LLaDA2.2-flash
- SGLang
How to use inclusionAI/LLaDA2.2-flash with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "inclusionAI/LLaDA2.2-flash" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "inclusionAI/LLaDA2.2-flash", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "inclusionAI/LLaDA2.2-flash" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "inclusionAI/LLaDA2.2-flash", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use inclusionAI/LLaDA2.2-flash with Docker Model Runner:
docker model run hf.co/inclusionAI/LLaDA2.2-flash
| import os | |
| import logging | |
| from typing import Any, Iterator, Union | |
| from transformers import PreTrainedTokenizerFast | |
| from transformers.convert_slow_tokenizer import bytes_to_unicode | |
| from .tool_declaration_ts import encode_tools_to_typescript_style | |
| logger = logging.getLogger(__name__) | |
| def deep_sort_dict(obj: Any) -> Any: | |
| """Deep sort dict keys recursively to ensure stable hashing and tokenization.""" | |
| if isinstance(obj, dict): | |
| return {k: deep_sort_dict(v) for k, v in sorted(obj.items())} | |
| if isinstance(obj, list): | |
| return [deep_sort_dict(item) for item in obj] | |
| return obj | |
| class CustomFastTokenizer(PreTrainedTokenizerFast): | |
| def __init__(self, *args, **kwargs): | |
| super().__init__(*args, **kwargs) | |
| # Byte-to-unicode mapping for downstream tasks requiring single-byte decoding | |
| self.byte_encoder = bytes_to_unicode() | |
| self.byte_decoder = {v: k for k, v in self.byte_encoder.items()} | |
| def _split_whitespaces_or_nonwhitespaces( | |
| s: str, max_consecutive_slice_len: int | |
| ) -> Iterator[str]: | |
| current_slice_len = 0 | |
| current_slice_is_space = s[0].isspace() if len(s) > 0 else False | |
| slice_start = 0 | |
| for i in range(len(s)): | |
| is_now_space = s[i].isspace() | |
| if current_slice_is_space ^ is_now_space: | |
| current_slice_len = 1 | |
| current_slice_is_space = is_now_space | |
| else: | |
| current_slice_len += 1 | |
| if current_slice_len > max_consecutive_slice_len: | |
| yield s[slice_start:i] | |
| slice_start = i | |
| current_slice_len = 1 | |
| yield s[slice_start:] | |
| def encode(self, text: Union[str, Any], *args, **kwargs) -> list[int]: | |
| if not isinstance(text, str) or args or kwargs: | |
| return super().encode(text, *args, **kwargs) | |
| # Chunking thresholds to prevent OOM on very long texts | |
| MAX_ENCODE_CHARS = 400_000 | |
| MAX_NO_WHITESPACES_CHARS = 25_000 | |
| all_substrs = [] | |
| for i in range(0, len(text), MAX_ENCODE_CHARS): | |
| chunk = text[i : i + MAX_ENCODE_CHARS] | |
| all_substrs.extend( | |
| self._split_whitespaces_or_nonwhitespaces( | |
| chunk, MAX_NO_WHITESPACES_CHARS | |
| ) | |
| ) | |
| t = [] | |
| for substr in all_substrs: | |
| t.extend(super().encode(substr, add_special_tokens=False)) | |
| return t | |
| def apply_chat_template(self, conversation, tools=None, **kwargs): | |
| tools = deep_sort_dict(tools) | |
| if tools: | |
| try: | |
| tools_ts_str = encode_tools_to_typescript_style(tools) | |
| kwargs["tools_ts_str"] = tools_ts_str | |
| except Exception as e: | |
| logger.error(f"Failed to convert tools to TypeScript style: {e}") | |
| return super().apply_chat_template( | |
| conversation=conversation, tools=tools, **kwargs | |
| ) | |