Text Generation
Transformers
Safetensors
qwen2
llama-factory
conversational
text-generation-inference
Instructions to use Alexjiuqiaoyu/HumanMirror with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Alexjiuqiaoyu/HumanMirror with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="Alexjiuqiaoyu/HumanMirror") messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("Alexjiuqiaoyu/HumanMirror") model = AutoModelForCausalLM.from_pretrained("Alexjiuqiaoyu/HumanMirror", device_map="auto") messages = [ {"role": "user", "content": "Who are you?"}, ] inputs = tokenizer.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(tokenizer.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use Alexjiuqiaoyu/HumanMirror with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "Alexjiuqiaoyu/HumanMirror" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Alexjiuqiaoyu/HumanMirror", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/Alexjiuqiaoyu/HumanMirror
- SGLang
How to use Alexjiuqiaoyu/HumanMirror with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "Alexjiuqiaoyu/HumanMirror" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Alexjiuqiaoyu/HumanMirror", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "Alexjiuqiaoyu/HumanMirror" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Alexjiuqiaoyu/HumanMirror", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use Alexjiuqiaoyu/HumanMirror with Docker Model Runner:
docker model run hf.co/Alexjiuqiaoyu/HumanMirror
File size: 2,501 Bytes
338f0c3 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 | # handler.py
import os
import torch
from typing import Dict, Any, Optional
from transformers import (
AutoTokenizer,
AutoModelForCausalLM,
GenerationConfig,
)
class EndpointHandler:
def __init__(self, model_path: str = None):
# Load tokenizer & model
self.tokenizer = AutoTokenizer.from_pretrained(model_path, use_fast=True)
self.model = AutoModelForCausalLM.from_pretrained(model_path)
# Ensure pad token
if self.tokenizer.pad_token_id is None:
self.tokenizer.pad_token = self.tokenizer.eos_token
# Put model on GPU if available
self.device = "cuda" if torch.cuda.is_available() else "cpu"
self.model.to(self.device)
def __call__(self, data: Dict[str, Any]) -> Dict[str, Any]:
# 1) Extract user input & parameters
user_input = data.get("inputs", "")
params = data.get("parameters", {})
# 2) Build prompt
prompt = f"User: {user_input}\nAssistant:"
# 3) Tokenize & prepare tensors
encoded = self.tokenizer(
prompt,
return_tensors="pt",
padding=False,
).to(self.device)
input_ids = encoded.input_ids
attention_mask = encoded.attention_mask
prompt_len = input_ids.shape[1]
# 4) Merge default gen args with overrides
gen_args = {
"max_new_tokens": params.get("max_new_tokens", 128),
"temperature": params.get("temperature", 1.0),
"top_p": params.get("top_p", 1.0),
"top_k": params.get("top_k", 50),
"do_sample": params.get("do_sample", True),
"repetition_penalty": params.get("repetition_penalty", 1.0),
"pad_token_id": self.tokenizer.pad_token_id,
"eos_token_id": self.tokenizer.eos_token_id,
}
# Build a GenerationConfig for cleaner API
gen_config = GenerationConfig(**gen_args)
# 5) Call generate
output = self.model.generate(
inputs=input_ids,
attention_mask=attention_mask,
generation_config=gen_config,
)
# 6) Decode only the new tokens
# output is shape [1, prompt_len + gen_len]
full = self.tokenizer.decode(output[0, prompt_len:], skip_special_tokens=True)
first_reply = full.split("\nUser:")[0].strip()
return {"generated_text": first_reply} |