Text Generation
Transformers
Safetensors
qwen2
llama-factory
conversational
text-generation-inference
Instructions to use Alexjiuqiaoyu/HumanMirror with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Alexjiuqiaoyu/HumanMirror with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="Alexjiuqiaoyu/HumanMirror") messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("Alexjiuqiaoyu/HumanMirror") model = AutoModelForCausalLM.from_pretrained("Alexjiuqiaoyu/HumanMirror", device_map="auto") messages = [ {"role": "user", "content": "Who are you?"}, ] inputs = tokenizer.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(tokenizer.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use Alexjiuqiaoyu/HumanMirror with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "Alexjiuqiaoyu/HumanMirror" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Alexjiuqiaoyu/HumanMirror", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/Alexjiuqiaoyu/HumanMirror
- SGLang
How to use Alexjiuqiaoyu/HumanMirror with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "Alexjiuqiaoyu/HumanMirror" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Alexjiuqiaoyu/HumanMirror", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "Alexjiuqiaoyu/HumanMirror" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Alexjiuqiaoyu/HumanMirror", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use Alexjiuqiaoyu/HumanMirror with Docker Model Runner:
docker model run hf.co/Alexjiuqiaoyu/HumanMirror
| # handler.py | |
| import os | |
| import torch | |
| from typing import Dict, Any, Optional | |
| from transformers import ( | |
| AutoTokenizer, | |
| AutoModelForCausalLM, | |
| GenerationConfig, | |
| ) | |
| class EndpointHandler: | |
| def __init__(self, model_path: str = None): | |
| # Load tokenizer & model | |
| self.tokenizer = AutoTokenizer.from_pretrained(model_path, use_fast=True) | |
| self.model = AutoModelForCausalLM.from_pretrained(model_path) | |
| # Ensure pad token | |
| if self.tokenizer.pad_token_id is None: | |
| self.tokenizer.pad_token = self.tokenizer.eos_token | |
| # Put model on GPU if available | |
| self.device = "cuda" if torch.cuda.is_available() else "cpu" | |
| self.model.to(self.device) | |
| def __call__(self, data: Dict[str, Any]) -> Dict[str, Any]: | |
| # 1) Extract user input & parameters | |
| user_input = data.get("inputs", "") | |
| params = data.get("parameters", {}) | |
| # 2) Build prompt | |
| prompt = f"User: {user_input}\nAssistant:" | |
| # 3) Tokenize & prepare tensors | |
| encoded = self.tokenizer( | |
| prompt, | |
| return_tensors="pt", | |
| padding=False, | |
| ).to(self.device) | |
| input_ids = encoded.input_ids | |
| attention_mask = encoded.attention_mask | |
| prompt_len = input_ids.shape[1] | |
| # 4) Merge default gen args with overrides | |
| gen_args = { | |
| "max_new_tokens": params.get("max_new_tokens", 128), | |
| "temperature": params.get("temperature", 1.0), | |
| "top_p": params.get("top_p", 1.0), | |
| "top_k": params.get("top_k", 50), | |
| "do_sample": params.get("do_sample", True), | |
| "repetition_penalty": params.get("repetition_penalty", 1.0), | |
| "pad_token_id": self.tokenizer.pad_token_id, | |
| "eos_token_id": self.tokenizer.eos_token_id, | |
| } | |
| # Build a GenerationConfig for cleaner API | |
| gen_config = GenerationConfig(**gen_args) | |
| # 5) Call generate | |
| output = self.model.generate( | |
| inputs=input_ids, | |
| attention_mask=attention_mask, | |
| generation_config=gen_config, | |
| ) | |
| # 6) Decode only the new tokens | |
| # output is shape [1, prompt_len + gen_len] | |
| full = self.tokenizer.decode(output[0, prompt_len:], skip_special_tokens=True) | |
| first_reply = full.split("\nUser:")[0].strip() | |
| return {"generated_text": first_reply} |