import gradio as gr from huggingface_hub import InferenceClient def build_prompt(system_message: str, history: list, message: str) -> str: prompt = f"{system_message}\n\n" for turn in history: if isinstance(turn, dict): role = turn.get("role", "") content = turn.get("content", "") else: # fallback: tuple format (user, assistant) role, content = ("user", turn[0]) if turn[0] else ("assistant", turn[1]) if role == "user": prompt += f"User: {content}\n" elif role == "assistant": prompt += f"Assistant: {content}\n" prompt += f"User: {message}\nAssistant:" return prompt def respond( message, history, system_message, max_tokens, temperature, top_p, hf_token: gr.OAuthToken, ): if hf_token is None: yield "⚠️ Vui lòng đăng nhập Hugging Face trước khi chat." return client = InferenceClient( token=hf_token.token, model="Hyggshi-AI/HOS-OSS-200M", ) prompt = build_prompt(system_message, history, message) response = "" try: for chunk in client.text_generation( prompt, max_new_tokens=max_tokens, stream=True, temperature=temperature, top_p=top_p, stop_sequences=["User:", "\nUser", "\n\nUser"], do_sample=True, ): response += chunk yield response except Exception as e: yield f"❌ Lỗi: {str(e)}" # ── UI ───────────────────────────────────────────────────────────────────────── chatbot = gr.ChatInterface( respond, title="HOS-OSS 200M", description="Hyggshi OS · Base LM · text_generation mode", additional_inputs=[ gr.Textbox( value="You are HOS-OSS, a helpful AI assistant made by Hyggshi.", label="System message", lines=3, ), gr.Slider(minimum=1, maximum=2048, value=512, step=1, label="Max new tokens"), gr.Slider(minimum=0.1, maximum=4.0, value=0.7, step=0.1, label="Temperature"), gr.Slider(minimum=0.1, maximum=1.0, value=0.95, step=0.05, label="Top-p"), ], ) with gr.Blocks(title="HOS-OSS 200M") as demo: with gr.Sidebar(): gr.Markdown("## 🔐 Đăng nhập") gr.LoginButton() gr.Markdown( "Cần tài khoản Hugging Face có quyền truy cập model " "`Hyggshi-AI/HOS-OSS-200M`." ) chatbot.render() if __name__ == "__main__": demo.launch()