File size: 2,744 Bytes
b643ad6
 
 
 
df259b7
418d19a
 
df259b7
 
 
 
 
 
418d19a
 
 
 
 
 
 
 
b643ad6
 
df259b7
b643ad6
 
 
 
 
 
418d19a
 
 
b643ad6
418d19a
 
 
 
b643ad6
418d19a
b643ad6
 
418d19a
 
 
 
 
 
 
 
 
 
 
 
 
 
b643ad6
 
418d19a
b643ad6
 
 
418d19a
 
b643ad6
418d19a
 
 
 
b643ad6
418d19a
 
 
b643ad6
 
 
418d19a
b643ad6
418d19a
b643ad6
418d19a
 
 
 
b643ad6
 
 
418d19a
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
import gradio as gr
from huggingface_hub import InferenceClient


def build_prompt(system_message: str, history: list, message: str) -> str:
    prompt = f"{system_message}\n\n"
    for turn in history:
        if isinstance(turn, dict):
            role = turn.get("role", "")
            content = turn.get("content", "")
        else:
            # fallback: tuple format (user, assistant)
            role, content = ("user", turn[0]) if turn[0] else ("assistant", turn[1])
        if role == "user":
            prompt += f"User: {content}\n"
        elif role == "assistant":
            prompt += f"Assistant: {content}\n"
    prompt += f"User: {message}\nAssistant:"
    return prompt


def respond(
    message,
    history,
    system_message,
    max_tokens,
    temperature,
    top_p,
    hf_token: gr.OAuthToken,
):
    if hf_token is None:
        yield "⚠️ Vui lΓ²ng Δ‘Δƒng nhαΊ­p Hugging Face trΖ°α»›c khi chat."
        return

    client = InferenceClient(
        token=hf_token.token,
        model="Hyggshi-AI/HOS-OSS-200M",
    )

    prompt = build_prompt(system_message, history, message)

    response = ""
    try:
        for chunk in client.text_generation(
            prompt,
            max_new_tokens=max_tokens,
            stream=True,
            temperature=temperature,
            top_p=top_p,
            stop_sequences=["User:", "\nUser", "\n\nUser"],
            do_sample=True,
        ):
            response += chunk
            yield response
    except Exception as e:
        yield f"❌ Lα»—i: {str(e)}"


# ── UI ─────────────────────────────────────────────────────────────────────────

chatbot = gr.ChatInterface(
    respond,
    title="HOS-OSS 200M",
    description="Hyggshi OS Β· Base LM Β· text_generation mode",
    additional_inputs=[
        gr.Textbox(
            value="You are HOS-OSS, a helpful AI assistant made by Hyggshi.",
            label="System message",
            lines=3,
        ),
        gr.Slider(minimum=1,   maximum=2048, value=512,  step=1,    label="Max new tokens"),
        gr.Slider(minimum=0.1, maximum=4.0,  value=0.7,  step=0.1,  label="Temperature"),
        gr.Slider(minimum=0.1, maximum=1.0,  value=0.95, step=0.05, label="Top-p"),
    ],
)

with gr.Blocks(title="HOS-OSS 200M") as demo:
    with gr.Sidebar():
        gr.Markdown("## πŸ” Đăng nhαΊ­p")
        gr.LoginButton()
        gr.Markdown(
            "Cần tài khoản Hugging Face có quyền truy cập model "
            "`Hyggshi-AI/HOS-OSS-200M`."
        )
    chatbot.render()

if __name__ == "__main__":
    demo.launch()