Download handler.py from shalev396/tiny-shakespeare-chat: direct link, hf CLI and curl.
- Browser
- Download file 1.04 kB
-
https://huggingface.co/shalev396/tiny-shakespeare-chat/resolve/main/handler.py
- Command line
-
hf download hf://shalev396/tiny-shakespeare-chat/handler.py
-
curl -L -o handler.py https://huggingface.co/shalev396/tiny-shakespeare-chat/resolve/main/handler.py
1.04 kB
| """Hugging Face Inference Endpoints entry point: deploy this repo as a CPU/GPU chat API. | |
| Request: {"inputs": "How fares the king?"} | |
| {"inputs": {"message": "...", "history": [["user", "bot"], ...]}, | |
| "parameters": {"max_new_tokens": 200, "temperature": 0.8, "top_k": 40, "seed": 1}} | |
| Response: [{"generated_text": "<reply>"}] | |
| """ | |
| import sys | |
| from pathlib import Path | |
| HERE = Path(__file__).resolve().parent | |
| sys.path.insert(0, str(HERE)) | |
| import model as M # noqa: E402 | |
| class EndpointHandler: | |
| def __init__(self, path: str = ""): | |
| self.predictor = M.load(path or HERE, "cuda" if M.cuda_available() else "cpu") | |
| def __call__(self, data: dict): | |
| inputs = data.pop("inputs", data) | |
| parameters = data.pop("parameters", None) or {} | |
| if isinstance(inputs, dict): | |
| message, history = inputs.get("message", ""), inputs.get("history") | |
| else: | |
| message, history = inputs, None | |
| return [{"generated_text": self.predictor.predict(message, history, **parameters)}] | |