Download chat_format.py from alby13/LibraMindMini: direct link, hf CLI and curl.
- Browser
- Download file 2.71 kB
-
https://huggingface.co/alby13/LibraMindMini/resolve/main/chat_format.py
- Command line
-
hf download hf://alby13/LibraMindMini/chat_format.py
-
curl -L -o chat_format.py https://huggingface.co/alby13/LibraMindMini/resolve/main/chat_format.py
2.71 kB
| """Chat template shared by data preparation, fine-tuning and the chat program (ChatML style). | |
| <|bos|><|im_start|>system\\n{system}<|im_end|>\\n<|im_start|>user\\n{text}<|im_end|>\\n<|im_start|>assistant\\n{reply}<|im_end|>\\n | |
| Messages are encoded piece by piece, so training and inference tokenize identically. | |
| Loss mask: 1 on assistant reply tokens and the <|im_end|> that closes them (so the model learns to stop). | |
| """ | |
| import numpy as np | |
| from tokenizers import Tokenizer | |
| BOS, IM_START, IM_END = "<|bos|>", "<|im_start|>", "<|im_end|>" | |
| ROLES = ("system", "user", "assistant", "tool") | |
| def chat_tokenizer(base_tokenizer_path): | |
| """The pretraining tokenizer plus the chat special tokens (appended as new ids).""" | |
| tok = Tokenizer.from_file(str(base_tokenizer_path)) | |
| tok.add_special_tokens([IM_START, IM_END]) | |
| return tok | |
| class ChatEncoder: | |
| def __init__(self, tok): | |
| self.tok = tok | |
| self.bos, self.im_start, self.im_end = (tok.token_to_id(t) for t in (BOS, IM_START, IM_END)) | |
| self.newline = tok.encode("\n", add_special_tokens=False).ids | |
| self.headers = {r: tok.encode(f"{r}\n", add_special_tokens=False).ids for r in ROLES} | |
| def encode_conversation(self, messages): | |
| """messages: list of {"role", "content"}. Returns (ids uint16, loss mask uint8).""" | |
| ids, mask = [self.bos], [0] | |
| for m in messages: | |
| head = [self.im_start] + self.headers[m["role"]] | |
| body = self.tok.encode(m["content"], add_special_tokens=False).ids + [self.im_end] | |
| train = 1 if m["role"] == "assistant" else 0 | |
| ids += head + body + self.newline | |
| mask += [0] * len(head) + [train] * len(body) + [0] * len(self.newline) | |
| return np.array(ids, dtype=np.uint16), np.array(mask, dtype=np.uint8) | |
| def encode_prompt(self, messages): | |
| """Conversation so far plus the assistant header, ready for generation.""" | |
| ids, _ = self.encode_conversation(messages) | |
| return np.concatenate([ids, [self.im_start] + self.headers["assistant"]]).astype(np.int64) | |
| def encode_reply(self, text): | |
| """An assistant reply as the model would produce it: its tokens followed by <|im_end|>.""" | |
| return np.array(self.tok.encode(text, add_special_tokens=False).ids + [self.im_end], dtype=np.int64) | |
| def decode_conversation(tok, ids): | |
| """Turn a tokenized conversation back into [{"role", "content"}] messages.""" | |
| text = tok.decode([int(i) for i in ids], skip_special_tokens=False).replace(BOS, "") | |
| msgs = [] | |
| for part in text.split(IM_START)[1:]: | |
| role, _, body = part.partition("\n") | |
| msgs.append({"role": role, "content": body.split(IM_END)[0]}) | |
| return msgs | |