File size: 2,711 Bytes
bcda938 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 | """Chat template shared by data preparation, fine-tuning and the chat program (ChatML style).
<|bos|><|im_start|>system\\n{system}<|im_end|>\\n<|im_start|>user\\n{text}<|im_end|>\\n<|im_start|>assistant\\n{reply}<|im_end|>\\n
Messages are encoded piece by piece, so training and inference tokenize identically.
Loss mask: 1 on assistant reply tokens and the <|im_end|> that closes them (so the model learns to stop).
"""
import numpy as np
from tokenizers import Tokenizer
BOS, IM_START, IM_END = "<|bos|>", "<|im_start|>", "<|im_end|>"
ROLES = ("system", "user", "assistant", "tool")
def chat_tokenizer(base_tokenizer_path):
"""The pretraining tokenizer plus the chat special tokens (appended as new ids)."""
tok = Tokenizer.from_file(str(base_tokenizer_path))
tok.add_special_tokens([IM_START, IM_END])
return tok
class ChatEncoder:
def __init__(self, tok):
self.tok = tok
self.bos, self.im_start, self.im_end = (tok.token_to_id(t) for t in (BOS, IM_START, IM_END))
self.newline = tok.encode("\n", add_special_tokens=False).ids
self.headers = {r: tok.encode(f"{r}\n", add_special_tokens=False).ids for r in ROLES}
def encode_conversation(self, messages):
"""messages: list of {"role", "content"}. Returns (ids uint16, loss mask uint8)."""
ids, mask = [self.bos], [0]
for m in messages:
head = [self.im_start] + self.headers[m["role"]]
body = self.tok.encode(m["content"], add_special_tokens=False).ids + [self.im_end]
train = 1 if m["role"] == "assistant" else 0
ids += head + body + self.newline
mask += [0] * len(head) + [train] * len(body) + [0] * len(self.newline)
return np.array(ids, dtype=np.uint16), np.array(mask, dtype=np.uint8)
def encode_prompt(self, messages):
"""Conversation so far plus the assistant header, ready for generation."""
ids, _ = self.encode_conversation(messages)
return np.concatenate([ids, [self.im_start] + self.headers["assistant"]]).astype(np.int64)
def encode_reply(self, text):
"""An assistant reply as the model would produce it: its tokens followed by <|im_end|>."""
return np.array(self.tok.encode(text, add_special_tokens=False).ids + [self.im_end], dtype=np.int64)
def decode_conversation(tok, ids):
"""Turn a tokenized conversation back into [{"role", "content"}] messages."""
text = tok.decode([int(i) for i in ids], skip_special_tokens=False).replace(BOS, "")
msgs = []
for part in text.split(IM_START)[1:]:
role, _, body = part.partition("\n")
msgs.append({"role": role, "content": body.split(IM_END)[0]})
return msgs
|