File size: 2,711 Bytes
bcda938
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
"""Chat template shared by data preparation, fine-tuning and the chat program (ChatML style).

  <|bos|><|im_start|>system\\n{system}<|im_end|>\\n<|im_start|>user\\n{text}<|im_end|>\\n<|im_start|>assistant\\n{reply}<|im_end|>\\n

Messages are encoded piece by piece, so training and inference tokenize identically.
Loss mask: 1 on assistant reply tokens and the <|im_end|> that closes them (so the model learns to stop).
"""
import numpy as np
from tokenizers import Tokenizer

BOS, IM_START, IM_END = "<|bos|>", "<|im_start|>", "<|im_end|>"
ROLES = ("system", "user", "assistant", "tool")


def chat_tokenizer(base_tokenizer_path):
    """The pretraining tokenizer plus the chat special tokens (appended as new ids)."""
    tok = Tokenizer.from_file(str(base_tokenizer_path))
    tok.add_special_tokens([IM_START, IM_END])
    return tok


class ChatEncoder:
    def __init__(self, tok):
        self.tok = tok
        self.bos, self.im_start, self.im_end = (tok.token_to_id(t) for t in (BOS, IM_START, IM_END))
        self.newline = tok.encode("\n", add_special_tokens=False).ids
        self.headers = {r: tok.encode(f"{r}\n", add_special_tokens=False).ids for r in ROLES}

    def encode_conversation(self, messages):
        """messages: list of {"role", "content"}. Returns (ids uint16, loss mask uint8)."""
        ids, mask = [self.bos], [0]
        for m in messages:
            head = [self.im_start] + self.headers[m["role"]]
            body = self.tok.encode(m["content"], add_special_tokens=False).ids + [self.im_end]
            train = 1 if m["role"] == "assistant" else 0
            ids += head + body + self.newline
            mask += [0] * len(head) + [train] * len(body) + [0] * len(self.newline)
        return np.array(ids, dtype=np.uint16), np.array(mask, dtype=np.uint8)

    def encode_prompt(self, messages):
        """Conversation so far plus the assistant header, ready for generation."""
        ids, _ = self.encode_conversation(messages)
        return np.concatenate([ids, [self.im_start] + self.headers["assistant"]]).astype(np.int64)

    def encode_reply(self, text):
        """An assistant reply as the model would produce it: its tokens followed by <|im_end|>."""
        return np.array(self.tok.encode(text, add_special_tokens=False).ids + [self.im_end], dtype=np.int64)


def decode_conversation(tok, ids):
    """Turn a tokenized conversation back into [{"role", "content"}] messages."""
    text = tok.decode([int(i) for i in ids], skip_special_tokens=False).replace(BOS, "")
    msgs = []
    for part in text.split(IM_START)[1:]:
        role, _, body = part.partition("\n")
        msgs.append({"role": role, "content": body.split(IM_END)[0]})
    return msgs