File size: 3,216 Bytes
24ebd71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
from __future__ import annotations

from itertools import product

from datasets import Dataset
from transformers import PreTrainedTokenizerFast

CONTEXT_LENGTH = 128

HEROES = [
    "a careful robot",
    "a brave mouse",
    "a curious child",
    "a small fox",
    "a patient inventor",
    "a lonely star",
]
PLACES = [
    "a moonlit castle",
    "a clockwork garden",
    "a floating library",
    "a quiet workshop",
    "a crystal forest",
    "an underwater city",
]
GOALS = [
    "find a lost key",
    "repair a broken bridge",
    "help a frightened friend",
    "learn why the bells stopped",
    "return a borrowed light",
]
LESSONS = [
    "courage can be quiet",
    "asking for help is wise",
    "patience can solve hard problems",
    "kindness changes a whole journey",
    "mistakes can become maps",
]


def build_examples() -> list[dict[str, str]]:
    examples: list[dict[str, str]] = []
    for index, (hero, place, goal, lesson) in enumerate(
        product(HEROES, PLACES, GOALS, LESSONS)
    ):
        object_name = ["lantern", "silver thread", "paper crown", "tiny compass"][index % 4]
        prompt = (
            f"Write a tiny story about {hero} in {place}. "
            f"The hero must {goal} and learn that {lesson}."
        )
        response = (
            f"In {place}, {hero} carried a {object_name}. The path seemed impossible, "
            f"but the hero chose to {goal}. A new friend noticed the effort and offered "
            f"one small clue. Together they finished before sunrise. From then on, the "
            f"hero remembered that {lesson}."
        )
        examples.append({"prompt": prompt, "response": response})
    return examples


def split_examples() -> tuple[list[dict[str, str]], list[dict[str, str]]]:
    examples = build_examples()
    train = [example for index, example in enumerate(examples) if index % 10 != 0]
    evaluation = [example for index, example in enumerate(examples) if index % 10 == 0]
    return train, evaluation


def encode_examples(
    examples: list[dict[str, str]],
    tokenizer: PreTrainedTokenizerFast,
) -> Dataset:
    rows = {"input_ids": [], "attention_mask": [], "labels": []}
    for example in examples:
        prefix = f"<bos>User: {example['prompt']}\nAssistant:"
        full_text = f"{prefix} {example['response']}<eos>"
        full = tokenizer(
            full_text,
            max_length=CONTEXT_LENGTH,
            truncation=True,
            padding="max_length",
            add_special_tokens=False,
        )
        prefix_ids = tokenizer(
            prefix,
            max_length=CONTEXT_LENGTH,
            truncation=True,
            add_special_tokens=False,
        )["input_ids"]
        labels = list(full["input_ids"])
        masked_prefix = min(len(prefix_ids), CONTEXT_LENGTH)
        labels[:masked_prefix] = [-100] * masked_prefix
        labels = [
            label if attention else -100
            for label, attention in zip(labels, full["attention_mask"], strict=True)
        ]
        rows["input_ids"].append(full["input_ids"])
        rows["attention_mask"].append(full["attention_mask"])
        rows["labels"].append(labels)
    return Dataset.from_dict(rows)