StrongDev2024 commited on
Commit
d3166c0
·
verified ·
1 Parent(s): 8ebf66d

Add Mini checkpoint

Browse files
Files changed (8) hide show
  1. README.md +41 -0
  2. config.json +9 -0
  3. model.py +168 -0
  4. model.safetensors +3 -0
  5. pretrained.py +14 -0
  6. tokenizer.py +81 -0
  7. tokenizer_config.json +13 -0
  8. vocab.json +509 -0
README.md ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: mit
3
+ language:
4
+ - en
5
+ library_name: pytorch
6
+ tags:
7
+ - text-generation
8
+ - chatbot
9
+ ---
10
+
11
+ # Mini
12
+
13
+ A small English chatbot trained from scratch. It is a decoder-only transformer with 6 layers, 4 attention heads, an embedding size of 192, and a context window of 512 tokens. The vocabulary is 507 words. It answers short questions it has seen in its training dialogues. It is not a general-purpose assistant.
14
+
15
+ ## Load
16
+
17
+ `model.py`, `tokenizer.py`, and `pretrained.py` from this repo need to be on the Python path.
18
+
19
+ ```python
20
+ from model import TinyGPT
21
+ from tokenizer import Tokenizer
22
+
23
+ model = TinyGPT.from_pretrained("StrongDev2024/mini", trust_remote_code=True)
24
+ tokenizer = Tokenizer.from_pretrained("StrongDev2024/mini")
25
+
26
+ ids = tokenizer.encode("<user> what is 2 + 2 <bot>")
27
+ import torch
28
+ out = model.generate(
29
+ torch.tensor([ids]),
30
+ max_new_tokens=40,
31
+ temperature=0.0,
32
+ stop_ids={tokenizer.token_to_id["<end>"], tokenizer.token_to_id["<user>"]},
33
+ )
34
+ print(tokenizer.decode(out[0, len(ids) :].tolist()))
35
+ ```
36
+
37
+ The prompt is `<user> your question <bot>`. Generation stops at `<end>`.
38
+
39
+ ## License
40
+
41
+ MIT
config.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model_type": "tiny-gpt",
3
+ "vocab_size": 507,
4
+ "block_size": 512,
5
+ "n_layer": 6,
6
+ "n_head": 4,
7
+ "n_embd": 192,
8
+ "dropout": 0.0
9
+ }
model.py ADDED
@@ -0,0 +1,168 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """A small decoder-only transformer.
2
+
3
+ It reads tokens from left to right and scores the next token.
4
+ Causal attention means a position may look at earlier tokens only.
5
+ """
6
+
7
+ import json
8
+ from pathlib import Path
9
+
10
+ import torch
11
+ import torch.nn as nn
12
+ import torch.nn.functional as F
13
+
14
+ from pretrained import resolve_pretrained_folder
15
+
16
+ # Keys TinyGPT.__init__ accepts. config.json may also carry Hub metadata.
17
+ CONFIG_KEYS = ("vocab_size", "block_size", "n_layer", "n_head", "n_embd", "dropout")
18
+
19
+
20
+ class CausalSelfAttention(nn.Module):
21
+ def __init__(self, n_embd, n_head, block_size, dropout):
22
+ super().__init__()
23
+ self.n_head = n_head
24
+ self.head_dim = n_embd // n_head
25
+ self.qkv = nn.Linear(n_embd, 3 * n_embd)
26
+ self.proj = nn.Linear(n_embd, n_embd)
27
+ self.dropout = nn.Dropout(dropout)
28
+ # Lower triangle is 1: each token may attend to itself and the past.
29
+ mask = torch.tril(torch.ones(block_size, block_size))
30
+ self.register_buffer("mask", mask.view(1, 1, block_size, block_size))
31
+
32
+ def forward(self, x, return_attn=False):
33
+ batch, time, channels = x.shape
34
+ qkv = self.qkv(x)
35
+ query, key, value = qkv.split(channels, dim=2)
36
+ query = query.view(batch, time, self.n_head, self.head_dim).transpose(1, 2)
37
+ key = key.view(batch, time, self.n_head, self.head_dim).transpose(1, 2)
38
+ value = value.view(batch, time, self.n_head, self.head_dim).transpose(1, 2)
39
+ scores = (query @ key.transpose(-2, -1)) / (self.head_dim ** 0.5)
40
+ scores = scores.masked_fill(self.mask[:, :, :time, :time] == 0, float("-inf"))
41
+ weights = F.softmax(scores, dim=-1)
42
+ mixed = (self.dropout(weights) @ value).transpose(1, 2).contiguous().view(batch, time, channels)
43
+ out = self.dropout(self.proj(mixed))
44
+ if return_attn:
45
+ return out, weights
46
+ return out
47
+
48
+
49
+ class Block(nn.Module):
50
+ def __init__(self, n_embd, n_head, block_size, dropout):
51
+ super().__init__()
52
+ self.ln1 = nn.LayerNorm(n_embd)
53
+ self.attn = CausalSelfAttention(n_embd, n_head, block_size, dropout)
54
+ self.ln2 = nn.LayerNorm(n_embd)
55
+ self.mlp = nn.Sequential(
56
+ nn.Linear(n_embd, 4 * n_embd),
57
+ nn.GELU(),
58
+ nn.Linear(4 * n_embd, n_embd),
59
+ nn.Dropout(dropout),
60
+ )
61
+
62
+ def forward(self, x, return_attn=False):
63
+ attended = self.attn(self.ln1(x), return_attn=return_attn)
64
+ if return_attn:
65
+ attended, weights = attended
66
+ x = x + attended
67
+ x = x + self.mlp(self.ln2(x))
68
+ if return_attn:
69
+ return x, weights
70
+ return x
71
+
72
+
73
+ class TinyGPT(nn.Module):
74
+ def __init__(self, vocab_size, block_size=128, n_layer=2, n_head=4, n_embd=128, dropout=0.1):
75
+ super().__init__()
76
+ if n_embd % n_head != 0:
77
+ raise ValueError("n_embd must be divisible by n_head")
78
+ self.block_size = block_size
79
+ self.tok_emb = nn.Embedding(vocab_size, n_embd)
80
+ self.pos_emb = nn.Embedding(block_size, n_embd)
81
+ self.drop = nn.Dropout(dropout)
82
+ self.blocks = nn.ModuleList(
83
+ [Block(n_embd, n_head, block_size, dropout) for _ in range(n_layer)]
84
+ )
85
+ self.ln_f = nn.LayerNorm(n_embd)
86
+ self.head = nn.Linear(n_embd, vocab_size, bias=False)
87
+ self.config = {
88
+ "vocab_size": vocab_size,
89
+ "block_size": block_size,
90
+ "n_layer": n_layer,
91
+ "n_head": n_head,
92
+ "n_embd": n_embd,
93
+ "dropout": dropout,
94
+ }
95
+
96
+ def forward(self, idx, targets=None, return_attn=False):
97
+ _batch, time = idx.shape
98
+ positions = torch.arange(time, device=idx.device)
99
+ x = self.drop(self.tok_emb(idx) + self.pos_emb(positions))
100
+ attentions = []
101
+ for block in self.blocks:
102
+ if return_attn:
103
+ x, weights = block(x, return_attn=True)
104
+ attentions.append(weights)
105
+ else:
106
+ x = block(x)
107
+ logits = self.head(self.ln_f(x))
108
+ loss = None
109
+ if targets is not None:
110
+ loss = F.cross_entropy(logits.view(-1, logits.size(-1)), targets.view(-1))
111
+ if return_attn:
112
+ return logits, loss, attentions
113
+ return logits, loss
114
+
115
+ @torch.no_grad()
116
+ def generate(self, idx, max_new_tokens, temperature=0.0, top_k=None, stop_ids=None):
117
+ """Append tokens until a stop token or the length limit.
118
+
119
+ temperature 0 always picks the most likely next token.
120
+ """
121
+ stop_ids = set(stop_ids or [])
122
+ for _ in range(max_new_tokens):
123
+ idx_cond = idx[:, -self.block_size :]
124
+ logits, _ = self(idx_cond)
125
+ logits = logits[:, -1, :]
126
+ if temperature <= 0:
127
+ next_id = torch.argmax(logits, dim=-1, keepdim=True)
128
+ else:
129
+ logits = logits / max(temperature, 1e-6)
130
+ if top_k is not None:
131
+ top_values, _ = torch.topk(logits, min(top_k, logits.size(-1)))
132
+ logits = logits.masked_fill(logits < top_values[:, [-1]], float("-inf"))
133
+ probs = F.softmax(logits, dim=-1)
134
+ next_id = torch.multinomial(probs, num_samples=1)
135
+ idx = torch.cat([idx, next_id], dim=1)
136
+ if int(next_id.item()) in stop_ids:
137
+ break
138
+ return idx
139
+
140
+ def save_pretrained(self, folder):
141
+ """Write config.json and model.safetensors for the Hugging Face Hub."""
142
+ folder = Path(folder)
143
+ folder.mkdir(parents=True, exist_ok=True)
144
+ payload = {"model_type": "tiny-gpt", **self.config}
145
+ (folder / "config.json").write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8")
146
+ from safetensors.torch import save_file
147
+
148
+ state = {name: value.detach().cpu().contiguous() for name, value in self.state_dict().items()}
149
+ save_file(state, str(folder / "model.safetensors"))
150
+
151
+ @classmethod
152
+ def from_pretrained(cls, path_or_repo, **_ignored):
153
+ """Load Mini from a local hub folder or a Hugging Face repo id.
154
+
155
+ trust_remote_code is accepted and ignored. Import this class from
156
+ model.py, then call from_pretrained with the repo id.
157
+ """
158
+ folder = resolve_pretrained_folder(path_or_repo)
159
+ raw = json.loads((folder / "config.json").read_text(encoding="utf-8"))
160
+ missing = [key for key in CONFIG_KEYS if key not in raw]
161
+ if missing:
162
+ raise FileNotFoundError(f"{folder / 'config.json'} is missing {', '.join(missing)}")
163
+ model = cls(**{key: raw[key] for key in CONFIG_KEYS})
164
+ from safetensors.torch import load_file
165
+
166
+ model.load_state_dict(load_file(str(folder / "model.safetensors")))
167
+ model.eval()
168
+ return model
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:15c9c12258b2842d082ed748517d15bcbf135aca2fea082707e6a05fdf4b679e
3
+ size 18149056
pretrained.py ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Find a local folder or a Hugging Face repo that holds Mini's files."""
2
+
3
+ from pathlib import Path
4
+
5
+
6
+ def resolve_pretrained_folder(path_or_repo):
7
+ """Return a directory with config.json. Download a Hub repo id first."""
8
+ folder = Path(path_or_repo)
9
+ if folder.is_dir():
10
+ return folder
11
+ from huggingface_hub import snapshot_download
12
+
13
+ downloaded = snapshot_download(repo_id=str(path_or_repo))
14
+ return Path(downloaded)
tokenizer.py ADDED
@@ -0,0 +1,81 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Turn text into token ids and token ids back into text.
2
+
3
+ The model never sees letters. It sees integers. This file builds that
4
+ mapping from the training dialogues.
5
+ """
6
+
7
+ import json
8
+ import re
9
+ from pathlib import Path
10
+
11
+ from pretrained import resolve_pretrained_folder
12
+
13
+ # Reserved tokens. They are written into the dialogue file and into chat prompts.
14
+ # <pad> evens out lengths if a batch needs filler.
15
+ # <unk> stands for a word that was not in the training file.
16
+ SPECIAL_TOKENS = ["<pad>", "<unk>", "<user>", "<bot>", "<end>"]
17
+
18
+ # Special tokens are kept whole. Other text splits into words and punctuation.
19
+ TOKEN_RE = re.compile(r"<pad>|<unk>|<user>|<bot>|<end>|\w+|[^\w\s]", re.UNICODE)
20
+
21
+
22
+ def tokenize(text):
23
+ """Split lowercased text into a list of token strings."""
24
+ return TOKEN_RE.findall(text.lower())
25
+
26
+
27
+ class Tokenizer:
28
+ def __init__(self, token_to_id):
29
+ self.token_to_id = dict(token_to_id)
30
+ self.id_to_token = {index: token for token, index in self.token_to_id.items()}
31
+ self.unk_id = self.token_to_id["<unk>"]
32
+
33
+ @classmethod
34
+ def build(cls, text):
35
+ """Assign an id to every special token, then to every token in the text."""
36
+ token_to_id = {token: index for index, token in enumerate(SPECIAL_TOKENS)}
37
+ for token in tokenize(text):
38
+ if token not in token_to_id:
39
+ token_to_id[token] = len(token_to_id)
40
+ return cls(token_to_id)
41
+
42
+ def encode(self, text):
43
+ """Text to a list of ids. Unknown words become <unk>."""
44
+ return [self.token_to_id.get(token, self.unk_id) for token in tokenize(text)]
45
+
46
+ def decode(self, ids):
47
+ """Ids to a readable string. Special tokens are left out of the reply."""
48
+ words = []
49
+ for token_id in ids:
50
+ token = self.id_to_token.get(int(token_id), "<unk>")
51
+ if token in SPECIAL_TOKENS:
52
+ continue
53
+ words.append(token)
54
+ text = " ".join(words)
55
+ text = re.sub(r"\s+([.,!?;:])", r"\1", text)
56
+ return text.strip()
57
+
58
+ def save_pretrained(self, folder):
59
+ """Write vocab.json and tokenizer_config.json for the Hub."""
60
+ folder = Path(folder)
61
+ folder.mkdir(parents=True, exist_ok=True)
62
+ (folder / "vocab.json").write_text(
63
+ json.dumps(self.token_to_id, ensure_ascii=False, indent=2) + "\n",
64
+ encoding="utf-8",
65
+ )
66
+ meta = {
67
+ "tokenizer_class": "Tokenizer",
68
+ "lowercase": True,
69
+ "special_tokens": SPECIAL_TOKENS,
70
+ "unk_token": "<unk>",
71
+ "pad_token": "<pad>",
72
+ }
73
+ (folder / "tokenizer_config.json").write_text(json.dumps(meta, indent=2) + "\n", encoding="utf-8")
74
+
75
+ @classmethod
76
+ def from_pretrained(cls, path_or_repo, **_ignored):
77
+ """Load the word vocabulary from a local hub folder or a Hub repo id."""
78
+ folder = resolve_pretrained_folder(path_or_repo)
79
+ raw = json.loads((folder / "vocab.json").read_text(encoding="utf-8"))
80
+ token_to_id = {token: int(index) for token, index in raw.items()}
81
+ return cls(token_to_id)
tokenizer_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "tokenizer_class": "Tokenizer",
3
+ "lowercase": true,
4
+ "special_tokens": [
5
+ "<pad>",
6
+ "<unk>",
7
+ "<user>",
8
+ "<bot>",
9
+ "<end>"
10
+ ],
11
+ "unk_token": "<unk>",
12
+ "pad_token": "<pad>"
13
+ }
vocab.json ADDED
@@ -0,0 +1,509 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "<pad>": 0,
3
+ "<unk>": 1,
4
+ "<user>": 2,
5
+ "<bot>": 3,
6
+ "<end>": 4,
7
+ "hello": 5,
8
+ "hi": 6,
9
+ ",": 7,
10
+ "i": 8,
11
+ "am": 9,
12
+ "mini": 10,
13
+ ".": 11,
14
+ "what": 12,
15
+ "do": 13,
16
+ "you": 14,
17
+ "want": 15,
18
+ "to": 16,
19
+ "talk": 17,
20
+ "about": 18,
21
+ "?": 19,
22
+ "hey": 20,
23
+ "there": 21,
24
+ "good": 22,
25
+ "morning": 23,
26
+ "how": 24,
27
+ "can": 25,
28
+ "help": 26,
29
+ "evening": 27,
30
+ "is": 28,
31
+ "your": 29,
32
+ "name": 30,
33
+ "my": 31,
34
+ "who": 32,
35
+ "are": 33,
36
+ "a": 34,
37
+ "small": 35,
38
+ "chatbot": 36,
39
+ "trained": 37,
40
+ "from": 38,
41
+ "scratch": 39,
42
+ "tell": 40,
43
+ "me": 41,
44
+ "doing": 42,
45
+ "well": 43,
46
+ "today": 44,
47
+ "ok": 45,
48
+ "glad": 46,
49
+ "hear": 47,
50
+ "that": 48,
51
+ "fine": 49,
52
+ "great": 50,
53
+ "not": 51,
54
+ "bad": 52,
55
+ "chat": 53,
56
+ "simple": 54,
57
+ "things": 55,
58
+ "like": 56,
59
+ "greetings": 57,
60
+ "names": 58,
61
+ "and": 59,
62
+ "questions": 60,
63
+ "short": 61,
64
+ "question": 62,
65
+ "will": 63,
66
+ "try": 64,
67
+ "answer": 65,
68
+ "need": 66,
69
+ "thank": 67,
70
+ "welcome": 68,
71
+ "thanks": 69,
72
+ "bye": 70,
73
+ "goodbye": 71,
74
+ "later": 72,
75
+ "see": 73,
76
+ "language": 74,
77
+ "speak": 75,
78
+ "english": 76,
79
+ "yes": 77,
80
+ "made": 78,
81
+ "was": 79,
82
+ "on": 80,
83
+ "dialogues": 81,
84
+ "where": 82,
85
+ "come": 83,
86
+ "1": 84,
87
+ "+": 85,
88
+ "2": 86,
89
+ "4": 87,
90
+ "3": 88,
91
+ "6": 89,
92
+ "color": 90,
93
+ "the": 91,
94
+ "sky": 92,
95
+ "blue": 93,
96
+ "grass": 94,
97
+ "green": 95,
98
+ "snow": 96,
99
+ "white": 97,
100
+ "pizza": 98,
101
+ "food": 99,
102
+ "cats": 100,
103
+ "dogs": 101,
104
+ "favorite": 102,
105
+ "joke": 103,
106
+ "why": 104,
107
+ "did": 105,
108
+ "robot": 106,
109
+ "sit": 107,
110
+ "down": 108,
111
+ "it": 109,
112
+ "had": 110,
113
+ "byte": 111,
114
+ "eat": 112,
115
+ "say": 113,
116
+ "something": 114,
117
+ "fun": 115,
118
+ "capital": 116,
119
+ "of": 117,
120
+ "france": 118,
121
+ "know": 119,
122
+ "yet": 120,
123
+ "ask": 121,
124
+ "quantum": 122,
125
+ "physics": 123,
126
+ "history": 124,
127
+ "write": 125,
128
+ "code": 126,
129
+ "soon": 127,
130
+ "nice": 128,
131
+ "meet": 129,
132
+ "too": 130,
133
+ "night": 131,
134
+ "sleep": 132,
135
+ "old": 133,
136
+ "new": 134,
137
+ "so": 135,
138
+ "very": 136,
139
+ "young": 137,
140
+ "live": 138,
141
+ "inside": 139,
142
+ "this": 140,
143
+ "program": 141,
144
+ "computer": 142,
145
+ "now": 143,
146
+ "understand": 144,
147
+ "animal": 145,
148
+ "cat": 146,
149
+ "sun": 147,
150
+ "yellow": 148,
151
+ "banana": 149,
152
+ "many": 150,
153
+ "legs": 151,
154
+ "does": 152,
155
+ "dog": 153,
156
+ "have": 154,
157
+ "has": 155,
158
+ "four": 156,
159
+ "8": 157,
160
+ "5": 158,
161
+ "10": 159,
162
+ "20": 160,
163
+ "count": 161,
164
+ "three": 162,
165
+ "one": 163,
166
+ "two": 164,
167
+ "weather": 165,
168
+ "cannot": 166,
169
+ "outside": 167,
170
+ "raining": 168,
171
+ "sad": 169,
172
+ "sorry": 170,
173
+ "here": 171,
174
+ "happy": 172,
175
+ "wonderful": 173,
176
+ "tired": 174,
177
+ "should": 175,
178
+ "rest": 176,
179
+ "water": 177,
180
+ "clear": 178,
181
+ "drink": 179,
182
+ "hot": 180,
183
+ "day": 181,
184
+ "no": 182,
185
+ "but": 183,
186
+ "12": 184,
187
+ "7": 185,
188
+ "14": 186,
189
+ "an": 187,
190
+ "apple": 188,
191
+ "red": 189,
192
+ "days": 190,
193
+ "in": 191,
194
+ "week": 192,
195
+ "seven": 193,
196
+ "opposite": 194,
197
+ "cold": 195,
198
+ "up": 196,
199
+ "number": 197,
200
+ "story": 198,
201
+ "sam": 199,
202
+ "kite": 200,
203
+ "wind": 201,
204
+ "took": 202,
205
+ "into": 203,
206
+ "tree": 204,
207
+ "helped": 205,
208
+ "climb": 206,
209
+ "free": 207,
210
+ "they": 208,
211
+ "carried": 209,
212
+ "home": 210,
213
+ "before": 211,
214
+ "named": 212,
215
+ "pip": 213,
216
+ "walk": 214,
217
+ "walked": 215,
218
+ "river": 216,
219
+ "happened": 217,
220
+ "next": 218,
221
+ "mended": 219,
222
+ "bent": 220,
223
+ "corner": 221,
224
+ "gentle": 222,
225
+ "flew": 223,
226
+ "stayed": 224,
227
+ "watched": 225,
228
+ "window": 226,
229
+ "another": 227,
230
+ "went": 228,
231
+ "market": 229,
232
+ "bought": 230,
233
+ "apples": 231,
234
+ "desk": 232,
235
+ "back": 233,
236
+ "noon": 234,
237
+ "he": 235,
238
+ "held": 236,
239
+ "string": 237,
240
+ "with": 238,
241
+ "both": 239,
242
+ "hands": 240,
243
+ "ran": 241,
244
+ "across": 242,
245
+ "strong": 243,
246
+ "climbed": 244,
247
+ "over": 245,
248
+ "fence": 246,
249
+ "stood": 247,
250
+ "by": 248,
251
+ "gate": 249,
252
+ "dance": 250,
253
+ "pulled": 251,
254
+ "harder": 252,
255
+ "slipped": 253,
256
+ "'": 254,
257
+ "s": 255,
258
+ "hand": 256,
259
+ "at": 257,
260
+ "end": 258,
261
+ "field": 259,
262
+ "caught": 260,
263
+ "high": 261,
264
+ "branch": 262,
265
+ "hung": 263,
266
+ "thin": 264,
267
+ "rope": 265,
268
+ "stopped": 266,
269
+ "running": 267,
270
+ "looked": 268,
271
+ "reach": 269,
272
+ "lowest": 270,
273
+ "close": 271,
274
+ "ground": 272,
275
+ "still": 273,
276
+ "while": 274,
277
+ "moved": 275,
278
+ "slowly": 276,
279
+ "leaves": 277,
280
+ "brushed": 278,
281
+ "his": 279,
282
+ "face": 280,
283
+ "bird": 281,
284
+ "hopped": 282,
285
+ "away": 283,
286
+ "reached": 284,
287
+ "lifted": 285,
288
+ "paper": 286,
289
+ "together": 287,
290
+ "against": 288,
291
+ "chest": 289,
292
+ "wound": 290,
293
+ "neat": 291,
294
+ "loop": 292,
295
+ "sat": 293,
296
+ "ate": 294,
297
+ "each": 295,
298
+ "said": 296,
299
+ "would": 297,
300
+ "fly": 298,
301
+ "again": 299,
302
+ "after": 300,
303
+ "grew": 301,
304
+ "quiet": 302,
305
+ "along": 303,
306
+ "path": 304,
307
+ "were": 305,
308
+ "left": 306,
309
+ "barked": 307,
310
+ "once": 308,
311
+ "yard": 309,
312
+ "then": 310,
313
+ "lay": 311,
314
+ "lights": 312,
315
+ "came": 313,
316
+ "windows": 314,
317
+ "put": 315,
318
+ "chair": 316,
319
+ "door": 317,
320
+ "piece": 318,
321
+ "same": 319,
322
+ "time": 320,
323
+ "let": 321,
324
+ "go": 322,
325
+ "rose": 323,
326
+ "steadied": 324,
327
+ "air": 325,
328
+ "laughed": 326,
329
+ "counted": 327,
330
+ "turns": 328,
331
+ "she": 329,
332
+ "likes": 330,
333
+ "warm": 331,
334
+ "places": 332,
335
+ "jumps": 333,
336
+ "onto": 334,
337
+ "sits": 335,
338
+ "beside": 336,
339
+ "folds": 337,
340
+ "her": 338,
341
+ "paws": 339,
342
+ "shuts": 340,
343
+ "eyes": 341,
344
+ "works": 342,
345
+ "types": 343,
346
+ "lines": 344,
347
+ "read": 345,
348
+ "them": 346,
349
+ "listens": 347,
350
+ "sound": 348,
351
+ "keys": 349,
352
+ "pours": 350,
353
+ "bowl": 351,
354
+ "drinks": 352,
355
+ "looks": 353,
356
+ "birds": 354,
357
+ "wire": 355,
358
+ "ears": 356,
359
+ "turn": 357,
360
+ "toward": 358,
361
+ "stays": 359,
362
+ "knows": 360,
363
+ "shut": 361,
364
+ "safe": 362,
365
+ "drinking": 363,
366
+ "washes": 364,
367
+ "paw": 365,
368
+ "afternoon": 366,
369
+ "moves": 367,
370
+ "floor": 368,
371
+ "follows": 369,
372
+ "patch": 370,
373
+ "light": 371,
374
+ "sleeps": 372,
375
+ "until": 373,
376
+ "reaches": 374,
377
+ "wall": 375,
378
+ "walks": 376,
379
+ "kitchen": 377,
380
+ "even": 378,
381
+ "when": 379,
382
+ "full": 380,
383
+ "says": 381,
384
+ "answers": 382,
385
+ "move": 383,
386
+ "set": 384,
387
+ "climbs": 385,
388
+ "watches": 386,
389
+ "sometimes": 387,
390
+ "hiss": 388,
391
+ "only": 389,
392
+ "tail": 390,
393
+ "house": 391,
394
+ "returns": 392,
395
+ "curls": 393,
396
+ "off": 394,
397
+ "room": 395,
398
+ "goes": 396,
399
+ "dark": 397,
400
+ "sunday": 398,
401
+ "narrow": 399,
402
+ "lined": 400,
403
+ "tall": 401,
404
+ "stones": 402,
405
+ "as": 403,
406
+ "found": 404,
407
+ "stone": 405,
408
+ "gray": 406,
409
+ "stripe": 407,
410
+ "flat": 408,
411
+ "for": 409,
412
+ "skipping": 410,
413
+ "kept": 411,
414
+ "pocket": 412,
415
+ "slow": 413,
416
+ "shone": 414,
417
+ "touched": 415,
418
+ "threw": 416,
419
+ "skipped": 417,
420
+ "twice": 418,
421
+ "sank": 419,
422
+ "tried": 420,
423
+ "times": 421,
424
+ "splash": 422,
425
+ "duck": 423,
426
+ "swam": 424,
427
+ "past": 425,
428
+ "care": 426,
429
+ "log": 427,
430
+ "shared": 428,
431
+ "bread": 429,
432
+ "asked": 430,
433
+ "moving": 431,
434
+ "sea": 432,
435
+ "drank": 433,
436
+ "bottle": 434,
437
+ "clouds": 435,
438
+ "gathered": 436,
439
+ "rain": 437,
440
+ "fell": 438,
441
+ "faded": 439,
442
+ "striped": 440,
443
+ "nothing": 441,
444
+ "memory": 442,
445
+ "skips": 443,
446
+ "saw": 444,
447
+ "stuck": 445,
448
+ "empty": 446,
449
+ "watching": 447,
450
+ "opened": 448,
451
+ "jumped": 449,
452
+ "rubbed": 450,
453
+ "leg": 451,
454
+ "table": 452,
455
+ "told": 453,
456
+ "everyone": 454,
457
+ "monday": 455,
458
+ "road": 456,
459
+ "bright": 457,
460
+ "cool": 458,
461
+ "bag": 459,
462
+ "list": 460,
463
+ "square": 461,
464
+ "row": 462,
465
+ "tables": 463,
466
+ "under": 464,
467
+ "long": 465,
468
+ "cloth": 466,
469
+ "roof": 467,
470
+ "woman": 468,
471
+ "sold": 469,
472
+ "wooden": 470,
473
+ "box": 471,
474
+ "picked": 472,
475
+ "paid": 473,
476
+ "coins": 474,
477
+ "loaves": 475,
478
+ "round": 476,
479
+ "brown": 477,
480
+ "loaf": 478,
481
+ "share": 479,
482
+ "way": 480,
483
+ "boy": 481,
484
+ "smaller": 482,
485
+ "than": 483,
486
+ "look": 484,
487
+ "nodded": 485,
488
+ "weak": 486,
489
+ "man": 487,
490
+ "smooth": 488,
491
+ "buy": 489,
492
+ "already": 490,
493
+ "slept": 491,
494
+ "wake": 492,
495
+ "enough": 493,
496
+ "turned": 494,
497
+ "sweet": 495,
498
+ "gates": 496,
499
+ "passed": 497,
500
+ "their": 498,
501
+ "own": 499,
502
+ "just": 500,
503
+ "board": 501,
504
+ "its": 502,
505
+ "could": 503,
506
+ "be": 504,
507
+ "seen": 505,
508
+ "smelled": 506
509
+ }