| import torch |
| import torch.nn as nn |
| import torch.optim as optim |
| from datasets import Dataset, load_dataset, concatenate_datasets |
| from tokenizers import Tokenizer |
| from huggingface_hub import PyTorchModelHubMixin, snapshot_download |
|
|
| tokenizer = Tokenizer.from_pretrained("LocalWisdom/PurpleGloVe") |
|
|
| |
| class GloVe(nn.Module, PyTorchModelHubMixin): |
| def __init__(self, config): |
| nn.Module.__init__(self) |
|
|
| self.context_window = config["context_window"] |
| self.x_max = config["x_max"] |
| self.alpha = config["alpha"] |
| self.vocab_size = config["vocab_size"] |
| self.embedding_dim = config["embedding_dim"] |
|
|
| self.wi = nn.Embedding(self.vocab_size, self.embedding_dim) |
| self.wj = nn.Embedding(self.vocab_size, self.embedding_dim) |
|
|
| self.bi = nn.Embedding(self.vocab_size, 1) |
| self.bj = nn.Embedding(self.vocab_size, 1) |
|
|
| def f(self, Xij): |
| return torch.clamp((Xij / self.x_max) ** self.alpha, max=1) |
|
|
| def forward(self, i, j, Xij): |
| wi = self.wi(i) |
| wj = self.wj(j) |
|
|
| bi = self.bi(i).squeeze(-1) |
| bj = self.bj(j).squeeze(-1) |
|
|
| weighting = self.f(Xij) |
|
|
| dot = (wi * wj).sum(dim=-1) |
| epsilon = 1e-8 |
| loss = weighting * (dot + bi + bj - torch.log(Xij + epsilon)) ** 2 |
| return torch.mean(loss) |
|
|
| |
| def embed(self, word_id): |
| |
| |
| |
| with torch.no_grad(): |
| word_id = torch.tensor([word_id]) |
| return self.wi(word_id) + self.wj(word_id) |
|
|
| PurpleGloVe = GloVe.from_pretrained("LocalWisdom/PurpleGloVe") |
|
|
| word_str = "king" |
| word = tokenizer.encode(word_str).ids[0] |
| word_e = PurpleGloVe.embed(word) |
| print(word_e) |
|
|