vsabolcec commited on
Commit
d664558
·
0 Parent(s):

Super-squash branch 'main' using huggingface_hub

Browse files
Files changed (4) hide show
  1. .gitattributes +35 -0
  2. README.md +200 -0
  3. mdclm.pt +3 -0
  4. mfwedu.pt +3 -0
.gitattributes ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,200 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ language:
4
+ - eng
5
+ - deu
6
+ - fra
7
+ - pol
8
+ - por
9
+ - spa
10
+ - ita
11
+ - cmn
12
+ - nld
13
+ - afr
14
+ - als
15
+ - amh
16
+ - arb
17
+ - ars
18
+ - ary
19
+ - arz
20
+ - asm
21
+ - azj
22
+ - bel
23
+ - ben
24
+ - bew
25
+ - bod
26
+ - bos
27
+ - bul
28
+ - cat
29
+ - ces
30
+ - ckb
31
+ - cym
32
+ - dan
33
+ - div
34
+ - ekk
35
+ - ell
36
+ - epo
37
+ - eus
38
+ - fas
39
+ - fil
40
+ - fin
41
+ - gle
42
+ - glg
43
+ - gmh
44
+ - guj
45
+ - heb
46
+ - hif
47
+ - hin
48
+ - hrv
49
+ - hun
50
+ - hye
51
+ - ind
52
+ - isl
53
+ - jpn
54
+ - kan
55
+ - kat
56
+ - kaz
57
+ - khk
58
+ - khm
59
+ - kir
60
+ - kmr
61
+ - kor
62
+ - lao
63
+ - lat
64
+ - lit
65
+ - ltz
66
+ - lvs
67
+ - mal
68
+ - mar
69
+ - mkd
70
+ - mlt
71
+ - mya
72
+ - nno
73
+ - nob
74
+ - npi
75
+ - nrm
76
+ - ory
77
+ - pan
78
+ - pbt
79
+ - plt
80
+ - ron
81
+ - rus
82
+ - sin
83
+ - slk
84
+ - slv
85
+ - snd
86
+ - som
87
+ - srp
88
+ - srp
89
+ - swe
90
+ - swh
91
+ - tam
92
+ - tat
93
+ - tel
94
+ - tgk
95
+ - tha
96
+ - tur
97
+ - uig
98
+ - ukr
99
+ - urd
100
+ - uzn
101
+ - uzn
102
+ - vie
103
+ - ydd
104
+ - zsm
105
+ ---
106
+
107
+ # FineWeb2-HQ-PlusPlus-Classifier
108
+
109
+ This repository contains the model weights of the trained deep learning quality classifiers distilled from [FineWeb-edu](https://huggingface.co/HuggingFaceFW/fineweb-edu-classifier) and [DCLM](https://huggingface.co/mlfoundations/fasttext-oh-eli5) for multilingual text quality scoring. The classifier uses [jhu-clsp/mmBERT-base](https://huggingface.co/jhu-clsp/mmBERT-base) embeddings to score the documents and supports English and additional 100 languages.
110
+
111
+ For more details, see our paper [Adapting English Quality Classifiers for Multilingual LLM Pretraining Data Selection](https://arxiv.org/abs/2610.11585).
112
+
113
+ ## Quickstart
114
+
115
+ Classifier uses a simple architecture that takes mean-pooled mmBERT-base embeddings as input. For the DCLM-based classifier, softmax should be applied on the output logit, whereas for the FineWeb-edu-based classifier, the raw logit score should be used.
116
+
117
+ ```python
118
+ import torch
119
+ import torch.nn.functional as F
120
+ from transformers import AutoModel, AutoTokenizer
121
+ import huggingface_hub
122
+
123
+ class BinaryClassifier(torch.nn.Module):
124
+ def __init__(self, embedding_dim=768, hidden_dim=3072):
125
+ super(BinaryClassifier, self).__init__()
126
+ self.classifier = torch.nn.Sequential(
127
+ torch.nn.Linear(embedding_dim, hidden_dim),
128
+ torch.nn.ReLU(),
129
+ torch.nn.Linear(hidden_dim, hidden_dim),
130
+ torch.nn.ReLU(),
131
+ torch.nn.Linear(hidden_dim, 1),
132
+ )
133
+
134
+ def forward(self, X):
135
+ return self.classifier(X)
136
+
137
+ def to_pt(self, file_name):
138
+ torch.save(self.state_dict(), file_name)
139
+
140
+ @classmethod
141
+ def from_pt(cls, file_name, embedding_dim=768, hidden_dim=3072):
142
+ state_dict = torch.load(
143
+ file_name, weights_only=True, map_location=torch.device("cpu")
144
+ )
145
+ classifier = BinaryClassifier(
146
+ embedding_dim=embedding_dim, hidden_dim=hidden_dim
147
+ )
148
+ classifier.load_state_dict(state_dict)
149
+ classifier.eval()
150
+ return classifier
151
+
152
+
153
+ if __name__ == "__main__":
154
+ embedding_model_name = "jhu-clsp/mmBERT-base"
155
+ embedding_tokenizer = AutoTokenizer.from_pretrained(embedding_model_name)
156
+ embedding_model = AutoModel.from_pretrained(
157
+ embedding_model_name,
158
+ dtype=torch.bfloat16,
159
+ ).cuda()
160
+
161
+ classifiers_dir = huggingface_hub.snapshot_download("epfml/FineWeb2-HQ-PlusPlus-Classifier")
162
+ mfwedu_model = BinaryClassifier.from_pt(f"{classifiers_dir}/mfwedu.pt").cuda()
163
+ mdclm_model = BinaryClassifier.from_pt(f"{classifiers_dir}/mdclm.pt").cuda()
164
+
165
+ def score_sample(text, tokenizer, embedding_model, classifier_model, apply_sigmoid=False):
166
+ inputs = tokenizer([text], return_tensors="pt").to("cuda")
167
+ embeddings = embedding_model(**inputs).last_hidden_state.float().mean(1)
168
+ if apply_sigmoid:
169
+ score = F.sigmoid(classifier_model(embeddings))
170
+ else:
171
+ score = classifier_model(embeddings)
172
+ return score.item()
173
+
174
+
175
+ text_en = "Question: How is bipolar disorder different from unipolar depression or 'regular' depression?\nAnswer: Both bipolar disorder and major depression are typically associated with depressive episodes. So both illnesses are accompanied by depressions. The difference is that in bipolar disorder people also have periods of elevation -- or severe irritability. We call these manic or hypomanic episodes."
176
+ mfwedu_score = score_sample(text_en, embedding_tokenizer, embedding_model, mfwedu_model, apply_sigmoid=False)
177
+ mdclm_score = score_sample(text_en, embedding_tokenizer, embedding_model, mdclm_model, apply_sigmoid=True)
178
+ print(f"{mfwedu_score:0.4f}") # 2.7353 (in [0-5])
179
+ print(f"{mdclm_score:0.4f}") # 0.8463 (in [0-1])
180
+
181
+
182
+ text_en = "Custom Wedding Gifts\nPersonalized photo frames, albums & keepsakes. Heirloom quality!\nCustom Engraved Journals\nHandmade in Florence Italy. Dozens of sizes and paper styles!"
183
+ mfwedu_score = score_sample(text_en, embedding_tokenizer, embedding_model, mfwedu_model, apply_sigmoid=False)
184
+ mdclm_score = score_sample(text_en, embedding_tokenizer, embedding_model, mdclm_model, apply_sigmoid=True)
185
+ print(f"{mfwedu_score:0.4f}") # -0.0370 (in [0-5])
186
+ print(f"{mdclm_score:0.4f}") # 0.0000 (in [0-1])
187
+ ```
188
+
189
+ ## Citation information
190
+ ```
191
+ @misc{sabolčec2026adaptingenglishqualityclassifiers,
192
+ title={Adapting English Quality Classifiers for Multilingual LLM Pretraining Data Selection},
193
+ author={Vinko Sabolčec and Bettina Messmer and Yassine Turki and Martin Jaggi},
194
+ year={2026},
195
+ eprint={2610.11585},
196
+ archivePrefix={arXiv},
197
+ primaryClass={cs.CL},
198
+ url={https://arxiv.org/abs/2610.11585},
199
+ }
200
+ ```
mdclm.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5d9349c13bb8b7ca10014ca6bfc19825ee00d2a1fa3262904c9cab604ef90099
3
+ size 47226125
mfwedu.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7781865837f21444ca96aa655ac6b7ce83686b2ac5516fde1fe3ff29de8b203f
3
+ size 47226137