| import pandas as pd |
| import joblib |
| from transformers import pipeline |
|
|
| |
| sample = pd.read_csv("sample_20.csv") |
| texts = sample["comment_text"] |
| true = sample["toxic"].values |
|
|
| |
| vectorizer = joblib.load("tfidf_vectorizer.joblib") |
| clf = joblib.load("toxic_classifier.joblib") |
| lr_preds = clf.predict(vectorizer.transform(texts)) |
|
|
| |
| bert_clf = pipeline( |
| "text-classification", |
| model="unitary/toxic-bert", |
| tokenizer="unitary/toxic-bert", |
| truncation=True |
| ) |
| bert_raw = [bert_clf(t)[0]["label"] for t in texts] |
| bert_preds = [1 if lbl == "toxic" else 0 for lbl in bert_raw] |
|
|
| |
| lr_acc = (lr_preds == true).mean() |
| bert_acc = (pd.Series(bert_preds).values == true).mean() |
|
|
| print("Logistic Regression accuracy on 20 samples:", lr_acc) |
| print("BERT accuracy on 20 samples:", bert_acc) |
|
|