Spaces:
Running on Zero
Running on Zero
| """TF-IDF + Logistic Regression baseline (traditional ML reference).""" | |
| import json | |
| import logging | |
| import pickle | |
| from pathlib import Path | |
| from typing import Optional | |
| import numpy as np | |
| import pandas as pd | |
| from sklearn.feature_extraction.text import TfidfVectorizer | |
| from sklearn.linear_model import LogisticRegression | |
| from sklearn.metrics import f1_score, accuracy_score, classification_report | |
| from sklearn.pipeline import Pipeline | |
| from . import config as cfg | |
| logger = logging.getLogger(__name__) | |
| def train_tfidf_baseline( | |
| train_df: pd.DataFrame, | |
| val_df: pd.DataFrame, | |
| test_df: pd.DataFrame, | |
| text_col: str = "full_text", | |
| label_col: str = "overall_label", | |
| output_dir: Optional[Path] = None, | |
| ): | |
| """Train TF-IDF + LogReg on overall 3-class sentiment. Returns metrics dict.""" | |
| if output_dir is None: | |
| output_dir = cfg.CHECKPOINT_DIR / "tfidf_baseline" | |
| output_dir = Path(output_dir) | |
| output_dir.mkdir(parents=True, exist_ok=True) | |
| pipeline = Pipeline([ | |
| ("tfidf", TfidfVectorizer( | |
| max_features=20000, | |
| ngram_range=(1, 2), | |
| min_df=2, | |
| max_df=0.95, | |
| stop_words="english", | |
| sublinear_tf=True, | |
| )), | |
| ("clf", LogisticRegression( | |
| max_iter=1000, | |
| class_weight="balanced", | |
| random_state=cfg.RANDOM_SEED, | |
| C=1.0, | |
| )), | |
| ]) | |
| X_train = train_df[text_col].fillna("").astype(str).values | |
| y_train = train_df[label_col].values | |
| X_val = val_df[text_col].fillna("").astype(str).values | |
| y_val = val_df[label_col].values | |
| X_test = test_df[text_col].fillna("").astype(str).values | |
| y_test = test_df[label_col].values | |
| logger.info("Training TF-IDF + LogReg on %d samples...", len(X_train)) | |
| pipeline.fit(X_train, y_train) | |
| metrics = {} | |
| for split_name, X, y in [("val", X_val, y_val), ("test", X_test, y_test)]: | |
| preds = pipeline.predict(X) | |
| metrics[f"{split_name}_macro_f1"] = float(f1_score(y, preds, average="macro", zero_division=0)) | |
| metrics[f"{split_name}_accuracy"] = float(accuracy_score(y, preds)) | |
| metrics[f"{split_name}_weighted_f1"] = float(f1_score(y, preds, average="weighted", zero_division=0)) | |
| metrics[f"{split_name}_report"] = classification_report( | |
| y, preds, target_names=["Negative", "Neutral", "Positive"], output_dict=True, zero_division=0 | |
| ) | |
| with open(output_dir / "model.pkl", "wb") as f: | |
| pickle.dump(pipeline, f) | |
| with open(output_dir / "metrics.json", "w") as f: | |
| json.dump({k: v for k, v in metrics.items() if not k.endswith("_report")}, f, indent=2) | |
| logger.info("TF-IDF baseline | val Macro-F1=%.4f | test Macro-F1=%.4f", | |
| metrics["val_macro_f1"], metrics["test_macro_f1"]) | |
| return pipeline, metrics | |