Spaces:
Sleeping
Sleeping
File size: 11,740 Bytes
2b28b2c 4a675a3 2b28b2c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 | """
Model Loader β StyleAIClassifierV5 (v9)
Pipeline completo:
1. Preprocesamiento del texto
2. Inferencia con SentenceTransformer + clasificador (3 pases TTA)
3. CalibraciΓ³n isotΓ³nica
4. AbstenciΓ³n en zona de incertidumbre [0.44, 0.56]
"""
import os
import re
import gc
import json
import random
import pickle
import warnings
from pathlib import Path
import numpy as np
import torch
import torch.nn as nn
import torch.nn.functional as F
from sentence_transformers import SentenceTransformer
from huggingface_hub import hf_hub_download
warnings.filterwarnings("ignore")
# ββ Repositorio Hugging Face con los pesos del modelo ββββββββββββββββββββββββββ
HF_REPO_ID = "Doffy143/mStyleDistance-finetunned"
# ββ Rutas locales (para desarrollo) βββββββββββββββββββββββββββββββββββββββββββ
_MODELS_DIR = Path(__file__).parent.parent / "models"
# ββ Archivos del modelo y sus nombres en el repositorio HF ββββββββββββββββββββ
_MODEL_FILES = {
"best_model.pt": _MODELS_DIR / "best_model.pt",
"isotonic_calibrator.pkl": _MODELS_DIR / "isotonic_calibrator.pkl",
"threshold_config.json": _MODELS_DIR / "threshold_config.json",
}
def _resolve_file(filename: str) -> str:
"""Busca el archivo local; si no existe, lo descarga desde Hugging Face."""
local_path = _MODEL_FILES[filename]
if local_path.exists():
print(f"[model_loader] '{filename}' encontrado localmente.")
return str(local_path)
print(f"[model_loader] '{filename}' no encontrado localmente, descargando desde HF...")
downloaded = hf_hub_download(
repo_id=HF_REPO_ID,
filename=filename,
cache_dir=os.environ.get("HF_HOME", None),
)
print(f"[model_loader] '{filename}' descargado en: {downloaded}")
return downloaded
# ββ Rutas resueltas (se calculan en load_model) ββββββββββββββββββββββββββββββ
MODEL_PATH = None
CALIBRATOR_PATH = None
THRESHOLD_PATH = None
# ββ HiperparΓ‘metros de inferencia (alineados con CONFIG del notebook v9) βββββββ
MODEL_NAME = "StyleDistance/mStyleDistance"
MAX_LENGTH = 256
DEVICE = torch.device("cuda" if torch.cuda.is_available() else "cpu")
# ββ Arquitectura (StyleAIClassifierV5 β igual que en el notebook v9) ββββββββββ
class StyleAIClassifierV5(nn.Module):
"""
Encoder SentenceTransformer + cabeza clasificadora 768β384β2.
IdΓ©ntico al notebook de entrenamiento v9.
"""
def __init__(self, encoder_model, num_classes: int = 2,
dropout: float = 0.20, hidden_dim: int = 384):
super().__init__()
self.encoder = encoder_model
self.embedding_dim = encoder_model.get_sentence_embedding_dimension()
self.classifier = nn.Sequential(
nn.LayerNorm(self.embedding_dim),
nn.Dropout(dropout),
nn.Linear(self.embedding_dim, hidden_dim),
nn.GELU(),
nn.LayerNorm(hidden_dim),
nn.Dropout(dropout),
nn.Linear(hidden_dim, num_classes),
)
def forward(self, input_ids, attention_mask):
features = {"input_ids": input_ids, "attention_mask": attention_mask}
embeddings = self.encoder(features)["sentence_embedding"].to(torch.float32)
return self.classifier(embeddings)
# ββ Preprocesamiento (igual al notebook v9) ββββββββββββββββββββββββββββββββββββ
def _preprocess(text: str) -> str:
"""Elimina URLs, @usuarios, #hashtags y normaliza espacios preservando pΓ‘rrafos."""
if not isinstance(text, str) or not text.strip():
return ""
text = re.sub(r"http\S+|www\S+|https\S+", "", text, flags=re.MULTILINE)
text = re.sub(r"@\w+|#\w+", "", text)
text = re.sub(r"[^\S\n]+", " ", text)
text = re.sub(r"\n\s*\n", "\n\n", text)
return text.strip()
# ββ Variantes TTA (sobre strings, no sobre tokens) ββββββββββββββββββββββββββββ
def _tta_truncate(text: str, ratio: float = 0.85) -> str:
"""Recorta el texto al ratio indicado de palabras."""
words = text.split()
if len(words) < 20:
return text
return " ".join(words[: int(len(words) * ratio)])
def _tta_span_mask(text: str, span_ratio: float = 0.10) -> str:
"""Enmascara un span aleatorio del texto."""
words = text.split()
if len(words) < 10:
return text
n = max(1, int(len(words) * span_ratio))
start = random.randint(0, max(0, len(words) - n))
words[start : start + n] = ["[MASK]"] * n
return " ".join(words)
# ββ TokenizaciΓ³n y predicciΓ³n en batch ββββββββββββββββββββββββββββββββββββββββ
def _encode_and_predict(model: StyleAIClassifierV5, texts: list[str],
tokenizer, max_length: int) -> list[float]:
"""Devuelve la probabilidad de clase IA (Γndice 1) para cada texto."""
encoding = tokenizer(
texts,
max_length=max_length,
padding="max_length",
truncation=True,
return_tensors="pt",
)
input_ids = encoding["input_ids"].to(DEVICE)
attention_mask = encoding["attention_mask"].to(DEVICE)
with torch.no_grad():
logits = model(input_ids, attention_mask)
probs = F.softmax(logits, dim=-1)
return probs[:, 1].cpu().tolist()
# ββ Singleton β modelo, calibrador y config se cargan una sola vez βββββββββββββ
_model = None
_tokenizer = None
_calibrator = None
_threshold_cfg: dict = {}
def load_model():
"""Carga el modelo, calibrador y configuraciΓ³n de umbral (singleton)."""
global _model, _tokenizer, _calibrator, _threshold_cfg
global MODEL_PATH, CALIBRATOR_PATH, THRESHOLD_PATH
if _model is not None:
return _model, _tokenizer, _calibrator, _threshold_cfg
# ββ Resolver rutas (local o Hugging Face) ββββββββββββββββββββββββββββββββββ
MODEL_PATH = _resolve_file("best_model.pt")
CALIBRATOR_PATH = _resolve_file("isotonic_calibrator.pkl")
THRESHOLD_PATH = _resolve_file("threshold_config.json")
print(f"[model_loader] Cargando encoder base '{MODEL_NAME}'...")
encoder = SentenceTransformer(MODEL_NAME)
_tokenizer = encoder.tokenizer
print(f"[model_loader] Cargando pesos desde '{MODEL_PATH}'...")
checkpoint = torch.load(MODEL_PATH, map_location=DEVICE, weights_only=False)
# El checkpoint guarda el estado completo del modelo (encoder + classifier).
# Las claves tienen el prefijo "encoder.0.model.*" porque SentenceTransformer
# almacena el transformer como encoder[0].auto_model. Construimos el modelo
# y cargamos el state_dict completo sobre Γ©l.
cfg_saved = checkpoint.get("config", {})
hidden_dim = cfg_saved.get("HIDDEN_DIM", 384)
dropout = cfg_saved.get("DROPOUT", 0.20)
_model = StyleAIClassifierV5(
encoder_model=encoder,
hidden_dim=hidden_dim,
dropout=dropout,
).float().to(DEVICE)
# Remapeo de claves: versiones antiguas de sentence-transformers guardaban los
# pesos del transformer bajo "encoder.0.model.*"; versiones nuevas los exponen
# como "encoder.0.auto_model.*". Normalizamos antes de cargar.
raw_sd = checkpoint["model_state_dict"]
remapped_sd = {}
for k, v in raw_sd.items():
new_k = k.replace("encoder.0.model.", "encoder.0.auto_model.")
remapped_sd[new_k] = v
missing, unexpected = _model.load_state_dict(remapped_sd, strict=False)
if missing or unexpected:
raise RuntimeError(
f"[model_loader] State dict no coincide con la arquitectura "
f"(posible mismatch de versiΓ³n de sentence-transformers/torch): "
f"{len(missing)} claves faltantes {missing[:5]}, "
f"{len(unexpected)} claves inesperadas {unexpected[:5]}"
)
print("[model_loader] State dict cargado sin discrepancias.")
_model.eval()
print(f"[model_loader] Cargando calibrador desde '{CALIBRATOR_PATH}'...")
with open(CALIBRATOR_PATH, "rb") as f:
_calibrator = pickle.load(f)
print(f"[model_loader] Cargando configuracion de umbral desde '{THRESHOLD_PATH}'...")
with open(THRESHOLD_PATH, "r") as f:
_threshold_cfg = json.load(f)
gc.collect()
if torch.cuda.is_available():
torch.cuda.empty_cache()
print("[model_loader] Pipeline listo.")
print(f"[model_loader] Dispositivo: {DEVICE}")
return _model, _tokenizer, _calibrator, _threshold_cfg
# ββ Inferencia principal βββββββββββββββββββββββββββββββββββββββββββββββββββββββ
def predict(text: str) -> dict:
"""
Ejecuta el pipeline completo: preproceso β TTA β calibraciΓ³n β abstenciΓ³n.
Retorna:
{
"label": "IA" | "Humano" | "Incierto",
"confidence": float (0β100),
"ai_prob": float (0β100),
"human_prob": float (0β100),
"word_count": int,
"abstained": bool
}
"""
model, tokenizer, calibrator, cfg = load_model()
# ββ 1. Preprocesamiento ββββββββββββββββββββββββββββββββββββββββββββββββββββ
cleaned = _preprocess(text)
# ββ 2. TTA: pesos [0.5, 0.25, 0.25] ββββββββββββββββββββββββββββββββββββββ
tta_texts = [cleaned, _tta_truncate(cleaned), _tta_span_mask(cleaned)]
tta_weights = cfg.get("tta_weights", [0.5, 0.25, 0.25])
# Batch ΓΊnico con los 3 pases TTA para eficiencia
raw_probs = _encode_and_predict(model, tta_texts, tokenizer, MAX_LENGTH)
# Promedio ponderado de las probabilidades AI de los tres pases
ai_raw = sum(w * p for w, p in zip(tta_weights, raw_probs))
# ββ 3. CalibraciΓ³n isotΓ³nica βββββββββββββββββββββββββββββββββββββββββββββββ
# IsotonicRegression.predict() espera array 1D de shape (n_samples,)
ai_calibrated = float(calibrator.predict(np.array([ai_raw]))[0])
# ββ 4. AbstenciΓ³n ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
abstain_low = cfg.get("abstain_low", 0.44)
abstain_high = cfg.get("abstain_high", 0.56)
threshold = cfg.get("threshold", 0.50)
abstained = abstain_low <= ai_calibrated <= abstain_high
if abstained:
label = "Incierto"
confidence = round((1 - abs(ai_calibrated - 0.5) * 2) * 100, 2)
elif ai_calibrated > threshold:
label = "IA"
confidence = round(ai_calibrated * 100, 2)
else:
label = "Humano"
confidence = round((1 - ai_calibrated) * 100, 2)
return {
"label": label,
"confidence": confidence,
"ai_prob": round(ai_calibrated * 100, 2),
"human_prob": round((1 - ai_calibrated) * 100, 2),
"word_count": len(cleaned.split()),
"abstained": abstained,
}
|