Sentiment_Analysis / src /weak_labeling.py
Lavender825's picture
Deploy 10W0715 dual-interface sentiment app
422d4ca
Raw
History Blame Contribute Delete
21.9 kB
"""Weak supervision label generation for ACSA v2: META-DEPENDENT labeling.
Key design principle: the labeling rules MUST depend on metadata so that a
text-only model (Baseline 3) cannot learn the full labeling function. Only a
model that sees both text AND metadata can reconstruct the labels perfectly.
Three meta-dependent labeling mechanisms:
1. Meta-Aware Aspect Detection: if the product's features_text mentions an
aspect (e.g. "100% Cotton" -> MATERIAL), we treat that aspect as "salient"
for this SKU. If the review text doesn't explicitly mention it via keyword
but the user gave a low/high rating, we infer aspect sentiment from rating.
Rationale: a product that advertises its material invites material-based
evaluation. A 1-star review on such a product likely reflects material
disappointment even if the word "fabric" never appears.
2. Price-Aware VALUE: products priced above 1.5x the category median with
low ratings get VALUE=Negative; products below 0.5x median with high
ratings get VALUE=Positive 鈥?even without "price"/"worth" in the text.
3. Category salience prior: Shoes 鈫?SIZE is extra-salient; Dresses/Tops 鈫? APPEARANCE is extra-salient. For salient aspects in the product's category,
we lower the threshold for marking them as mentioned.
For each review row, for each aspect, we assign one of:
0 = Not_Mentioned, 1 = Positive, 2 = Negative
Output columns: aspect_<NAME> (int), evidence_<NAME> (JSON list)
"""
import json
import logging
import re
from typing import Dict, List, Optional, Set, Tuple
import numpy as np
import pandas as pd
from tqdm import tqdm
from . import config as cfg
from .aspect_dict import compile_aspect_patterns
logger = logging.getLogger(__name__)
_SENT_SPLIT_RE = re.compile(r"(?<=[.!?])\s+")
# Category -> which aspects are extra-salient (lowered threshold for detection)
# High-precision phrase rules for clothing reviews. These catch common cases
# that generic sentence sentiment may score as neutral.
EXPLICIT_PHRASE_RULES = [
("SIZE", 2, [
r"\bruns?\s+(a\s+little\s+)?small\b",
r"\bruns?\s+(a\s+little\s+)?large\b",
r"\btoo\s+(small|large|big|tight|loose|short|long)\b",
r"\b(size|fit|fits|fitting)\b[^.!?]{0,35}\b(small|large|big|tight|loose)\b",
r"\bwrong\s+size\b",
]),
("SIZE", 1, [
r"\btrue\s+to\s+size\b",
r"\bperfect\s+fit\b",
r"\bfit(s)?\s+perfectly\b",
]),
("MATERIAL", 2, [
r"\bheavier\s+than\s+expected\b",
r"\btoo\s+heavy\b",
r"\b(feels?|felt)\s+heavy\b",
r"\b(stiff|scratchy|itchy|rough|bulky)\b",
r"\bsee[- ]through\b",
r"\btoo\s+thin\b",
]),
("MATERIAL", 1, [
r"\bsoft\s+fabric\b",
r"\bfabric\s+(feels?\s+)?soft\b",
r"\bbreathable\b",
r"\blightweight\b",
]),
("QUALITY", 2, [
r"\bfell\s+apart\b",
r"\bfalling\s+apart\b",
r"\b(broke|broken|ripped|tore|torn)\b",
r"\bloose\s+threads?\b",
r"\b(pilling|pilled|faded|shrunk)\b",
r"\bshrunk\s+after\s+wash(ing)?\b",
]),
("QUALITY", 1, [
r"\bwell[- ]made\b",
r"\bgood\s+quality\b",
r"\bhigh\s+quality\b",
r"\bdurable\b",
]),
("APPEARANCE", 2, [
r"\b(color|colour)\s+(is\s+)?off\b",
r"\bdifferent\s+from\s+(the\s+)?(picture|photo)\b",
r"\bmisleading\s+photo\b",
r"\blooks?\s+cheap\b",
r"\bugly\b",
]),
("APPEARANCE", 1, [
r"\bas\s+(pictured|shown|described|advertised)\b",
r"\bmatches\s+the\s+picture\b",
r"\b(beautiful|gorgeous|vibrant|stunning)\b",
]),
("STYLE", 2, [
r"\bunflattering\b",
r"\bshapeless\b",
r"\bawkward\s+cut\b",
]),
("STYLE", 1, [
r"\bflattering\b",
r"\bstylish\b",
r"\breceived\s+compliments\b",
]),
("VALUE", 2, [
r"\bnot\s+worth(\s+(it|the\s+(money|price)))?\b",
r"\bnot\s+worth\s+what\s+i\s+paid\b",
r"\b(overpriced|over-priced|too\s+expensive)\b",
r"\b(waste|wasted)\s+of\s+money\b",
r"\bsave\s+your\s+money\b",
r"\bnot\s+a\s+good\s+value\b",
r"\bpoor\s+value\b",
r"\brip[-\s]?off\b",
]),
("VALUE", 1, [
r"\bworth\s+(it|the\s+(money|price))\b",
r"\bworth\s+every\s+penny\b",
r"\b(good|great|excellent)\s+value\b",
r"\bvalue\s+for\s+money\b",
r"\b(good|great)\s+deal\b",
r"\b(reasonable|affordable)\s+price\b",
r"\bgood\s+for\s+the\s+price\b",
r"\bgreat\s+for\s+the\s+price\b",
]),
]
# Terms mined from product descriptions are useful model inputs, but many are
# too generic to be aspect evidence in a review. Filtering happens only inside
# this labeler; data/aspect_dict.json remains unchanged for reproducibility.
LABEL_NOISE_TERMS = {
"breathable", "comfort", "easy", "gift", "hand", "high", "keep",
"lightweight", "long", "look", "only", "please", "pockets", "pull",
"quality", "size", "spandex",
}
# Local sentiment must be aspect-specific. A global word such as
# "uncomfortable" cannot legitimately make every aspect Negative.
ASPECT_LOCAL_CUES = {
"SIZE": {
1: [r"\b(true\s+to\s+size|perfect\s+fit|fits?\s+perfectly)\b"],
2: [r"\b(too\s+(small|large|big|tight|loose|short|long)|runs?\s+(small|large)|wrong\s+size)\b"],
},
"MATERIAL": {
1: [r"\b(soft\s+(fabric|material)|breathable\s+(fabric|material)|fabric\s+feels?\s+soft)\b"],
2: [r"\b(scratchy|itchy|rough|stiff|bulky|see[- ]through|too\s+(heavy|thin))\b"],
},
"QUALITY": {
1: [r"\b(well[- ]made|good\s+quality|high\s+quality|durable|sturdy)\b"],
2: [r"\b(fell\s+apart|falling\s+apart|broke|broken|ripped|torn|pilling|pilled|faded|shrunk|loose\s+threads?)\b"],
},
"APPEARANCE": {
1: [r"\b(beautiful|gorgeous|vibrant|stunning|matches?\s+(the\s+)?(picture|photo))\b"],
2: [r"\b(color|colour)\s+(is\s+)?off\b", r"\b(different\s+from\s+(the\s+)?(picture|photo)|misleading\s+photo|ugly)\b"],
},
"STYLE": {
1: [r"\b(flattering|stylish|fashionable|received\s+compliments?)\b"],
2: [r"\b(unflattering|shapeless|awkward\s+cut)\b"],
},
"VALUE": {
1: [r"\b(worth\s+(it|the\s+(money|price))|worth\s+every\s+penny|(good|great|excellent)\s+value|value\s+for\s+money|(good|great)\s+deal|(reasonable|affordable)\s+price|(good|great)\s+for\s+the\s+price)\b"],
2: [r"\b(not\s+worth(\s+(it|the\s+(money|price)))?|overpriced|over-priced|too\s+expensive|waste(d)?\s+of\s+money|save\s+your\s+money|not\s+a\s+good\s+value|poor\s+value|rip[-\s]?off)\b"],
},
}
CATEGORY_SALIENCE = {
"Shoes": {"SIZE", "QUALITY"},
"Boots": {"SIZE", "QUALITY"},
"Sandals": {"SIZE", "QUALITY", "APPEARANCE"},
"Sneakers": {"SIZE", "QUALITY"},
"Dresses": {"APPEARANCE", "STYLE", "SIZE"},
"Tops": {"APPEARANCE", "MATERIAL"},
"T-Shirts": {"MATERIAL", "SIZE"},
"Sweaters": {"MATERIAL", "QUALITY"},
"Jeans": {"SIZE", "QUALITY"},
"Pants": {"SIZE", "QUALITY"},
"Jackets": {"QUALITY", "MATERIAL"},
"Coats": {"QUALITY", "MATERIAL"},
"Jewelry": {"VALUE", "APPEARANCE", "QUALITY"},
"Watches": {"VALUE", "QUALITY"},
"Handbags": {"QUALITY", "VALUE"},
"Socks": {"MATERIAL", "SIZE"},
"Underwear": {"MATERIAL", "SIZE"},
"Lingerie": {"MATERIAL", "SIZE", "APPEARANCE"},
"Swimwear": {"SIZE", "APPEARANCE"},
"Activewear": {"MATERIAL", "SIZE"},
}
def split_sentences(text: str) -> List[str]:
if not text:
return []
return [s.strip() for s in _SENT_SPLIT_RE.split(text.strip()) if s.strip()]
class WeakLabeler:
"""Apply weak supervision rules to assign aspect-level sentiment labels."""
def __init__(self, aspect_dict: Dict[str, List[str]],
category_median_prices: Optional[Dict[str, float]] = None):
self.aspect_dict = aspect_dict
self.aspects = list(aspect_dict.keys())
self.label_aspect_dict = {
aspect: [term for term in terms if term.strip().lower() not in LABEL_NOISE_TERMS]
for aspect, terms in aspect_dict.items()
}
self.patterns = compile_aspect_patterns(self.label_aspect_dict)
self.category_median_prices = category_median_prices or {}
from vaderSentiment.vaderSentiment import SentimentIntensityAnalyzer
self.vader = SentimentIntensityAnalyzer()
# ---- atomic signals ----
def _sentence_sentiment(self, sentence: str) -> int:
"""1 (pos), 2 (neg), 0 (neutral/unknown)."""
score = self.vader.polarity_scores(sentence)["compound"]
if score >= cfg.VADER_POSITIVE_THRESHOLD:
return 1
if score <= cfg.VADER_NEGATIVE_THRESHOLD:
return 2
return 0
def _rating_sentiment(self, rating: int) -> int:
if rating >= cfg.RATING_POSITIVE_THRESHOLD:
return 1
if rating <= cfg.RATING_NEGATIVE_THRESHOLD:
return 2
return 0
def _meta_prior(self, avg_rating: float) -> int:
if not np.isfinite(avg_rating):
return 0
if avg_rating >= 4.3:
return 1
if avg_rating <= 2.7:
return 2
return 0
def _local_cue_sentiment(self, aspect: str, sentence: str) -> int:
"""Return sentiment only for cues that belong to this aspect."""
text = str(sentence or "").lower()
for pattern in ASPECT_LOCAL_CUES.get(aspect, {}).get(2, []):
if re.search(pattern, text, flags=re.IGNORECASE):
return 2
for pattern in ASPECT_LOCAL_CUES.get(aspect, {}).get(1, []):
if re.search(pattern, text, flags=re.IGNORECASE):
return 1
return 0
def _resolve_text_sentiment(self, aspect: str, sentence: str) -> int:
"""Use aspect-specific local evidence before sentence-level sentiment."""
local = self._local_cue_sentiment(aspect, sentence)
if local != 0:
return local
return self._sentence_sentiment(sentence)
def _allow_meta_inference(self, aspect: str, rating: int, meta_avg_rating: float,
features_salient: Set[str], category_salient: Set[str]) -> bool:
"""Allow metadata-only labels only for high-confidence extreme ratings.
Text evidence always has priority. Metadata fills otherwise unlabeled
cases, preserving a real but bounded role for product context.
"""
if rating <= 1:
return aspect in features_salient or aspect in category_salient
if rating >= 5:
return (
aspect in features_salient
and np.isfinite(meta_avg_rating)
and meta_avg_rating >= 4.3
)
return False
# ---- META-DEPENDENT mechanisms ----
def _detect_salient_aspects_from_features(
self, features_text: str
) -> Set[str]:
"""Check which aspects are mentioned in the product's features_text.
Returns a set of aspect names that the product explicitly advertises.
"""
if not features_text:
return set()
salient = set()
ft_lower = features_text.lower()
for aspect in self.aspects:
if self.patterns[aspect].search(ft_lower):
salient.add(aspect)
return salient
def _get_category_salient_aspects(self, leaf_category: str) -> Set[str]:
"""Which aspects are extra-salient for this product category."""
if not leaf_category:
return set()
for cat_key, aspects in CATEGORY_SALIENCE.items():
if cat_key.lower() in leaf_category.lower():
return aspects
return set()
def _price_value_signal(
self, price: float, rating: int, leaf_category: str
) -> int:
"""Conservative price-aware VALUE labeling.\n Infer VALUE only when category-relative price and rating form an\n extreme high-confidence signal. Returns 0, 1, or 2.\n """
if not np.isfinite(price) or price <= 0:
return 0
median = self.category_median_prices.get(leaf_category)
if median is None or median <= 0:
# Fall back to global median if category not found
median = self.category_median_prices.get("__global__")
if median is None or median <= 0:
return 0
if price > median * 1.75 and rating <= 1:
return 2 # Clearly expensive + very bad rating -> VALUE Negative
if price < median * 0.45 and rating >= 5:
return 1 # Clearly cheap + excellent rating -> VALUE Positive
if price > median * 2.25 and rating <= 2:
return 2 # Very expensive + low rating -> VALUE Negative
return 0
def _apply_explicit_phrase_rules(
self,
text: str,
labels: Dict[str, int],
evidence: Dict[str, List[str]],
) -> None:
"""Apply high-precision clothing phrase rules after sentence matching.
Negative rules can override positive/default labels because explicit
complaints are usually more informative for ACSA than global rating.
Positive rules fill only unlabeled aspects to avoid washing out a
complaint already found in text.
"""
lower = str(text or "").lower()
for aspect, sentiment, patterns in EXPLICIT_PHRASE_RULES:
for pat in patterns:
m = re.search(pat, lower, flags=re.IGNORECASE)
if not m:
continue
snippet = m.group(0)
if labels[aspect] == 0 or sentiment == 2:
labels[aspect] = sentiment
evidence[aspect].append(f"[phrase-rule: {snippet}]")
break
# ---- per-review labeling ----
def label_review(
self,
text: str,
rating: int,
meta_avg_rating: float = float("nan"),
features_text: str = "",
leaf_category: str = "",
price: float = float("nan"),
) -> Tuple[Dict[str, int], Dict[str, List[str]]]:
labels = {a: 0 for a in self.aspects}
evidence = {a: [] for a in self.aspects}
sentences = split_sentences(text)
# --- META-DEPENDENT: detect salient aspects from product features ---
features_salient = self._detect_salient_aspects_from_features(features_text)
category_salient = self._get_category_salient_aspects(leaf_category)
all_salient = features_salient | category_salient
# --- Phase 1: text-based detection ---
for sent in sentences:
for aspect in self.aspects:
if not self.patterns[aspect].search(sent):
continue
sentiment = self._resolve_text_sentiment(aspect, sent)
if sentiment == 0:
continue
evidence[aspect].append(sent)
if labels[aspect] == 0:
labels[aspect] = sentiment
elif sentiment == 2:
labels[aspect] = 2
# --- Phase 1b: high-precision clothing phrase rules ---
self._apply_explicit_phrase_rules(text, labels, evidence)
# --- Phase 2: controlled metadata inference ---
# Text labels remain authoritative. An unlabeled aspect receives a
# metadata label only when an extreme review rating and a product signal
# agree. This avoids turning every product attribute into sentiment.
if cfg.META_PRIOR_BOOST:
for aspect in self.aspects:
if labels[aspect] != 0:
if aspect in features_salient:
evidence[aspect].append("[meta-corroborated: feature source]")
elif aspect in category_salient:
evidence[aspect].append("[meta-corroborated: category source]")
continue
if self._allow_meta_inference(
aspect, rating, meta_avg_rating, features_salient, category_salient
):
labels[aspect] = 2 if rating <= 1 else 1
source = "feature" if aspect in features_salient else "category"
evidence[aspect].append(
f"[meta-inferred: {source} source, rating={rating}]"
)
# --- Phase 3: price-aware VALUE inference ---
# Price can label VALUE when its category-relative signal and review
# rating agree, even if the text does not use a price keyword.
if labels["VALUE"] == 0:
price_signal = self._price_value_signal(price, rating, leaf_category)
if price_signal != 0:
labels["VALUE"] = price_signal
evidence["VALUE"].append(
f"[price-inferred: price={price:.1f}, "
f"cat_median={self.category_median_prices.get(leaf_category, '?')}, "
f"rating={rating}]"
)
return labels, evidence
# ---- dataframe-level ----
def label_dataframe(
self,
df: pd.DataFrame,
text_col: str = "full_text",
rating_col: str = "rating",
meta_avg_rating_col: str = "average_rating",
features_text_col: str = "features_text",
leaf_category_col: str = "leaf_category",
price_col: str = "price",
) -> pd.DataFrame:
all_labels = {a: [] for a in self.aspects}
all_evidence = {a: [] for a in self.aspects}
has_meta_rating = meta_avg_rating_col in df.columns
has_features = features_text_col in df.columns
has_category = leaf_category_col in df.columns
has_price = price_col in df.columns
if not has_meta_rating:
logger.info("[weak_labeling] No %s column; meta prior disabled.", meta_avg_rating_col)
if not has_features:
logger.info("[weak_labeling] No %s column; feature-based salience disabled.", features_text_col)
if not has_price:
logger.info("[weak_labeling] No %s column; price-aware VALUE disabled.", price_col)
for _, row in tqdm(df.iterrows(), total=len(df), desc="weak labeling"):
meta_avg = float(row[meta_avg_rating_col]) if has_meta_rating else float("nan")
feat_text = str(row[features_text_col]) if has_features else ""
cat = str(row[leaf_category_col]) if has_category else ""
price = float(row[price_col]) if has_price else float("nan")
labels, evidence = self.label_review(
str(row[text_col]),
int(row[rating_col]),
meta_avg_rating=meta_avg,
features_text=feat_text,
leaf_category=cat,
price=price,
)
for a in self.aspects:
all_labels[a].append(labels[a])
all_evidence[a].append(evidence[a])
result = df.copy()
for a in self.aspects:
result[f"aspect_{a}"] = all_labels[a]
result[f"evidence_{a}"] = [json.dumps(e) for e in all_evidence[a]]
return result
def compute_category_median_prices(df: pd.DataFrame,
price_col: str = "price",
cat_col: str = "leaf_category") -> Dict[str, float]:
"""Compute median price per leaf category. Used by Price-Aware VALUE labeling."""
result = {}
if price_col not in df.columns or cat_col not in df.columns:
return result
prices = pd.to_numeric(df[price_col], errors="coerce")
global_median = float(prices.median())
if np.isfinite(global_median):
result["__global__"] = global_median
for cat, group in df.groupby(cat_col):
cat_prices = pd.to_numeric(group[price_col], errors="coerce").dropna()
if len(cat_prices) >= 5:
result[str(cat)] = float(cat_prices.median())
logger.info("Computed median prices for %d categories (global=%.1f)",
len(result) - 1, result.get("__global__", 0))
return result
def label_distribution_summary(df: pd.DataFrame) -> pd.DataFrame:
rows = []
for a in cfg.ASPECTS:
col = f"aspect_{a}"
if col not in df.columns:
continue
vc = df[col].value_counts().sort_index()
rows.append({
"aspect": a,
"not_mentioned": int(vc.get(0, 0)),
"positive": int(vc.get(1, 0)),
"negative": int(vc.get(2, 0)),
"mentioned_pct": round(100 * (1 - vc.get(0, 0) / max(len(df), 1)), 2),
})
return pd.DataFrame(rows)
def save_audit_sample(df: pd.DataFrame, n: int = cfg.AUDIT_SAMPLE_SIZE, path=None):
if path is None:
path = cfg.REPORT_DIR / "weak_label_audit_sample.csv"
cols = ["full_text", "rating", "overall_label", "parent_asin",
"average_rating", "leaf_category", "price", "features_text"]
cols += [f"aspect_{a}" for a in cfg.ASPECTS]
cols += [f"evidence_{a}" for a in cfg.ASPECTS]
cols = [c for c in cols if c in df.columns]
by_rating = df.groupby("rating", group_keys=False)
sample = by_rating.apply(lambda x: x.sample(n=min(max(n // 5, 1), len(x)),
random_state=cfg.RANDOM_SEED))
sample = sample.reset_index(drop=True)
if len(sample) > n:
sample = sample.sample(n=n, random_state=cfg.RANDOM_SEED)
keep = [c for c in cols if c in sample.columns]
sample[keep].to_csv(path, index=False)
logger.info("Saved %d audit samples to %s", len(sample), path)
return sample