""" EDA trên sampled_1m_reviews.jsonl và sampled_meta.jsonl """ import json, time, math from collections import Counter, defaultdict from pathlib import Path REVIEWS_PATH = Path("/workspace/amazon/sampled_1m_reviews.jsonl") META_PATH = Path("/workspace/amazon/sampled_meta.jsonl") def hist(values, bins, total=None): """ASCII histogram.""" total = total or len(values) for lo, hi in bins: cnt = sum(1 for v in values if lo <= v <= hi) bar = "#" * int(cnt / total * 40) print(f" {lo:>5}–{hi:<6} {cnt:>8,} ({cnt/total*100:5.1f}%) {bar}") def percentiles(vals, ps=(10,25,50,75,90,95,99)): s = sorted(vals) n = len(s) parts = {f"p{p}": s[min(int(n*p/100), n-1)] for p in ps} parts["min"] = s[0]; parts["max"] = s[-1]; parts["mean"] = sum(s)/n return parts # ═══════════════════════════════════════════════════════════════════════════ # 1. REVIEWS EDA # ═══════════════════════════════════════════════════════════════════════════ print("\n" + "="*60) print("1. REVIEWS EDA (sampled_1m_reviews.jsonl)") print("="*60) t0 = time.time() user_reviews = defaultdict(list) # uid -> [ts, ...] user_stars = defaultdict(list) # uid -> [stars, ...] item_counts = Counter() star_dist = Counter() year_dist = Counter() verified_cnt = 0 text_lens = [] total_reviews = 0 import datetime with open(REVIEWS_PATH, "r", encoding="utf-8", errors="replace") as f: for line in f: line = line.strip() if not line: continue try: rec = json.loads(line) except: continue total_reviews += 1 uid = rec.get("reviewerID", "") asin = rec.get("asin", "") ts = rec.get("unixReviewTime", 0) or 0 star = int(rec.get("overall", 0)) user_reviews[uid].append(ts) user_stars[uid].append(star) item_counts[asin] += 1 star_dist[star] += 1 if ts: year_dist[datetime.datetime.fromtimestamp(ts).year] += 1 txt = rec.get("reviewText", "") or "" if txt.strip(): text_lens.append(len(txt.split())) print(f"Total reviews : {total_reviews:,}") print(f"Unique users : {len(user_reviews):,}") print(f"Unique items : {len(item_counts):,}") # Rating distribution print("\nRating distribution:") for s in range(1, 6): cnt = star_dist[s] bar = "#" * int(cnt / total_reviews * 40) print(f" {s}★ {cnt:>8,} ({cnt/total_reviews*100:5.1f}%) {bar}") # Year distribution print("\nYear distribution:") for yr in sorted(year_dist): cnt = year_dist[yr] bar = "#" * int(cnt / total_reviews * 40) print(f" {yr} {cnt:>8,} ({cnt/total_reviews*100:5.1f}%) {bar}") # Review text length if text_lens: p = percentiles(text_lens) print(f"\nReview text length (words):") print(f" min={p['min']} p25={p['p25']} p50={p['p50']} p75={p['p75']} p90={p['p90']} p99={p['p99']} max={p['max']} mean={p['mean']:.1f}") # Per-user interaction count user_counts_list = [len(v) for v in user_reviews.values()] p = percentiles(user_counts_list) print(f"\nInteractions per user:") print(f" min={p['min']} p25={p['p25']} p50={p['p50']} p75={p['p75']} p90={p['p90']} p99={p['p99']} max={p['max']} mean={p['mean']:.1f}") print(" Buckets:") hist(user_counts_list, [(5,9),(10,19),(20,49),(50,99),(100,499),(500,9999)], len(user_counts_list)) # Per-user avg rating user_avg_stars = [sum(v)/len(v) for v in user_stars.values()] p = percentiles(user_avg_stars) print(f"\nAvg rating per user:") print(f" min={p['min']:.2f} p25={p['p25']:.2f} p50={p['p50']:.2f} p75={p['p75']:.2f} mean={p['mean']:.2f}") # Item popularity (long tail) pop = sorted(item_counts.values(), reverse=True) n_items = len(pop) top1_pct = sum(pop[:max(1,n_items//100)]) top10_pct = sum(pop[:max(1,n_items//10)]) print(f"\nItem popularity (long tail):") print(f" Top 1% items cover {top1_pct/total_reviews*100:.1f}% of reviews") print(f" Top 10% items cover {top10_pct/total_reviews*100:.1f}% of reviews") print(f" Items with 1 review: {sum(1 for v in pop if v==1):,}") print(f" Items with >=10: {sum(1 for v in pop if v>=10):,}") print(f" Items with >=100: {sum(1 for v in pop if v>=100):,}") # Temporal span per user user_spans = [] for uid, tss in user_reviews.items(): if len(tss) >= 2: user_spans.append((max(tss) - min(tss)) / (3600*24)) # days p = percentiles(user_spans) print(f"\nUser history span (days, users with ≥2 reviews):") print(f" p25={p['p25']:.0f}d p50={p['p50']:.0f}d p75={p['p75']:.0f}d p90={p['p90']:.0f}d max={p['max']:.0f}d mean={p['mean']:.1f}d") print(f"\n(Reviews EDA: {time.time()-t0:.1f}s)") # ═══════════════════════════════════════════════════════════════════════════ # 2. META EDA # ═══════════════════════════════════════════════════════════════════════════ print("\n" + "="*60) print("2. META EDA (sampled_meta.jsonl)") print("="*60) t1 = time.time() cat_top = Counter() brand_top = Counter() price_vals = [] also_buy_len = [] also_view_len= [] desc_lens = [] title_lens = [] total_meta = 0 has_fields = Counter() with open(META_PATH, "r", encoding="utf-8", errors="replace") as f: for line in f: line = line.strip() if not line: continue try: rec = json.loads(line) except: continue total_meta += 1 # categories cats = rec.get("category") or [] if len(cats) > 0: has_fields["category"] += 1 cat_top[cats[0]] += 1 # brand brand = (rec.get("brand") or "").strip() if brand: has_fields["brand"] += 1 brand_top[brand] += 1 # price p_raw = str(rec.get("price") or "").replace("$","").strip() if p_raw and p_raw.lower() not in ("", "none"): try: pv = float(p_raw.split("-")[0].replace(",","").strip()) if 0 < pv < 100000: price_vals.append(pv) has_fields["price"] += 1 except: pass # also_buy / also_view ab = rec.get("also_buy") or [] av = rec.get("also_view") or [] if ab: also_buy_len.append(len(ab)); has_fields["also_buy"] += 1 if av: also_view_len.append(len(av)); has_fields["also_view"] += 1 # description length desc = rec.get("description") or [] if isinstance(desc, list): desc = " ".join(desc) desc = str(desc).strip() if desc: has_fields["description"] += 1 desc_lens.append(len(desc.split())) # title length title = (rec.get("title") or "").strip() if title: has_fields["title"] += 1 title_lens.append(len(title.split())) print(f"Total meta records: {total_meta:,}") print(f"\nField coverage:") for field in ["title","brand","category","price","description","also_buy","also_view"]: cnt = has_fields[field] print(f" {field:<14} {cnt:>8,} ({cnt/total_meta*100:.1f}%)") # Category distribution print(f"\nTop-20 categories (level-1):") for cat, cnt in cat_top.most_common(20): bar = "#" * int(cnt / total_meta * 30) print(f" {cat[:50]:<52} {cnt:>7,} ({cnt/total_meta*100:.1f}%) {bar}") # Brand distribution print(f"\nTop-15 brands:") for brand, cnt in brand_top.most_common(15): print(f" {brand[:40]:<42} {cnt:>7,}") print(f" ... ({len(brand_top):,} unique brands total)") # Price if price_vals: p = percentiles(price_vals) print(f"\nPrice distribution (USD, records with valid price):") print(f" mean=${p['mean']:.2f} p25=${p['p25']:.2f} p50=${p['p50']:.2f} p75=${p['p75']:.2f} p90=${p['p90']:.2f} p99=${p['p99']:.2f} max=${p['max']:.2f}") print(" Buckets:") hist(price_vals, [(0,9),(10,24),(25,49),(50,99),(100,499),(500,99999)], len(price_vals)) # also_buy / also_view if also_buy_len: p = percentiles(also_buy_len) print(f"\nalso_buy per item (when present): mean={p['mean']:.1f} p50={p['p50']} p90={p['p90']} max={p['max']}") if also_view_len: p = percentiles(also_view_len) print(f"also_view per item (when present): mean={p['mean']:.1f} p50={p['p50']} p90={p['p90']} max={p['max']}") # Co-purchase graph density total_edges = sum(also_buy_len) + sum(also_view_len) print(f"\nCo-purchase graph:") print(f" also_buy edges: {sum(also_buy_len):,}") print(f" also_view edges: {sum(also_view_len):,}") print(f" total edges : {total_edges:,}") print(f" avg degree : {total_edges/total_meta:.1f}") # Description & title length if desc_lens: p = percentiles(desc_lens) print(f"\nDescription length (words): p25={p['p25']} p50={p['p50']} p75={p['p75']} p90={p['p90']} mean={p['mean']:.1f}") if title_lens: p = percentiles(title_lens) print(f"Title length (words): p25={p['p25']} p50={p['p50']} p75={p['p75']} p90={p['p90']} mean={p['mean']:.1f}") print(f"\n(Meta EDA: {time.time()-t1:.1f}s)") # ═══════════════════════════════════════════════════════════════════════════ # 3. CROSS-ANALYSIS # ═══════════════════════════════════════════════════════════════════════════ print("\n" + "="*60) print("3. CROSS-ANALYSIS") print("="*60) # Category distribution of reviewed items (via meta lookup) meta_asin_cat = {} with open(META_PATH, "r", encoding="utf-8", errors="replace") as f: for line in f: line = line.strip() if not line: continue try: rec = json.loads(line) except: continue asin = rec.get("asin","") cats = rec.get("category") or [] if asin and cats: meta_asin_cat[asin] = cats[0] # Map reviews to categories review_cat_dist = Counter() n_mapped = 0 with open(REVIEWS_PATH, "r", encoding="utf-8", errors="replace") as f: for line in f: line = line.strip() if not line: continue try: rec = json.loads(line) except: continue asin = rec.get("asin","") cat = meta_asin_cat.get(asin) if cat: review_cat_dist[cat] += 1 n_mapped += 1 print(f"\nCategory distribution of reviews (top-15):") for cat, cnt in review_cat_dist.most_common(15): bar = "#" * int(cnt / total_reviews * 30) print(f" {cat[:50]:<52} {cnt:>7,} ({cnt/total_reviews*100:.1f}%) {bar}") print(f"\nDone. Total EDA time: {time.time()-t0:.1f}s")