| |
| """ |
| Deep quality audit of litdata_english_clean. |
| |
| Checks EVERY row for: |
| 1. Content bias (topic distribution, over-represented domains) |
| 2. Unwanted context (ads, spam, SEO, cookie notices, legal boilerplate, etc.) |
| 3. English quality (grammar structure, vocabulary richness, readability) |
| 4. LLM learning value (diverse sentence structures, good knowledge density) |
| 5. Toxic/harmful content flags |
| 6. Residual noise (leftover URLs, code, non-English fragments) |
| |
| Goal: ensure the data teaches the LLM to understand English very well |
| so it can later do SFT on any dataset with strong comprehension. |
| """ |
|
|
| import json |
| import os |
| import re |
| import time |
| import string |
| from pathlib import Path |
| from collections import Counter, defaultdict |
|
|
| import numpy as np |
| from tokenizers import Tokenizer |
|
|
| ROOT = Path(__file__).resolve().parent.parent.parent |
| BLOCK_SIZE = 1025 |
| DTYPE = np.int32 |
| EOS_TOKEN_ID = 0 |
|
|
| print("Loading tokenizer...") |
| tokenizer = Tokenizer.from_file( |
| str(ROOT / "Base" / "checkpoints" / "EleutherAI" / "pythia-160m" / "tokenizer.json") |
| ) |
|
|
| |
| |
| |
|
|
| def read_all_tokens(litdata_dir): |
| with open(litdata_dir / "index.json") as f: |
| index = json.load(f) |
| chunks = index["chunks"] |
| total_tokens = sum(c["dim"] for c in chunks) |
| print(f" Reading {len(chunks)} chunks ({total_tokens:,} tokens)...") |
| all_tokens = np.empty(total_tokens, dtype=DTYPE) |
| pos = 0 |
| for i, chunk in enumerate(chunks): |
| chunk_path = litdata_dir / chunk["filename"] |
| n_blocks = chunk["chunk_size"] |
| header_ints = 1 + n_blocks + 1 |
| header_bytes = header_ints * 4 |
| with open(chunk_path, "rb") as f: |
| f.seek(header_bytes) |
| data = np.fromfile(f, dtype=DTYPE, count=chunk["dim"]) |
| all_tokens[pos:pos + len(data)] = data |
| pos += len(data) |
| print(f" Read {len(chunks)} chunks ({pos:,} tokens)") |
| return all_tokens[:pos] |
|
|
|
|
| def split_documents(token_stream): |
| eos_positions = np.where(token_stream == EOS_TOKEN_ID)[0] |
| docs = [] |
| start = 0 |
| for eos_pos in eos_positions: |
| if eos_pos > start: |
| docs.append(token_stream[start:eos_pos]) |
| start = eos_pos + 1 |
| if start < len(token_stream): |
| docs.append(token_stream[start:]) |
| return docs |
|
|
|
|
| |
| |
| |
|
|
| |
| RE_COOKIE = re.compile(r'(cookie|cookies)\s+(policy|consent|notice|preferences|settings)', re.I) |
| RE_PRIVACY = re.compile(r'(privacy\s+policy|terms\s+of\s+(service|use)|legal\s+disclaimer)', re.I) |
| RE_SUBSCRIBE = re.compile(r'(subscribe|sign\s*up|newsletter|unsubscribe|opt[\s-]*out)', re.I) |
| RE_CLICKBAIT = re.compile(r'(you\s+won\'?t\s+believe|click\s+here|read\s+more|share\s+this|trending\s+now|sponsored|advertisement)', re.I) |
| RE_SEO_SPAM = re.compile(r'(best\s+\d+\s+\w+\s+for|top\s+\d+\s+\w+|buy\s+now|free\s+shipping|limited\s+time\s+offer|discount\s+code)', re.I) |
| RE_NAVIGATION = re.compile(r'(home\s*>\s*|breadcrumb|sidebar|footer|header|menu|navigation|skip\s+to\s+content)', re.I) |
| RE_SOCIAL = re.compile(r'(follow\s+us\s+on|share\s+on\s+(facebook|twitter|linkedin|instagram)|like\s+us\s+on|tweet\s+this)', re.I) |
| RE_COMMENT_SECTION = re.compile(r'(leave\s+a\s+(comment|reply)|post\s+a\s+comment|\d+\s+comments?\s|logged\s+in\s+as)', re.I) |
| RE_COPYRIGHT = re.compile(r'(all\s+rights\s+reserved|copyright\s+\d{4}|\(c\)\s*\d{4})', re.I) |
| RE_BOILERPLATE_LOGIN = re.compile(r'(log\s*in|sign\s*in|create\s+account|forgot\s+password|remember\s+me)', re.I) |
|
|
| |
| RE_RESIDUAL_URL = re.compile(r'https?://\S+|www\.\S+', re.I) |
| RE_RESIDUAL_EMAIL = re.compile(r'\b[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}\b') |
| RE_RESIDUAL_CODE = re.compile(r'(function\s*\(|var\s+\w+\s*=|console\.log|document\.get|if\s*\(\s*\w+\s*[!=]==)', re.I) |
| RE_CURLY_BRACES = re.compile(r'\{[^}]{5,}\}') |
| RE_HEX_COLORS = re.compile(r'#[0-9a-fA-F]{6}\b') |
|
|
| |
| RE_CJK = re.compile(r'[\u4e00-\u9fff\u3040-\u309f\u30a0-\u30ff\uac00-\ud7af]{3,}') |
| RE_ARABIC = re.compile(r'[\u0600-\u06ff]{5,}') |
| RE_CYRILLIC = re.compile(r'[\u0400-\u04ff]{5,}') |
| RE_DEVANAGARI = re.compile(r'[\u0900-\u097f]{5,}') |
|
|
| |
| RE_ALLCAPS_LINE = re.compile(r'^[A-Z\s]{20,}$', re.M) |
| RE_EXCESSIVE_NUMBERS = re.compile(r'(?:\d+[\s,.-]*){10,}') |
| RE_LIST_SPAM = re.compile(r'(?:^\s*[-*]\s*.{3,50}\n){10,}', re.M) |
|
|
| |
| TOXIC_TERMS = [ |
| 'kill yourself', 'kys', 'hate speech', 'racial slur', |
| 'white supremac', 'nazi', 'nigger', 'faggot', |
| ] |
|
|
|
|
| |
| |
| |
|
|
| TOPIC_KEYWORDS = { |
| "Science": ["experiment", "hypothesis", "molecule", "atom", "chemical", "physics", |
| "biology", "evolution", "species", "organism", "cell", "dna", "gene", |
| "electron", "neutron", "quantum", "telescope", "galaxy", "planet"], |
| "Mathematics": ["equation", "theorem", "algebra", "calculus", "geometry", "integer", |
| "fraction", "polynomial", "derivative", "integral", "matrix", "probability"], |
| "History": ["century", "civilization", "empire", "dynasty", "revolution", "colonial", |
| "medieval", "ancient", "historian", "archaeological", "monarchy", "treaty"], |
| "Geography": ["continent", "climate", "ocean", "mountain", "river", "latitude", |
| "longitude", "ecosystem", "peninsula", "volcano", "earthquake", "terrain"], |
| "Literature": ["novel", "poem", "author", "literary", "character", "narrative", |
| "fiction", "metaphor", "protagonist", "shakespeare", "prose", "genre"], |
| "Technology": ["software", "hardware", "computer", "algorithm", "database", "internet", |
| "programming", "digital", "server", "network", "processor", "encryption"], |
| "Medicine/Health": ["patient", "symptom", "diagnosis", "treatment", "disease", "infection", |
| "surgery", "therapy", "vaccine", "antibiotic", "clinical", "chronic"], |
| "Law/Politics": ["constitution", "legislation", "democracy", "parliament", "judiciary", |
| "amendment", "election", "government", "policy", "regulation", "statute"], |
| "Economics/Business": ["market", "economy", "inflation", "revenue", "investment", "stock", |
| "profit", "trade", "gdp", "fiscal", "monetary", "corporation"], |
| "Education": ["student", "teacher", "curriculum", "classroom", "university", "academic", |
| "learning", "examination", "school", "pedagogy", "literacy", "enrollment"], |
| "Philosophy/Religion": ["philosophy", "ethics", "moral", "theological", "spiritual", |
| "consciousness", "existence", "metaphysics", "belief", "virtue"], |
| "Arts/Culture": ["painting", "sculpture", "museum", "gallery", "architecture", "cinema", |
| "music", "composer", "portrait", "exhibition", "artistic", "cultural"], |
| "Environment": ["pollution", "conservation", "deforestation", "renewable", "sustainability", |
| "biodiversity", "emissions", "habitat", "endangered", "recycling"], |
| "Psychology": ["behavior", "cognitive", "emotion", "personality", "anxiety", "depression", |
| "consciousness", "motivation", "perception", "neuroscience", "memory"], |
| "Sports": ["championship", "tournament", "athlete", "league", "stadium", "coach", |
| "scorer", "goalkeeper", "referee", "olympics", "medal", "cricket"], |
| } |
|
|
|
|
| |
| |
| |
|
|
| def compute_readability(text): |
| """Simplified Flesch-Kincaid readability approximation.""" |
| words = text.split() |
| if len(words) < 10: |
| return 0.0 |
| sentences = max(len(re.findall(r'[.!?]+', text)), 1) |
| |
| syllables = sum(len(re.findall(r'[aeiouy]+', w.lower())) for w in words) |
| syllables = max(syllables, len(words)) |
| words_per_sent = len(words) / sentences |
| syl_per_word = syllables / len(words) |
| |
| fre = 206.835 - 1.015 * words_per_sent - 84.6 * syl_per_word |
| return max(0, min(100, fre)) |
|
|
|
|
| def classify_sentence_types(text): |
| """Classify sentences into types for diversity check.""" |
| sents = re.split(r'(?<=[.!?])\s+', text[:5000]) |
| types = Counter() |
| for s in sents: |
| s = s.strip() |
| if not s: |
| continue |
| if s.endswith('?'): |
| types['question'] += 1 |
| elif s.endswith('!'): |
| types['exclamation'] += 1 |
| elif any(s.lower().startswith(w) for w in ['because', 'since', 'although', 'while', 'if', 'when', 'whereas']): |
| types['complex'] += 1 |
| elif any(s.lower().startswith(w) for w in ['the', 'a ', 'an ', 'this', 'that', 'these', 'those']): |
| types['declarative'] += 1 |
| elif any(s.lower().startswith(w) for w in ['for example', 'such as', 'in other words', 'namely']): |
| types['explanatory'] += 1 |
| elif any(s.lower().startswith(w) for w in ['however', 'nevertheless', 'moreover', 'furthermore', 'therefore']): |
| types['transitional'] += 1 |
| else: |
| types['other'] += 1 |
| return types |
|
|
|
|
| def detect_topic(text_lower): |
| """Classify document into topics based on keyword density.""" |
| topics_found = [] |
| for topic, keywords in TOPIC_KEYWORDS.items(): |
| hits = sum(1 for kw in keywords if kw in text_lower) |
| if hits >= 2: |
| topics_found.append((topic, hits)) |
| topics_found.sort(key=lambda x: -x[1]) |
| return topics_found |
|
|
|
|
| |
| |
| |
|
|
| def main(): |
| input_dir = ROOT / "Base" / "data" / "litdata_english_clean" |
|
|
| print(f"\n{'='*75}") |
| print(f" DEEP QUALITY AUDIT: litdata_english_clean") |
| print(f" Input: {input_dir}") |
| print(f"{'='*75}") |
|
|
| |
| t0 = time.time() |
| token_stream = read_all_tokens(input_dir) |
| doc_tokens = split_documents(token_stream) |
| print(f" Found {len(doc_tokens):,} documents") |
| del token_stream |
|
|
| print(f" Decoding ALL {len(doc_tokens):,} documents...") |
| texts = [] |
| t1 = time.time() |
| for i, toks in enumerate(doc_tokens): |
| text = tokenizer.decode(toks.tolist(), skip_special_tokens=False) |
| texts.append(text) |
| if (i + 1) % 20000 == 0 or i == len(doc_tokens) - 1: |
| print(f" Decoded {i+1:,}/{len(doc_tokens):,}") |
| del doc_tokens |
| print(f" Decoded in {time.time()-t1:.1f}s") |
|
|
| total_docs = len(texts) |
| print(f"\n Auditing {total_docs:,} documents across ALL rows...\n") |
|
|
| |
| |
| |
| t2 = time.time() |
|
|
| |
| topic_counter = Counter() |
| topic_per_doc = [] |
| sentence_type_totals = Counter() |
| readability_scores = [] |
| word_counts = [] |
| vocab_richness = [] |
| avg_sentence_lengths = [] |
|
|
| |
| issues = { |
| "cookie_privacy": [], |
| "subscribe_newsletter": [], |
| "clickbait_seo": [], |
| "navigation_boilerplate": [], |
| "social_media": [], |
| "comment_section": [], |
| "copyright_legal": [], |
| "login_boilerplate": [], |
| "residual_urls": [], |
| "residual_emails": [], |
| "residual_code": [], |
| "non_english_fragments": [], |
| "allcaps_heavy": [], |
| "excessive_numbers": [], |
| "list_spam": [], |
| "toxic_content": [], |
| "too_short": [], |
| "too_repetitive": [], |
| "low_readability": [], |
| "single_topic_bias": [], |
| } |
|
|
| |
| flagged_docs = set() |
| flag_reasons = defaultdict(list) |
|
|
| for i, text in enumerate(texts): |
| text_lower = text.lower() |
| words = text.split() |
| word_count = len(words) |
| word_counts.append(word_count) |
|
|
| |
| topics = detect_topic(text_lower) |
| if topics: |
| for t_name, _ in topics[:2]: |
| topic_counter[t_name] += 1 |
| topic_per_doc.append(topics[0][0]) |
| else: |
| topic_counter["Uncategorized"] += 1 |
| topic_per_doc.append("Uncategorized") |
|
|
| |
| stypes = classify_sentence_types(text) |
| for k, v in stypes.items(): |
| sentence_type_totals[k] += v |
|
|
| |
| fre = compute_readability(text) |
| readability_scores.append(fre) |
|
|
| |
| if word_count > 20: |
| unique_ratio = len(set(w.lower() for w in words)) / word_count |
| vocab_richness.append(unique_ratio) |
| else: |
| vocab_richness.append(0) |
|
|
| |
| sents = re.split(r'[.!?]+', text) |
| sents = [s for s in sents if len(s.strip().split()) > 2] |
| if sents: |
| avg_sl = sum(len(s.split()) for s in sents) / len(sents) |
| avg_sentence_lengths.append(avg_sl) |
| else: |
| avg_sentence_lengths.append(0) |
|
|
| |
| is_flagged = False |
|
|
| |
| if RE_COOKIE.search(text) or RE_PRIVACY.search(text): |
| m = RE_COOKIE.findall(text) + RE_PRIVACY.findall(text) |
| |
| if len(m) >= 2 or (len(m) >= 1 and word_count < 100): |
| issues["cookie_privacy"].append(i) |
| if word_count < 100: |
| is_flagged = True |
| flag_reasons[i].append("cookie/privacy boilerplate") |
|
|
| |
| m = RE_SUBSCRIBE.findall(text) |
| if len(m) >= 2: |
| issues["subscribe_newsletter"].append(i) |
| if word_count < 100 and len(m) >= 2: |
| is_flagged = True |
| flag_reasons[i].append("subscribe/newsletter spam") |
|
|
| |
| m = RE_CLICKBAIT.findall(text) |
| if m: |
| issues["clickbait_seo"].append(i) |
| if len(m) >= 3: |
| is_flagged = True |
| flag_reasons[i].append("clickbait/SEO content") |
|
|
| |
| m = RE_NAVIGATION.findall(text) |
| if len(m) >= 3: |
| issues["navigation_boilerplate"].append(i) |
| if word_count < 80: |
| is_flagged = True |
| flag_reasons[i].append("navigation boilerplate") |
|
|
| |
| m = RE_SOCIAL.findall(text) |
| if m: |
| issues["social_media"].append(i) |
|
|
| |
| m = RE_COMMENT_SECTION.findall(text) |
| if m: |
| issues["comment_section"].append(i) |
|
|
| |
| m = RE_COPYRIGHT.findall(text) |
| if m: |
| issues["copyright_legal"].append(i) |
|
|
| |
| m = RE_BOILERPLATE_LOGIN.findall(text) |
| if len(m) >= 3: |
| issues["login_boilerplate"].append(i) |
| if word_count < 80: |
| is_flagged = True |
| flag_reasons[i].append("login boilerplate") |
|
|
| |
| m = RE_RESIDUAL_URL.findall(text) |
| if m: |
| issues["residual_urls"].append(i) |
| is_flagged = True |
| flag_reasons[i].append(f"residual URLs ({len(m)})") |
|
|
| |
| m = RE_RESIDUAL_EMAIL.findall(text) |
| if m: |
| issues["residual_emails"].append(i) |
|
|
| |
| m = RE_RESIDUAL_CODE.findall(text) |
| if len(m) >= 3: |
| issues["residual_code"].append(i) |
| if len(m) >= 5: |
| is_flagged = True |
| flag_reasons[i].append(f"residual code ({len(m)} matches)") |
|
|
| |
| has_cjk = bool(RE_CJK.search(text)) |
| has_arabic = bool(RE_ARABIC.search(text)) |
| has_cyrillic = bool(RE_CYRILLIC.search(text)) |
| has_devanagari = bool(RE_DEVANAGARI.search(text)) |
| if has_cjk or has_arabic or has_cyrillic or has_devanagari: |
| issues["non_english_fragments"].append(i) |
| scripts = [] |
| if has_cjk: scripts.append("CJK") |
| if has_arabic: scripts.append("Arabic") |
| if has_cyrillic: scripts.append("Cyrillic") |
| if has_devanagari: scripts.append("Devanagari") |
| is_flagged = True |
| flag_reasons[i].append(f"non-English ({', '.join(scripts)})") |
|
|
| |
| caps_lines = RE_ALLCAPS_LINE.findall(text) |
| if len(caps_lines) >= 3: |
| issues["allcaps_heavy"].append(i) |
|
|
| |
| if RE_EXCESSIVE_NUMBERS.search(text): |
| issues["excessive_numbers"].append(i) |
|
|
| |
| if RE_LIST_SPAM.search(text): |
| issues["list_spam"].append(i) |
|
|
| |
| for term in TOXIC_TERMS: |
| if term in text_lower: |
| issues["toxic_content"].append(i) |
| is_flagged = True |
| flag_reasons[i].append(f"toxic: '{term}'") |
| break |
|
|
| |
| if word_count < 50: |
| issues["too_short"].append(i) |
| is_flagged = True |
| flag_reasons[i].append(f"too short ({word_count} words)") |
|
|
| |
| if word_count > 50 and vocab_richness[-1] < 0.20: |
| issues["too_repetitive"].append(i) |
| is_flagged = True |
| flag_reasons[i].append(f"very repetitive (unique ratio: {vocab_richness[-1]:.3f})") |
|
|
| |
| if fre < 10 and word_count > 50: |
| issues["low_readability"].append(i) |
|
|
| if is_flagged: |
| flagged_docs.add(i) |
|
|
| if (i + 1) % 10000 == 0 or i == total_docs - 1: |
| print(f" Audited {i+1:,}/{total_docs:,} | flagged so far: {len(flagged_docs):,}") |
|
|
| audit_time = time.time() - t2 |
| print(f" Audit completed in {audit_time:.1f}s") |
|
|
| |
| |
| |
| report = [] |
| report.append(f"\n{'='*75}") |
| report.append(f" LITDATA_ENGLISH_CLEAN - DEEP QUALITY AUDIT REPORT") |
| report.append(f"{'='*75}") |
| report.append(f"\n Total documents audited: {total_docs:,}") |
| report.append(f" Total flagged for review: {len(flagged_docs):,} ({len(flagged_docs)/total_docs*100:.2f}%)") |
| report.append(f" Audit time: {audit_time:.1f}s") |
|
|
| |
| report.append(f"\n\n TOPIC DISTRIBUTION (all {total_docs:,} docs)") |
| report.append(f" {'-'*65}") |
| total_categorized = sum(topic_counter.values()) |
| sorted_topics = sorted(topic_counter.items(), key=lambda x: -x[1]) |
| max_topic_count = sorted_topics[0][1] if sorted_topics else 0 |
| for topic, count in sorted_topics: |
| pct = count / total_categorized * 100 |
| bar = "#" * int(pct / 2) |
| report.append(f" {topic:<25} {count:>7,} ({pct:5.1f}%) {bar}") |
|
|
| |
| if sorted_topics: |
| top_pct = sorted_topics[0][1] / total_categorized * 100 |
| if top_pct > 30: |
| report.append(f"\n ** WARNING: '{sorted_topics[0][0]}' dominates at {top_pct:.1f}% - potential topic bias **") |
| else: |
| report.append(f"\n OK: No single topic exceeds 30% - good diversity") |
|
|
| |
| report.append(f"\n\n ENGLISH QUALITY METRICS (all {total_docs:,} docs)") |
| report.append(f" {'-'*65}") |
|
|
| avg_readability = sum(readability_scores)/len(readability_scores) |
| avg_vocab = sum(vocab_richness)/len(vocab_richness) |
| avg_words = sum(word_counts)/len(word_counts) |
| avg_sent_len = sum(avg_sentence_lengths)/max(len([x for x in avg_sentence_lengths if x > 0]), 1) |
|
|
| report.append(f" Avg Flesch Reading Ease: {avg_readability:.1f}") |
| if avg_readability >= 60: |
| report.append(f" -> Standard/Easy (good for general English learning)") |
| elif avg_readability >= 30: |
| report.append(f" -> College level (moderately complex)") |
| else: |
| report.append(f" -> Very difficult (may hinder learning)") |
|
|
| report.append(f" Avg vocabulary richness: {avg_vocab:.4f} (unique words / total words)") |
| report.append(f" Avg document length: {avg_words:.0f} words") |
| report.append(f" Avg sentence length: {avg_sent_len:.1f} words/sentence") |
|
|
| |
| short_docs = sum(1 for w in word_counts if w < 50) |
| medium_docs = sum(1 for w in word_counts if 50 <= w < 200) |
| standard_docs = sum(1 for w in word_counts if 200 <= w < 1000) |
| long_docs = sum(1 for w in word_counts if 1000 <= w < 5000) |
| very_long_docs = sum(1 for w in word_counts if w >= 5000) |
|
|
| report.append(f"\n Document length distribution:") |
| report.append(f" < 50 words: {short_docs:>7,} ({short_docs/total_docs*100:.1f}%)") |
| report.append(f" 50-199 words: {medium_docs:>7,} ({medium_docs/total_docs*100:.1f}%)") |
| report.append(f" 200-999 words: {standard_docs:>7,} ({standard_docs/total_docs*100:.1f}%)") |
| report.append(f" 1,000-4,999 words: {long_docs:>7,} ({long_docs/total_docs*100:.1f}%)") |
| report.append(f" 5,000+ words: {very_long_docs:>7,} ({very_long_docs/total_docs*100:.1f}%)") |
|
|
| |
| very_easy = sum(1 for r in readability_scores if r >= 80) |
| easy = sum(1 for r in readability_scores if 60 <= r < 80) |
| college = sum(1 for r in readability_scores if 30 <= r < 60) |
| hard = sum(1 for r in readability_scores if 10 <= r < 30) |
| very_hard = sum(1 for r in readability_scores if r < 10) |
|
|
| report.append(f"\n Readability distribution:") |
| report.append(f" Very Easy (80-100): {very_easy:>7,} ({very_easy/total_docs*100:.1f}%)") |
| report.append(f" Easy (60-79): {easy:>7,} ({easy/total_docs*100:.1f}%)") |
| report.append(f" College (30-59): {college:>7,} ({college/total_docs*100:.1f}%)") |
| report.append(f" Hard (10-29): {hard:>7,} ({hard/total_docs*100:.1f}%)") |
| report.append(f" Very Hard (0-9): {very_hard:>7,} ({very_hard/total_docs*100:.1f}%)") |
|
|
| |
| report.append(f"\n\n SENTENCE TYPE DIVERSITY") |
| report.append(f" {'-'*65}") |
| total_sents = sum(sentence_type_totals.values()) |
| for stype, count in sorted(sentence_type_totals.items(), key=lambda x: -x[1]): |
| pct = count / max(total_sents, 1) * 100 |
| report.append(f" {stype:<20} {count:>10,} ({pct:5.1f}%)") |
| report.append(f" {'TOTAL':<20} {total_sents:>10,}") |
|
|
| if sentence_type_totals.get('question', 0) / max(total_sents, 1) < 0.01: |
| report.append(f" ** NOTE: Very few questions - adding Q&A data in SFT will help **") |
|
|
| |
| report.append(f"\n\n UNWANTED CONTENT DETECTION (all {total_docs:,} docs scanned)") |
| report.append(f" {'-'*65}") |
| issue_order = [ |
| ("cookie_privacy", "Cookie/Privacy boilerplate"), |
| ("subscribe_newsletter", "Subscribe/Newsletter prompts"), |
| ("clickbait_seo", "Clickbait/SEO content"), |
| ("navigation_boilerplate", "Navigation boilerplate"), |
| ("social_media", "Social media prompts"), |
| ("comment_section", "Comment section artifacts"), |
| ("copyright_legal", "Copyright/Legal notices"), |
| ("login_boilerplate", "Login/Account boilerplate"), |
| ("residual_urls", "Residual URLs"), |
| ("residual_emails", "Residual email addresses"), |
| ("residual_code", "Residual code fragments"), |
| ("non_english_fragments", "Non-English script fragments"), |
| ("allcaps_heavy", "Heavy ALL-CAPS usage"), |
| ("excessive_numbers", "Excessive number sequences"), |
| ("list_spam", "Long list-only content"), |
| ("toxic_content", "Toxic/harmful content"), |
| ("too_short", "Too short (< 50 words)"), |
| ("too_repetitive", "Very repetitive content"), |
| ("low_readability", "Extremely low readability"), |
| ] |
|
|
| total_issues = 0 |
| for key, label in issue_order: |
| count = len(issues[key]) |
| total_issues += count |
| pct = count / total_docs * 100 |
| status = "OK" if count == 0 else "CLEAN" if pct < 0.1 else "LOW" if pct < 1 else "MEDIUM" if pct < 5 else "HIGH" |
| marker = " *" if count > 0 and pct >= 1 else "" |
| report.append(f" {label:<35} {count:>6,} ({pct:5.2f}%) [{status}]{marker}") |
|
|
| report.append(f"\n Total issue instances: {total_issues:,}") |
|
|
| |
| report.append(f"\n\n FLAGGED DOCUMENTS FOR REVIEW: {len(flagged_docs):,}") |
| report.append(f" {'-'*65}") |
|
|
| if flagged_docs: |
| |
| reason_counter = Counter() |
| for doc_idx, reasons in flag_reasons.items(): |
| for r in reasons: |
| reason_counter[r.split('(')[0].strip()] += 1 |
|
|
| report.append(f" Flag reason summary:") |
| for reason, count in sorted(reason_counter.items(), key=lambda x: -x[1]): |
| report.append(f" {reason:<40} {count:>6,}") |
|
|
| |
| report.append(f"\n Worst flagged documents (up to 15 examples):") |
| report.append(f" {'-'*65}") |
| |
| worst = sorted(flag_reasons.items(), key=lambda x: -len(x[1]))[:15] |
| for doc_idx, reasons in worst: |
| text_preview = texts[doc_idx][:200].replace('\n', ' ') |
| wc = len(texts[doc_idx].split()) |
| report.append(f"\n Doc #{doc_idx} ({wc} words) - Flags: {', '.join(reasons)}") |
| report.append(f" \"{text_preview}...\"") |
| else: |
| report.append(f" No documents flagged - dataset is clean!") |
|
|
| |
| report.append(f"\n\n LLM LEARNING VALUE ASSESSMENT") |
| report.append(f" {'-'*65}") |
|
|
| good_count = 0 |
| for i in range(total_docs): |
| if i not in flagged_docs: |
| if word_counts[i] >= 100 and readability_scores[i] >= 30 and vocab_richness[i] >= 0.3: |
| good_count += 1 |
|
|
| good_pct = good_count / total_docs * 100 |
| report.append(f" High-quality docs (100+ words, readable, diverse vocab): {good_count:,} ({good_pct:.1f}%)") |
| report.append(f" Flagged docs (potential issues): {len(flagged_docs):,} ({len(flagged_docs)/total_docs*100:.1f}%)") |
|
|
| |
| if good_pct >= 95: |
| grade = "A" |
| assessment = "Excellent - dataset will teach strong English comprehension" |
| elif good_pct >= 85: |
| grade = "B" |
| assessment = "Good - dataset is solid, minor cleanup would help" |
| elif good_pct >= 70: |
| grade = "C" |
| assessment = "Fair - dataset needs targeted cleanup of flagged docs" |
| else: |
| grade = "D" |
| assessment = "Needs work - significant cleanup required" |
|
|
| report.append(f"\n GRADE: {grade}") |
| report.append(f" ASSESSMENT: {assessment}") |
|
|
| |
| report.append(f"\n RECOMMENDATIONS FOR OPTIMAL LLM ENGLISH LEARNING:") |
| if len(issues["too_short"]) > 0: |
| report.append(f" - Remove {len(issues['too_short']):,} docs under 50 words (too short to teach patterns)") |
| if len(issues["residual_urls"]) > 0: |
| report.append(f" - Strip {len(issues['residual_urls']):,} docs still containing URLs") |
| if len(issues["non_english_fragments"]) > 0: |
| report.append(f" - Remove {len(issues['non_english_fragments']):,} docs with non-English script fragments") |
| if len(issues["too_repetitive"]) > 0: |
| report.append(f" - Remove {len(issues['too_repetitive']):,} very repetitive docs") |
| if len(issues["toxic_content"]) > 0: |
| report.append(f" - URGENT: Remove {len(issues['toxic_content']):,} docs with toxic content") |
| if len(issues["residual_code"]) > 0: |
| report.append(f" - Review {len(issues['residual_code']):,} docs with residual code fragments") |
| if sorted_topics and sorted_topics[0][1] / total_categorized * 100 > 30: |
| report.append(f" - Consider balancing topics ('{sorted_topics[0][0]}' is over-represented)") |
| if sentence_type_totals.get('question', 0) / max(total_sents, 1) < 0.03: |
| report.append(f" - Dataset has few questions - SFT with Q&A pairs will complement this well") |
| if len(flagged_docs) == 0: |
| report.append(f" - Dataset is clean and ready for pretraining!") |
| elif len(flagged_docs) < 100: |
| report.append(f" - Only {len(flagged_docs)} docs flagged - minor cleanup recommended") |
| report.append(f" - Shall I auto-remove flagged docs and rebuild? (would lose minimal data)") |
|
|
| report.append(f"\n{'='*75}") |
|
|
| |
| full_report = '\n'.join(report) |
| print(full_report) |
|
|
| |
| report_path = input_dir / "DEEP_AUDIT_REPORT.txt" |
| with open(report_path, "w", encoding="utf-8") as f: |
| f.write(full_report) |
| print(f"\n Report saved to: {report_path}") |
|
|
| |
| if flagged_docs: |
| flagged_path = input_dir / "flagged_docs.json" |
| flagged_data = { |
| "total_docs": total_docs, |
| "flagged_count": len(flagged_docs), |
| "flagged_indices": sorted(flagged_docs), |
| "reasons": {str(k): v for k, v in flag_reasons.items()}, |
| } |
| with open(flagged_path, "w", encoding="utf-8") as f: |
| json.dump(flagged_data, f, indent=2) |
| print(f" Flagged indices saved to: {flagged_path}") |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|