import os import csv import json import re import time import uuid import asyncio import aiohttp from bs4 import BeautifulSoup from urllib.parse import urljoin, urlparse from collections import Counter import textstat import concurrent.futures from dotenv import load_dotenv load_dotenv() # PATCH for missing aiohttp.SocketTimeoutError if not hasattr(aiohttp, 'SocketTimeoutError'): setattr(aiohttp, 'SocketTimeoutError', aiohttp.ClientTimeout) # PATCH for missing aiohttp.client_exceptions.NonHttpUrlClientError import aiohttp.client_exceptions if not hasattr(aiohttp.client_exceptions, 'NonHttpUrlClientError'): class NonHttpUrlClientError(Exception): pass setattr(aiohttp.client_exceptions, 'NonHttpUrlClientError', NonHttpUrlClientError) # PATCH for missing aiohttp.client_exceptions.InvalidUrlClientError if not hasattr(aiohttp.client_exceptions, 'InvalidUrlClientError'): class InvalidUrlClientError(Exception): pass setattr(aiohttp.client_exceptions, 'InvalidUrlClientError', InvalidUrlClientError) # Playwright imports from playwright.async_api import async_playwright PLAYWRIGHT_AVAILABLE = True # Optional grammar check try: import language_tool_python LT_AVAILABLE = True except Exception: LT_AVAILABLE = False print("⚠️ language_tool_python not available") HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"} # ============================== # SET YOUR OPENAI API KEY HERE # ============================== OPENAI_API_KEY = os.environ.get("OPENAI_API_KEY") if OPENAI_API_KEY: print("✅ OpenAI API Key loaded from environment") else: print("⚠️ OPENAI_API_KEY not set - AI features will be disabled") # ============================== # OPENAI CLIENT - COMPATIBLE VERSION (0.28.1) # ============================== _openai_client = None try: import openai OPENAI_AVAILABLE = True _openai_client = openai.OpenAI(api_key=os.environ.get("OPENAI_API_KEY")) print("✅ OpenAI client initialized successfully") except Exception as e: OPENAI_AVAILABLE = False print(f"⚠️ OpenAI not available: {e}") # ============================== # IMPROVED URL DISCOVERY WITH BETTER ERROR HANDLING # ============================== async def get_sitemap_links_parallel(base_url): """Get all URLs from sitemap with parallel processing - IMPROVED""" sitemap_urls = [ urljoin(base_url, "sitemap.xml"), urljoin(base_url, "sitemap_index.xml"), urljoin(base_url, "sitemap-0.xml"), urljoin(base_url, "sitemap.txt"), urljoin(base_url, "sitemap") ] async def fetch_sitemap(url): try: print(f" Trying sitemap: {url}") async with aiohttp.ClientSession() as session: async with session.get(url, headers=HEADERS, timeout=15) as response: if response.status == 200: content_type = response.headers.get('content-type', '').lower() text = await response.text() # Check if it's XML sitemap if 'xml' in content_type or ' { const allLinks = Array.from(document.links) .map(link => link.href) .filter(href => href && href.includes(baseDomain)) .filter(href => !href.includes('#') && !href.includes('javascript:')) // FILTER OUT NON-HTML FILES .filter(href => !href.includes('.xml') && !href.includes('sitemap') && !href.includes('rss') && !href.includes('feed')) .filter(href => !href.match(/\\.(pdf|jpg|png|gif|css|js|json)$/i)) .slice(0, 50); return [...new Set(allLinks)]; // Remove duplicates }""", urlparse(base_url).netloc) await browser.close() if links: print(f" Found {len(links)} internal HTML links") result_urls = [base_url] + links[:max_pages-1] return result_urls else: print(" No internal links found, using only homepage") return [base_url] except Exception as e: print(f"❌ URL discovery failed: {e}") print(" Using fallback: homepage only") return [base_url] # Always return at least homepage # ============================== # PARALLEL PLAYWRIGHT FETCHER # ============================== async def fetch_page_playwright(url, browser, timeout=25000): """Fetch a single page with Playwright""" context = await browser.new_context( viewport={'width': 1920, 'height': 1080}, user_agent='Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36', java_script_enabled=True ) # Block unnecessary resources for speed await context.route("**/*.{png,jpg,jpeg,gif,svg,webp}", lambda route: route.abort()) await context.route("**/*.css", lambda route: route.abort()) page = await context.new_page() try: # Longer timeout print(f" 📡 Fetching: {url}") await page.goto(url, wait_until='domcontentloaded', timeout=timeout) # Wait for critical content await page.wait_for_selector('body', timeout=15000) # Extract comprehensive SEO data seo_data = await page.evaluate("""() => { // Get all meta tags const metas = {}; document.querySelectorAll('meta').forEach(meta => { const name = meta.getAttribute('name') || meta.getAttribute('property'); if (name) metas[name] = meta.getAttribute('content'); }); // Get all images with detailed info const images = Array.from(document.images).map(img => ({ src: img.src, alt: img.alt || '', naturalWidth: img.naturalWidth, naturalHeight: img.naturalHeight, complete: img.complete })); // Get all links const links = Array.from(document.links).map(link => ({ href: link.href, text: link.textContent?.slice(0, 100) || '', rel: link.rel })); // Get schema data const schemas = []; document.querySelectorAll('script[type="application/ld+json"]').forEach(script => { try { if (script.textContent) { const data = JSON.parse(script.textContent); schemas.push(data); } } catch (e) {} }); return { url: window.location.href, title: document.title, metas: metas, description: metas.description || metas['og:description'] || '', canonical: document.querySelector('link[rel="canonical"]')?.href || '', robots: metas.robots || '', viewport: metas.viewport || '', h1_count: document.querySelectorAll('h1').length, h2_count: document.querySelectorAll('h2').length, h3_count: document.querySelectorAll('h3').length, images: images, links: links, schemas: schemas, html: document.documentElement.outerHTML, opengraph: { title: metas['og:title'] || '', description: metas['og:description'] || '', image: metas['og:image'] || '', url: metas['og:url'] || '' }, twitter: { title: metas['twitter:title'] || '', description: metas['twitter:description'] || '', image: metas['twitter:image'] || '', card: metas['twitter:card'] || '' } }; }""") print(f" ✅ Success: {url}") return seo_data except Exception as e: print(f" ❌ Failed: {url} - {str(e)[:100]}...") return None finally: await context.close() async def fetch_all_pages_parallel(urls, max_concurrent=1): """Fetch multiple pages in parallel""" if not urls: print("❌ No URLs to fetch!") return [] print(f"🚀 Launching {max_concurrent} browsers for parallel fetching...") print(f"📡 Fetching {len(urls)} pages...") async with async_playwright() as p: # Launch browser with optimized settings browser = await p.chromium.launch( headless=True, args=[ '--disable-gpu', '--disable-dev-shm-usage', '--disable-setuid-sandbox', '--no-first-run', '--no-sandbox', '--no-zygote', '--deterministic-fetch', '--max_old_space_size=4096' ] ) # Create semaphore for concurrency control semaphore = asyncio.Semaphore(max_concurrent) async def fetch_with_semaphore(url): async with semaphore: return await fetch_page_playwright(url, browser) # Fetch all pages in parallel with progress tasks = [fetch_with_semaphore(url) for url in urls] results = [] for i, task in enumerate(asyncio.as_completed(tasks)): result = await task results.append(result) if (i + 1) % 2 == 0 or (i + 1) == len(urls): print(f" 📊 Progress: {i + 1}/{len(urls)} pages completed") await browser.close() # Filter out failed fetches successful_results = [r for r in results if r is not None] print(f"✅ Successfully fetched {len(successful_results)} out of {len(urls)} pages") return successful_results # ============================== # PARALLEL SEO ANALYSIS - FIXED SCHEMA EXTRACTION # ============================== async def analyze_pages_parallel(playwright_data_list, domain): """Analyze all pages in parallel""" if not playwright_data_list: print("❌ No data to analyze!") return [] print("🔬 Analyzing pages in parallel...") def analyze_single_page(seo_data): """Analyze a single page's SEO data""" try: html = seo_data.get('html', '') soup = BeautifulSoup(html, 'html.parser') # Extract text content text = soup.get_text(separator=" ", strip=True) # Images analysis images_data = seo_data.get('images', []) total_images = len(images_data) missing_alt = len([img for img in images_data if not img.get('alt')]) # Links analysis links_data = seo_data.get('links', []) internal_links = len([link for link in links_data if domain in link.get('href', '')]) external_links = len([link for link in links_data if domain not in link.get('href', '')]) # ============================== # FIXED SCHEMA EXTRACTION - PROPERLY INDENTED # ============================== schemas = seo_data.get('schemas', []) schema_types = [] for schema in schemas: try: # Handle different schema formats if isinstance(schema, dict): # Direct schema object if '@type' in schema: schema_types.append(schema['@type']) # Schema with @graph if '@graph' in schema and isinstance(schema['@graph'], list): for item in schema['@graph']: if isinstance(item, dict) and '@type' in item: schema_types.append(item['@type']) elif isinstance(schema, list): # Array of schemas for item in schema: if isinstance(item, dict) and '@type' in item: schema_types.append(item['@type']) except Exception as e: print(f" Schema parsing error: {e}") # Also check for microdata and other schema formats in HTML try: # Check for microdata microdata = soup.find_all(attrs={"itemtype": True}) for item in microdata: itemtype = item.get('itemtype', '') if itemtype: schema_types.append(itemtype.split('/')[-1]) # Get just the type name # Check for other schema script tags schema_scripts = soup.find_all('script', type=lambda x: x and 'ld+json' in x) for script in schema_scripts: try: if script.string: data = json.loads(script.string) if isinstance(data, dict) and '@type' in data: schema_types.append(data['@type']) elif isinstance(data, list): for item in data: if isinstance(item, dict) and '@type' in item: schema_types.append(item['@type']) except: pass except Exception as e: print(f" HTML schema extraction error: {e}") # Deduplicate schema types schema_types = list(set(schema_types)) # ============================== # END OF FIXED SCHEMA EXTRACTION # ============================== # Metrics try: readability_score = textstat.flesch_reading_ease(text) except Exception: readability_score = 0 word_count = len(text.split()) # Grammar errors (optional) grammar_errors = 0 if LT_AVAILABLE: try: tool = language_tool_python.LanguageTool('en-US') grammar_errors = len(tool.check(text[:1000])) tool.close() except Exception: pass # Keyword density top_keywords = keyword_density(text) text_to_html_ratio = round((len(text) / len(html)) * 100, 2) if html else 0 # Compile page data page = { "url": seo_data.get('url', ''), "title": seo_data.get('title', ''), "meta_description": seo_data.get('description', ''), "h1_count": seo_data.get('h1_count', 0), "h2_count": seo_data.get('h2_count', 0), "h3_count": seo_data.get('h3_count', 0), "heading_order": get_heading_order(soup), "missing_alt_tags": missing_alt, "total_images": total_images, "small_images": len([img for img in images_data if img.get('naturalWidth', 0) < 100]), "large_images": len([img for img in images_data if img.get('naturalWidth', 0) > 2000]), "ideal_images": len([img for img in images_data if 100 <= img.get('naturalWidth', 0) <= 2000]), "internal_links": internal_links, "external_links": external_links, "canonical_tag": bool(seo_data.get('canonical')), "robots_meta": seo_data.get('robots', ''), "viewport_present": 'width' in seo_data.get('viewport', ''), "schema_types": ", ".join(schema_types) if schema_types else "No schema found", "opengraph_tags": count_opengraph_tags(seo_data.get('metas', {})), "twitter_tags": count_twitter_tags(seo_data.get('metas', {})), "word_count": word_count, "readability_score": readability_score, "grammar_errors": grammar_errors, "text_to_html_ratio": text_to_html_ratio, "top_keywords": top_keywords, "load_time": 0, } return page except Exception as e: print(f"❌ Analysis error for {seo_data.get('url', 'unknown')}: {e}") return None # Run analysis in parallel using ThreadPoolExecutor loop = asyncio.get_event_loop() with concurrent.futures.ThreadPoolExecutor(max_workers=5) as executor: tasks = [ loop.run_in_executor(executor, analyze_single_page, data) for data in playwright_data_list ] results = await asyncio.gather(*tasks) # Filter out failed analyses successful_results = [r for r in results if r is not None] print(f"✅ Successfully analyzed {len(successful_results)} pages") return successful_results def get_heading_order(soup): """Extract heading order from BeautifulSoup""" headings = soup.find_all(re.compile('^h[1-6]$')) return ", ".join([h.name for h in headings]) def count_opengraph_tags(metas): """Count OpenGraph tags""" return len([k for k in metas.keys() if k.startswith('og:')]) def count_twitter_tags(metas): """Count Twitter card tags""" return len([k for k in metas.keys() if k.startswith('twitter:')]) def keyword_density(text): """Calculate keyword density""" words = re.findall(r'\b\w+\b', (text or "").lower()) freq = Counter(w for w in words if len(w) > 3) total = sum(freq.values()) or 1 items = sorted([(k, round(v / total * 100, 2)) for k, v in freq.items() if v > 1], key=lambda x: -x[1])[:10] return ", ".join([f"{k}:{p}%" for k, p in items]) # ============================== # ULTRA-SPECIFIC AI SUGGESTIONS - WITH EXACT REPLACEMENTS # ============================== async def generate_page_suggestions_async(page_data): """Generate ULTRA-SPECIFIC AI suggestions with exact replacements""" api_key = os.environ.get("OPENAI_API_KEY") if not api_key or api_key == "YOUR_OPENAI_API_KEY_HERE" or not OPENAI_AVAILABLE: return "AI disabled - set valid OPENAI_API_KEY" # Extract detailed page data url = page_data.get('url', 'Unknown URL') title = page_data.get('title', '') meta_description = page_data.get('meta_description', '') seo_score = page_data.get('seo_score', 0) h1_count = page_data.get('h1_count', 0) h2_count = page_data.get('h2_count', 0) h3_count = page_data.get('h3_count', 0) word_count = page_data.get('word_count', 0) readability_score = page_data.get('readability_score', 0) missing_alt_tags = page_data.get('missing_alt_tags', 0) total_images = page_data.get('total_images', 0) schema_types = page_data.get('schema_types', '') internal_links = page_data.get('internal_links', 0) external_links = page_data.get('external_links', 0) opengraph_tags = page_data.get('opengraph_tags', 0) twitter_tags = page_data.get('twitter_tags', 0) top_keywords = page_data.get('top_keywords', '') heading_order = page_data.get('heading_order', '') # ULTRA-SPECIFIC PROMPT - Demands exact replacements prompt = f""" You are an expert technical SEO consultant. Analyze this page and provide EXACT, ACTIONABLE recommendations with SPECIFIC REPLACEMENTS. CRITICAL REQUIREMENTS: - Provide EXACT replacement text for bad titles, meta descriptions, etc. - Give SPECIFIC OpenGraph and Twitter Card markup when missing - Provide EXACT schema markup code when missing - Give SPECIFIC H1 text when missing - Provide EXACT alt text examples for images PAGE DATA: URL: {url} Current SEO Score: {seo_score}/100 CURRENT CONTENT: - Title: "{title}" ({len(title)} chars) - Meta Description: "{meta_description}" ({len(meta_description)} chars) - H1 Count: {h1_count} | H2 Count: {h2_count} | H3 Count: {h3_count} - Word Count: {word_count} words - Readability: {readability_score}/100 - Missing Alt Tags: {missing_alt_tags} of {total_images} images - Schema: {schema_types} - Internal Links: {internal_links} | External Links: {external_links} - OpenGraph Tags: {opengraph_tags} | Twitter Cards: {twitter_tags} - Top Keywords: {top_keywords} - Heading Structure: {heading_order} Provide recommendations in this EXACT format: HIGH IMPACT: 1. TITLE OPTIMIZATION: Current: "{title}" ({len(title)} chars) REPLACE WITH: "[Exact new title text - 55-60 characters]" 2. META DESCRIPTION: Current: "{meta_description}" ({len(meta_description)} chars) REPLACE WITH: "[Exact new meta description - 150-155 characters]" 3. H1 TAG: Current: {h1_count} H1 tags ADD THIS EXACT H1: "[Exact H1 text with primary keyword]" MEDIUM IMPACT: 4. OPENGRAPH TAGS (Missing {8 - opengraph_tags} tags): ADD THIS EXACT MARKUP: 5. TWITTER CARDS (Missing {5 - twitter_tags} tags): ADD THIS EXACT MARKUP: 6. SCHEMA MARKUP: Current: {schema_types} ADD THIS EXACT SCHEMA: [Provide complete JSON-LD schema code] LOW IMPACT: 7. IMAGE ALT TEXT: Missing alt text for {missing_alt_tags} images EXAMPLE ALT TEXTS: - "[Exact alt text for first image]" - "[Exact alt text for second image]" 8. CONTENT IMPROVEMENT: Current: {word_count} words, {readability_score}/100 readability ADD THIS EXACT CONTENT SECTION: "[Specific content to add with exact paragraph]" Provide EXACT text replacements - no generic advice! """ try: print(f" 🤖 Generating ULTRA-SPECIFIC suggestions for: {url[:50]}...") # gpt-5-nano is a reasoning-family model: it only accepts # max_completion_tokens (not max_tokens) and only the default # temperature (1), so no temperature override is passed. # Uses OpenAI 0.28.1 syntax (matches installed SDK version). # response = await asyncio.get_event_loop().run_in_executor( # None, # lambda: openai.ChatCompletion.create( # model="gpt-5-nano", # messages=[ # {"role": "system", "content": "You are a technical SEO expert who provides EXACT replacement text and markup. Always give specific examples and complete code snippets. No generic advice allowed."}, # {"role": "user", "content": prompt} # ], # max_completion_tokens=800, # ) # ) response = await asyncio.get_event_loop().run_in_executor( None, lambda: _openai_client.chat.completions.create( model="gpt-5-nano", messages=[ {"role": "system", "content": "You are a technical SEO expert who provides EXACT replacement text and markup. Always give specific examples and complete code snippets. No generic advice allowed."}, {"role": "user", "content": prompt} ], max_completion_tokens=6000, reasoning_effort="low", ) ) ai_suggestion = response.choices[0].message.content.strip() print(f" ✅ ULTRA-SPECIFIC suggestions generated for: {url[:50]}...") return ai_suggestion except Exception as e: print(f" ❌ AI failed for {url[:50]}: {str(e)[:100]}") return generate_ultra_specific_fallback(page_data) def generate_ultra_specific_fallback(page_data): """Generate ultra-specific fallback suggestions with exact examples""" suggestions = [] url = page_data.get('url', '') title = page_data.get('title', '') meta_desc = page_data.get('meta_description', '') h1_count = page_data.get('h1_count', 0) word_count = page_data.get('word_count', 0) missing_alt = page_data.get('missing_alt_tags', 0) schema_types = page_data.get('schema_types', '') opengraph_tags = page_data.get('opengraph_tags', 0) twitter_tags = page_data.get('twitter_tags', 0) # Extract domain for context domain = urlparse(url).netloc.replace('www.', '') site_name = domain.split('.')[0].title() # HIGH IMPACT - EXACT REPLACEMENTS if h1_count == 0: page_topic = url.split('/')[-1].replace('-', ' ').title() suggestions.append(f"HIGH: ADD EXACT H1: '{page_topic} - Complete Guide | {site_name}'") title_len = len(title) if title_len < 45 or title_len > 65: page_topic = url.split('/')[-1].replace('-', ' ').title() suggestions.append(f"HIGH: REPLACE TITLE: '{page_topic} - Complete {site_name} Guide 2024'") if not meta_desc or len(meta_desc) < 50: page_topic = url.split('/')[-1].replace('-', ' ').title() suggestions.append(f"HIGH: REPLACE META: 'Learn everything about {page_topic.lower()} with our complete guide. Get expert tips, best practices, and step-by-step instructions from {site_name}.'") # MEDIUM IMPACT - EXACT MARKUP if opengraph_tags < 4: page_topic = url.split('/')[-1].replace('-', ' ').title() suggestions.append(f"MEDIUM: ADD OPENGraph:\n\n\n") if twitter_tags < 3: page_topic = url.split('/')[-1].replace('-', ' ').title() suggestions.append(f"MEDIUM: ADD TWITTER CARDS:\n\n\n") if schema_types == "No schema found": page_topic = url.split('/')[-1].replace('-', ' ').title() suggestions.append(f"MEDIUM: ADD SCHEMA:\n") # LOW IMPACT - EXACT EXAMPLES if missing_alt > 0: page_topic = url.split('/')[-1].replace('-', ' ').title() suggestions.append(f"LOW: ADD ALT TEXTS:\n- \"{page_topic} diagram and explanation\"\n- \"Step-by-step {page_topic.lower()} process visualization\"\n- \"{site_name} {page_topic} tutorial screenshot\"") if word_count < 800: suggestions.append(f"LOW: ADD CONTENT SECTION:\n\"In this comprehensive guide, we'll cover the essential aspects of {page_topic.lower()} including best practices, common pitfalls to avoid, and actionable strategies you can implement immediately. Whether you're a beginner or looking to advanced your skills, this guide provides the foundation you need for success.\"") return "\n\n".join(suggestions) if suggestions else "All major elements optimized - focus on internal linking and user experience" async def generate_all_page_suggestions_parallel(pages): """Generate ULTRA-SPECIFIC AI suggestions for ALL pages in parallel""" suggestions = {} print(f"🤖 Generating ULTRA-SPECIFIC AI suggestions for {len(pages)} pages in parallel...") # Create tasks for ALL pages tasks = [] for i, page in enumerate(pages): task = generate_page_suggestions_async(page) tasks.append((i, task)) # Run ALL AI calls concurrently if tasks: coroutines = [task for _, task in tasks] results = await asyncio.gather(*coroutines, return_exceptions=True) # Map results back to pages for result_idx, (page_idx, _) in enumerate(tasks): result = results[result_idx] if isinstance(result, Exception): print(f" ❌ AI failed for page {page_idx}, using ultra-specific fallback") suggestions[page_idx] = generate_ultra_specific_fallback(pages[page_idx]) else: suggestions[page_idx] = result return suggestions async def add_comprehensive_suggestions_async(results): """Add comprehensive ULTRA-SPECIFIC AI suggestions to all pages""" api_key = os.environ.get("OPENAI_API_KEY") if not api_key or api_key == "YOUR_OPENAI_API_KEY_HERE" or not OPENAI_AVAILABLE: print("⚠️ OPENAI_API_KEY not set — AI suggestions disabled.") for p in results: p["ai_suggestions"] = "AI suggestions disabled - set valid OPENAI_API_KEY" return print("🚀 Generating comprehensive ULTRA-SPECIFIC AI suggestions for all pages...") # Get ULTRA-SPECIFIC AI suggestions suggestions_dict = await generate_all_page_suggestions_parallel(results) # Apply suggestions to pages for i, p in enumerate(results): ai_suggestion = suggestions_dict.get(i, "No AI suggestions generated") p["ai_suggestions"] = ai_suggestion # ============================== # DETECT PAGE TYPE FUNCTION # ============================== def detect_page_type(url, page_data): """Detect if page is homepage, article, category, etc.""" parsed = urlparse(url) path = parsed.path.strip('/') # Check if it's homepage if not path or path == '' or path == 'index.html' or path == 'index.php': return 'homepage' # Check for common article patterns article_patterns = ['/blog/', '/article/', '/news/', '/post/', '/2024/', '/2025/'] if any(pattern in url for pattern in article_patterns): return 'article' # Check for category/listing pages category_patterns = ['/category/', '/tag/', '/topic/'] if any(pattern in url for pattern in category_patterns): return 'category' # Default return 'standard' # ============================== # COMPREHENSIVE SCORING FUNCTION - FIXED VERSION # ============================== def calculate_seo_score(page): """Calculate comprehensive SEO score with realistic thresholds""" score = 0 max_score = 100 # Detect page type once for use throughout url = page.get('url', '') page_type = detect_page_type(url, page) # ===== TITLE OPTIMIZATION (10 points) ===== title = page.get('title', '') if title: title_len = len(title) if 50 <= title_len <= 60: # Perfect score += 10 elif 45 <= title_len <= 65: # Good score += 8 elif 30 <= title_len <= 70: # Acceptable score += 6 elif title_len > 0: # Exists but poor score += 3 # ===== META DESCRIPTION (8 points) ===== meta_desc = page.get('meta_description', '') if meta_desc: meta_len = len(meta_desc) if 120 <= meta_len <= 155: # Perfect score += 8 elif 100 <= meta_len <= 160: # Good score += 6 elif 70 <= meta_len <= 170: # Acceptable score += 4 elif meta_len > 0: # Exists but poor score += 2 # ===== HEADING STRUCTURE (12 points) ===== h1_count = page.get('h1_count', 0) heading_order = page.get('heading_order', '') # H1 Score (6 points) if h1_count == 1: # Perfect score += 6 elif h1_count == 0: # Critical score += 0 elif h1_count == 2: # Minor issue score += 4 else: # Multiple H1s score += 1 # FIXED: Heading Hierarchy (6 points) - More flexible if heading_order: headings = [h.strip() for h in heading_order.split(',')] heading_levels = [] for heading in headings: if heading.startswith('h'): try: level = int(heading[1]) heading_levels.append(level) except: continue # More flexible heading structure scoring has_h1 = 1 in heading_levels has_h2 = 2 in heading_levels has_h3 = 3 in heading_levels # Check if headings follow a logical order if has_h1 and (has_h2 or has_h3): score += 6 # Full points for logical structure elif has_h1: score += 4 # Has H1 but no subheadings elif has_h2 or has_h3: score += 2 # No H1 but has other headings # ===== IMAGE OPTIMIZATION (15 points) ===== total_images = page.get('total_images', 0) missing_alt_tags = page.get('missing_alt_tags', 0) small_images = page.get('small_images', 0) large_images = page.get('large_images', 0) ideal_images = page.get('ideal_images', 0) # Alt Text Score (5 points) if total_images == 0: score += 5 else: alt_ratio = (total_images - missing_alt_tags) / total_images if alt_ratio >= 0.95: score += 5 elif alt_ratio >= 0.80: score += 4 elif alt_ratio >= 0.60: score += 3 elif alt_ratio >= 0.40: score += 2 elif alt_ratio > 0: score += 1 # FIXED: Image Size Optimization (5 points) - Forgive small images if total_images > 0: # Don't penalize small images that might be icons/logos # Assume first 2 small images could be logo/favicon/icons forgiven_small_images = max(0, small_images - 2) adjusted_total = total_images - forgiven_small_images if adjusted_total > 0: # Recalculate ideal ratio without penalized small images ideal_ratio = ideal_images / adjusted_total if adjusted_total > 0 else 1 if ideal_ratio >= 0.7: # Slightly lowered threshold score += 5 elif ideal_ratio >= 0.5: score += 4 elif ideal_ratio >= 0.3: score += 3 elif ideal_ratio >= 0.15: score += 2 else: score += 1 else: score += 5 # All images are forgiven (probably icons/logos) else: score += 5 # Image Quantity (5 points) if total_images == 0: score += 3 elif 3 <= total_images <= 20: score += 5 elif total_images <= 50: score += 4 elif total_images <= 100: score += 3 else: score += 2 # ===== FIXED LINK STRUCTURE (10 points) ===== internal_links = page.get('internal_links', 0) external_links = page.get('external_links', 0) # Internal Links (6 points) - More realistic thresholds if internal_links >= 30: # Was 80 score += 6 elif internal_links >= 20: # Was 60 score += 5 elif internal_links >= 10: # Was 40 score += 4 elif internal_links >= 5: # Was 20 score += 3 elif internal_links >= 3: # Was 10 score += 2 elif internal_links >= 1: score += 1 # External Links (4 points) - Keep as is if external_links >= 10: score += 4 elif external_links >= 7: score += 3 elif external_links >= 4: score += 2 elif external_links >= 1: score += 1 # ===== TECHNICAL SEO (20 points) ===== # Canonical Tag (3 points) if page.get('canonical_tag', False): score += 3 # Robots Meta (3 points) robots_meta = page.get('robots_meta', '') if robots_meta: robots_lower = robots_meta.lower() if 'noindex' not in robots_lower and 'nofollow' not in robots_lower: score += 3 elif 'noindex' in robots_lower: score += 0 else: score += 2 else: score += 2 # Viewport (3 points) if page.get('viewport_present', False): score += 3 # FIXED: Schema Markup (4 points) - Quality over quantity schema_types = page.get('schema_types', '') if schema_types and schema_types.strip() and schema_types != "No schema found": schema_count = len([s for s in schema_types.split(', ') if s.strip()]) # Check for important schema types important_schemas = ['Organization', 'WebSite', 'Article', 'Product', 'LocalBusiness'] has_important = any(schema in schema_types for schema in important_schemas) if has_important and schema_count >= 2: score += 4 # Has important schema plus others elif has_important: score += 3 # Has at least one important schema elif schema_count >= 2: score += 3 # Multiple schemas even if not "important" elif schema_count == 1: score += 2 # Has some schema else: # No schema, but don't penalize too heavily for certain page types if page_type not in ['article', 'product']: # Pages that really should have schema score += 1 # Small penalty instead of zero # OpenGraph Tags (4 points) opengraph_tags = page.get('opengraph_tags', 0) if opengraph_tags >= 10: score += 4 elif opengraph_tags >= 7: score += 3 elif opengraph_tags >= 5: score += 2 elif opengraph_tags >= 3: score += 1 # Twitter Cards (3 points) twitter_tags = page.get('twitter_tags', 0) if twitter_tags >= 5: score += 3 elif twitter_tags >= 3: score += 2 elif twitter_tags >= 1: score += 1 # ===== CONTENT QUALITY (25 points) ===== word_count = page.get('word_count', 0) readability_score = page.get('readability_score', 0) grammar_errors = page.get('grammar_errors', 0) text_to_html_ratio = page.get('text_to_html_ratio', 0) top_keywords = page.get('top_keywords', '') # FIXED: Word Count (6 points) - Page-type aware if page_type == 'homepage': # Homepages can be concise if word_count >= 500: score += 6 elif word_count >= 300: score += 5 elif word_count >= 200: score += 4 elif word_count >= 100: score += 3 elif word_count >= 50: score += 2 else: score += 1 elif page_type == 'article': # Articles need depth if word_count >= 2000: score += 6 elif word_count >= 1200: score += 5 elif word_count >= 800: score += 4 elif word_count >= 500: score += 3 elif word_count >= 300: score += 2 else: score += 1 else: # Standard pages if word_count >= 1500: score += 6 elif word_count >= 800: score += 5 elif word_count >= 500: score += 4 elif word_count >= 300: score += 3 elif word_count >= 150: score += 2 else: score += 1 # Readability (6 points) if readability_score >= 50: score += 6 elif readability_score >= 45: score += 5 elif readability_score >= 40: score += 4 elif readability_score >= 35: score += 3 elif readability_score >= 20: score += 2 elif readability_score >= 10: score += 1 # Grammar (4 points) if grammar_errors == 0: score += 4 elif grammar_errors <= 2: score += 3 elif grammar_errors <= 5: score += 2 elif grammar_errors <= 10: score += 1 # Text to HTML Ratio (5 points) if text_to_html_ratio >= 25: score += 5 elif text_to_html_ratio >= 20: score += 4 elif text_to_html_ratio >= 15: score += 3 elif text_to_html_ratio >= 10: score += 2 elif text_to_html_ratio >= 5: score += 1 # Keyword Optimization (4 points) if top_keywords: keyword_entries = [k for k in top_keywords.split(', ') if ':' in k and float(k.split(':')[1][:-1]) > 1.0] keyword_count = len(keyword_entries) if keyword_count >= 8: score += 4 elif keyword_count >= 5: score += 3 elif keyword_count >= 3: score += 2 elif keyword_count >= 1: score += 1 return min(score, max_score) # ============================== # ENHANCED MAIN FUNCTION WITH ULTRA-SPECIFIC AI SUGGESTIONS # ============================== async def run_seo_and_suggestions_async(base_url, max_pages=20, tmp_dir="/tmp", use_ai=True, max_concurrent=3): """ FULLY PARALLEL SEO analysis with Playwright - ENHANCED WITH ULTRA-SPECIFIC AI SUGGESTIONS """ if not base_url: raise ValueError("base_url is required") print(f"🎯 Starting ULTRA-SPECIFIC SEO analysis for: {base_url}") print(f"⚡ Optimized: {max_concurrent} concurrent browsers, {max_pages} max pages") print(f"🤖 ULTRA-SPECIFIC AI Suggestions: {'ENABLED' if use_ai and OPENAI_AVAILABLE and os.environ.get('OPENAI_API_KEY') else 'DISABLED'}") start_time = time.time() domain = urlparse(base_url).netloc try: # STEP 1: Discover URLs in parallel urls = await discover_urls_parallel(base_url, max_pages) if not urls: print("❌ No URLs discovered, using homepage only as fallback") urls = [base_url] print(f"🔍 Found {len(urls)} URLs to analyze") # STEP 2: Fetch all pages in parallel with Playwright playwright_data = await fetch_all_pages_parallel(urls, max_concurrent) if not playwright_data: print("❌ No pages could be fetched, creating error report") # Create error report os.makedirs(tmp_dir, exist_ok=True) filename = os.path.join(tmp_dir, f"seo_report_error_{uuid.uuid4().hex}.csv") error_data = [{ "url": base_url, "error": "Failed to fetch any pages. Site may be blocking bots or require authentication.", "seo_suggestions": "Check if the site is accessible and not blocking headless browsers." }] with open(filename, "w", newline="", encoding="utf-8") as f: writer = csv.DictWriter(f, fieldnames=error_data[0].keys()) writer.writeheader() writer.writerows(error_data) return error_data, filename # STEP 3: Analyze all pages in parallel results = await analyze_pages_parallel(playwright_data, domain) if not results: print("❌ No pages could be analyzed, creating error report") # Create error report os.makedirs(tmp_dir, exist_ok=True) filename = os.path.join(tmp_dir, f"seo_report_error_{uuid.uuid4().hex}.csv") error_data = [{ "url": base_url, "error": "Failed to analyze any pages. There may be issues with the page content.", "seo_suggestions": "Check if the site has proper HTML structure and content." }] with open(filename, "w", newline="", encoding="utf-8") as f: writer = csv.DictWriter(f, fieldnames=error_data[0].keys()) writer.writeheader() writer.writerows(error_data) return error_data, filename # STEP 4: Calculate SEO scores print("📊 Calculating SEO scores...") for p in results: p["seo_score"] = calculate_seo_score(p) # STEP 5: Generate ULTRA-SPECIFIC AI suggestions in parallel if use_ai: await add_comprehensive_suggestions_async(results) else: for p in results: p["ai_suggestions"] = "AI suggestions disabled - set use_ai=True and OPENAI_API_KEY" # STEP 6: Save to CSV os.makedirs(tmp_dir, exist_ok=True) filename = os.path.join(tmp_dir, f"seo_report_ultra_specific_{uuid.uuid4().hex}.csv") if results: keys = list(results[0].keys()) with open(filename, "w", newline="", encoding="utf-8") as f: writer = csv.DictWriter(f, fieldnames=keys) writer.writeheader() writer.writerows(results) elapsed_time = time.time() - start_time print(f"✅ ULTRA-SPECIFIC SEO analysis complete! Analyzed {len(results)} pages in {elapsed_time:.1f}s") print(f"📊 Report saved to: {filename}") # Count ULTRA-SPECIFIC AI suggestions ultra_specific_count = sum(1 for p in results if p.get('ai_suggestions') and 'AI disabled' not in p.get('ai_suggestions', '') and 'AI Error' not in p.get('ai_suggestions', '')) print(f"🤖 ULTRA-SPECIFIC AI Suggestions: {ultra_specific_count}/{len(results)} pages") return results, filename except Exception as e: print(f"❌ Unexpected error during SEO analysis: {e}") # Create error report os.makedirs(tmp_dir, exist_ok=True) filename = os.path.join(tmp_dir, f"seo_report_error_{uuid.uuid4().hex}.csv") error_data = [{ "url": base_url, "error": f"Unexpected error: {str(e)}", "seo_suggestions": "Please check the website URL and try again." }] with open(filename, "w", newline="", encoding="utf-8") as f: writer = csv.DictWriter(f, fieldnames=error_data[0].keys()) writer.writeheader() writer.writerows(error_data) return error_data, filename # ============================== # ASYNC WRAPPER FOR FASTAPI # ============================== async def run_seo_analysis_fastapi(base_url, max_pages=5, use_ai=True, max_concurrent=2, download=False): """ Async wrapper for FastAPI compatibility """ try: results, csv_path = await run_seo_and_suggestions_async( base_url=base_url, max_pages=max_pages, tmp_dir="/tmp", use_ai=use_ai, max_concurrent=max_concurrent ) return results, csv_path except Exception as e: filename = f"/tmp/seo_report_error_{uuid.uuid4().hex}.csv" error_data = [{ "url": base_url, "error": f"Analysis failed: {str(e)}", "seo_suggestions": "Please check the website URL and try again." }] with open(filename, "w", newline="", encoding="utf-8") as f: writer = csv.DictWriter(f, fieldnames=error_data[0].keys()) writer.writeheader() writer.writerows(error_data) return error_data, filename