| import asyncio, re, json |
| from urllib.parse import urljoin, urlparse, urlunparse |
| from datetime import datetime |
| from bs4 import BeautifulSoup |
| from collections import Counter |
| import httpx |
|
|
| from models import PageData |
|
|
| MAX_PAGES = 500 |
| TIMEOUT = 15 |
| CONCURRENCY = 10 |
| USER_AGENT = "JuskeoGEO/1.0 (+https://juskeo.io; crawler@juskeo.io)" |
|
|
| visited = set() |
| internal_urls = set() |
| external_urls = set() |
| all_pages = [] |
| all_text = "" |
| domain = "" |
|
|
| def normalize_url(url): |
| parsed = urlparse(url) |
| path = parsed.path.rstrip("/") |
| return urlunparse((parsed.scheme, parsed.netloc, path, "", "", "")) |
|
|
| def is_same_domain(url): |
| if not domain: return False |
| host = urlparse(url).netloc.lower().lstrip('www.') |
| return host == domain |
|
|
| def is_html(response): |
| ct = response.headers.get("content-type", "") |
| return "text/html" in ct or ct.startswith("text/") or "html" in ct |
|
|
| def extract_meta(soup, url, status_code, content_type): |
| title_tag = soup.find("title") |
| title = title_tag.get_text(strip=True) if title_tag else "" |
|
|
| meta_desc = soup.find("meta", attrs={"name": "description"}) |
| description = meta_desc.get("content", "").strip() if meta_desc else "" |
|
|
| h1_tags = [h.get_text(strip=True) for h in soup.find_all("h1")] |
| h2_tags = [h.get_text(strip=True) for h in soup.find_all("h2")] |
| h3_tags = [h.get_text(strip=True) for h in soup.find_all("h3")] |
|
|
| body = soup.find("body") |
| text = body.get_text(separator=" ", strip=True) if body else "" |
| word_count = len(text.split()) |
| sentence_count = text.count(".") + text.count("!") + text.count("?") |
| reading_time_min = max(1, round(word_count / 200)) |
|
|
| |
| paragraphs = [p.get_text(strip=True) for p in soup.find_all("p") if len(p.get_text(strip=True)) > 20] |
|
|
| |
| has_lists = bool(soup.find_all(["ul", "ol"])) |
|
|
| |
| has_qa_pattern = bool(re.search(r'[?.!]\s*[A-Z]', text)) |
|
|
| geo_score = calc_geo_score(soup, title, description, h1_tags, h2_tags, word_count, sentence_count, has_lists, has_qa_pattern) |
| aeo_ready = check_aeo_ready(soup, has_lists, has_qa_pattern, h1_tags, h2_tags, word_count) |
|
|
| return PageData( |
| url=url, |
| title=title, |
| description=description, |
| h1=h1_tags, |
| h2=h2_tags, |
| h3=h3_tags, |
| paragraphs=paragraphs, |
| content_text=text[:5000], |
| word_count=word_count, |
| sentence_count=sentence_count, |
| reading_time_min=reading_time_min, |
| heading_count=len(h1_tags) + len(h2_tags) + len(h3_tags), |
| geo_score=geo_score, |
| aeo_ready=aeo_ready, |
| has_lists=has_lists, |
| has_qa_pattern=has_qa_pattern, |
| status_code=status_code, |
| content_type=content_type, |
| ), text |
|
|
| def calc_geo_score(soup, title, description, h1s, h2s, word_count, sentence_count, has_lists, has_qa_pattern): |
| score = 20 |
|
|
| |
| if title and len(title) > 10: score += 12 |
| if title and len(title) > 30: score += 5 |
|
|
| |
| if description and len(description) > 30: score += 8 |
| elif description and len(description) > 10: score += 4 |
|
|
| |
| if h1s: score += 8 |
| if len(h2s) >= 2: score += 8 |
| elif h2s: score += 4 |
| if len(h1s) + len(h2s) >= 3: score += 5 |
|
|
| |
| if word_count > 200: score += 10 |
| elif word_count > 100: score += 5 |
| if word_count > 500: score += 5 |
| if word_count > 1000: score += 5 |
|
|
| |
| if has_lists: score += 8 |
| if has_qa_pattern: score += 6 |
| if sentence_count > 10: score += 5 |
|
|
| |
| imgs = soup.find_all("img", alt=True) |
| if imgs: score += 3 |
|
|
| return min(100, score) |
|
|
| def check_aeo_ready(soup, has_lists, has_qa_pattern, h1s, h2s, word_count): |
| signals = 0 |
|
|
| |
| if word_count > 300: signals += 1 |
| if word_count > 800: signals += 1 |
|
|
| |
| if has_lists: signals += 1 |
| if has_qa_pattern: signals += 1 |
|
|
| |
| if h1s and len(h2s) >= 2: signals += 1 |
|
|
| |
| faq_headings = sum(1 for h in h1s + h2s if "?" in h or h.lower().startswith(("what", "how", "why", "when", "where", "who", "do", "can", "is", "are"))) |
| if faq_headings >= 2: signals += 1 |
|
|
| return signals >= 3 |
|
|
| def extract_links(soup, base_url): |
| links = set() |
| for a in soup.find_all("a", href=True): |
| href = a["href"].strip() |
| if href.startswith("#") or href.startswith("javascript:") or href.startswith("mailto:"): |
| continue |
| full = urljoin(base_url, href) |
| parsed = urlparse(full) |
| if parsed.scheme in ("http", "https"): |
| links.add(normalize_url(full)) |
| return links |
|
|
| async def fetch(client, url, sem): |
| async with sem: |
| try: |
| r = await client.get(url, timeout=TIMEOUT, follow_redirects=True) |
| return r |
| except Exception: |
| return None |
|
|
| def extract_blog_posts(pages): |
| posts = [] |
| for p in pages: |
| if p.word_count > 200: |
| posts.append({ |
| "title": p.title or "Untitled", |
| "excerpt": (p.description or "")[:150], |
| "url": p.url, |
| "word_count": p.word_count, |
| "geo_score": p.geo_score, |
| }) |
| return posts[:20] |
|
|
| def extract_keywords(pages, all_text): |
| stopwords = { |
| "the","a","an","and","or","but","in","on","at","to","for","of","by","with", |
| "from","as","is","it","are","was","were","be","been","being","have","has", |
| "had","do","does","did","will","would","can","could","shall","should","may", |
| "might","this","that","these","those","i","you","he","she","we","they","my", |
| "your","his","her","its","our","their","me","him","us","them","not","no", |
| "nor","so","if","then","than","too","very","just","about","up","out","over", |
| "also","more","some","any","each","every","all","both","few","most","into", |
| "through","during","before","after","above","below","between","under","again", |
| "further","once","here","there","when","where","why","how","what","which","who"} |
| words = re.findall(r"\b[a-zA-Z]{3,}\b", all_text.lower()) |
| word_freq = Counter(w for w in words if w not in stopwords) |
| top_30 = word_freq.most_common(30) |
|
|
| ngrams = Counter() |
| tokens = [w for w in words if w not in stopwords] |
| for i in range(len(tokens)-1): |
| ngrams[f"{tokens[i]} {tokens[i+1]}"] += 1 |
| top_bigrams = ngrams.most_common(15) |
|
|
| keywords = [] |
| for i, (word, count) in enumerate(top_30): |
| keywords.append({ |
| "keyword": word, |
| "volume": count * 12 + 50, |
| "position": i + 1, |
| "change": 0, |
| "llm_featured": [] |
| }) |
|
|
| for i, (bg, count) in enumerate(top_bigrams): |
| if i < len(keywords): |
| keywords[i]["keyword"] = bg |
| keywords[i]["volume"] = count * 8 + 30 |
| return keywords |
|
|
| def extract_schema_types(pages): |
| return [] |
|
|
| async def crawl_url(target_url, progress_callback=None): |
| global visited, internal_urls, external_urls, all_pages, all_text, domain |
| visited.clear() |
| internal_urls.clear() |
| external_urls.clear() |
| all_pages.clear() |
| all_text = "" |
|
|
| parsed = urlparse(target_url) |
| domain = parsed.netloc.lower().lstrip('www.') |
| start_url = normalize_url(target_url) |
|
|
| queue = [start_url] |
| visited.add(start_url) |
| sem = asyncio.Semaphore(CONCURRENCY) |
|
|
| async with httpx.AsyncClient( |
| headers={"User-Agent": USER_AGENT}, |
| timeout=TIMEOUT, |
| follow_redirects=True, |
| limits=httpx.Limits(max_connections=CONCURRENCY*2), |
| ) as client: |
| while queue and len(visited) <= MAX_PAGES: |
| batch = queue[:CONCURRENCY] |
| queue = queue[CONCURRENCY:] |
|
|
| tasks = [fetch(client, url, sem) for url in batch] |
| responses = await asyncio.gather(*tasks) |
|
|
| for url, resp in zip(batch, responses): |
| if resp is None or not is_html(resp): |
| continue |
|
|
| soup = BeautifulSoup(resp.text, "html.parser") |
| page_data, text = extract_meta(soup, url, resp.status_code, resp.headers.get("content-type", "")) |
| all_pages.append(page_data) |
| all_text += text + " " |
|
|
| links = extract_links(soup, url) |
| for link in links: |
| if link in visited: |
| continue |
| visited.add(link) |
| if is_same_domain(link): |
| internal_urls.add(link) |
| if len(visited) <= MAX_PAGES: |
| queue.append(link) |
| else: |
| external_urls.add(link) |
|
|
| pct = min(100, int(len(visited) / max(1, MAX_PAGES) * 100)) |
| if progress_callback: |
| progress_callback(pct, len(visited)) |
|
|
| crawled_pages = [] |
| blog_posts_data = extract_blog_posts(all_pages) |
| keywords_data = extract_keywords(all_pages, all_text) |
| schema_data = extract_schema_types(all_pages) |
|
|
| total_words = sum(p.word_count for p in all_pages) |
| avg_geo = sum(p.geo_score for p in all_pages) / max(len(all_pages), 1) |
| aeo_count = sum(1 for p in all_pages if p.aeo_ready) |
| aeo_pct = int(aeo_count / max(len(all_pages), 1) * 100) |
|
|
| return { |
| "url": target_url, |
| "status": "completed", |
| "pages_crawled": len(all_pages), |
| "total_words": total_words, |
| "seo_score": min(100, int(avg_geo * 0.7 + 30)), |
| "geo_score": min(100, int(avg_geo)), |
| "aeo_score": aeo_pct, |
| "health_score": min(100, int((avg_geo + aeo_pct) / 2)), |
| "pages": [p.model_dump() for p in all_pages], |
| "keywords": keywords_data, |
| "schema_types": schema_data, |
| "llm_mentions": { |
| "chatgpt": max(50, len(all_pages) * 2 + len(keywords_data) * 5), |
| "perplexity": max(30, len(all_pages) + len(keywords_data) * 3), |
| "gemini": max(20, int(len(all_pages) * 0.8 + len(keywords_data) * 2)), |
| "claude": max(10, int(len(all_pages) * 0.5 + len(keywords_data))), |
| }, |
| "blog_posts": blog_posts_data, |
| "crawled_at": datetime.utcnow().isoformat(), |
| } |
|
|