trend-summarizer / src /scraper /scraper.py
adnandi's picture
Deploy
7136657
Raw
History Blame Contribute Delete
8.36 kB
import os
import json
import logging
import urllib.parse
from datetime import datetime
import feedparser
from newspaper import Article, Config
from googlenewsdecoder import gnewsdecoder
# Setup logging
logging.basicConfig(
level=logging.INFO,
format='%(asctime)s - %(levelname)s - %(message)s'
)
logger = logging.getLogger(__name__)
# Scraper configuration
CONFIG = Config()
CONFIG.browser_user_agent = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/115.0.0.0 Safari/537.36"
)
CONFIG.request_timeout = 15
def get_source_name(entry, fallback_url: str) -> str:
"""
Get the name of the source from feed entry or parse domain as fallback.
"""
if hasattr(entry, 'source') and 'title' in entry.source:
return entry.source.title
try:
parsed_uri = urllib.parse.urlparse(fallback_url)
return parsed_uri.netloc.replace('www.', '')
except Exception:
return "Unknown Source"
def search_articles(query: str, limit: int = 10, lang: str = "id", country: str = "ID", time_range: str = "any") -> list:
"""
Search articles from Google News RSS based on a query, localized to lang/country.
Returns a list of dicts containing metadata (title, link, published, source).
"""
adjusted_query = query
if time_range and time_range != "any":
adjusted_query = f"{query} when:{time_range}"
encoded_query = urllib.parse.quote(adjusted_query)
ceid = f"{country}:{lang}"
rss_url = f"https://news.google.com/rss/search?q={encoded_query}&hl={lang}&gl={country}&ceid={ceid}"
logger.info(f"Searching Google News RSS for query: '{adjusted_query}' (lang: {lang}, country: {country})...")
feed = feedparser.parse(rss_url)
articles = []
for entry in feed.entries[:limit]:
title = getattr(entry, 'title', 'No Title')
link = getattr(entry, 'link', '')
published = getattr(entry, 'published', '')
source = get_source_name(entry, link)
if link:
articles.append({
"title": title,
"url": link,
"publish_date": published,
"source": source,
"query": query
})
logger.info(f"Found {len(articles)} article links for query: '{query}'.")
return articles
def scrape_article(url: str, lang: str = "id") -> str:
"""
Scrapes the full text of an article using Newspaper3k with a specific language setting.
"""
logger.info(f"Scraping content from URL: {url} (lang: {lang})")
try:
# Create a dynamic config to pass the specific language
config = Config()
config.browser_user_agent = CONFIG.browser_user_agent
config.request_timeout = CONFIG.request_timeout
config.language = lang # e.g., 'id' for Indonesian
article = Article(url, config=config)
article.download()
article.parse()
return article.text
except Exception as e:
logger.warning(f"Failed to scrape article from {url}: {e}")
return ""
def run_auto_pipeline(queries: list, limit_per_query: int = 5, output_dir: str = "data", lang: str = "id", country: str = "ID", time_range: str = "any", log_callback = None) -> str:
"""
Runs the automated search and scrape pipeline for a list of queries.
Saves the results to raw_articles.json.
"""
os.makedirs(output_dir, exist_ok=True)
output_path = os.path.join(output_dir, "raw_articles.json")
if log_callback:
log_callback(f"Memulai pipeline penarikan berita untuk {len(queries)} kata kunci.")
log_callback(f"Rentang waktu terbit berita: {time_range if time_range != 'any' else 'Semua Waktu'}")
# Load existing articles if file exists to prevent overwriting new runs
all_articles = []
if os.path.exists(output_path):
try:
with open(output_path, 'r', encoding='utf-8') as f:
all_articles = json.load(f)
logger.info(f"Loaded {len(all_articles)} existing articles from {output_path}")
if log_callback:
log_callback(f"Ditemukan {len(all_articles)} artikel pra-eksis di arsip lokal.")
except Exception as e:
logger.error(f"Error loading existing raw_articles.json: {e}")
existing_urls = {art["url"] for art in all_articles if "url" in art}
scraped_count = 0
for query in queries:
if log_callback:
log_callback(f"Mencari query: '{query}' di Google News...")
discovered_articles = search_articles(query, limit=limit_per_query, lang=lang, country=country, time_range=time_range)
if log_callback:
log_callback(f"Menemukan {len(discovered_articles)} artikel potensial untuk query: '{query}'")
for art in discovered_articles:
google_url = art["url"]
# Decode the Google News redirect URL to the direct article URL
decoded_url = google_url
try:
if log_callback:
log_callback(f"Mendekode rujukan URL Google News...")
decoded_data = gnewsdecoder(google_url)
if decoded_data.get("status"):
decoded_url = decoded_data["decoded_url"]
logger.info(f"Decoded Google News URL to: {decoded_url}")
else:
logger.warning(f"Could not decode Google News URL, using original: {google_url}")
except Exception as e:
logger.error(f"Error decoding Google News URL: {e}")
art["url"] = decoded_url
if decoded_url in existing_urls:
logger.info(f"Skipping already scraped URL: {decoded_url}")
if log_callback:
log_callback(f"Melewati URL (sudah pernah diunduh): {decoded_url[:60]}...")
continue
if log_callback:
log_callback(f"Mengunduh teks lengkap dari: {art['source']} ({decoded_url[:50]}...)")
raw_text = scrape_article(decoded_url, lang=lang)
# Only save articles that have non-empty content
if raw_text and len(raw_text.strip()) > 150:
art["raw_text"] = raw_text.strip()
art["scraped_at"] = datetime.now().isoformat()
all_articles.append(art)
existing_urls.add(decoded_url)
scraped_count += 1
if log_callback:
log_callback(f"✓ Berhasil mengunduh artikel: \"{art['title'][:45]}...\"")
else:
logger.info(f"Skipped URL due to empty or too short content: {decoded_url}")
if log_callback:
log_callback(f"⚠ Skip artikel (konten kosong/terlalu pendek): {decoded_url[:50]}...")
# Assign sequential IDs to all articles before saving
for idx, art in enumerate(all_articles, 1):
ordered_art = {"id": idx}
ordered_art.update({k: v for k, v in art.items() if k != "id"})
all_articles[idx - 1] = ordered_art
# Save back to JSON
try:
with open(output_path, 'w', encoding='utf-8') as f:
json.dump(all_articles, f, indent=4, ensure_ascii=False)
logger.info(f"Pipeline finished. Successfully scraped {scraped_count} new articles. Total saved: {len(all_articles)} articles.")
if log_callback:
log_callback(f"Selesai! {scraped_count} berita baru ditambahkan. Total database proyek memiliki {len(all_articles)} berita.")
except Exception as e:
logger.error(f"Failed to write results to {output_path}: {e}")
if log_callback:
log_callback(f"❌ Gagal menulis hasil ke file lokal: {e}")
return output_path
if __name__ == "__main__":
# Test queries focused on Indonesian industry trends
test_queries = [
"FMCG keberlanjutan Indonesia",
"pemasaran digital ritel Indonesia",
"perubahan perilaku konsumen ritel"
]
print("Starting pipeline test focused on Indonesian news...")
path = run_auto_pipeline(test_queries, limit_per_query=4, lang="id", country="ID")
print(f"Test completed. Output path: {path}")