Spaces:
Sleeping
Sleeping
| import pandas as pd | |
| import requests | |
| import time | |
| from bs4 import BeautifulSoup | |
| from selenium import webdriver | |
| from selenium.webdriver.chrome.options import Options | |
| from selenium.webdriver.chrome.service import Service | |
| # ====================================== | |
| # CREATE DRIVER | |
| # ====================================== | |
| def create_driver(): | |
| chrome_options = Options() | |
| chrome_options.add_argument("--headless") | |
| chrome_options.add_argument("--no-sandbox") | |
| chrome_options.add_argument("--disable-dev-shm-usage") | |
| chrome_options.add_argument("--disable-gpu") | |
| chrome_options.binary_location = "/usr/bin/chromium" | |
| service = Service("/usr/bin/chromedriver") | |
| driver = webdriver.Chrome( | |
| service=service, | |
| options=chrome_options | |
| ) | |
| return driver | |
| # ====================================== | |
| # STATIC SCRAPER | |
| # ====================================== | |
| def extract_static(url): | |
| try: | |
| headers = { | |
| "User-Agent": ( | |
| "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " | |
| "AppleWebKit/537.36 (KHTML, like Gecko) " | |
| "Chrome/124.0.0.0 Safari/537.36" | |
| ) | |
| } | |
| response = requests.get( | |
| url, | |
| headers=headers, | |
| timeout=15 | |
| ) | |
| soup = BeautifulSoup( | |
| response.text, | |
| "html.parser" | |
| ) | |
| headline = "" | |
| if soup.title: | |
| headline = soup.title.text.strip() | |
| paragraphs = [] | |
| for p in soup.find_all("p"): | |
| text = p.get_text(" ", strip=True) | |
| if len(text) > 40: | |
| paragraphs.append(text) | |
| article = "\n".join(paragraphs) | |
| if len(article) > 100: | |
| return headline, article | |
| return None, None | |
| except: | |
| return None, None | |
| # ====================================== | |
| # SELENIUM SCRAPER | |
| # ====================================== | |
| def extract_selenium(driver, url): | |
| try: | |
| driver.get(url) | |
| time.sleep(3) | |
| soup = BeautifulSoup( | |
| driver.page_source, | |
| "html.parser" | |
| ) | |
| headline = "" | |
| h1 = soup.find("h1") | |
| if h1: | |
| headline = h1.text.strip() | |
| paragraphs = [] | |
| for p in soup.find_all("p"): | |
| text = p.get_text(" ", strip=True) | |
| if len(text) > 40: | |
| paragraphs.append(text) | |
| article = "\n".join(paragraphs) | |
| if len(article) > 100: | |
| return headline, article | |
| return None, None | |
| except: | |
| return None, None | |
| # ====================================== | |
| # MAIN SCRAPER | |
| # ====================================== | |
| def scrape_news(file_path): | |
| df = pd.read_excel(file_path) | |
| links = df["Link"].dropna().tolist() | |
| driver = create_driver() | |
| output_file = "scraped_news.txt" | |
| with open(output_file, "w", encoding="utf-8") as f: | |
| for idx, link in enumerate(links, start=1): | |
| headline, article = extract_static(link) | |
| if not headline or not article: | |
| headline, article = extract_selenium( | |
| driver, | |
| link | |
| ) | |
| if headline and article: | |
| f.write("\n") | |
| f.write("=" * 100) | |
| f.write("\n") | |
| f.write(f"ARTICLE {idx}\n") | |
| f.write("=" * 100) | |
| f.write("\n\n") | |
| f.write(f"TITLE:\n{headline}\n\n") | |
| f.write(f"URL:\n{link}\n\n") | |
| f.write("CONTENT:\n\n") | |
| f.write(article) | |
| f.write("\n\n") | |
| driver.quit() | |
| return output_file |