finsight-ai / web_scraping_and_preprocessing.py
mustdo's picture
feat: complete Docker and FastAPI deployment configuration
dc2aa6c
Raw
History Blame Contribute Delete
4.17 kB
# -*- coding: utf-8 -*-
"""Web Scraping and Preprocessing.ipynb
Automatically generated by Colab.
Original file is located at
https://colab.research.google.com/drive/1AHby4fTck0P3S3AKBi_Mw7rmeRN1fGRi
"""
!pip install requests beautifulsoup4 pandas
import requests
from bs4 import BeautifulSoup
import pandas as pd
YAHOO_URL = "https://finance.yahoo.com/news/"
BUSINESS_INSIDER_URL = "https://www.businessinsider.com/finance"
def scrape_yahoonews():
response = requests.get(YAHOO_URL, headers={"User-Agent": "Mozilla/5.0"})
if response.status_code != 200:
print("Failed to fetch news")
return []
soup = BeautifulSoup(response.text, "html.parser")
news_data = []
for article in soup.select("a:has(h3)"):
headline_tag = article.find("h3")
if not headline_tag:
continue
headline = headline_tag.text.strip()
link = article["href"]
if not link.startswith("https"):
link = "https://finance.yahoo.com" + link
article_response = requests.get(link, headers={"User-Agent": "Mozilla/5.0"})
article_soup = BeautifulSoup(article_response.text, "html.parser")
paragraphs = article_soup.find_all("p")
content = " ".join([p.text for p in paragraphs])
news_data.append({"headline": headline, "content": content, "link": link})
return news_data
yahoo_news = scrape_yahoonews()
df_yahoo = pd.DataFrame(yahoo_news)
df_yahoo.head()
def clean_yahoo_text(text):
text = text.replace("Oops, something went wrong", "").strip()
return text
df_yahoo["content"] = df_yahoo["content"].apply(clean_yahoo_text)
df_yahoo.to_csv("scraped_yahoo_data.csv", index=False)
def scrape_businessinsider():
headers = {
"User-Agent": "Mozilla/5.0"
}
response = requests.get(BUSINESS_INSIDER_URL, headers=headers)
if response.status_code != 200:
print("Failed to fetch Business Insider news.")
return []
soup = BeautifulSoup(response.text, "html.parser")
news_data = []
for link in soup.find_all("a", href=True):
article_url = link["href"]
title = link.get_text(strip=True)
if not title or "subscribe" in title.lower() or "login" in title.lower() or "/video" in article_url or "/photo" in article_url:
continue
if not article_url.startswith("https"):
article_url = "https://www.businessinsider.com" + article_url
try:
article_response = requests.get(article_url, headers=headers)
article_soup = BeautifulSoup(article_response.text, "html.parser")
paragraphs = article_soup.find_all("p")
content = " ".join([p.get_text(strip=True) for p in paragraphs if len(p.get_text(strip=True)) > 30])
if content:
news_data.append({"headline": title, "content": content, "link": article_url})
except Exception as e:
print(f"Error fetching article {article_url}: {e}")
return news_data
insider_news = scrape_businessinsider()
df_insider = pd.DataFrame(insider_news)
df_insider.head()
df_insider.to_csv("scraped_business_insider_news.csv", index=False)
df_combined = pd.concat([df_yahoo, df_insider], ignore_index=True)
df_combined.to_csv("financial_news_combined.csv", index=False)
import pandas as pd
import re
import nltk
from nltk.corpus import stopwords
from nltk.tokenize import word_tokenize
from nltk.stem import WordNetLemmatizer
nltk.download("stopwords")
nltk.download("punkt")
nltk.download("wordnet")
nltk.download("punkt_tab")
df = pd.read_csv("financial_news_combined.csv")
lemmatizer = WordNetLemmatizer()
def text_preprocessing(text):
if pd.isna(text):
return ""
text = text.lower()
text = re.sub(r"http\S+|www\S+", "", text)
text = re.sub(r"[^a-zA-Z\s]", "", text)
words = word_tokenize(text)
words = [lemmatizer.lemmatize(word) for word in words if word not in stopwords.words("english")]
return " ".join(words)
df["preprocessed_content"] = df["content"].astype(str).apply(text_preprocessing)
df.head()
df.to_csv("financial_news_preprocessed.csv", index=False)