Spaces:
Sleeping
Sleeping
File size: 6,514 Bytes
3632ab3 0fd1259 edca2c5 dbe9781 edca2c5 330965a edca2c5 330965a 9804413 edca2c5 0fd1259 06aecf1 0fd1259 d6c0b32 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 0fd1259 d6c0b32 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 0fd1259 d6c0b32 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 0fd1259 edca2c5 0fd1259 d6c0b32 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 edca2c5 06aecf1 0fd1259 edca2c5 0fd1259 edca2c5 3632ab3 0fd1259 3632ab3 edca2c5 3632ab3 edca2c5 3632ab3 0fd1259 d6c0b32 1d04cea ba01da7 edca2c5 ba01da7 edca2c5 ba01da7 edca2c5 ba01da7 edca2c5 ba01da7 edca2c5 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 | import time
import requests
from bs4 import BeautifulSoup
from urllib.parse import quote_plus
# Global session with headers
session = requests.Session()
session.headers.update({
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/114.0.0.0 Safari/537.36",
"Accept-Language": "en-US,en;q=0.9"
})
def fetch_soup(url):
try:
resp = session.get(url, timeout=10)
resp.raise_for_status()
time.sleep(1) # polite delay
return BeautifulSoup(resp.text, "html.parser")
except requests.exceptions.HTTPError as e:
print(f"[ERROR] HTTP error while fetching: {url}\n{e}")
return BeautifulSoup("", "html.parser")
except Exception as e:
print(f"[ERROR] General error while fetching: {url}\n{e}")
return BeautifulSoup("", "html.parser")
def scrape_amazon(query):
print(f"Scraping Amazon for query: {query}")
url = f"https://www.amazon.in/s?k={quote_plus(query)}"
soup = fetch_soup(url)
products = []
for item in soup.select("div[data-component-type='s-search-result']"):
name_el = item.select_one("h2 span")
price_el = item.select_one("span.a-price-whole")
img_el = item.select_one("img.s-image")
link_el = item.select_one("a")
products.append({
"name": name_el.get_text(strip=True) if name_el else "N/A",
"price": price_el.get_text(strip=True) if price_el else "N/A",
"img": img_el["src"] if img_el else "",
"link": "https://www.amazon.in" + link_el["href"] if link_el else ""
})
return products
def scrape_1mg(query):
print(f"Scraping 1mg for query: {query}")
url = f"https://www.1mg.com/search/all?name={quote_plus(query)}"
soup = fetch_soup(url)
products = []
for product in soup.select("div.style__horizontal-card___1Zwmt"):
name_el = product.select_one("div.style__product-description___1vPQe")
price_el = product.select_one("div.style__price-tag___B2csA")
link_el = product.select_one("a")
img_el = product.select_one("img")
products.append({
"name": name_el.get_text(strip=True) if name_el else "N/A",
"price": price_el.get_text(strip=True) if price_el else "N/A",
"img": img_el["src"] if img_el else "",
"link": "https://www.1mg.com" + link_el["href"] if link_el else ""
})
return products
def scrape_netmeds(query):
print(f"Scraping Netmeds for query: {query}")
url = f"https://www.netmeds.com/catalogsearch/result/{quote_plus(query)}/all"
soup = fetch_soup(url)
products = []
for product in soup.select("div.cat-item"):
name_el = product.select_one("h3")
price_el = product.select_one("span.final-price")
link_el = product.select_one("a")
img_el = product.select_one("img")
products.append({
"name": name_el.get_text(strip=True) if name_el else "N/A",
"price": price_el.get_text(strip=True) if price_el else "N/A",
"img": img_el["src"] if img_el else "",
"link": "https://www.netmeds.com" + link_el["href"] if link_el else ""
})
return products
def scrape_pharmeasy(query):
print(f"Scraping PharmEasy for query: {query}")
url = f"https://pharmeasy.in/search/all?name={quote_plus(query)}"
soup = fetch_soup(url)
products = []
for product in soup.select("div.ProductCard_medicineUnitContainer__m2_zO"):
name_el = product.select_one("h1")
price_el = product.select_one("div.ProductCard_ourPrice__yU5GB")
link_el = product.select_one("a")
img_el = product.select_one("img")
products.append({
"name": name_el.get_text(strip=True) if name_el else "N/A",
"price": price_el.get_text(strip=True) if price_el else "N/A",
"img": img_el["src"] if img_el else "",
"link": "https://pharmeasy.in" + link_el["href"] if link_el else ""
})
return products
def search_product(query):
amazon_products = [{"source": "Amazon", **p} for p in scrape_amazon(query)]
one_mg_products = [{"source": "1mg", **p} for p in scrape_1mg(query)]
netmeds_products = [{"source": "Netmeds", **p} for p in scrape_netmeds(query)]
pharmeasy_products = [{"source": "PharmEasy", **p} for p in scrape_pharmeasy(query)]
combined = []
max_len = max(len(amazon_products), len(one_mg_products), len(netmeds_products), len(pharmeasy_products))
for i in range(max_len):
combined.append({
"name": (
amazon_products[i]["name"] if i < len(amazon_products) else
one_mg_products[i]["name"] if i < len(one_mg_products) else
netmeds_products[i]["name"] if i < len(netmeds_products) else
pharmeasy_products[i]["name"] if i < len(pharmeasy_products) else ""
),
"img": (
amazon_products[i]["img"] if i < len(amazon_products) else
one_mg_products[i]["img"] if i < len(one_mg_products) else
netmeds_products[i]["img"] if i < len(netmeds_products) else
pharmeasy_products[i]["img"] if i < len(pharmeasy_products) else ""
),
"amazon": amazon_products[i] if i < len(amazon_products) else {},
"one_mg": one_mg_products[i] if i < len(one_mg_products) else {},
"netmeds": netmeds_products[i] if i < len(netmeds_products) else {},
"pharmeasy": pharmeasy_products[i] if i < len(pharmeasy_products) else {},
})
return combined
def search_product_with_progress(query):
combined = search_product(query)
# Fetch side effects using OpenFDA API
api_url = f'https://api.fda.gov/drug/event.json?api_key=ah8n0nI6468qpS6cMfqvBM0sdGs4nO1sT3Ie7LjJ&search=patient.drug.medicinalproduct:"{query}"'
try:
response = requests.get(api_url)
data = response.json()
if data.get('results'):
side_effects = [
reaction.get('reactionmeddrapt', 'N/A')
for reaction in data['results'][0].get('patient', {}).get('reaction', [])
]
else:
side_effects = ["No side effects found."]
except Exception as e:
print(f"Error fetching side effects: {e}")
side_effects = ["Error fetching side effects."]
return combined, side_effects
|