Spaces:
Sleeping
Sleeping
| import httpx | |
| import asyncio | |
| from bs4 import BeautifulSoup | |
| from django.utils import timezone | |
| import random | |
| async def fetch_product_details(client, link, basic_data): | |
| try: | |
| clean_path = link.replace('../../../', '') | |
| if 'catalogue/' not in clean_path: | |
| clean_path = f"catalogue/{clean_path}" | |
| full_url = f"http://books.toscrape.com/{clean_path}" | |
| response = await client.get(full_url, timeout=10.0) | |
| if response.status_code == 200: | |
| prod_soup = BeautifulSoup(response.content, 'html.parser') | |
| table = prod_soup.find('table', class_='table-striped') | |
| if table: | |
| rows = table.find_all('tr') | |
| for row in rows: | |
| header = row.find('th').get_text(strip=True) | |
| value = row.find('td').get_text(strip=True) | |
| if header == 'UPC': | |
| basic_data['sku'] = value | |
| return basic_data | |
| except Exception: | |
| return None | |
| return None | |
| async def scrape_products(b_url): | |
| b_url = b_url.rstrip('/') | |
| try: | |
| random_page = random.randint(1, 50) | |
| url = f"{b_url}/catalogue/page-{random_page}.html" | |
| headers = {'User-Agent': 'Mozilla/5.0'} | |
| async with httpx.AsyncClient(headers=headers, follow_redirects=True) as client: | |
| response = await client.get(url, timeout=15.0) | |
| soup = BeautifulSoup(response.content, 'html.parser') | |
| products_elements = soup.find_all('article', class_='product_pod') | |
| tasks = [] | |
| for item in products_elements: | |
| title = item.h3.a['title'] | |
| relative_link = item.h3.a['href'] | |
| price_text = item.find('p', class_='price_color').text | |
| price = ''.join(c for c in price_text if c.isdigit() or c == '.') | |
| stock_text = item.find('p', class_='instock availability').text | |
| basic_data = { | |
| 'title': title, | |
| 'price': float(price) if price else 0.0, | |
| 'stock_status': "In stock" in stock_text, | |
| 'last_scraped_at': timezone.now() | |
| } | |
| tasks.append(fetch_product_details(client, relative_link, basic_data)) | |
| scraped_data = await asyncio.gather(*tasks) | |
| return [data for data in scraped_data if data and data.get('sku')] | |
| except Exception as e: | |
| print(f"Scraping Error: {e}") | |
| return [] |