| import requests |
| from bs4 import BeautifulSoup |
| from fpdf import FPDF |
| from huggingface_hub import HfApi, upload_file |
| import os |
|
|
| HF_TOKEN = os.environ.get("HF_TOKEN", "") |
| DATASET_REPO = "sosa123454321/Notary-PDF-Dataset" |
|
|
| def scrape_article_to_pdf(url, output_name): |
| print(f"Scraping {url}...") |
| try: |
| response = requests.get(url) |
| soup = BeautifulSoup(response.content, 'html.parser') |
| |
| |
| for script in soup(["script", "style"]): |
| script.decompose() |
| |
| text = soup.get_text() |
| lines = (line.strip() for line in text.splitlines()) |
| chunks = (phrase.strip() for line in lines for phrase in line.split(" ")) |
| clean_text = '\n'.join(chunk for chunk in chunks if chunk) |
|
|
| |
| pdf = FPDF() |
| pdf.add_page() |
| pdf.set_font("Arial", size=10) |
| |
| |
| pdf.multi_cell(0, 10, txt=clean_text.encode('latin-1', 'replace').decode('latin-1')) |
| |
| pdf_path = f"{output_name}.pdf" |
| pdf.output(pdf_path) |
| print(f"Saved to {pdf_path}") |
| |
| |
| print(f"Uploading to {DATASET_REPO}...") |
| upload_file( |
| path_or_fileobj=pdf_path, |
| path_in_repo=f"documents/{pdf_path}", |
| repo_id=DATASET_REPO, |
| repo_type="dataset", |
| token=HF_TOKEN |
| ) |
| print("Upload successful!") |
| return pdf_path |
| except Exception as e: |
| print(f"Error: {e}") |
| return None |
|
|
| if __name__ == "__main__": |
| |
| test_url = "https://www.notary662th.ir/induction-manual" |
| scrape_article_to_pdf(test_url, "notary_induction_persian") |
|
|