| |
| """ |
| scrape_v3.py โ Fixed DBE corpus scraper |
| Key fix: education.gov.za uses LinkClick.aspx?fileticket= for ALL downloads. |
| Now also includes NSC past exam papers (2008-2025) โ the largest Sepedi source. |
| |
| Usage: |
| cd ~/leotsha_project |
| python3 scrape_v3.py |
| |
| Output: ~/leotsha_project/corpus/ |
| """ |
|
|
| import os, re, csv, time, requests |
| from urllib.parse import urljoin, urlparse, parse_qs, urlencode, urlunparse |
|
|
| BASE_DIR = os.path.expanduser("~/leotsha_project/corpus") |
|
|
| HEADERS = { |
| "User-Agent": "Mozilla/5.0 (X11; Linux x86_64; rv:120.0) Gecko/20100101 Firefox/120.0", |
| "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", |
| "Accept-Language": "en-ZA,en;q=0.9", |
| } |
|
|
| |
| SEED_URLS = [ |
| |
| ("https://www.education.gov.za/Curriculum/LearningandTeachingSupportMaterials(LTSM)/MindtheGapStudyGuides.aspx", "eduintel/mind_the_gap"), |
| ("https://www.education.gov.za/Curriculum/LearningandTeachingSupportMaterials(LTSM)/Workbooks.aspx", "eduintel/workbooks"), |
| ("https://www.education.gov.za/Curriculum/LearningandTeachingSupportMaterials(LTSM)/GradedReadersandBigBookHL.aspx","eduintel/graded_readers"), |
| ("https://www.education.gov.za/Curriculum/LearningandTeachingSupportMaterials(LTSM)/IIALResources.aspx", "eduintel/iial"), |
| ("https://www.education.gov.za/Curriculum/LearningandTeachingSupportMaterials(LTSM)/AttheCrossroadsTextbooks.aspx","eduintel/crossroads"), |
| ("https://www.education.gov.za/Curriculum/LearningandTeachingSupportMaterials(LTSM)/DigitalContent.aspx", "eduintel/digital"), |
| ("https://www.education.gov.za/Curriculum/LearningandTeachingSupportMaterials(LTSM)/LTSMNationalCatalogue.aspx", "eduintel/catalogues"), |
| ("https://www.education.gov.za/SelfStudyGuidesGrade10-12.aspx", "eduintel/self_study"), |
|
|
| |
| ("https://www.education.gov.za/Curriculum/CurriculumAssessmentPolicyStatements(CAPS).aspx", "eduintel/caps"), |
| ("https://www.education.gov.za/Curriculum/NationalCurriculumStatementsGradesR-12.aspx", "eduintel/caps"), |
|
|
| |
| ("https://www.education.gov.za/Curriculum/NationalSeniorCertificate(NSC)Examinations.aspx", "eduintel/nsc_exams"), |
| ("https://www.education.gov.za/Curriculum/NationalSeniorCertificate(NSC)Examinations/2025NovemberExamPapers.aspx", "eduintel/nsc_exams/2025"), |
| ("https://www.education.gov.za/Curriculum/NationalSeniorCertificate(NSC)Examinations/2024MayJuneExamPapers.aspx", "eduintel/nsc_exams/2024"), |
| ("https://www.education.gov.za/2024NSCNovemberpastpapers.aspx", "eduintel/nsc_exams/2024"), |
| ("https://www.education.gov.za/Curriculum/NationalSeniorCertificate(NSC)Examinations/2023NSCNovemberpastpapers.aspx","eduintel/nsc_exams/2023"), |
| ("https://www.education.gov.za/Curriculum/NationalSeniorCertificate(NSC)Examinations/2023MayJuneExamPapers.aspx", "eduintel/nsc_exams/2023"), |
| ("https://www.education.gov.za/Curriculum/NationalSeniorCertificate(NSC)Examinations/2022MayJuneExamPapers.aspx", "eduintel/nsc_exams/2022"), |
| ("https://www.education.gov.za/Home/2017NSCNovemberpastpapers.aspx", "eduintel/nsc_exams/2017"), |
| ("https://www.education.gov.za/Curriculum/NationalSeniorCertificate(NSC)Examinations/2021NSCExamPapers.aspx", "eduintel/nsc_exams/2021"), |
| ("https://www.education.gov.za/Curriculum/NationalSeniorCertificate(NSC)Examinations/2020NSCExamPapers.aspx", "eduintel/nsc_exams/2020"), |
|
|
| |
| ("https://www.education.gov.za/Curriculum/NationalSeniorCertificate(NSC)Examinations/2021MayJuneNSCExams.aspx", "eduintel/nsc_exams/2021"), |
|
|
| |
| ("https://www.education.gov.za/Curriculum/AnnualNationalAssessments.aspx", "eduintel/ana"), |
|
|
| |
| ("https://www.education.gov.za/Resources/Publications.aspx", "eduintel/other"), |
| ("https://www.education.gov.za/Resources/Policies.aspx", "eduintel/other"), |
| ("https://www.education.gov.za/Resources/Legislation/WhitePapers.aspx", "eduintel/other"), |
| ("https://www.education.gov.za/Resources/Manuals.aspx", "eduintel/other"), |
| ("https://www.education.gov.za/Curriculum/NationalCirculars.aspx", "eduintel/other"), |
| ("https://www.education.gov.za/Informationfor/Researcher.aspx", "eduintel/other"), |
| ("https://www.education.gov.za/Informationfor/Teachers.aspx", "eduintel/other"), |
|
|
| |
| ("https://www.justice.gov.za/legislation/constitution/index.html", "wardintel"), |
| ("https://www.gov.za/about-government/government-system/local-government", "wardintel"), |
| ] |
|
|
| SEPEDI_KEYWORDS = [ |
| "sepedi", "sesotho sa leboa", "northern sotho", "nsotho", |
| "n.sotho", "sesotho", "leboa", "sotho sa leboa" |
| ] |
|
|
| GRADE_KEYWORDS = [ |
| "grade 10", "grade 11", "grade 12", "gr 10", "gr 11", "gr 12", |
| "grade10", "grade11", "grade12", "g10", "g11", "g12", |
| "fet", "home language", " hl ", "fal", "first additional", |
| "caps", "mind the gap", "self study", "workbook", |
| "nsc", "matric", "exam", "paper", "memo", "memorand", |
| "sepedi", "sesotho", "language" |
| ] |
|
|
| def is_sepedi(url, text): |
| c = (url + " " + text).lower() |
| return any(k in c for k in SEPEDI_KEYWORDS) |
|
|
| def is_relevant(url, text): |
| c = (url + " " + text).lower() |
| return any(k in c for k in GRADE_KEYWORDS + SEPEDI_KEYWORDS) |
|
|
| def safe_filename(url, text=""): |
| """Generate a safe filename from URL or link text.""" |
| parsed = urlparse(url) |
|
|
| |
| if "linkclick.aspx" in url.lower() and text: |
| clean = re.sub(r'[^\w\s\-]', '', text).strip() |
| clean = re.sub(r'\s+', '_', clean)[:80] |
| if clean: |
| return clean + ".pdf" |
|
|
| |
| name = os.path.basename(parsed.path) |
| name = re.sub(r'[^\w\-_\. ]', '_', name) |
| return name if name and name != "aspx" else "document.pdf" |
|
|
| def make_download_url(url): |
| """Convert a LinkClick view URL to a forcedownload URL.""" |
| if "linkclick.aspx" in url.lower() and "forcedownload" not in url.lower(): |
| sep = "&" if "?" in url else "?" |
| return url + sep + "forcedownload=true" |
| return url |
|
|
| def fetch_page(url): |
| try: |
| r = requests.get(url, headers=HEADERS, timeout=25) |
| r.raise_for_status() |
| return r.text |
| except Exception as e: |
| print(f" โ Fetch failed: {e}") |
| return None |
|
|
| def extract_links(html, base_url): |
| """ |
| Extract all downloadable links from DBE pages. |
| KEY FIX: DBE uses LinkClick.aspx?fileticket= for ALL file downloads. |
| Also captures direct .pdf, .doc, .docx, .zip links. |
| """ |
| links = [] |
|
|
| |
| linkclick = re.findall( |
| r'href=["\']([^"\']*LinkClick\.aspx\?[^"\']*fileticket[^"\']*)["\']', |
| html, re.IGNORECASE |
| ) |
| for href in linkclick: |
| abs_url = urljoin(base_url, href) |
| |
| if "fileticket" in abs_url.lower(): |
| links.append(abs_url) |
|
|
| |
| direct = re.findall( |
| r'href=["\']([^"\']*\.(pdf|doc|docx|pptx|xlsx|zip))["\']', |
| html, re.IGNORECASE |
| ) |
| for href, _ in direct: |
| links.append(urljoin(base_url, href)) |
|
|
| return list(set(links)) |
|
|
| def extract_link_text(html, url): |
| """Find the anchor text for a given URL.""" |
| escaped = re.escape(url.split("&forcedownload")[0].split("&tabid")[0][:60]) |
| match = re.search( |
| escaped + r'[^"\']*["\'][^>]*>([^<]{1,80})', |
| html, re.IGNORECASE |
| ) |
| if match: |
| return match.group(1).strip() |
| return "" |
|
|
| def download_file(url, dest): |
| try: |
| dl_url = make_download_url(url) |
| r = requests.get(dl_url, headers=HEADERS, timeout=45, stream=True) |
| r.raise_for_status() |
|
|
| |
| ct = r.headers.get("content-type", "") |
| if "html" in ct and "pdf" not in ct: |
| |
| if r.url != dl_url: |
| r = requests.get(r.url, headers=HEADERS, timeout=45, stream=True) |
|
|
| os.makedirs(os.path.dirname(dest), exist_ok=True) |
| with open(dest, "wb") as f: |
| for chunk in r.iter_content(8192): |
| f.write(chunk) |
|
|
| size = os.path.getsize(dest) |
| if size < 1000: |
| os.remove(dest) |
| return None |
| return size // 1024 |
| except Exception as e: |
| print(f" โ {e}") |
| return None |
|
|
| def main(): |
| print("\n" + "โ"*64) |
| print(" SEDIBA SYSTEMS โ DBE Corpus Scraper v3") |
| print(f" Pages: {len(SEED_URLS)} | Output: {BASE_DIR}") |
| print(" Fix: LinkClick.aspx detection + NSC exam papers added") |
| print("โ"*64 + "\n") |
|
|
| |
| folders = set(f for _, f in SEED_URLS) |
| folders.update(["eduintel/nsc_exams", "eduintel/caps", "wardintel"]) |
| for f in folders: |
| os.makedirs(os.path.join(BASE_DIR, f), exist_ok=True) |
|
|
| all_docs = [] |
| seen_urls = set() |
| log_rows = [] |
|
|
| |
| print("PHASE 1 โ Crawling pages...\n") |
|
|
| for seed_url, default_folder in SEED_URLS: |
| print(f" Crawling: {seed_url.split('/')[-1]}") |
| html = fetch_page(seed_url) |
| if not html: |
| time.sleep(3) |
| continue |
|
|
| links = extract_links(html, seed_url) |
| new_docs = 0 |
|
|
| for link in links: |
| if link in seen_urls: |
| continue |
| seen_urls.add(link) |
|
|
| text = extract_link_text(html, link) |
| priority = is_sepedi(link, text) |
| relevant = is_relevant(link, text) |
|
|
| all_docs.append({ |
| "url": link, |
| "text": text, |
| "source": seed_url, |
| "folder": default_folder, |
| "priority": priority, |
| "relevant": relevant, |
| }) |
| new_docs += 1 |
|
|
| sepedi_count = sum(1 for d in all_docs[-new_docs:] if d["priority"]) if new_docs else 0 |
| print(f" โ {new_docs} docs found ({sepedi_count} Sepedi priority)") |
| time.sleep(2) |
|
|
| total = len(all_docs) |
| relevant = [d for d in all_docs if d["relevant"]] |
| priority = [d for d in all_docs if d["priority"]] |
|
|
| print(f"\n Total found: {total}") |
| print(f" Gr10-12 relevant: {len(relevant)}") |
| print(f" โ
Sepedi: {len(priority)}") |
|
|
| |
| print("\nPHASE 2 โ Downloading...\n") |
|
|
| downloaded = skipped = failed = 0 |
|
|
| |
| to_dl = sorted(relevant, key=lambda d: (not d["priority"], d["text"].lower())) |
|
|
| for doc in to_dl: |
| fname = safe_filename(doc["url"], doc["text"]) |
| dest = os.path.join(BASE_DIR, doc["folder"], fname) |
|
|
| tag = "โ
SEPEDI " if doc["priority"] else " Gr10-12" |
| label = (doc["text"] or fname)[:55] |
| print(f" [{tag}] {label}") |
|
|
| if os.path.exists(dest) and os.path.getsize(dest) > 1000: |
| print(f" โญ Already exists") |
| skipped += 1 |
| log_rows.append({**doc, "status":"exists", "size_kb": os.path.getsize(dest)//1024, "local_path":dest}) |
| continue |
|
|
| size_kb = download_file(doc["url"], dest) |
| if size_kb: |
| print(f" โ {doc['folder']}/ ({size_kb} KB)") |
| log_rows.append({**doc, "status":"downloaded", "size_kb":size_kb, "local_path":dest}) |
| downloaded += 1 |
| else: |
| print(f" โ Failed") |
| log_rows.append({**doc, "status":"failed", "size_kb":0, "local_path":""}) |
| failed += 1 |
| time.sleep(1.5) |
|
|
| |
| for doc in all_docs: |
| if not doc["relevant"]: |
| log_rows.append({**doc, "status":"skipped", "size_kb":0, "local_path":""}) |
|
|
| |
| log_path = os.path.join(BASE_DIR, "_index.csv") |
| with open(log_path, "w", newline="", encoding="utf-8") as f: |
| w = csv.DictWriter(f, fieldnames=[ |
| "status","priority","relevant","folder","text","url","source","size_kb","local_path" |
| ]) |
| w.writeheader() |
| w.writerows(log_rows) |
|
|
| print("\n" + "โ"*64) |
| print(" COMPLETE") |
| print(f" Found: {total} documents across {len(SEED_URLS)} pages") |
| print(f" Sepedi โ
: {len(priority)}") |
| print(f" Downloaded: {downloaded}") |
| print(f" Skipped: {skipped}") |
| print(f" Failed: {failed}") |
| print(f" Index: {log_path}") |
| print("\n Key folders:") |
| print(f" โ
corpus/eduintel/nsc_exams/ โ Sepedi HL exam papers 2008-2025") |
| print(f" โ
corpus/eduintel/mind_the_gap/ โ Sepedi HL study guides") |
| print(f" corpus/eduintel/caps/ โ CAPS curriculum docs") |
| print(f" corpus/wardintel/ โ Constitution + governance") |
| print("โ"*64 + "\n") |
|
|
| if __name__ == "__main__": |
| main() |
|
|