| # import os | |
| # import json | |
| # import shutil | |
| # from fastapi import APIRouter, HTTPException, UploadFile, Form, Request | |
| # from crewai import Crew, Process | |
| # from agents.design_phase import scraping_built_in_agent, scraping_built_in_task,scraping_bs4_agent,scraping_bs4_task | |
| # from schemas import DNAMetadata, OutlineInput | |
| # from tools import scrape_with_bs4, crawl_parse_url,crawl_bs_url,extract_pdf_content | |
| # router = APIRouter(prefix="/design", tags=["Design"]) | |
| # ################################# | |
| # # Built-in # | |
| # ################################# | |
| # @router.post("/scraper_built_in") | |
| # # async def run_training(request: Request, file: UploadFile, data: str = Form(...)): | |
| # """Uploads keywords JSON + metadata JSON, runs CrewAI search, returns download link.""" | |
| # # β Parse metadata JSON | |
| # try: | |
| # parsed_data = json.loads(data) | |
| # metadata = OutlineInput(**parsed_data) | |
| # except Exception as e: | |
| # raise HTTPException(status_code=400, detail=f"Invalid JSON in 'data': {e}") | |
| # # β Save uploaded file temporarily | |
| # save_path = f"/tmp/{file.filename}" | |
| # with open(save_path, "wb") as buffer: | |
| # shutil.copyfileobj(file.file, buffer) | |
| # # β Validate file extension | |
| # if not save_path.lower().endswith(".json"): | |
| # raise HTTPException(status_code=400, detail="File must be a JSON file") | |
| # # β Load file content | |
| # try: | |
| # with open(save_path, "r", encoding="utf-8") as f: | |
| # urls_data = json.load(f) | |
| # except json.JSONDecodeError: | |
| # raise HTTPException(status_code=400, detail="Invalid JSON file content") | |
| # # β Initialize Crew | |
| # crew = Crew( | |
| # agents=[scraping_built_in_agent], | |
| # tasks=[scraping_built_in_task], | |
| # process=Process.sequential, | |
| # ) | |
| # # β Build static user metadata | |
| # user_inputs = DNAMetadata( | |
| # topic=metadata.topic, | |
| # domain=metadata.domain, | |
| # content_type=metadata.content_type, | |
| # audience=metadata.audience, | |
| # material_type=metadata.material_type, | |
| # ).dict() | |
| # all_results = [] | |
| # # β Iterate through each topic unit and result link | |
| # for unit in urls_data["results"]: | |
| # unit_title = unit["unit_title"] | |
| # subtopic_title = unit["subtopic_title"] | |
| # query = unit["query"] | |
| # for result_item in unit["results"]: | |
| # url = result_item["url"] | |
| # print(f"π Running scrape for [{subtopic_title}] | URL: {url}") | |
| # merged_input = { | |
| # **user_inputs, | |
| # "url": url, | |
| # "unit_title": unit_title, | |
| # "subtopic_title": subtopic_title, | |
| # "query": query, | |
| # } | |
| # try: | |
| # result = crew.kickoff(inputs=merged_input) | |
| # all_results.append(result.dict()) | |
| # # usage = result.token_usage # CrewAI Ψ¨ΩΨΨ³Ψ¨ΩΨ§ Ψ¬Ψ§ΩΨ² | |
| # # total_prompt += usage["prompt_tokens"] | |
| # # total_completion += usage["completion_tokens"] | |
| # # total_tokens += usage["total_tokens"] | |
| # except Exception as e: | |
| # print(f"β οΈ Error while processing '{url}': {e}") | |
| # # β Save aggregated results | |
| # output_data = {"results": all_results} | |
| # output_file = "/tmp/search_results.json" | |
| # with open(output_file, "w", encoding="utf-8") as f: | |
| # json.dump(output_data, f, ensure_ascii=False, indent=2) | |
| # # β Build download URL | |
| # base_url = str(request.base_url).rstrip("/") | |
| # download_link = ( | |
| # f"{base_url}/design/download?filename={os.path.basename(output_file)}" | |
| # ) | |
| # return { | |
| # "message": "Scraping process completed successfully π", | |
| # "total_queries": len(all_results), | |
| # # "total_prompt": total_prompt, | |
| # # "total_completion": total_completion, | |
| # # "total_tokens": total_tokens, | |
| # "download_link": download_link, | |
| # "result": all_results, | |
| # "json_dict": output_data, | |
| # } | |
| # ################################# | |
| # # BS4 # | |
| # ################################# | |
| # @router.post("/scraper_bs4_agent") | |
| # async def run_training(request: Request, file: UploadFile, data: str = Form(...)): | |
| # """Uploads keywords JSON + metadata JSON, runs CrewAI search, returns download link.""" | |
| # # β Parse metadata JSON | |
| # try: | |
| # parsed_data = json.loads(data) | |
| # metadata = OutlineInput(**parsed_data) | |
| # except Exception as e: | |
| # raise HTTPException(status_code=400, detail=f"Invalid JSON in 'data': {e}") | |
| # # β Save uploaded file temporarily | |
| # save_path = f"/tmp/{file.filename}" | |
| # with open(save_path, "wb") as buffer: | |
| # shutil.copyfileobj(file.file, buffer) | |
| # # β Validate file extension | |
| # if not save_path.lower().endswith(".json"): | |
| # raise HTTPException(status_code=400, detail="File must be a JSON file") | |
| # # β Load file content | |
| # try: | |
| # with open(save_path, "r", encoding="utf-8") as f: | |
| # urls_data = json.load(f) | |
| # except json.JSONDecodeError: | |
| # raise HTTPException(status_code=400, detail="Invalid JSON file content") | |
| # # β Initialize Crew | |
| # crew = Crew( | |
| # agents=[scraping_bs4_agent], | |
| # tasks=[scraping_bs4_task], | |
| # process=Process.sequential, | |
| # ) | |
| # # β Build static user metadata | |
| # user_inputs = DNAMetadata( | |
| # topic=metadata.topic, | |
| # domain=metadata.domain, | |
| # content_type=metadata.content_type, | |
| # audience=metadata.audience, | |
| # material_type=metadata.material_type, | |
| # ).dict() | |
| # all_results = [] | |
| # # β Iterate through each topic unit and result link | |
| # for unit in urls_data["results"]: | |
| # unit_title = unit["unit_title"] | |
| # subtopic_title = unit["subtopic_title"] | |
| # query = unit["query"] | |
| # for result_item in unit["results"]: | |
| # url = result_item["url"] | |
| # print(f"π Running scrape for [{subtopic_title}] | URL: {url}") | |
| # merged_input = { | |
| # **user_inputs, | |
| # "url": url, | |
| # "unit_title": unit_title, | |
| # "subtopic_title": subtopic_title, | |
| # "query": query, | |
| # } | |
| # try: | |
| # result = crew.kickoff(inputs=merged_input) | |
| # all_results.append(result.dict()) | |
| # # usage = result.token_usage # CrewAI Ψ¨ΩΨΨ³Ψ¨ΩΨ§ Ψ¬Ψ§ΩΨ² | |
| # # total_prompt += usage["prompt_tokens"] | |
| # # total_completion += usage["completion_tokens"] | |
| # # total_tokens += usage["total_tokens"] | |
| # except Exception as e: | |
| # print(f"β οΈ Error while processing '{url}': {e}") | |
| # # β Save aggregated results | |
| # output_data = {"results": all_results} | |
| # output_file = "/tmp/search_results.json" | |
| # with open(output_file, "w", encoding="utf-8") as f: | |
| # json.dump(output_data, f, ensure_ascii=False, indent=2) | |
| # # β Build download URL | |
| # base_url = str(request.base_url).rstrip("/") | |
| # download_link = ( | |
| # f"{base_url}/design/download?filename={os.path.basename(output_file)}" | |
| # ) | |
| # return { | |
| # "message": "Scraping process completed successfully π", | |
| # "total_queries": len(all_results), | |
| # # "total_prompt": total_prompt, | |
| # # "total_completion": total_completion, | |
| # # "total_tokens": total_tokens, | |
| # "download_link": download_link, | |
| # "result": all_results, | |
| # "json_dict": output_data, | |
| # } | |
| # # ################################# | |
| # # # crawlee # | |
| # # ################################# | |
| # # @router.post("/scraper_crawlee_agent") | |
| # # async def run_training(request: Request, file: UploadFile, data: str = Form(...)): | |
| # # """Uploads keywords JSON + metadata JSON, runs CrewAI search, returns download link.""" | |
| # # # β Parse metadata JSON | |
| # # try: | |
| # # parsed_data = json.loads(data) | |
| # # metadata = OutlineInput(**parsed_data) | |
| # # except Exception as e: | |
| # # raise HTTPException(status_code=400, detail=f"Invalid JSON in 'data': {e}") | |
| # # # β Save uploaded file temporarily | |
| # # save_path = f"/tmp/{file.filename}" | |
| # # with open(save_path, "wb") as buffer: | |
| # # shutil.copyfileobj(file.file, buffer) | |
| # # # β Validate file extension | |
| # # if not save_path.lower().endswith(".json"): | |
| # # raise HTTPException(status_code=400, detail="File must be a JSON file") | |
| # # # β Load file content | |
| # # try: | |
| # # with open(save_path, "r", encoding="utf-8") as f: | |
| # # urls_data = json.load(f) | |
| # # except json.JSONDecodeError: | |
| # # raise HTTPException(status_code=400, detail="Invalid JSON file content") | |
| # # # β Initialize Crew | |
| # # crew = Crew( | |
| # # agents=[scraping_built_in_agent], | |
| # # tasks=[scraping_built_in_task], | |
| # # process=Process.sequential, | |
| # # ) | |
| # # # β Build static user metadata | |
| # # user_inputs = DNAMetadata( | |
| # # topic=metadata.topic, | |
| # # domain=metadata.domain, | |
| # # content_type=metadata.content_type, | |
| # # audience=metadata.audience, | |
| # # material_type=metadata.material_type, | |
| # # ).dict() | |
| # # all_results = [] | |
| # # # β Iterate through each topic unit and result link | |
| # # for unit in urls_data["results"]: | |
| # # unit_title = unit["unit_title"] | |
| # # subtopic_title = unit["subtopic_title"] | |
| # # query = unit["query"] | |
| # # for result_item in unit["results"]: | |
| # # url = result_item["url"] | |
| # # print(f"π Running scrape for [{subtopic_title}] | URL: {url}") | |
| # # merged_input = { | |
| # # **user_inputs, | |
| # # "url": url, | |
| # # "unit_title": unit_title, | |
| # # "subtopic_title": subtopic_title, | |
| # # "query": query, | |
| # # } | |
| # # try: | |
| # # result = crew.kickoff(inputs=merged_input) | |
| # # all_results.append(result.dict()) | |
| # # except Exception as e: | |
| # # print(f"β οΈ Error while processing '{url}': {e}") | |
| # # # β Save aggregated results | |
| # # output_data = {"results": all_results} | |
| # # output_file = "/tmp/search_results.json" | |
| # # with open(output_file, "w", encoding="utf-8") as f: | |
| # # json.dump(output_data, f, ensure_ascii=False, indent=2) | |
| # # # β Build download URL | |
| # # base_url = str(request.base_url).rstrip("/") | |
| # # download_link = ( | |
| # # f"{base_url}/design/download?filename={os.path.basename(output_file)}" | |
| # # ) | |
| # # return { | |
| # # "message": "Scraping process completed successfully π", | |
| # # "total_queries": len(all_results), | |
| # # "download_link": download_link, | |
| # # "result": all_results, | |
| # # "json_dict": output_data, | |
| # # } | |
| # ############################## | |
| # # =================================================================== | |
| # # SHARED HELPER | |
| # # =================================================================== | |
| # async def process_json_scrape(request: Request, file: UploadFile, data: str, mode: str): | |
| # """ | |
| # mode = 'bs4' or 'crawlee' | |
| # """ | |
| # # ---- Parse metadata JSON ---- | |
| # try: | |
| # metadata = json.loads(data) | |
| # except Exception as e: | |
| # raise HTTPException(status_code=400, detail=f"Invalid metadata JSON: {e}") | |
| # # ---- Save JSON file ---- | |
| # save_path = f"/tmp/{file.filename}" | |
| # with open(save_path, "wb") as buffer: | |
| # shutil.copyfileobj(file.file, buffer) | |
| # if not save_path.endswith(".json"): | |
| # raise HTTPException(status_code=400, detail="Uploaded file must be .json") | |
| # # ---- Load JSON content ---- | |
| # try: | |
| # with open(save_path, "r", encoding="utf-8") as f: | |
| # urls_data = json.load(f) | |
| # except Exception: | |
| # raise HTTPException(status_code=400, detail="Invalid JSON file content") | |
| # all_results = [] | |
| # # ---- Loop through your structure ---- | |
| # for unit in urls_data["results"]: | |
| # unit_title = unit["unit_title"] | |
| # subtopic_title = unit["subtopic_title"] | |
| # query = unit["query"] | |
| # for result_item in unit["results"]: | |
| # url = result_item["url"] | |
| # print(f"π Scraping | {subtopic_title} | {url}") | |
| # try: | |
| # # ----- PDF case ----- | |
| # if url.lower().endswith(".pdf"): | |
| # scraped = { | |
| # "page_url": url, | |
| # "title": "", | |
| # "content": extract_pdf_content(url), | |
| # "img_url": [], | |
| # "video_url": [], | |
| # "audio_url": [], | |
| # "pdf_url": [], | |
| # "agent_recommendation_rank": 4.2, | |
| # "agent_recommendation_notes": "Scraped successfully using Crawlee + ParselCrawler.", | |
| # "header": "Web Scraping Test", | |
| # "sub_header": "Crawlee Version", | |
| # } | |
| # elif mode == "bs4": | |
| # scraped = scrape_with_bs4(url) | |
| # elif mode == "crawlee bs": | |
| # scraped = await crawl_bs_url(url) | |
| # elif mode == "crawlee parsel": | |
| # scraped = await crawl_parse_url(url) | |
| # all_results.append( | |
| # { | |
| # "unit_title": unit_title, | |
| # "subtopic_title": subtopic_title, | |
| # "query": query, | |
| # "parts": scraped, | |
| # } | |
| # ) | |
| # except Exception as e: | |
| # all_results.append( | |
| # { | |
| # "unit_title": unit_title, | |
| # "subtopic_title": subtopic_title, | |
| # "query": query, | |
| # "url": url, | |
| # "error": str(e), | |
| # } | |
| # ) | |
| # # ---- Save Output ---- | |
| # output_file = "/tmp/scrape_results.json" | |
| # with open(output_file, "w", encoding="utf-8") as f: | |
| # json.dump({"results": all_results}, f, ensure_ascii=False, indent=2) | |
| # # ---- Download link ---- | |
| # base_url = str(request.base_url).rstrip("/") | |
| # download_link = f"{base_url}/design/download?filename=scrape_results.json" | |
| # return { | |
| # "message": f"Scraping completed using {mode.upper()} β", | |
| # "total_links": len(all_results), | |
| # "download_link": download_link, | |
| # "results": all_results, | |
| # } | |
| # # =================================================================== | |
| # # ROUTE 1 β BS4 SCRAPER | |
| # # =================================================================== | |
| # @router.post("/scraper_bs4") | |
| # async def scraper_bs4(request: Request, file: UploadFile, data: str = Form(...)): | |
| # return await process_json_scrape(request, file, data, mode="bs4") | |
| # # =================================================================== | |
| # # ROUTE 2 β CRAWLEE SCRAPER PARSEL | |
| # # =================================================================== | |
| # @router.post("/scraper_crawlee_parsel") | |
| # async def scraper_crawlee(request: Request, file: UploadFile, data: str = Form(...)): | |
| # return await process_json_scrape(request, file, data, mode="crawlee parsel") | |
| # # =================================================================== | |
| # # ROUTE 2 β CRAWLEE SCRAPER BESUTIFULSOUP | |
| # # =================================================================== | |
| # @router.post("/scraper_crawlee_bs") | |
| # async def scraper_crawlee(request: Request, file: UploadFile, data: str = Form(...)): | |
| # return await process_json_scrape(request, file, data, mode="crawlee bs") | |