# import os # import json # import shutil # from fastapi import APIRouter, HTTPException, UploadFile, Form, Request # from crewai import Crew, Process # from agents.design_phase import scraping_built_in_agent, scraping_built_in_task,scraping_bs4_agent,scraping_bs4_task # from schemas import DNAMetadata, OutlineInput # from tools import scrape_with_bs4, crawl_parse_url,crawl_bs_url,extract_pdf_content # router = APIRouter(prefix="/design", tags=["Design"]) # ################################# # # Built-in # # ################################# # @router.post("/scraper_built_in") # # async def run_training(request: Request, file: UploadFile, data: str = Form(...)): # """Uploads keywords JSON + metadata JSON, runs CrewAI search, returns download link.""" # # ✅ Parse metadata JSON # try: # parsed_data = json.loads(data) # metadata = OutlineInput(**parsed_data) # except Exception as e: # raise HTTPException(status_code=400, detail=f"Invalid JSON in 'data': {e}") # # ✅ Save uploaded file temporarily # save_path = f"/tmp/{file.filename}" # with open(save_path, "wb") as buffer: # shutil.copyfileobj(file.file, buffer) # # ✅ Validate file extension # if not save_path.lower().endswith(".json"): # raise HTTPException(status_code=400, detail="File must be a JSON file") # # ✅ Load file content # try: # with open(save_path, "r", encoding="utf-8") as f: # urls_data = json.load(f) # except json.JSONDecodeError: # raise HTTPException(status_code=400, detail="Invalid JSON file content") # # ✅ Initialize Crew # crew = Crew( # agents=[scraping_built_in_agent], # tasks=[scraping_built_in_task], # process=Process.sequential, # ) # # ✅ Build static user metadata # user_inputs = DNAMetadata( # topic=metadata.topic, # domain=metadata.domain, # content_type=metadata.content_type, # audience=metadata.audience, # material_type=metadata.material_type, # ).dict() # all_results = [] # # ✅ Iterate through each topic unit and result link # for unit in urls_data["results"]: # unit_title = unit["unit_title"] # subtopic_title = unit["subtopic_title"] # query = unit["query"] # for result_item in unit["results"]: # url = result_item["url"] # print(f"🔍 Running scrape for [{subtopic_title}] | URL: {url}") # merged_input = { # **user_inputs, # "url": url, # "unit_title": unit_title, # "subtopic_title": subtopic_title, # "query": query, # } # try: # result = crew.kickoff(inputs=merged_input) # all_results.append(result.dict()) # # usage = result.token_usage # CrewAI بيحسبها جاهز # # total_prompt += usage["prompt_tokens"] # # total_completion += usage["completion_tokens"] # # total_tokens += usage["total_tokens"] # except Exception as e: # print(f"⚠️ Error while processing '{url}': {e}") # # ✅ Save aggregated results # output_data = {"results": all_results} # output_file = "/tmp/search_results.json" # with open(output_file, "w", encoding="utf-8") as f: # json.dump(output_data, f, ensure_ascii=False, indent=2) # # ✅ Build download URL # base_url = str(request.base_url).rstrip("/") # download_link = ( # f"{base_url}/design/download?filename={os.path.basename(output_file)}" # ) # return { # "message": "Scraping process completed successfully 🚀", # "total_queries": len(all_results), # # "total_prompt": total_prompt, # # "total_completion": total_completion, # # "total_tokens": total_tokens, # "download_link": download_link, # "result": all_results, # "json_dict": output_data, # } # ################################# # # BS4 # # ################################# # @router.post("/scraper_bs4_agent") # async def run_training(request: Request, file: UploadFile, data: str = Form(...)): # """Uploads keywords JSON + metadata JSON, runs CrewAI search, returns download link.""" # # ✅ Parse metadata JSON # try: # parsed_data = json.loads(data) # metadata = OutlineInput(**parsed_data) # except Exception as e: # raise HTTPException(status_code=400, detail=f"Invalid JSON in 'data': {e}") # # ✅ Save uploaded file temporarily # save_path = f"/tmp/{file.filename}" # with open(save_path, "wb") as buffer: # shutil.copyfileobj(file.file, buffer) # # ✅ Validate file extension # if not save_path.lower().endswith(".json"): # raise HTTPException(status_code=400, detail="File must be a JSON file") # # ✅ Load file content # try: # with open(save_path, "r", encoding="utf-8") as f: # urls_data = json.load(f) # except json.JSONDecodeError: # raise HTTPException(status_code=400, detail="Invalid JSON file content") # # ✅ Initialize Crew # crew = Crew( # agents=[scraping_bs4_agent], # tasks=[scraping_bs4_task], # process=Process.sequential, # ) # # ✅ Build static user metadata # user_inputs = DNAMetadata( # topic=metadata.topic, # domain=metadata.domain, # content_type=metadata.content_type, # audience=metadata.audience, # material_type=metadata.material_type, # ).dict() # all_results = [] # # ✅ Iterate through each topic unit and result link # for unit in urls_data["results"]: # unit_title = unit["unit_title"] # subtopic_title = unit["subtopic_title"] # query = unit["query"] # for result_item in unit["results"]: # url = result_item["url"] # print(f"🔍 Running scrape for [{subtopic_title}] | URL: {url}") # merged_input = { # **user_inputs, # "url": url, # "unit_title": unit_title, # "subtopic_title": subtopic_title, # "query": query, # } # try: # result = crew.kickoff(inputs=merged_input) # all_results.append(result.dict()) # # usage = result.token_usage # CrewAI بيحسبها جاهز # # total_prompt += usage["prompt_tokens"] # # total_completion += usage["completion_tokens"] # # total_tokens += usage["total_tokens"] # except Exception as e: # print(f"⚠️ Error while processing '{url}': {e}") # # ✅ Save aggregated results # output_data = {"results": all_results} # output_file = "/tmp/search_results.json" # with open(output_file, "w", encoding="utf-8") as f: # json.dump(output_data, f, ensure_ascii=False, indent=2) # # ✅ Build download URL # base_url = str(request.base_url).rstrip("/") # download_link = ( # f"{base_url}/design/download?filename={os.path.basename(output_file)}" # ) # return { # "message": "Scraping process completed successfully 🚀", # "total_queries": len(all_results), # # "total_prompt": total_prompt, # # "total_completion": total_completion, # # "total_tokens": total_tokens, # "download_link": download_link, # "result": all_results, # "json_dict": output_data, # } # # ################################# # # # crawlee # # # ################################# # # @router.post("/scraper_crawlee_agent") # # async def run_training(request: Request, file: UploadFile, data: str = Form(...)): # # """Uploads keywords JSON + metadata JSON, runs CrewAI search, returns download link.""" # # # ✅ Parse metadata JSON # # try: # # parsed_data = json.loads(data) # # metadata = OutlineInput(**parsed_data) # # except Exception as e: # # raise HTTPException(status_code=400, detail=f"Invalid JSON in 'data': {e}") # # # ✅ Save uploaded file temporarily # # save_path = f"/tmp/{file.filename}" # # with open(save_path, "wb") as buffer: # # shutil.copyfileobj(file.file, buffer) # # # ✅ Validate file extension # # if not save_path.lower().endswith(".json"): # # raise HTTPException(status_code=400, detail="File must be a JSON file") # # # ✅ Load file content # # try: # # with open(save_path, "r", encoding="utf-8") as f: # # urls_data = json.load(f) # # except json.JSONDecodeError: # # raise HTTPException(status_code=400, detail="Invalid JSON file content") # # # ✅ Initialize Crew # # crew = Crew( # # agents=[scraping_built_in_agent], # # tasks=[scraping_built_in_task], # # process=Process.sequential, # # ) # # # ✅ Build static user metadata # # user_inputs = DNAMetadata( # # topic=metadata.topic, # # domain=metadata.domain, # # content_type=metadata.content_type, # # audience=metadata.audience, # # material_type=metadata.material_type, # # ).dict() # # all_results = [] # # # ✅ Iterate through each topic unit and result link # # for unit in urls_data["results"]: # # unit_title = unit["unit_title"] # # subtopic_title = unit["subtopic_title"] # # query = unit["query"] # # for result_item in unit["results"]: # # url = result_item["url"] # # print(f"🔍 Running scrape for [{subtopic_title}] | URL: {url}") # # merged_input = { # # **user_inputs, # # "url": url, # # "unit_title": unit_title, # # "subtopic_title": subtopic_title, # # "query": query, # # } # # try: # # result = crew.kickoff(inputs=merged_input) # # all_results.append(result.dict()) # # except Exception as e: # # print(f"⚠️ Error while processing '{url}': {e}") # # # ✅ Save aggregated results # # output_data = {"results": all_results} # # output_file = "/tmp/search_results.json" # # with open(output_file, "w", encoding="utf-8") as f: # # json.dump(output_data, f, ensure_ascii=False, indent=2) # # # ✅ Build download URL # # base_url = str(request.base_url).rstrip("/") # # download_link = ( # # f"{base_url}/design/download?filename={os.path.basename(output_file)}" # # ) # # return { # # "message": "Scraping process completed successfully 🚀", # # "total_queries": len(all_results), # # "download_link": download_link, # # "result": all_results, # # "json_dict": output_data, # # } # ############################## # # =================================================================== # # SHARED HELPER # # =================================================================== # async def process_json_scrape(request: Request, file: UploadFile, data: str, mode: str): # """ # mode = 'bs4' or 'crawlee' # """ # # ---- Parse metadata JSON ---- # try: # metadata = json.loads(data) # except Exception as e: # raise HTTPException(status_code=400, detail=f"Invalid metadata JSON: {e}") # # ---- Save JSON file ---- # save_path = f"/tmp/{file.filename}" # with open(save_path, "wb") as buffer: # shutil.copyfileobj(file.file, buffer) # if not save_path.endswith(".json"): # raise HTTPException(status_code=400, detail="Uploaded file must be .json") # # ---- Load JSON content ---- # try: # with open(save_path, "r", encoding="utf-8") as f: # urls_data = json.load(f) # except Exception: # raise HTTPException(status_code=400, detail="Invalid JSON file content") # all_results = [] # # ---- Loop through your structure ---- # for unit in urls_data["results"]: # unit_title = unit["unit_title"] # subtopic_title = unit["subtopic_title"] # query = unit["query"] # for result_item in unit["results"]: # url = result_item["url"] # print(f"🔍 Scraping | {subtopic_title} | {url}") # try: # # ----- PDF case ----- # if url.lower().endswith(".pdf"): # scraped = { # "page_url": url, # "title": "", # "content": extract_pdf_content(url), # "img_url": [], # "video_url": [], # "audio_url": [], # "pdf_url": [], # "agent_recommendation_rank": 4.2, # "agent_recommendation_notes": "Scraped successfully using Crawlee + ParselCrawler.", # "header": "Web Scraping Test", # "sub_header": "Crawlee Version", # } # elif mode == "bs4": # scraped = scrape_with_bs4(url) # elif mode == "crawlee bs": # scraped = await crawl_bs_url(url) # elif mode == "crawlee parsel": # scraped = await crawl_parse_url(url) # all_results.append( # { # "unit_title": unit_title, # "subtopic_title": subtopic_title, # "query": query, # "parts": scraped, # } # ) # except Exception as e: # all_results.append( # { # "unit_title": unit_title, # "subtopic_title": subtopic_title, # "query": query, # "url": url, # "error": str(e), # } # ) # # ---- Save Output ---- # output_file = "/tmp/scrape_results.json" # with open(output_file, "w", encoding="utf-8") as f: # json.dump({"results": all_results}, f, ensure_ascii=False, indent=2) # # ---- Download link ---- # base_url = str(request.base_url).rstrip("/") # download_link = f"{base_url}/design/download?filename=scrape_results.json" # return { # "message": f"Scraping completed using {mode.upper()} ✔", # "total_links": len(all_results), # "download_link": download_link, # "results": all_results, # } # # =================================================================== # # ROUTE 1 — BS4 SCRAPER # # =================================================================== # @router.post("/scraper_bs4") # async def scraper_bs4(request: Request, file: UploadFile, data: str = Form(...)): # return await process_json_scrape(request, file, data, mode="bs4") # # =================================================================== # # ROUTE 2 — CRAWLEE SCRAPER PARSEL # # =================================================================== # @router.post("/scraper_crawlee_parsel") # async def scraper_crawlee(request: Request, file: UploadFile, data: str = Form(...)): # return await process_json_scrape(request, file, data, mode="crawlee parsel") # # =================================================================== # # ROUTE 2 — CRAWLEE SCRAPER BESUTIFULSOUP # # =================================================================== # @router.post("/scraper_crawlee_bs") # async def scraper_crawlee(request: Request, file: UploadFile, data: str = Form(...)): # return await process_json_scrape(request, file, data, mode="crawlee bs")