# from crewai import Agent, Task # from modules import llm_g # from tools import pdf_tool, scraping_tool # from tools import WebScrapingCrawleeTool # from schemas import UnitSubtopicOutputModel # web_scraper = WebScrapingCrawleeTool() # scraping_crawlee_agent = Agent( # role="Educational Content Scraping & Knowledge Extraction Agent", # goal="\n".join( # [ # "Collect and extract complete, structured, and educationally valuable content " # "from Arabic and English websites and PDFs that i will give to you related to the course topic: {topic}.", # "Focus on content that match the course domain ({domain}), content type ({content_type}), " # "and audience ({audience}).", # "Prioritize materials that can serve as strong foundations for creating {material_type} " # "learning materials (conceptual, structural, procedural, and real-world).", # "Extract full text including all sections, examples, and details, ensuring high accuracy for Arabic text.", # "Assess each source’s credibility and educational relevance, ranking them by usefulness for course design.", # "Provide concise expert notes and recommendations that will assist curriculum developers and instructional designers " # "in selecting the best materials for building a complete learning unit.", # ] # ), # backstory="\n".join( # [ # "You are a specialized educational data researcher trained to explore, extract, and organize academic and professional content.", # "You excel at discovering high-quality Arabic and English resources that align with specific course development objectives.", # "Your mission is to help course designers collect trustworthy, pedagogically sound materials that will form the backbone of educational units.", # "You understand how to evaluate the quality, relevance, and credibility of both web pages and PDFs.", # "You are particularly skilled at preserving Arabic text integrity and extracting complete structured information for learning materials.", # ] # ), # llm=llm_g, # tools=[web_scraper], # verbose=True, # ) # scraping_crawlee_task = Task( # description="\n".join( # [ # "Your task is to extract and organize full educational content from the following source:", # "", # "URL: {url}", # "Unit Title: {unit_title}", # "Subtopic Title: {subtopic_title}", # "Query Used: {query}", # "", # "This link is part of the course topic '{topic}' under the domain '{domain}'.", # "The extracted content should help create educational materials for the audience '{audience}', focusing on '{content_type}' learning goals.", # "", # "For the given URL:", # " - Extract the full title, structured text, and any available media (images, videos, audios, PDFs).", # " - Maintain the Arabic text structure and readability.", # " - Evaluate its reliability and educational value in relation to {material_type} material categories.", # " - Assign an agent recommendation rank (0–5) based on credibility and relevance.", # " - Provide short expert notes justifying the ranking and explaining how the content can contribute to the course design.", # "", # "Ensure no important content, examples, or explanations are omitted from extraction.", # "Output will be a json format with no task output or raw data only the formatted json dictionary.", # ] # ), # expected_output=( # "Return ONLY a valid Python dictionary.\n" # "- Do not include explanations, markdown, or code fences.\n" # "- The dictionary must be UTF-8 safe and directly usable in Python with ast.literal_eval.\n" # "- Keys must be wrapped in double quotes.\n\n" # "Format example:\n" # "{\n" # ' "unit_title": "من الفكرة إلى نموذج العمل: بناء الأساس الريادي",\n' # ' "subtopic_title": "مفهوم ريادة الأعمال وأهميتها الاقتصادية والاجتماعية",\n' # ' "query": "دور ريادة الأعمال في التنمية الاجتماعية",\n' # ' "parts": [\n' # " {\n" # ' "page_url": "https://example.com/page1",\n' # ' "title": "Understanding Entrepreneurship in the Arab World",\n' # ' "content": "Full educational content extracted from the site.",\n' # ' "img_url": ["https://example.com/image1.jpg"],\n' # ' "video_url": ["https://example.com/video1.mp4"],\n' # ' "audio_url": ["https://example.com/audio1.mp3"],\n' # ' "pdf_url": ["https://example.com/file1.pdf"],\n' # ' "agent_recommendation_rank": 4.8,\n' # ' "agent_recommendation_notes": "Rich Arabic content, relevant to conceptual materials."\n' # " }\n" # " ]\n" # "}\n\n" # "Make the output compatible with Python's ast library (use r1 = result.dict()['raw']; f_result = ast.literal_eval(r1)).\n" # "Ensure valid JSON syntax with no unterminated strings or extra text.\n" # "Output ONLY the dictionary — no thoughts, explanations, or markdown formatting." # ), # agent=scraping_crawlee_agent, # output_json=UnitSubtopicOutputModel, # ) # #