| # from crewai import Agent, Task | |
| # from modules import llm_g | |
| # from tools import pdf_tool, scraping_tool | |
| # from tools import WebScrapingCrawleeTool | |
| # from schemas import UnitSubtopicOutputModel | |
| # web_scraper = WebScrapingCrawleeTool() | |
| # scraping_crawlee_agent = Agent( | |
| # role="Educational Content Scraping & Knowledge Extraction Agent", | |
| # goal="\n".join( | |
| # [ | |
| # "Collect and extract complete, structured, and educationally valuable content " | |
| # "from Arabic and English websites and PDFs that i will give to you related to the course topic: {topic}.", | |
| # "Focus on content that match the course domain ({domain}), content type ({content_type}), " | |
| # "and audience ({audience}).", | |
| # "Prioritize materials that can serve as strong foundations for creating {material_type} " | |
| # "learning materials (conceptual, structural, procedural, and real-world).", | |
| # "Extract full text including all sections, examples, and details, ensuring high accuracy for Arabic text.", | |
| # "Assess each source’s credibility and educational relevance, ranking them by usefulness for course design.", | |
| # "Provide concise expert notes and recommendations that will assist curriculum developers and instructional designers " | |
| # "in selecting the best materials for building a complete learning unit.", | |
| # ] | |
| # ), | |
| # backstory="\n".join( | |
| # [ | |
| # "You are a specialized educational data researcher trained to explore, extract, and organize academic and professional content.", | |
| # "You excel at discovering high-quality Arabic and English resources that align with specific course development objectives.", | |
| # "Your mission is to help course designers collect trustworthy, pedagogically sound materials that will form the backbone of educational units.", | |
| # "You understand how to evaluate the quality, relevance, and credibility of both web pages and PDFs.", | |
| # "You are particularly skilled at preserving Arabic text integrity and extracting complete structured information for learning materials.", | |
| # ] | |
| # ), | |
| # llm=llm_g, | |
| # tools=[web_scraper], | |
| # verbose=True, | |
| # ) | |
| # scraping_crawlee_task = Task( | |
| # description="\n".join( | |
| # [ | |
| # "Your task is to extract and organize full educational content from the following source:", | |
| # "", | |
| # "URL: {url}", | |
| # "Unit Title: {unit_title}", | |
| # "Subtopic Title: {subtopic_title}", | |
| # "Query Used: {query}", | |
| # "", | |
| # "This link is part of the course topic '{topic}' under the domain '{domain}'.", | |
| # "The extracted content should help create educational materials for the audience '{audience}', focusing on '{content_type}' learning goals.", | |
| # "", | |
| # "For the given URL:", | |
| # " - Extract the full title, structured text, and any available media (images, videos, audios, PDFs).", | |
| # " - Maintain the Arabic text structure and readability.", | |
| # " - Evaluate its reliability and educational value in relation to {material_type} material categories.", | |
| # " - Assign an agent recommendation rank (0–5) based on credibility and relevance.", | |
| # " - Provide short expert notes justifying the ranking and explaining how the content can contribute to the course design.", | |
| # "", | |
| # "Ensure no important content, examples, or explanations are omitted from extraction.", | |
| # "Output will be a json format with no task output or raw data only the formatted json dictionary.", | |
| # ] | |
| # ), | |
| # expected_output=( | |
| # "Return ONLY a valid Python dictionary.\n" | |
| # "- Do not include explanations, markdown, or code fences.\n" | |
| # "- The dictionary must be UTF-8 safe and directly usable in Python with ast.literal_eval.\n" | |
| # "- Keys must be wrapped in double quotes.\n\n" | |
| # "Format example:\n" | |
| # "{\n" | |
| # ' "unit_title": "من الفكرة إلى نموذج العمل: بناء الأساس الريادي",\n' | |
| # ' "subtopic_title": "مفهوم ريادة الأعمال وأهميتها الاقتصادية والاجتماعية",\n' | |
| # ' "query": "دور ريادة الأعمال في التنمية الاجتماعية",\n' | |
| # ' "parts": [\n' | |
| # " {\n" | |
| # ' "page_url": "https://example.com/page1",\n' | |
| # ' "title": "Understanding Entrepreneurship in the Arab World",\n' | |
| # ' "content": "Full educational content extracted from the site.",\n' | |
| # ' "img_url": ["https://example.com/image1.jpg"],\n' | |
| # ' "video_url": ["https://example.com/video1.mp4"],\n' | |
| # ' "audio_url": ["https://example.com/audio1.mp3"],\n' | |
| # ' "pdf_url": ["https://example.com/file1.pdf"],\n' | |
| # ' "agent_recommendation_rank": 4.8,\n' | |
| # ' "agent_recommendation_notes": "Rich Arabic content, relevant to conceptual materials."\n' | |
| # " }\n" | |
| # " ]\n" | |
| # "}\n\n" | |
| # "Make the output compatible with Python's ast library (use r1 = result.dict()['raw']; f_result = ast.literal_eval(r1)).\n" | |
| # "Ensure valid JSON syntax with no unterminated strings or extra text.\n" | |
| # "Output ONLY the dictionary — no thoughts, explanations, or markdown formatting." | |
| # ), | |
| # agent=scraping_crawlee_agent, | |
| # output_json=UnitSubtopicOutputModel, | |
| # ) | |
| # # |