ContiAI / agents /design_phase /scraper /extractor_crawlee.py
ziadsameh32's picture
Add login page
325b94c
Raw
History Blame Contribute Delete
5.63 kB
# from crewai import Agent, Task
# from modules import llm_g
# from tools import pdf_tool, scraping_tool
# from tools import WebScrapingCrawleeTool
# from schemas import UnitSubtopicOutputModel
# web_scraper = WebScrapingCrawleeTool()
# scraping_crawlee_agent = Agent(
# role="Educational Content Scraping & Knowledge Extraction Agent",
# goal="\n".join(
# [
# "Collect and extract complete, structured, and educationally valuable content "
# "from Arabic and English websites and PDFs that i will give to you related to the course topic: {topic}.",
# "Focus on content that match the course domain ({domain}), content type ({content_type}), "
# "and audience ({audience}).",
# "Prioritize materials that can serve as strong foundations for creating {material_type} "
# "learning materials (conceptual, structural, procedural, and real-world).",
# "Extract full text including all sections, examples, and details, ensuring high accuracy for Arabic text.",
# "Assess each source’s credibility and educational relevance, ranking them by usefulness for course design.",
# "Provide concise expert notes and recommendations that will assist curriculum developers and instructional designers "
# "in selecting the best materials for building a complete learning unit.",
# ]
# ),
# backstory="\n".join(
# [
# "You are a specialized educational data researcher trained to explore, extract, and organize academic and professional content.",
# "You excel at discovering high-quality Arabic and English resources that align with specific course development objectives.",
# "Your mission is to help course designers collect trustworthy, pedagogically sound materials that will form the backbone of educational units.",
# "You understand how to evaluate the quality, relevance, and credibility of both web pages and PDFs.",
# "You are particularly skilled at preserving Arabic text integrity and extracting complete structured information for learning materials.",
# ]
# ),
# llm=llm_g,
# tools=[web_scraper],
# verbose=True,
# )
# scraping_crawlee_task = Task(
# description="\n".join(
# [
# "Your task is to extract and organize full educational content from the following source:",
# "",
# "URL: {url}",
# "Unit Title: {unit_title}",
# "Subtopic Title: {subtopic_title}",
# "Query Used: {query}",
# "",
# "This link is part of the course topic '{topic}' under the domain '{domain}'.",
# "The extracted content should help create educational materials for the audience '{audience}', focusing on '{content_type}' learning goals.",
# "",
# "For the given URL:",
# " - Extract the full title, structured text, and any available media (images, videos, audios, PDFs).",
# " - Maintain the Arabic text structure and readability.",
# " - Evaluate its reliability and educational value in relation to {material_type} material categories.",
# " - Assign an agent recommendation rank (0–5) based on credibility and relevance.",
# " - Provide short expert notes justifying the ranking and explaining how the content can contribute to the course design.",
# "",
# "Ensure no important content, examples, or explanations are omitted from extraction.",
# "Output will be a json format with no task output or raw data only the formatted json dictionary.",
# ]
# ),
# expected_output=(
# "Return ONLY a valid Python dictionary.\n"
# "- Do not include explanations, markdown, or code fences.\n"
# "- The dictionary must be UTF-8 safe and directly usable in Python with ast.literal_eval.\n"
# "- Keys must be wrapped in double quotes.\n\n"
# "Format example:\n"
# "{\n"
# ' "unit_title": "من الفكرة إلى نموذج العمل: بناء الأساس الريادي",\n'
# ' "subtopic_title": "مفهوم ريادة الأعمال وأهميتها الاقتصادية والاجتماعية",\n'
# ' "query": "دور ريادة الأعمال في التنمية الاجتماعية",\n'
# ' "parts": [\n'
# " {\n"
# ' "page_url": "https://example.com/page1",\n'
# ' "title": "Understanding Entrepreneurship in the Arab World",\n'
# ' "content": "Full educational content extracted from the site.",\n'
# ' "img_url": ["https://example.com/image1.jpg"],\n'
# ' "video_url": ["https://example.com/video1.mp4"],\n'
# ' "audio_url": ["https://example.com/audio1.mp3"],\n'
# ' "pdf_url": ["https://example.com/file1.pdf"],\n'
# ' "agent_recommendation_rank": 4.8,\n'
# ' "agent_recommendation_notes": "Rich Arabic content, relevant to conceptual materials."\n'
# " }\n"
# " ]\n"
# "}\n\n"
# "Make the output compatible with Python's ast library (use r1 = result.dict()['raw']; f_result = ast.literal_eval(r1)).\n"
# "Ensure valid JSON syntax with no unterminated strings or extra text.\n"
# "Output ONLY the dictionary — no thoughts, explanations, or markdown formatting."
# ),
# agent=scraping_crawlee_agent,
# output_json=UnitSubtopicOutputModel,
# )
# #