File size: 5,626 Bytes
325b94c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
# from crewai import Agent, Task
# from modules import llm_g
# from tools import pdf_tool, scraping_tool
# from tools import WebScrapingCrawleeTool
# from schemas import UnitSubtopicOutputModel


# web_scraper = WebScrapingCrawleeTool()

# scraping_crawlee_agent = Agent(
#     role="Educational Content Scraping & Knowledge Extraction Agent",
#     goal="\n".join(
#         [
#             "Collect and extract complete, structured, and educationally valuable content "
#             "from Arabic and English websites and PDFs that i will give to you related to the course topic: {topic}.",
            # "Focus on content that match the course domain ({domain}), content type ({content_type}), "
            # "and audience ({audience}).",
#             "Prioritize materials that can serve as strong foundations for creating {material_type} "
#             "learning materials (conceptual, structural, procedural, and real-world).",
#             "Extract full text including all sections, examples, and details, ensuring high accuracy for Arabic text.",
#             "Assess each source’s credibility and educational relevance, ranking them by usefulness for course design.",
#             "Provide concise expert notes and recommendations that will assist curriculum developers and instructional designers "
#             "in selecting the best materials for building a complete learning unit.",
#         ]
#     ),
#     backstory="\n".join(
#         [
#             "You are a specialized educational data researcher trained to explore, extract, and organize academic and professional content.",
#             "You excel at discovering high-quality Arabic and English resources that align with specific course development objectives.",
#             "Your mission is to help course designers collect trustworthy, pedagogically sound materials that will form the backbone of educational units.",
#             "You understand how to evaluate the quality, relevance, and credibility of both web pages and PDFs.",
#             "You are particularly skilled at preserving Arabic text integrity and extracting complete structured information for learning materials.",
#         ]
#     ),
#     llm=llm_g,
#     tools=[web_scraper],
#     verbose=True,
# )


# scraping_crawlee_task = Task(
#     description="\n".join(
#         [
#             "Your task is to extract and organize full educational content from the following source:",
#             "",
#             "URL: {url}",
#             "Unit Title: {unit_title}",
#             "Subtopic Title: {subtopic_title}",
#             "Query Used: {query}",
#             "",
#             "This link is part of the course topic '{topic}' under the domain '{domain}'.",
#             "The extracted content should help create educational materials for the audience '{audience}', focusing on '{content_type}' learning goals.",
#             "",
#             "For the given URL:",
#             "  - Extract the full title, structured text, and any available media (images, videos, audios, PDFs).",
#             "  - Maintain the Arabic text structure and readability.",
#             "  - Evaluate its reliability and educational value in relation to {material_type} material categories.",
#             "  - Assign an agent recommendation rank (0–5) based on credibility and relevance.",
#             "  - Provide short expert notes justifying the ranking and explaining how the content can contribute to the course design.",
#             "",
#             "Ensure no important content, examples, or explanations are omitted from extraction.",
            # "Output will be a json format with no task output or raw data only the formatted json dictionary.",

#         ]
#     ),
#     expected_output=(
#         "Return ONLY a valid Python dictionary.\n"
#         "- Do not include explanations, markdown, or code fences.\n"
#         "- The dictionary must be UTF-8 safe and directly usable in Python with ast.literal_eval.\n"
#         "- Keys must be wrapped in double quotes.\n\n"
#         "Format example:\n"
#         "{\n"
#         '  "unit_title": "من الفكرة إلى نموذج العمل: بناء الأساس الريادي",\n'
#         '  "subtopic_title": "مفهوم ريادة الأعمال وأهميتها الاقتصادية والاجتماعية",\n'
#         '  "query": "دور ريادة الأعمال في التنمية الاجتماعية",\n'
#         '  "parts": [\n'
#         "    {\n"
#         '      "page_url": "https://example.com/page1",\n'
#         '      "title": "Understanding Entrepreneurship in the Arab World",\n'
#         '      "content": "Full educational content extracted from the site.",\n'
#         '      "img_url": ["https://example.com/image1.jpg"],\n'
#         '      "video_url": ["https://example.com/video1.mp4"],\n'
#         '      "audio_url": ["https://example.com/audio1.mp3"],\n'
#         '      "pdf_url": ["https://example.com/file1.pdf"],\n'
#         '      "agent_recommendation_rank": 4.8,\n'
#         '      "agent_recommendation_notes": "Rich Arabic content, relevant to conceptual materials."\n'
#         "    }\n"
#         "  ]\n"
#         "}\n\n"
#         "Make the output compatible with Python's ast library (use r1 = result.dict()['raw']; f_result = ast.literal_eval(r1)).\n"
#         "Ensure valid JSON syntax with no unterminated strings or extra text.\n"
#         "Output ONLY the dictionary — no thoughts, explanations, or markdown formatting."
#     ),
#     agent=scraping_crawlee_agent,
#     output_json=UnitSubtopicOutputModel,
# )
# #