Spaces:
Sleeping
Sleeping
| """AI-powered challenge evaluators (legacy + per-type). | |
| Each ``evaluate_*`` function consumes the request + scenario row, runs | |
| the appropriate AI (or pattern check), and returns the JSON shape the | |
| frontend expects. Where the result can be computed without AI (e.g. a | |
| plain flag check), we do so directly. | |
| """ | |
| import re | |
| from typing import Optional | |
| import httpx | |
| from fastapi import BackgroundTasks, HTTPException | |
| from app.core.constants import CYBER_SECURITY_TOPICS | |
| from app.core.security import ( | |
| normalize_str, | |
| normalize_vuln_key, | |
| ip_matches, | |
| ioc_matches, | |
| timestamp_close, | |
| ) | |
| from app.core.config import GROQ_API_URL, GROQ_API_KEY, MISTRAL_API_URL, MISTRAL_MODEL, MISTRAL_API_KEY | |
| from app.core.text import parse_json_safe | |
| from app.services.supabase_service import fetch_scenario_by_id | |
| from app.services.scenario_service import handle_background_replacement | |
| from app.services.completion_service import record_user_completion | |
| # --------------------------------------------------------------------------- # | |
| # Legacy: red-team / generic / blue-code-fix AI evaluator # | |
| # --------------------------------------------------------------------------- # | |
| async def evaluate_training(req, background_tasks: BackgroundTasks) -> dict: | |
| """The original /api/training/evaluate (red-team payload + blue fix).""" | |
| challenge = req.originalChallenge | |
| user_code = req.userCode | |
| eval_prompt = f"""أنت مهندس أمن سيبراني خبير ومراجع أكواد. | |
| مهمتك: تقييم الكود أو الاستغلال الذي قدمه المستخدم. | |
| دور المستخدم هو: {req.teamRole} | |
| إذا كان دور المستخدم هو "blue" (مدافع): | |
| - إذا كان الكود المعدل يسد الثغرة الأمنية ويحل المشكلة بشكل صحيح، أرجع secured: true. | |
| - إذا لم تحل المشكلة أو كان خاطئاً، أرجع secured: false. | |
| إذا كان دور المستخدم هو "red" (مهاجم): | |
| - إذا كان الكود (أو Payload) يستغل الثغرة بنجاح، أرجع secured: true (نقصد بها النجاح). | |
| - إذا كان الاستغلال فاشلاً، أرجع secured: false. | |
| أرجع JSON فقط: | |
| {{ | |
| "secured": true/false, | |
| "feedback": "تقييمك باللغة العربية" | |
| }} | |
| التحدي الأصلي: | |
| - الثغرة: {challenge.get("vulnerabilityLocation", "")} | |
| - الإجابة المتوقعة: {challenge.get("expectedAnswer", "")} | |
| - الشرح: {challenge.get("explanation", "")} | |
| كود/استغلال المستخدم: | |
| {user_code}""" | |
| async with httpx.AsyncClient(timeout=60) as client: | |
| resp = await client.post( | |
| GROQ_API_URL, | |
| json={ | |
| "model": "llama-3.3-70b-versatile", | |
| "messages": [ | |
| {"role": "system", "content": "أنت مقيّم أكواد أمني. أعد JSON فقط."}, | |
| {"role": "user", "content": eval_prompt}, | |
| ], | |
| "temperature": 0.2, | |
| "max_tokens": 1024, | |
| }, | |
| headers={ | |
| "Content-Type": "application/json", | |
| "Authorization": f"Bearer {GROQ_API_KEY}", | |
| }, | |
| ) | |
| if resp.status_code != 200: | |
| # Log error server-side only, don't leak to client | |
| import logging | |
| logging.error(f"Groq API error: {resp.status_code}") | |
| try: | |
| logging.error(f"Groq API error body: {resp.text[:200]}") | |
| except Exception: | |
| pass | |
| raise HTTPException(status_code=500, detail="AI evaluation service temporarily unavailable") | |
| content = resp.json()["choices"][0]["message"]["content"] | |
| evaluation = parse_json_safe(content) | |
| if evaluation.get("secured") is True: | |
| scenario_id = challenge.get("scenarioId") or challenge.get("id") | |
| if scenario_id: | |
| module_name = challenge.get("type", "") | |
| topic_info = CYBER_SECURITY_TOPICS.get( | |
| module_name, {"path": "cryptography", "category": "encryption"} | |
| ) | |
| path = topic_info.get("path", "cryptography") | |
| category = topic_info.get("category", "encryption") | |
| difficulty = challenge.get("difficulty", "متوسط") | |
| background_tasks.add_task( | |
| handle_background_replacement, | |
| scenario_id, | |
| req.teamRole, | |
| module_name, | |
| path, | |
| category, | |
| difficulty, | |
| ) | |
| return {"evaluation": evaluation} | |
| # --------------------------------------------------------------------------- # | |
| # Code-fix AI evaluator (single source of truth, shared with 1v1) # | |
| # --------------------------------------------------------------------------- # | |
| async def ai_evaluate_code_fix(challenge_id: str, fixed_code: str, team_role: str = "blue") -> dict: | |
| """Single source of truth for both /evaluate-code-fix and 1v1's blue verifier.""" | |
| row = await fetch_scenario_by_id(team_role or "blue", challenge_id, challenge_type="code-fixing") | |
| if not row: | |
| return {"secured": False, "feedback": "التحدي غير موجود"} | |
| vulnerable_code = row.get("vulnerable_code") or "" | |
| vuln_type = row.get("vulnerability_type") or "" | |
| vuln_desc = row.get("vulnerability_description") or "" | |
| language = row.get("language") or "" | |
| eval_prompt = f"""أنت مهندس أمن سيبراني خبير ومراجع أكواد. | |
| مهمتك: تقييم الكود المُعدل الذي قدمه المتدرب لتصحيح ثغرة أمنية. | |
| معلومات التحدي: | |
| - اللغة: {language} | |
| - نوع الثغرة: {vuln_type} | |
| - وصف الثغرة: {vuln_desc} | |
| الكود الأصلي (المصاب بالثغرة): | |
| ```{language} | |
| {vulnerable_code} | |
| ``` | |
| الكود المُعدل من المتدرب: | |
| ```{language} | |
| {fixed_code} | |
| ``` | |
| قيّم الكود المُعدل وتحقق من: | |
| 1. هل الثغرة أُصلحت فعلياً؟ | |
| 2. هل الكود صحيح نحويًا (syntax)؟ | |
| 3. هل الحل يتبع أفضل الممارسات الأمنية؟ | |
| قاعدة صارمة: اكتب حقل "feedback" باللغة العربية فقط. ممنوع منعاً باتاً استخدام أي كلمة بلغة أخرى. | |
| أرجع JSON صالحاً فقط بالشكل التالي (بدون أي نص قبله أو بعده): | |
| {{ | |
| "secured": true/false, | |
| "feedback": "تقييمك المختصر بالعربية فقط يشرح هل الثغرة أُصلحت ولماذا", | |
| "vulnerability_fixed": true/false, | |
| "code_valid": true/false | |
| }}""" | |
| print(f"[ai-evaluate-code-fix] challenge={challenge_id} lang={language}") | |
| async def _call_ai(api_url: str, api_key: str, model: str) -> Optional[dict]: | |
| """Call an AI provider and return parsed evaluation dict, or None on failure.""" | |
| try: | |
| async with httpx.AsyncClient(timeout=60) as client: | |
| resp = await client.post( | |
| api_url, | |
| json={ | |
| "model": model, | |
| "messages": [ | |
| {"role": "system", "content": "أنت مقيّم أكواد أمني. أعد JSON فقط."}, | |
| {"role": "user", "content": eval_prompt}, | |
| ], | |
| "temperature": 0.2, | |
| "max_tokens": 1024, | |
| }, | |
| headers={ | |
| "Content-Type": "application/json", | |
| "Authorization": f"Bearer {api_key}", | |
| }, | |
| ) | |
| except Exception: | |
| return None | |
| if resp.status_code != 200: | |
| print(f"[ai-evaluate-code-fix] {api_url} returned {resp.status_code}") | |
| return None | |
| try: | |
| content = resp.json()["choices"][0]["message"]["content"] | |
| evaluation = parse_json_safe(content) | |
| except Exception: | |
| return None | |
| if not isinstance(evaluation, dict): | |
| return None | |
| return evaluation | |
| # Tier 1: Groq | |
| evaluation = await _call_ai(GROQ_API_URL, GROQ_API_KEY, "llama-3.3-70b-versatile") | |
| if evaluation is not None: | |
| return evaluation | |
| # Tier 2: Mistral (fallback) | |
| print("[ai-evaluate-code-fix] Groq failed, trying Mistral...") | |
| evaluation = await _call_ai(MISTRAL_API_URL, MISTRAL_API_KEY, MISTRAL_MODEL) | |
| if evaluation is not None: | |
| return evaluation | |
| return {"secured": False, "feedback": "AI evaluation error: 401"} | |
| async def evaluate_code_fix(req, background_tasks: BackgroundTasks) -> dict: | |
| """Public /evaluate-code-fix handler (regular training).""" | |
| challenge_id = req.challengeId | |
| fixed_code = req.fixedCode | |
| if not challenge_id: | |
| return {"success": False, "error": "challengeId مفقود"} | |
| if not fixed_code or not fixed_code.strip(): | |
| return {"success": False, "error": "لم تُرسل كود مُعدل"} | |
| evaluation = await ai_evaluate_code_fix(challenge_id, fixed_code, req.teamRole or "blue") | |
| if "error" in evaluation and "secured" not in evaluation: | |
| return {"success": False, "error": evaluation.get("error", "AI evaluation failed")} | |
| if evaluation.get("secured") is True: | |
| row = await fetch_scenario_by_id(req.teamRole or "blue", challenge_id, challenge_type="code-fixing") | |
| if row: | |
| background_tasks.add_task( | |
| handle_background_replacement, | |
| challenge_id, | |
| req.teamRole or "blue", | |
| row.get("module", ""), | |
| "cryptography", | |
| row.get("difficulty", "متوسط"), | |
| row.get("difficulty", "متوسط"), | |
| ) | |
| background_tasks.add_task( | |
| record_user_completion, | |
| req.userId, | |
| req.teamRole or "blue", | |
| "code-fixing", | |
| row.get("module", ""), | |
| challenge_id, | |
| int(row.get("xp_reward") or 150), | |
| ) | |
| return {"evaluation": evaluation} | |
| # --------------------------------------------------------------------------- # | |
| # Log analysis — exact 4-field match + AI feedback via Mistral # | |
| # --------------------------------------------------------------------------- # | |
| async def evaluate_log_analysis(req, background_tasks: BackgroundTasks) -> dict: | |
| challenge_id = req.challengeId | |
| if not challenge_id: | |
| return {"success": False, "error": "challengeId مفقود"} | |
| row = await fetch_scenario_by_id(req.teamRole or "blue", challenge_id, challenge_type="log-analysis") | |
| if not row: | |
| return {"success": False, "error": "التحدي غير موجود"} | |
| expected_attack = normalize_str(row.get("expected_attack_type", "")) | |
| expected_ip = row.get("expected_attacker_ip") or "" | |
| expected_ts = row.get("expected_timestamp") or "" | |
| expected_ioc = row.get("expected_ioc") or "" | |
| correct_fields = [] | |
| if normalize_str(req.attackType) == expected_attack: | |
| correct_fields.append("نوع الهجوم") | |
| if ip_matches(req.attackerIp, expected_ip): | |
| correct_fields.append("عنوان IP المهاجم") | |
| if timestamp_close(req.timestamp, expected_ts): | |
| correct_fields.append("الطابع الزمني") | |
| if ioc_matches(req.ioc, expected_ioc): | |
| correct_fields.append("مؤشر الاختراق (IOC)") | |
| total = 4 | |
| score = int(len(correct_fields) * 100 / total) | |
| passed = score >= 75 # need 3 of 4 | |
| # AI feedback (Arabic-only) via Mistral | |
| feedback_text = await _log_analysis_ai_feedback( | |
| row, expected_attack, expected_ip, expected_ts, expected_ioc, | |
| req, correct_fields, total, passed, | |
| ) | |
| if not feedback_text: | |
| if passed: | |
| feedback_text = f"تحليل ممتاز! حددت {len(correct_fields)} من {total} حقول بشكل صحيح. {row.get('vulnerability_description', '')}" | |
| else: | |
| missed = [f for f in ["نوع الهجوم", "عنوان IP المهاجم", "الطابع الزمني", "مؤشر الاختراق (IOC)"] if f not in correct_fields] | |
| feedback_text = f"تم تحديد {len(correct_fields)} من {total} حقول. الحقول التي تحتاج مراجعة: {', '.join(missed)}. راجع السجل مرة أخرى وابحث عن الأنماط المشبوهة." | |
| xp_awarded = int((row.get("xp_reward") or 150) * score / 100) | |
| if passed: | |
| background_tasks.add_task( | |
| handle_background_replacement, | |
| challenge_id, | |
| req.teamRole or "blue", | |
| row.get("module", "forensics"), | |
| "forensics", | |
| row.get("difficulty", "متوسط"), | |
| row.get("difficulty", "متوسط"), | |
| ) | |
| background_tasks.add_task( | |
| record_user_completion, | |
| req.userId, | |
| req.teamRole or "blue", | |
| "log-analysis", | |
| row.get("module", "forensics"), | |
| challenge_id, | |
| xp_awarded, | |
| ) | |
| return { | |
| "evaluation": { | |
| "passed": passed, | |
| "score": score, | |
| "correct_fields": correct_fields, | |
| "feedback": feedback_text, | |
| "xp_awarded": xp_awarded, | |
| } | |
| } | |
| async def _log_analysis_ai_feedback( | |
| row, expected_attack, expected_ip, expected_ts, expected_ioc, | |
| req, correct_fields, total, passed, | |
| ) -> str: | |
| """Use Mistral to generate a short Arabic SOC-style feedback paragraph.""" | |
| correct_str = "، ".join(correct_fields) if correct_fields else "لا شيء" | |
| missed = [f for f in ["نوع الهجوم", "عنوان IP المهاجم", "الطابع الزمني", "مؤشر الاختراق (IOC)"] if f not in correct_fields] | |
| missed_str = "، ".join(missed) if missed else "لا شيء" | |
| feedback_prompt = f"""أنت محلل خبير في مركز عمليات الأمن السيبراني (SOC). راجع إجابة المتدرب. | |
| ═══════════════════════════════════════ | |
| قواعد اللغة (مطلوبة بصرامة): | |
| - اكتب بالعربية الفصحى فقط، بدون أي كلمات إنجليزية. | |
| - ممنوع منعاً باتاً استخدام: nor، and، or، the، to، of، in، على الإطلاق. | |
| - استخدم: و، أو، ثم، لكن، بل، لأن. | |
| - لا تذكر أسماء حقول تقنية (attack_type، IP، timestamp) — استخدم الأسماء العربية فقط. | |
| ═══════════════════════════════════════ | |
| معلومات التحدي: | |
| - العنوان: {row.get("title", "")} | |
| - نوع السجل: {row.get("log_type", "")} | |
| - نوع الهجوم الصحيح: {expected_attack} | |
| - عنوان الـ IP الصحيح: {expected_ip} | |
| - الطابع الزمني الصحيح: {expected_ts} | |
| - مؤشر الاختراق الصحيح: {expected_ioc} | |
| إجابة المتدرب: | |
| - نوع الهجوم: {req.attackType or "(فارغ)"} | |
| - عنوان الـ IP: {req.attackerIp or "(فارغ)"} | |
| - الطابع الزمني: {req.timestamp or "(فارغ)"} | |
| - مؤشر الاختراق: {req.ioc or "(فارغ)"} | |
| - التحليل الحر: {req.explanation or "(لم يكتب شيئاً)"} | |
| نتيجة التقييم: {len(correct_fields)} من {total} حقول صحيحة. | |
| الحقول الصحيحة: {correct_str} | |
| الحقول الخاطئة أو الفارغة: {missed_str} | |
| ═══════════════════════════════════════ | |
| التعليمات: | |
| - اكتب فقرة واحدة إلى ثلاث فقرات قصيرة بالعربية فقط. | |
| - ابدأ بجملة افتتاحية تصف النتيجة (مثلاً: "أصبت في X من Y" أو "لم تتمكن من كشف أي حقل بشكل صحيح"). | |
| - إذا كانت النتيجة 3 أو 4 من 4: امدح المتدرب وأضف سياقاً أمنياً مختصراً. | |
| - إذا كانت النتيجة أقل من 3: وضّح الحقول التي أخطأ فيها وقل كيف يكتشفها مستقبلاً، مع ربط بالإجابة الصحيحة. | |
| - لا تتجاوز 150 كلمة. | |
| ═══════════════════════════════════════ | |
| أرجع JSON فقط بدون أي شرح إضافي: | |
| {{"feedback": "النص هنا"}}""" | |
| try: | |
| async with httpx.AsyncClient(timeout=60) as client: | |
| fb_resp = await client.post( | |
| MISTRAL_API_URL, | |
| json={ | |
| "model": MISTRAL_MODEL, | |
| "messages": [ | |
| {"role": "system", "content": "أنت محلل خبير في مركز عمليات الأمن السيبراني (SOC). أرجع JSON فقط، اكتب بالعربية الفصحى حصراً، ولا تستخدم أي كلمات إنجليزية على الإطلاق."}, | |
| {"role": "user", "content": feedback_prompt}, | |
| ], | |
| "temperature": 0.3, | |
| "max_tokens": 1024, | |
| "response_format": {"type": "json_object"}, | |
| }, | |
| headers={ | |
| "Content-Type": "application/json", | |
| "Authorization": f"Bearer {MISTRAL_API_KEY}", | |
| }, | |
| ) | |
| if fb_resp.status_code == 200: | |
| fb_content = fb_resp.json()["choices"][0]["message"]["content"] | |
| fb_data = parse_json_safe(fb_content) | |
| return fb_data.get("feedback", "") | |
| print(f"[log-analysis eval] Mistral HTTP {fb_resp.status_code}: {fb_resp.text[:200]}") | |
| except Exception as e: | |
| print(f"[log-analysis eval] AI feedback failed: {e}") | |
| return "" | |
| # --------------------------------------------------------------------------- # | |
| # Vulnerability hunter — exact canonical-key match # | |
| # --------------------------------------------------------------------------- # | |
| # Common alias → canonical form. Pre-filters well-known abbreviations so | |
| # we don't pay for a Mistral call when the user types "sqli" or "xss". | |
| # If the alias matches, the answer is treated as exact-match (full XP). | |
| _VULN_ALIASES: dict[str, str] = { | |
| "sqli": "sql-injection", | |
| "sql": "sql-injection", | |
| "sqlinj": "sql-injection", | |
| "xss": "xss", | |
| "crosssitescripting": "xss", | |
| "csrf": "csrf", | |
| "xsrf": "csrf", | |
| "ssrf": "ssrf", | |
| "xxe": "xxe", | |
| "idor": "idor", | |
| "lfi": "local-file-inclusion", | |
| "rfi": "remote-file-inclusion", | |
| "rce": "remote-code-execution", | |
| "lpe": "local-privilege-escalation", | |
| "privesc": "privilege-escalation", | |
| "bof": "buffer-overflow", | |
| "uaf": "use-after-free", | |
| "formatstring": "format-string", | |
| "integeroverflow": "integer-overflow", | |
| "npd": "null-pointer-dereference", | |
| "doublefree": "double-free", | |
| "offbyone": "off-by-one", | |
| "uninit": "uninitialized-memory", | |
| "race": "race-condition", | |
| "toctou": "time-of-check-time-of-use", | |
| "proto": "prototype-pollution", | |
| "deserialization": "insecure-deserialization", | |
| "misconfig": "security-misconfiguration", | |
| "bac": "broken-access-control", | |
| "openredirect": "open-redirect", | |
| "ssti": "server-side-template-injection", | |
| } | |
| def _alias_resolve(s: str) -> str: | |
| """Lowercase + strip dashes/spaces. Returns the normalised token or the | |
| canonical alias if it matches ``_VULN_ALIASES``.""" | |
| if not s: | |
| return "" | |
| norm = re.sub(r"[\s_\-]+", "", (s or "").lower().strip()) | |
| return _VULN_ALIASES.get(norm, norm) | |
| async def _grade_vuln_hunter_with_mistral( | |
| user_answer: str, | |
| expected: str, | |
| ) -> dict: | |
| """Ask Mistral to grade the student's vuln-name answer semantically. | |
| Returns ``{similarity, is_match, reason, xp_ratio}``. ``xp_ratio`` is | |
| clamped between 0.4 and 1.0 when ``is_match`` is true (so partial | |
| credit is always at least 40% of the reward). Returns | |
| ``{"similarity": 0.0, "is_match": False, "reason": "...", "xp_ratio": 0}`` | |
| on any Mistral error so the caller can fall back to a hard fail. | |
| """ | |
| if not MISTRAL_API_KEY: | |
| return {"similarity": 0.0, "is_match": False, "reason": "Mistral not configured", "xp_ratio": 0} | |
| system_prompt = ( | |
| "You are a senior cybersecurity instructor grading a student's answer " | |
| "in an Arabic-first training platform. The student is asked to name a " | |
| "specific vulnerability class shown in a code exhibit. Compare the " | |
| "STUDENT_ANSWER to the EXPECTED_ANSWER and decide if they refer to the " | |
| "same vulnerability.\n\n" | |
| "Aliases are equivalent (e.g. 'sqli' = 'sql-injection', " | |
| "'xss' = 'cross-site-scripting', 'bof' = 'buffer-overflow', " | |
| "'rce' = 'remote-code-execution'). Different specific attack families " | |
| "are NOT the same (e.g. 'sql-injection' is NOT 'xss').\n\n" | |
| "Return raw JSON only, no markdown:\n" | |
| '{"similarity": 0.0-1.0, "is_match": true|false, "reason": "..."}' | |
| ) | |
| user_prompt = ( | |
| f"EXPECTED_ANSWER: {expected}\n" | |
| f"STUDENT_ANSWER: {user_answer}\n\n" | |
| "Are these the same vulnerability? Grade it." | |
| ) | |
| try: | |
| async with httpx.AsyncClient(timeout=20) as client: | |
| r = await client.post( | |
| MISTRAL_API_URL, | |
| json={ | |
| "model": MISTRAL_MODEL, | |
| "messages": [ | |
| {"role": "system", "content": system_prompt}, | |
| {"role": "user", "content": user_prompt}, | |
| ], | |
| "temperature": 0.0, | |
| "max_tokens": 300, | |
| "response_format": {"type": "json_object"}, | |
| }, | |
| headers={ | |
| "Content-Type": "application/json", | |
| "Authorization": f"Bearer {MISTRAL_API_KEY}", | |
| }, | |
| ) | |
| if r.status_code != 200: | |
| return {"similarity": 0.0, "is_match": False, "reason": f"Mistral HTTP {r.status_code}", "xp_ratio": 0} | |
| content = r.json().get("choices", [{}])[0].get("message", {}).get("content", "") | |
| data = parse_json_safe(content) or {} | |
| similarity = float(data.get("similarity", 0) or 0) | |
| is_match = bool(data.get("is_match", False)) or similarity >= 0.65 | |
| reason = str(data.get("reason", "")).strip() or "تم التقييم بواسطة Mistral" | |
| xp_ratio = 0 | |
| if is_match: | |
| xp_ratio = max(0.4, min(1.0, similarity)) if similarity >= 0.95 else max(0.4, min(0.85, similarity * 0.9)) | |
| return {"similarity": round(similarity, 3), "is_match": is_match, "reason": reason, "xp_ratio": xp_ratio} | |
| except Exception as e: | |
| return {"similarity": 0.0, "is_match": False, "reason": f"Mistral error: {e}", "xp_ratio": 0} | |
| async def evaluate_vuln_hunter(req, background_tasks: BackgroundTasks) -> dict: | |
| challenge_id = req.challengeId | |
| if not challenge_id: | |
| return {"success": False, "error": "challengeId مفقود"} | |
| if not (req.vulnerabilityType or "").strip(): | |
| return {"success": False, "error": "لم تُرسل اسم الثغرة"} | |
| row = await fetch_scenario_by_id(req.teamRole or "blue", challenge_id, challenge_type="vulnerability-hunter") | |
| if not row: | |
| return {"success": False, "error": "التحدي غير موجود"} | |
| expected = normalize_vuln_key(row.get("vulnerability_type", "")) | |
| user = normalize_vuln_key(req.vulnerabilityType) | |
| # Also resolve common aliases ("sqli" → "sql-injection", "xss" → "xss"). | |
| # This catches well-known abbreviations without paying for a Mistral call. | |
| user_resolved = _alias_resolve(req.vulnerabilityType) | |
| expected_resolved = _alias_resolve(row.get("vulnerability_type", "")) | |
| correct = (bool(expected) and (user == expected)) or ( | |
| bool(expected_resolved) and (user_resolved == expected_resolved) | |
| ) | |
| xp_reward = int(row.get("xp_reward") or 150) | |
| ai_grade: dict = {"similarity": 0.0, "is_match": False, "reason": "", "xp_ratio": 0} | |
| if not correct: | |
| # Ask Mistral to compare semantically when the alias/exact match fails. | |
| # This recognises near-misses and ambiguous phrasing so the student | |
| # gets partial credit instead of a hard fail. | |
| ai_grade = await _grade_vuln_hunter_with_mistral( | |
| req.vulnerabilityType, row.get("vulnerability_type", "") | |
| ) | |
| passed = correct or ai_grade["is_match"] | |
| if ai_grade["is_match"]: | |
| xp_awarded = int(xp_reward * ai_grade["xp_ratio"]) | |
| elif correct: | |
| xp_awarded = xp_reward | |
| else: | |
| xp_awarded = 0 | |
| if passed: | |
| background_tasks.add_task( | |
| handle_background_replacement, | |
| challenge_id, | |
| req.teamRole or "blue", | |
| row.get("module", "vulnerability-hunter"), | |
| "cryptography", | |
| row.get("difficulty", "متوسط"), | |
| row.get("difficulty", "متوسط"), | |
| ) | |
| background_tasks.add_task( | |
| record_user_completion, | |
| req.userId, | |
| req.teamRole or "blue", | |
| "vulnerability-hunter", | |
| row.get("module", "vulnerability-hunter"), | |
| challenge_id, | |
| xp_awarded, | |
| ) | |
| if correct: | |
| feedback = ( | |
| "إجابة صحيحة! الثغرة هي فعلاً " | |
| f"{expected}. {row.get('vulnerability_description', '')}" | |
| ).strip() | |
| elif ai_grade["is_match"]: | |
| feedback = ( | |
| f"إجابة قريبة ({int(ai_grade['similarity'] * 100)}% تشابه). " | |
| f"الثغرة المتوقعة: {row.get('vulnerability_type', 'غير معروف')}. " | |
| f"تم احتساب {xp_awarded} من {xp_reward} XP. {ai_grade.get('reason', '')}" | |
| ).strip() | |
| else: | |
| feedback = ( | |
| f"الثغرة المتوقعة هي: {row.get('vulnerability_type', 'غير معروف')}. " | |
| f"{row.get('vulnerability_description', '')}" | |
| ).strip() | |
| return { | |
| "success": True, | |
| "evaluation": { | |
| "passed": passed, | |
| "score": int((100 if correct else (ai_grade["similarity"] * 100)) if passed else 0), | |
| "feedback": feedback, | |
| "xp_awarded": xp_awarded, | |
| "similarity": ai_grade["similarity"], | |
| "ai_graded": ai_grade["is_match"] and not correct, | |
| "ai_reason": ai_grade.get("reason", ""), | |
| "vulnerability_type": row.get("vulnerability_type", ""), | |
| "vulnerability_class": row.get("vulnerability_class", ""), | |
| }, | |
| } | |
| # --------------------------------------------------------------------------- # | |
| # Web Exploitation — flag + AI evaluation (Red Team) # | |
| # --------------------------------------------------------------------------- # | |
| async def _grade_web_exploit_with_mistral( | |
| payload: str, | |
| vuln_type: str, | |
| vuln_description: str, | |
| http_request: str, | |
| http_response: str, | |
| expected_flag: str, | |
| ) -> dict: | |
| """Ask Mistral to evaluate whether the submitted payload successfully | |
| exploits the web vulnerability described in the challenge. | |
| Returns ``{"secured": True|False, "feedback": "..."}``. On any Mistral | |
| error, returns a hard fail so the student can retry. | |
| """ | |
| if not MISTRAL_API_KEY: | |
| return {"secured": False, "feedback": "Mistral غير مهيأ للتقييم"} | |
| system_prompt = ( | |
| "أنت مصحح آلي صارم لتحديات اختبار الاختراق. مهمتك الوحيدة: تحديد هل حمولة الطالب " | |
| "تستغل الثغرة المطلوبة بنجاح أم لا. قرارك ثنائي فقط: true أو false.\n\n" | |
| "قواعد صارمة:\n" | |
| "1. إذا كانت الحمولة قادرة على استغلال الثغرة الموضحة (بغض النظر عن محتوى العلم)، " | |
| "فأرجع secured: true.\n" | |
| "2. إذا كانت الحمولة لا تستغل الثغرة أو غير ذات صلة، فأرجع secured: false.\n" | |
| "3. لا يوجد خيار ثالث. لا تكن متردداً. لا تطلب اختباراً عملياً.\n" | |
| "4. secured: true = الحمولة صحيحة وتستغل الثغرة. secured: false = الحمولة خاطئة.\n\n" | |
| "أرجع JSON فقط:\n" | |
| '{"secured": true, "feedback": "إجابة صحيحة! الحمولة تستغل الثغرة بنجاح."}\n' | |
| '{"secured": false, "feedback": "...سبب الخطأ..."}' | |
| ) | |
| user_prompt = ( | |
| f"نوع الثغرة: {vuln_type}\n" | |
| f"وصف الثغرة: {vuln_description}\n" | |
| f"الطلب الأصلي (HTTP Request):\n{http_request}\n\n" | |
| f"الاستجابة الأصلية (HTTP Response):\n{http_response}\n\n" | |
| f"العلم المتوقع: {expected_flag}\n" | |
| f"---\n" | |
| f"حمولة الطالب: {payload}\n\n" | |
| "هل تستغل هذه الحمولة الثغرة بنجاح؟ أعد true أو false فقط." | |
| ) | |
| try: | |
| async with httpx.AsyncClient(timeout=25) as client: | |
| r = await client.post( | |
| MISTRAL_API_URL, | |
| json={ | |
| "model": MISTRAL_MODEL, | |
| "messages": [ | |
| {"role": "system", "content": system_prompt}, | |
| {"role": "user", "content": user_prompt}, | |
| ], | |
| "temperature": 0.0, | |
| "max_tokens": 400, | |
| "response_format": {"type": "json_object"}, | |
| }, | |
| headers={ | |
| "Content-Type": "application/json", | |
| "Authorization": f"Bearer {MISTRAL_API_KEY}", | |
| }, | |
| ) | |
| if r.status_code != 200: | |
| return {"secured": False, "feedback": f"Mistral HTTP {r.status_code}"} | |
| content = r.json().get("choices", [{}])[0].get("message", {}).get("content", "") | |
| data = parse_json_safe(content) or {} | |
| secured = bool(data.get("secured", False)) | |
| feedback = str(data.get("feedback", "")).strip() or "تم التقييم بواسطة Mistral" | |
| return {"secured": secured, "feedback": feedback} | |
| except Exception as e: | |
| return {"secured": False, "feedback": f"خطأ في التقييم: {e}"} | |
| async def ai_evaluate_web_exploit( | |
| challenge_id: str, payload: str, team_role: str = "red" | |
| ) -> dict: | |
| """Evaluate a web exploitation payload submission — flag match first, then AI.""" | |
| row = await fetch_scenario_by_id( | |
| team_role, challenge_id, challenge_type="web-exploitation" | |
| ) | |
| if not row: | |
| return {"secured": False, "feedback": "التحدي غير موجود"} | |
| expected = row.get("flag_preview", "") | |
| if not expected: | |
| return {"secured": False, "feedback": "لا يوجد علم مرجعي لهذا التحدي"} | |
| # Normalize: strip CyberArena{} wrapper, lowercase, strip whitespace | |
| def _clean(s: str) -> str: | |
| s = (s or "").strip() | |
| if s.startswith("CyberArena{") and s.endswith("}"): | |
| s = s[len("CyberArena{"):-1] | |
| if s.startswith("CyberArena{") and s.endswith("}"): | |
| s = s[len("CyberArena{"):-1] | |
| return s.lower().strip() | |
| user_clean = _clean(payload) | |
| expected_clean = _clean(expected) | |
| correct = bool(user_clean) and bool(expected_clean) and ( | |
| user_clean == expected_clean | |
| or user_clean in expected_clean | |
| or expected_clean in user_clean | |
| ) | |
| if correct: | |
| return { | |
| "secured": True, | |
| "feedback": "إجابة صحيحة! تم استغلال الثغرة بنجاح.", | |
| } | |
| # Flag didn't match — fall back to AI semantic evaluation | |
| return await _grade_web_exploit_with_mistral( | |
| payload=payload, | |
| vuln_type=row.get("vulnerability_type", ""), | |
| vuln_description=row.get("vulnerability_description", ""), | |
| http_request=row.get("http_request", ""), | |
| http_response=row.get("http_response", ""), | |
| expected_flag=expected, | |
| ) | |
| async def evaluate_web_exploit(req, background_tasks: BackgroundTasks) -> dict: | |
| """Public /evaluate-web-exploit handler (regular training).""" | |
| challenge_id = req.challengeId | |
| payload = req.payload | |
| if not challenge_id: | |
| return {"success": False, "error": "challengeId مفقود"} | |
| if not payload or not str(payload).strip(): | |
| return {"success": False, "error": "لم ترسل الحمولة أو العلم"} | |
| evaluation = await ai_evaluate_web_exploit( | |
| challenge_id, str(payload), req.teamRole or "red" | |
| ) | |
| if evaluation.get("secured") is True: | |
| row = await fetch_scenario_by_id( | |
| req.teamRole or "red", challenge_id, challenge_type="web-exploitation" | |
| ) | |
| if row: | |
| background_tasks.add_task( | |
| handle_background_replacement, | |
| challenge_id, | |
| req.teamRole or "red", | |
| row.get("module", "web-exploitation"), | |
| row.get("topic", "xss"), | |
| row.get("difficulty", "متوسط"), | |
| row.get("difficulty", "متوسط"), | |
| ) | |
| background_tasks.add_task( | |
| record_user_completion, | |
| req.userId, | |
| req.teamRole or "red", | |
| "web-exploitation", | |
| row.get("topic", "xss"), | |
| challenge_id, | |
| int(row.get("xp_reward") or 200), | |
| ) | |
| return {"evaluation": evaluation} | |