Spaces:
Running
Running
| """ | |
| Python Code Evaluator — SE-Group1 | |
| COS60011 Technology Design Project | |
| """ | |
| # --------------------------------------------------------------------------- | |
| # IMPORTS (all at top) | |
| # --------------------------------------------------------------------------- | |
| import os | |
| import re | |
| import ast | |
| import time | |
| from collections import deque | |
| from dotenv import load_dotenv | |
| import gradio as gr | |
| from google import genai | |
| load_dotenv() | |
| # --------------------------------------------------------------------------- | |
| # MODULE 3 — Pre-processing Module | |
| # --------------------------------------------------------------------------- | |
| def build_prompt(description: str, code: str) -> str: | |
| prompt = f"""<start_of_turn>user | |
| ### ROLE | |
| You are an expert Python code reviewer. Your sole task is to determine whether the | |
| provided Python code correctly implements the behaviour described in the user's | |
| requirements description. | |
| ### CONTEXT | |
| Developers sometimes write code that does not fully satisfy the requirements they | |
| were given. You will analyse the semantic relationship between a natural-language | |
| description and a Python code snippet, then produce a structured evaluation report. | |
| ### INPUT DATA | |
| **Requirements Description:** | |
| {description.strip()} | |
| **Python Code:** | |
| ```python | |
| {code.strip()} | |
| ``` | |
| ### TASK | |
| 1. Read the requirements description carefully. | |
| 2. Analyse the Python code line by line. | |
| 3. Determine whether the code fulfils ALL requirements stated in the description. | |
| 4. Estimate an accuracy percentage (0–100) reflecting how completely the code | |
| matches the description. | |
| 5. List any specific requirements that are missing or incorrectly implemented. | |
| ### CONSTRAINTS | |
| - Do NOT execute the code. | |
| - Base your evaluation solely on static code analysis and logical reasoning. | |
| - Keep feedback concise, clear, and actionable. | |
| - Your response MUST follow the output format exactly. | |
| ### OUTPUT FORMAT | |
| Respond only with the following structure — no extra text before or after: | |
| RESULT: <PASS or FAIL> | |
| ACCURACY: <integer 0-100>% | |
| SUMMARY: <one sentence overall assessment> | |
| ISSUES: | |
| - <issue 1, or "None" if code fully matches the description> | |
| - <issue 2> | |
| ... | |
| <end_of_turn> | |
| <start_of_turn>model | |
| """ | |
| return prompt | |
| # --------------------------------------------------------------------------- | |
| # MODULE 4 — Generation Module | |
| # --------------------------------------------------------------------------- | |
| MODEL_ID = "gemma-4-31b-it" | |
| GOOGLE_API_KEY = os.environ.get("GOOGLE_API_KEY") | |
| def generate_response(prompt: str) -> str: | |
| if not GOOGLE_API_KEY: | |
| return ( | |
| "⚠️ Error: API Key not found! " | |
| "Please configure GOOGLE_API_KEY in Settings → Variables and secrets." | |
| ) | |
| try: | |
| client = genai.Client(api_key=GOOGLE_API_KEY) | |
| response = client.models.generate_content(model=MODEL_ID, contents=prompt) | |
| return response.text | |
| except Exception as e: | |
| return f"❌ An error occurred during generation: {str(e)}" | |
| # --------------------------------------------------------------------------- | |
| # MODULE 5 — Output Module | |
| # --------------------------------------------------------------------------- | |
| def parse_output(raw: str) -> dict: | |
| if "<start_of_turn>model" in raw: | |
| raw = raw.split("<start_of_turn>model")[-1] | |
| result_match = re.search(r"RESULT:\s*(PASS|FAIL)", raw, re.IGNORECASE) | |
| accuracy_match = re.search(r"ACCURACY:\s*(\d{1,3})%?", raw, re.IGNORECASE) | |
| summary_match = re.search(r"SUMMARY:\s*(.+)", raw, re.IGNORECASE) | |
| issues_match = re.search(r"ISSUES:\s*([\s\S]+)", raw, re.IGNORECASE) | |
| result = result_match.group(1).upper() if result_match else "UNKNOWN" | |
| accuracy = int(accuracy_match.group(1)) if accuracy_match else -1 | |
| summary = summary_match.group(1).strip() if summary_match else "Could not extract summary." | |
| if issues_match: | |
| raw_issues = issues_match.group(1).strip() | |
| issues = [ | |
| line.lstrip("-•* ").strip() | |
| for line in raw_issues.splitlines() | |
| if line.strip() and line.strip() not in ("-", "•") | |
| ] | |
| else: | |
| issues = ["Could not extract issues from model response."] | |
| return { | |
| "result": result, | |
| "accuracy": accuracy, | |
| "summary": summary, | |
| "issues": issues, | |
| "raw": raw.strip(), | |
| } | |
| def format_for_display(parsed: dict) -> tuple[str, str, str]: | |
| emoji = "✅" if parsed["result"] == "PASS" else ("❌" if parsed["result"] == "FAIL" else "⚠️") | |
| verdict = f"{emoji} {parsed['result']}" | |
| acc_str = f"{parsed['accuracy']}%" if parsed["accuracy"] >= 0 else "N/A" | |
| metrics = f"Accuracy: {acc_str}\n\nSummary: {parsed['summary']}" | |
| issues_text = "\n".join(f"• {issue}" for issue in parsed["issues"]) | |
| return verdict, metrics, issues_text | |
| # --------------------------------------------------------------------------- | |
| # MODULE 2 — Validation & Flow Management Module | |
| # --------------------------------------------------------------------------- | |
| MIN_DESCRIPTION_CHARS = 20 | |
| MAX_DESCRIPTION_CHARS = 3000 | |
| MIN_CODE_CHARS = 10 | |
| MAX_CODE_CHARS = 8000 | |
| MAX_CODE_LINES = 300 | |
| FORBIDDEN_PATTERNS = [ | |
| r"ignore (all |previous |above )?instructions", | |
| r"disregard (all |previous |above )?instructions", | |
| r"you are now", | |
| r"act as (a |an )?", | |
| r"<\s*(script|iframe|object|embed)", | |
| r"system\s*prompt", | |
| r"jailbreak", | |
| ] | |
| PYTHON_KEYWORDS = { | |
| "def", "class", "import", "from", "return", "if", "else", "elif", | |
| "for", "while", "try", "except", "with", "lambda", "yield", "pass", | |
| "raise", "assert", "in", "not", "and", "or", "True", "False", "None", | |
| "print", "len", "range", "self", | |
| } | |
| class RateLimiter: | |
| def __init__(self, max_calls: int = 5, window_seconds: int = 60): | |
| self.max_calls = max_calls | |
| self.window_seconds = window_seconds | |
| self._timestamps: deque = deque() | |
| def is_allowed(self) -> tuple[bool, str]: | |
| now = time.time() | |
| while self._timestamps and now - self._timestamps[0] > self.window_seconds: | |
| self._timestamps.popleft() | |
| if len(self._timestamps) >= self.max_calls: | |
| wait = int(self.window_seconds - (now - self._timestamps[0])) + 1 | |
| return False, ( | |
| f"⏳ Rate limit reached — {self.max_calls} requests in " | |
| f"{self.window_seconds}s. Please wait ~{wait}s and try again." | |
| ) | |
| self._timestamps.append(now) | |
| return True, "" | |
| _rate_limiter = RateLimiter(max_calls=5, window_seconds=60) | |
| def _check_forbidden(text: str) -> str | None: | |
| lower = text.lower() | |
| for pattern in FORBIDDEN_PATTERNS: | |
| if re.search(pattern, lower): | |
| return ( | |
| "Input contains disallowed content. " | |
| "Please remove prompt-injection or HTML patterns and try again." | |
| ) | |
| return None | |
| def _looks_like_python(code: str) -> tuple[bool, str]: | |
| tokens = set(re.findall(r"[A-Za-z_]\w*", code)) | |
| if not tokens.intersection(PYTHON_KEYWORDS): | |
| return False, ( | |
| "🐍 The code doesn't appear to be Python — no recognisable Python " | |
| "keywords found (e.g. def, class, import, return). " | |
| "Please submit Python code only." | |
| ) | |
| try: | |
| ast.parse(code) | |
| except SyntaxError as exc: | |
| line_hint = f" (line {exc.lineno})" if exc.lineno else "" | |
| return False, ( | |
| f"🐍 Python syntax error{line_hint}: {exc.msg}. " | |
| "Please fix the syntax error before evaluating." | |
| ) | |
| return True, "" | |
| def validate_inputs(description: str, code: str) -> list[str]: | |
| errors: list[str] = [] | |
| if not description or not description.strip(): | |
| errors.append("📋 Requirements description is required.") | |
| if not code or not code.strip(): | |
| errors.append("🐍 Python code is required.") | |
| if errors: | |
| return errors | |
| desc, code_ = description.strip(), code.strip() | |
| if len(desc) < MIN_DESCRIPTION_CHARS: | |
| errors.append(f"📋 Description too short ({len(desc)} chars) — minimum is {MIN_DESCRIPTION_CHARS} characters.") | |
| if len(code_) < MIN_CODE_CHARS: | |
| errors.append(f"🐍 Code too short ({len(code_)} chars) — minimum is {MIN_CODE_CHARS} characters.") | |
| if len(desc) > MAX_DESCRIPTION_CHARS: | |
| errors.append(f"📋 Description too long ({len(desc):,} chars) — max is {MAX_DESCRIPTION_CHARS:,} characters.") | |
| if len(code_) > MAX_CODE_CHARS: | |
| errors.append(f"🐍 Code too long ({len(code_):,} chars) — max is {MAX_CODE_CHARS:,} characters.") | |
| if len(code_.splitlines()) > MAX_CODE_LINES: | |
| errors.append(f"🐍 Code has too many lines ({len(code_.splitlines())}) — max is {MAX_CODE_LINES} lines.") | |
| if err := _check_forbidden(desc): | |
| errors.append(f"📋 {err}") | |
| if err := _check_forbidden(code_): | |
| errors.append(f"🐍 {err}") | |
| if not errors: | |
| is_python, py_error = _looks_like_python(code_) | |
| if not is_python: | |
| errors.append(py_error) | |
| if not errors: | |
| allowed, rate_msg = _rate_limiter.is_allowed() | |
| if not allowed: | |
| errors.append(rate_msg) | |
| return errors | |
| def validate_and_evaluate(description: str, code: str): | |
| errors = validate_inputs(description, code) | |
| if errors: | |
| return "", "", "", "\n".join(f"{i+1}. {e}" for i, e in enumerate(errors)) | |
| if not GOOGLE_API_KEY: | |
| return "", "", "", "❌ GOOGLE_API_KEY secret not set in Space Settings." | |
| prompt = build_prompt(description, code) | |
| try: | |
| raw_output = generate_response(prompt) | |
| except Exception as exc: | |
| return "", "", "", f"❌ Model error: {exc}" | |
| parsed = parse_output(raw_output) | |
| verdict, metrics, issues = format_for_display(parsed) | |
| return verdict, metrics, issues, "" | |
| # --------------------------------------------------------------------------- | |
| # MODULE 1 — UI Module | |
| # --------------------------------------------------------------------------- | |
| CUSTOM_CSS = """ | |
| @import url('https://fonts.googleapis.com/css2?family=Space+Mono:wght@400;700&family=DM+Sans:ital,wght@0,300;0,400;0,500;0,600;1,400&display=swap'); | |
| *, *::before, *::after { box-sizing: border-box; } | |
| body, .gradio-container { | |
| font-family: 'DM Sans', sans-serif !important; | |
| background: #0D0F14 !important; | |
| color: #E8EAF0 !important; | |
| } | |
| .gradio-container { | |
| max-width: 1140px !important; | |
| margin: 0 auto !important; | |
| padding: 32px 24px !important; | |
| } | |
| .eval-header { | |
| display: flex; | |
| align-items: center; | |
| gap: 16px; | |
| padding: 0 0 28px; | |
| border-bottom: 1px solid #2E3140; | |
| margin-bottom: 28px; | |
| } | |
| .eval-header .logo { | |
| width: 44px; height: 44px; | |
| border-radius: 10px; | |
| background: linear-gradient(135deg, #534AB7 0%, #1D9E75 100%); | |
| display: flex; align-items: center; justify-content: center; | |
| flex-shrink: 0; | |
| font-size: 22px; line-height: 1; | |
| } | |
| .eval-header h1 { | |
| font-size: 20px !important; | |
| font-weight: 600 !important; | |
| letter-spacing: -0.3px !important; | |
| color: #E8EAF0 !important; | |
| margin: 0 !important; | |
| } | |
| .eval-header .model-badge { | |
| margin-left: auto; | |
| font-family: 'Space Mono', monospace; | |
| font-size: 10px; | |
| background: #1E2028; | |
| border: 1px solid #3A3E52; | |
| color: #6A6E80; | |
| padding: 4px 10px; | |
| border-radius: 4px; | |
| letter-spacing: 1.5px; | |
| white-space: nowrap; | |
| } | |
| .gradio-textbox textarea, | |
| .gradio-code textarea, | |
| .gradio-textbox input { | |
| background: #161820 !important; | |
| border: 1px solid #2E3140 !important; | |
| border-radius: 10px !important; | |
| color: #E8EAF0 !important; | |
| font-family: 'DM Sans', sans-serif !important; | |
| font-size: 13.5px !important; | |
| line-height: 1.7 !important; | |
| padding: 14px 16px !important; | |
| transition: border-color 0.15s !important; | |
| resize: vertical !important; | |
| } | |
| .gradio-textbox textarea:focus, | |
| .gradio-code textarea:focus { | |
| border-color: #534AB7 !important; | |
| outline: none !important; | |
| box-shadow: 0 0 0 3px rgba(83, 74, 183, 0.15) !important; | |
| } | |
| .gradio-textbox textarea::placeholder { | |
| color: #4A4E60 !important; | |
| } | |
| #code-input textarea { | |
| font-family: 'Space Mono', monospace !important; | |
| font-size: 12.5px !important; | |
| line-height: 1.65 !important; | |
| } | |
| .gradio-textbox label span, | |
| .gradio-code label span { | |
| font-family: 'DM Sans', sans-serif !important; | |
| font-size: 13px !important; | |
| font-weight: 700 !important; | |
| color: #E8EAF0 !important; | |
| text-transform: uppercase !important; | |
| letter-spacing: 0.8px !important; | |
| } | |
| #eval-btn { | |
| background: #534AB7 !important; | |
| border: none !important; | |
| color: #fff !important; | |
| font-family: 'DM Sans', sans-serif !important; | |
| font-size: 14px !important; | |
| font-weight: 500 !important; | |
| padding: 12px 32px !important; | |
| border-radius: 8px !important; | |
| cursor: pointer !important; | |
| transition: background 0.15s, transform 0.1s !important; | |
| letter-spacing: -0.1px !important; | |
| } | |
| #eval-btn:hover { background: #7F77DD !important; transform: translateY(-1px) !important; } | |
| #eval-btn:active { transform: translateY(0) !important; } | |
| #clear-btn { | |
| background: transparent !important; | |
| border: 1px solid #3A3E52 !important; | |
| color: #9DA0B0 !important; | |
| font-family: 'DM Sans', sans-serif !important; | |
| font-size: 13px !important; | |
| padding: 12px 20px !important; | |
| border-radius: 8px !important; | |
| cursor: pointer !important; | |
| transition: border-color 0.15s, color 0.15s !important; | |
| } | |
| #clear-btn:hover { border-color: #7F77DD !important; color: #E8EAF0 !important; } | |
| .results-heading { | |
| font-size: 11px !important; | |
| font-weight: 500 !important; | |
| color: #6A6E80 !important; | |
| text-transform: uppercase !important; | |
| letter-spacing: 1px !important; | |
| padding: 0 0 16px !important; | |
| border-bottom: 1px solid #2E3140 !important; | |
| margin-bottom: 20px !important; | |
| } | |
| #verdict-out textarea { | |
| background: #161820 !important; border: 1px solid #2E3140 !important; | |
| border-radius: 10px !important; font-family: 'Space Mono', monospace !important; | |
| font-size: 26px !important; font-weight: 700 !important; | |
| text-align: center !important; padding: 20px !important; | |
| color: #E8EAF0 !important; cursor: default !important; | |
| } | |
| #accuracy-out textarea { | |
| background: #161820 !important; border: 1px solid #2E3140 !important; | |
| border-radius: 10px !important; font-family: 'Space Mono', monospace !important; | |
| font-size: 26px !important; font-weight: 700 !important; | |
| text-align: center !important; padding: 20px !important; | |
| color: #E8EAF0 !important; cursor: default !important; | |
| } | |
| #summary-out textarea { | |
| background: #161820 !important; border: 1px solid #2E3140 !important; | |
| border-radius: 10px !important; font-family: 'DM Sans', sans-serif !important; | |
| font-size: 13.5px !important; line-height: 1.7 !important; | |
| padding: 16px !important; color: #C0C3D0 !important; cursor: default !important; | |
| } | |
| #issues-out textarea { | |
| background: #161820 !important; border: 1px solid #2E3140 !important; | |
| border-radius: 10px !important; font-family: 'DM Sans', sans-serif !important; | |
| font-size: 13.5px !important; line-height: 1.8 !important; | |
| padding: 16px !important; color: #C0C3D0 !important; | |
| cursor: default !important; min-height: 120px !important; | |
| } | |
| #error-out textarea { | |
| background: #1A0E0E !important; | |
| border: 1px solid #5a2020 !important; | |
| border-radius: 10px !important; | |
| font-family: 'DM Sans', sans-serif !important; | |
| font-size: 13.5px !important; | |
| padding: 14px 16px !important; | |
| color: #F0997B !important; | |
| cursor: default !important; | |
| min-height: 120px !important; | |
| width: 100% !important; | |
| } | |
| .divider { height: 1px; background: #2E3140; margin: 20px 0; } | |
| footer { display: none !important; } | |
| ::-webkit-scrollbar { width: 6px; height: 6px; } | |
| ::-webkit-scrollbar-track { background: transparent; } | |
| ::-webkit-scrollbar-thumb { background: #3A3E52; border-radius: 3px; } | |
| ::-webkit-scrollbar-thumb:hover { background: #534AB7; } | |
| """ | |
| HEADER_HTML = """ | |
| <div class="eval-header"> | |
| <div class="logo">🔍</div> | |
| <div><h1>Python Code Evaluator</h1></div> | |
| <div class="model-badge">GEMMA-4 · 31B-IT</div> | |
| </div> | |
| """ | |
| RESULTS_HEADING_HTML = """ | |
| <div class="divider"></div> | |
| <div class="results-heading">⬡ Evaluation Results</div> | |
| """ | |
| INPUT_HINT_HTML = """ | |
| <div style="font-size:12px;color:#4A4E60;margin-top:-4px;padding-bottom:4px;"> | |
| Tip — paste your description and code above, then click | |
| <strong style="color:#7F77DD">Evaluate</strong>. | |
| </div> | |
| """ | |
| def build_ui(evaluate_fn): | |
| def _split_metrics(metrics_str: str): | |
| acc, summary = "", "" | |
| if not metrics_str: | |
| return acc, summary | |
| for line in metrics_str.splitlines(): | |
| if line.startswith("Accuracy:"): | |
| acc = line.replace("Accuracy:", "").strip() | |
| elif line.startswith("Summary:"): | |
| summary = line.replace("Summary:", "").strip() | |
| return acc, summary | |
| def _ui_evaluate(description: str, code: str): | |
| verdict, metrics, issues, error = evaluate_fn(description, code) | |
| accuracy, summary = _split_metrics(metrics) | |
| if error: | |
| # Error occurred — show only the error box, hide all result boxes | |
| return ( | |
| gr.update(value="", visible=False), # verdict | |
| gr.update(value="", visible=False), # accuracy | |
| gr.update(value="", visible=False), # summary | |
| gr.update(value="", visible=False), # issues | |
| gr.update(value=error, visible=True), # error ← only this shows | |
| ) | |
| # No error — show all result boxes, hide the error box | |
| if "PASS" in verdict.upper(): | |
| verdict_display = "✅ PASS" | |
| elif "FAIL" in verdict.upper(): | |
| verdict_display = "❌ FAIL" | |
| else: | |
| verdict_display = verdict or "" | |
| return ( | |
| gr.update(value=verdict_display, visible=True), # verdict | |
| gr.update(value=accuracy, visible=True), # accuracy | |
| gr.update(value=summary, visible=True), # summary | |
| gr.update(value=issues, visible=True), # issues | |
| gr.update(value="", visible=False), # error ← hidden when no error | |
| ) | |
| with gr.Blocks(title="Python Code Evaluator") as demo: | |
| gr.HTML(HEADER_HTML) | |
| with gr.Row(equal_height=True): | |
| description_input = gr.Textbox( | |
| label="Requirements Description", | |
| placeholder=( | |
| "Describe what the Python code is supposed to do…\n\n" | |
| "Example: Write a function that accepts a list of integers " | |
| "and returns the sum of all even numbers. It should handle " | |
| "empty lists by returning 0 and ignore non-integer values." | |
| ), | |
| lines=10, | |
| max_lines=20, | |
| elem_id="desc-input", | |
| ) | |
| code_input = gr.Textbox( | |
| label="Python Code", | |
| placeholder=( | |
| "# Paste your Python code here…\n\n" | |
| "# Note: Ensure proper indentation for accurate evaluation.\n\n" | |
| "def example(numbers):\n" | |
| " total = 0\n" | |
| " for item in numbers:\n" | |
| " if item % 2 == 0:\n" | |
| " total += item\n" | |
| " return total\n" | |
| ), | |
| lines=10, | |
| max_lines=30, | |
| elem_id="code-input", | |
| ) | |
| gr.HTML(INPUT_HINT_HTML) | |
| with gr.Row(): | |
| eval_btn = gr.Button("Evaluate", variant="primary", | |
| elem_id="eval-btn", scale=0) | |
| clear_btn = gr.Button("Clear", variant="secondary", | |
| elem_id="clear-btn", scale=0) | |
| gr.HTML(RESULTS_HEADING_HTML) | |
| with gr.Row(equal_height=True): | |
| verdict_out = gr.Textbox(label="Verdict", interactive=False, elem_id="verdict-out", scale=1) | |
| accuracy_out = gr.Textbox(label="Accuracy", interactive=False, elem_id="accuracy-out", scale=1) | |
| summary_out = gr.Textbox(label="Summary", interactive=False, elem_id="summary-out", scale=2, lines=3) | |
| issues_out = gr.Textbox(label="Issues Detected", interactive=False, lines=5, elem_id="issues-out") | |
| with gr.Row(): | |
| error_out = gr.Textbox( | |
| label="Status / Errors", | |
| interactive=False, | |
| visible=True, | |
| elem_id="error-out", | |
| scale=1, | |
| ) | |
| gr.HTML(""" | |
| <div style="text-align:right;font-size:11px;color:#4A4E60;margin-top:8px;"> | |
| Press <kbd style="background:#1E2028;border:1px solid #3A3E52;border-radius:4px; | |
| padding:2px 6px;font-family:'Space Mono',monospace;font-size:10px;color:#9DA0B0;"> | |
| Ctrl+Enter</kbd> inside any input to evaluate | |
| </div> | |
| """) | |
| outputs = [verdict_out, accuracy_out, summary_out, issues_out, error_out] | |
| eval_btn.click(fn=_ui_evaluate, inputs=[description_input, code_input], outputs=outputs) | |
| description_input.submit(fn=_ui_evaluate, inputs=[description_input, code_input], outputs=outputs) | |
| code_input.submit(fn=_ui_evaluate, inputs=[description_input, code_input], outputs=outputs) | |
| def _clear(): | |
| return "", "", "", "", "", "" | |
| clear_btn.click( | |
| fn=_clear, | |
| inputs=[], | |
| outputs=[description_input, code_input, | |
| verdict_out, accuracy_out, summary_out, issues_out], | |
| ) | |
| return demo | |
| # --------------------------------------------------------------------------- | |
| # ENTRY POINT | |
| # --------------------------------------------------------------------------- | |
| if __name__ == "__main__": | |
| app = build_ui(validate_and_evaluate) | |
| app.launch(server_name="0.0.0.0", server_port=7860, share=False, | |
| show_error=True, css=CUSTOM_CSS) |