import os import re import ast import sys import json import time from collections import deque import gradio as gr from transformers import pipeline # --------------------------------------------------------------------------- # Model Initialization — GPT-2 via HuggingFace Transformers # --------------------------------------------------------------------------- print("Loading GPT-2 model… this may take a moment on first run.") generator = pipeline("text-generation", model="gpt2", max_new_tokens=400) print("Model loaded.") # --------------------------------------------------------------------------- # MODULE 2 — Validation & Flow Management # --------------------------------------------------------------------------- MIN_DESCRIPTION_CHARS = 20 MAX_DESCRIPTION_CHARS = 3000 MIN_CODE_CHARS = 10 MAX_CODE_CHARS = 8000 MAX_CODE_LINES = 300 FORBIDDEN_PATTERNS = [ r"ignore (all |previous |above )?instructions", r"disregard (all |previous |above )?instructions", r"you are now", r"act as (a |an )?", r"<\s*(script|iframe|object|embed)", r"system\s*prompt", r"jailbreak", ] PYTHON_KEYWORDS = { "def", "class", "import", "from", "return", "if", "else", "elif", "for", "while", "try", "except", "with", "lambda", "yield", "pass", "raise", "assert", "in", "not", "and", "or", "True", "False", "None", "print", "len", "range", "self", } class RateLimiter: def __init__(self, max_calls: int = 5, window_seconds: int = 60): self.max_calls = max_calls self.window_seconds = window_seconds self._timestamps: deque = deque() def is_allowed(self) -> tuple[bool, str]: now = time.time() while self._timestamps and now - self._timestamps[0] > self.window_seconds: self._timestamps.popleft() if len(self._timestamps) >= self.max_calls: wait = int(self.window_seconds - (now - self._timestamps[0])) + 1 return False, ( f"⏳ Rate limit reached — {self.max_calls} requests in " f"{self.window_seconds}s. Please wait ~{wait}s and try again." ) self._timestamps.append(now) return True, "" _rate_limiter = RateLimiter(max_calls=5, window_seconds=60) def _check_forbidden(text: str) -> str | None: lower = text.lower() for pattern in FORBIDDEN_PATTERNS: if re.search(pattern, lower): return ( "Input contains disallowed content. " "Please remove prompt-injection or HTML patterns and try again." ) return None def _looks_like_python(code: str) -> tuple[bool, str]: tokens = set(re.findall(r"[A-Za-z_]\w*", code)) if not tokens.intersection(PYTHON_KEYWORDS): return False, ( "🐍 The code doesn't appear to be Python — no recognisable Python " "keywords found (e.g. def, class, import, return). " "Please submit Python code only." ) try: ast.parse(code) except SyntaxError as exc: line_hint = f" (line {exc.lineno})" if exc.lineno else "" return False, ( f"🐍 Python syntax error{line_hint}: {exc.msg}. " "Please fix the syntax error before evaluating." ) return True, "" def validate_inputs(description: str, code: str) -> list[str]: errors: list[str] = [] if not description or not description.strip(): errors.append("📋 Requirements description is required.") if not code or not code.strip(): errors.append("🐍 Python code is required.") if errors: return errors desc, code_ = description.strip(), code.strip() if len(desc) < MIN_DESCRIPTION_CHARS: errors.append(f"📋 Description too short ({len(desc)} chars) — minimum is {MIN_DESCRIPTION_CHARS} characters.") if len(code_) < MIN_CODE_CHARS: errors.append(f"🐍 Code too short ({len(code_)} chars) — minimum is {MIN_CODE_CHARS} characters.") if len(desc) > MAX_DESCRIPTION_CHARS: errors.append(f"📋 Description too long ({len(desc):,} chars) — max is {MAX_DESCRIPTION_CHARS:,} characters.") if len(code_) > MAX_CODE_CHARS: errors.append(f"🐍 Code too long ({len(code_):,} chars) — max is {MAX_CODE_CHARS:,} characters.") if len(code_.splitlines()) > MAX_CODE_LINES: errors.append(f"🐍 Code has too many lines ({len(code_.splitlines())}) — max is {MAX_CODE_LINES} lines.") if err := _check_forbidden(desc): errors.append(f"📋 {err}") if err := _check_forbidden(code_): errors.append(f"🐍 {err}") if not errors: is_python, py_error = _looks_like_python(code_) if not is_python: errors.append(py_error) if not errors: allowed, rate_msg = _rate_limiter.is_allowed() if not allowed: errors.append(rate_msg) return errors # --------------------------------------------------------------------------- # MODULE 3 — Prompt Builder # --------------------------------------------------------------------------- def build_prompt(description: str, code: str) -> str: return f"""You are a Python code reviewer. Evaluate whether the code meets the requirements. Requirements: {description} Code: ```python {code} ``` Reply ONLY with a valid JSON object (no extra text) using exactly these keys: - "result": "Pass" or "Fail" - "accuracy": integer 0-100 - "summary": one sentence explanation - "issues": numbered list of issues as a single string, e.g. "1. Warning - missing docstring\\n2. Error - off-by-one" JSON: """ # --------------------------------------------------------------------------- # MODULE 4 — Generation & JSON Parsing # --------------------------------------------------------------------------- def _extract_json(text: str) -> dict: """Try to extract a JSON object from the raw model output.""" # Find the first { ... } block match = re.search(r'\{.*?\}', text, re.DOTALL) if match: try: return json.loads(match.group()) except json.JSONDecodeError: pass # Fallback: attempt to parse keys manually result = "Pass" if re.search(r'"result"\s*:\s*"Pass"', text, re.I) else "Fail" acc_m = re.search(r'"accuracy"\s*:\s*(\d+)', text) accuracy = int(acc_m.group(1)) if acc_m else 50 sum_m = re.search(r'"summary"\s*:\s*"([^"]+)"', text) summary = sum_m.group(1) if sum_m else "Unable to parse summary from model output." iss_m = re.search(r'"issues"\s*:\s*"([^"]*)"', text, re.DOTALL) issues = iss_m.group(1).replace("\\n", "\n") if iss_m else "No issues extracted." return {"result": result, "accuracy": accuracy, "summary": summary, "issues": issues} def generate_response(prompt: str) -> dict: outputs = generator(prompt, do_sample=False, temperature=1.0) raw_text = outputs[0]["generated_text"] # Only look at the text appended after the prompt new_text = raw_text[len(prompt):] return _extract_json(new_text) # --------------------------------------------------------------------------- # Core orchestration # --------------------------------------------------------------------------- def format_for_display(raw: dict) -> tuple[str, str, str]: emoji = "✅" if raw["result"].upper() == "PASS" else "❌" verdict = f"{emoji} {raw['result']}" acc_str = f"{raw['accuracy']}%" if raw.get("accuracy", -1) >= 0 else "N/A" metrics = f"Accuracy: {acc_str}\n\nSummary: {raw['summary']}" return verdict, metrics, raw.get("issues", "") def validate_and_evaluate(description: str, code: str): errors = validate_inputs(description, code) if errors: return "", "", "", "\n".join(f"{i+1}. {e}" for i, e in enumerate(errors)) prompt = build_prompt(description, code) try: raw = generate_response(prompt) except Exception as exc: return "", "", "", f"❌ Model error: {exc}" verdict, metrics, issues = format_for_display(raw) return verdict, metrics, issues, "" # --------------------------------------------------------------------------- # MODULE 1 — UI # --------------------------------------------------------------------------- CUSTOM_CSS = """ @import url('https://fonts.googleapis.com/css2?family=Space+Mono:wght@400;700&family=DM+Sans:ital,wght@0,300;0,400;0,500;0,600;1,400&display=swap'); *, *::before, *::after { box-sizing: border-box; } body, .gradio-container { font-family: 'DM Sans', sans-serif !important; background: #0D0F14 !important; color: #E8EAF0 !important; } .gradio-container { max-width: 1140px !important; margin: 0 auto !important; padding: 32px 24px !important; } .eval-header { display: flex; align-items: center; gap: 16px; padding: 0 0 28px; border-bottom: 1px solid #2E3140; margin-bottom: 28px; } .eval-header .logo { width: 44px; height: 44px; border-radius: 10px; background: linear-gradient(135deg, #534AB7 0%, #1D9E75 100%); display: flex; align-items: center; justify-content: center; flex-shrink: 0; font-size: 22px; line-height: 1; } .eval-header h1 { font-size: 20px !important; font-weight: 600 !important; letter-spacing: -0.3px !important; color: #E8EAF0 !important; margin: 0 !important; } .eval-header p { font-size: 13px !important; color: #9DA0B0 !important; margin: 2px 0 0 !important; } .eval-header .model-badge { margin-left: auto; font-family: 'Space Mono', monospace; font-size: 10px; background: #1E2028; border: 1px solid #3A3E52; color: #6A6E80; padding: 4px 10px; border-radius: 4px; letter-spacing: 1.5px; white-space: nowrap; } .gradio-textbox textarea, .gradio-code textarea, .gradio-textbox input { background: #161820 !important; border: 1px solid #2E3140 !important; border-radius: 10px !important; color: #E8EAF0 !important; font-family: 'DM Sans', sans-serif !important; font-size: 13.5px !important; line-height: 1.7 !important; padding: 14px 16px !important; transition: border-color 0.15s !important; resize: vertical !important; } .gradio-textbox textarea:focus, .gradio-code textarea:focus { border-color: #534AB7 !important; outline: none !important; box-shadow: 0 0 0 3px rgba(83, 74, 183, 0.15) !important; } .gradio-textbox textarea::placeholder { color: #4A4E60 !important; } #code-input { min-height: 300px; } #code-input .cm-editor { min-height: 300px; } .gradio-textbox label span, .gradio-code label span { font-family: 'DM Sans', sans-serif !important; font-size: 13px !important; font-weight: 700 !important; color: #E8EAF0 !important; text-transform: uppercase !important; letter-spacing: 0.8px !important; } #eval-btn { background: #534AB7 !important; border: none !important; color: #fff !important; font-family: 'DM Sans', sans-serif !important; font-size: 14px !important; font-weight: 500 !important; padding: 12px 32px !important; border-radius: 8px !important; cursor: pointer !important; transition: background 0.15s, transform 0.1s !important; } #eval-btn:hover { background: #7F77DD !important; transform: translateY(-1px) !important; } #clear-btn { background: transparent !important; border: 1px solid #3A3E52 !important; color: #9DA0B0 !important; font-family: 'DM Sans', sans-serif !important; font-size: 13px !important; padding: 12px 20px !important; border-radius: 8px !important; cursor: pointer !important; } #clear-btn:hover { border-color: #7F77DD !important; color: #E8EAF0 !important; } .results-heading { font-size: 11px !important; font-weight: 500 !important; color: #6A6E80 !important; text-transform: uppercase !important; letter-spacing: 1px !important; padding: 0 0 16px !important; border-bottom: 1px solid #2E3140 !important; margin-bottom: 20px !important; } #verdict-out textarea, #accuracy-out textarea { background: #161820 !important; border: 1px solid #2E3140 !important; border-radius: 10px !important; font-family: 'Space Mono', monospace !important; font-size: 26px !important; font-weight: 700 !important; text-align: center !important; padding: 20px !important; color: #E8EAF0 !important; cursor: default !important; } #summary-out textarea, #issues-out textarea { background: #161820 !important; border: 1px solid #2E3140 !important; border-radius: 10px !important; font-family: 'DM Sans', sans-serif !important; font-size: 13.5px !important; line-height: 1.7 !important; padding: 16px !important; color: #C0C3D0 !important; cursor: default !important; } #issues-out textarea { min-height: 120px !important; line-height: 1.8 !important; } #error-out textarea { background: #1A0E0E !important; border: 1px solid #5a2020 !important; border-radius: 10px !important; font-family: 'DM Sans', sans-serif !important; font-size: 13.5px !important; padding: 14px 16px !important; color: #F0997B !important; cursor: default !important; } .divider { height: 1px; background: #2E3140; margin: 20px 0; } footer { display: none !important; } ::-webkit-scrollbar { width: 6px; height: 6px; } ::-webkit-scrollbar-track { background: transparent; } ::-webkit-scrollbar-thumb { background: #3A3E52; border-radius: 3px; } ::-webkit-scrollbar-thumb:hover { background: #534AB7; } """ HEADER_HTML = """
Powered by GPT-2 (HuggingFace)