Spaces:
Sleeping
Sleeping
| import os | |
| import re | |
| import ast | |
| import sys | |
| import json | |
| import time | |
| from collections import deque | |
| import gradio as gr | |
| from transformers import pipeline | |
| # --------------------------------------------------------------------------- | |
| # Model Initialization — GPT-2 via HuggingFace Transformers | |
| # --------------------------------------------------------------------------- | |
| print("Loading GPT-2 model… this may take a moment on first run.") | |
| generator = pipeline("text-generation", model="gpt2", max_new_tokens=400) | |
| print("Model loaded.") | |
| # --------------------------------------------------------------------------- | |
| # MODULE 2 — Validation & Flow Management | |
| # --------------------------------------------------------------------------- | |
| MIN_DESCRIPTION_CHARS = 20 | |
| MAX_DESCRIPTION_CHARS = 3000 | |
| MIN_CODE_CHARS = 10 | |
| MAX_CODE_CHARS = 8000 | |
| MAX_CODE_LINES = 300 | |
| FORBIDDEN_PATTERNS = [ | |
| r"ignore (all |previous |above )?instructions", | |
| r"disregard (all |previous |above )?instructions", | |
| r"you are now", | |
| r"act as (a |an )?", | |
| r"<\s*(script|iframe|object|embed)", | |
| r"system\s*prompt", | |
| r"jailbreak", | |
| ] | |
| PYTHON_KEYWORDS = { | |
| "def", "class", "import", "from", "return", "if", "else", "elif", | |
| "for", "while", "try", "except", "with", "lambda", "yield", "pass", | |
| "raise", "assert", "in", "not", "and", "or", "True", "False", "None", | |
| "print", "len", "range", "self", | |
| } | |
| class RateLimiter: | |
| def __init__(self, max_calls: int = 5, window_seconds: int = 60): | |
| self.max_calls = max_calls | |
| self.window_seconds = window_seconds | |
| self._timestamps: deque = deque() | |
| def is_allowed(self) -> tuple[bool, str]: | |
| now = time.time() | |
| while self._timestamps and now - self._timestamps[0] > self.window_seconds: | |
| self._timestamps.popleft() | |
| if len(self._timestamps) >= self.max_calls: | |
| wait = int(self.window_seconds - (now - self._timestamps[0])) + 1 | |
| return False, ( | |
| f"⏳ Rate limit reached — {self.max_calls} requests in " | |
| f"{self.window_seconds}s. Please wait ~{wait}s and try again." | |
| ) | |
| self._timestamps.append(now) | |
| return True, "" | |
| _rate_limiter = RateLimiter(max_calls=5, window_seconds=60) | |
| def _check_forbidden(text: str) -> str | None: | |
| lower = text.lower() | |
| for pattern in FORBIDDEN_PATTERNS: | |
| if re.search(pattern, lower): | |
| return ( | |
| "Input contains disallowed content. " | |
| "Please remove prompt-injection or HTML patterns and try again." | |
| ) | |
| return None | |
| def _looks_like_python(code: str) -> tuple[bool, str]: | |
| tokens = set(re.findall(r"[A-Za-z_]\w*", code)) | |
| if not tokens.intersection(PYTHON_KEYWORDS): | |
| return False, ( | |
| "🐍 The code doesn't appear to be Python — no recognisable Python " | |
| "keywords found (e.g. def, class, import, return). " | |
| "Please submit Python code only." | |
| ) | |
| try: | |
| ast.parse(code) | |
| except SyntaxError as exc: | |
| line_hint = f" (line {exc.lineno})" if exc.lineno else "" | |
| return False, ( | |
| f"🐍 Python syntax error{line_hint}: {exc.msg}. " | |
| "Please fix the syntax error before evaluating." | |
| ) | |
| return True, "" | |
| def validate_inputs(description: str, code: str) -> list[str]: | |
| errors: list[str] = [] | |
| if not description or not description.strip(): | |
| errors.append("📋 Requirements description is required.") | |
| if not code or not code.strip(): | |
| errors.append("🐍 Python code is required.") | |
| if errors: | |
| return errors | |
| desc, code_ = description.strip(), code.strip() | |
| if len(desc) < MIN_DESCRIPTION_CHARS: | |
| errors.append(f"📋 Description too short ({len(desc)} chars) — minimum is {MIN_DESCRIPTION_CHARS} characters.") | |
| if len(code_) < MIN_CODE_CHARS: | |
| errors.append(f"🐍 Code too short ({len(code_)} chars) — minimum is {MIN_CODE_CHARS} characters.") | |
| if len(desc) > MAX_DESCRIPTION_CHARS: | |
| errors.append(f"📋 Description too long ({len(desc):,} chars) — max is {MAX_DESCRIPTION_CHARS:,} characters.") | |
| if len(code_) > MAX_CODE_CHARS: | |
| errors.append(f"🐍 Code too long ({len(code_):,} chars) — max is {MAX_CODE_CHARS:,} characters.") | |
| if len(code_.splitlines()) > MAX_CODE_LINES: | |
| errors.append(f"🐍 Code has too many lines ({len(code_.splitlines())}) — max is {MAX_CODE_LINES} lines.") | |
| if err := _check_forbidden(desc): | |
| errors.append(f"📋 {err}") | |
| if err := _check_forbidden(code_): | |
| errors.append(f"🐍 {err}") | |
| if not errors: | |
| is_python, py_error = _looks_like_python(code_) | |
| if not is_python: | |
| errors.append(py_error) | |
| if not errors: | |
| allowed, rate_msg = _rate_limiter.is_allowed() | |
| if not allowed: | |
| errors.append(rate_msg) | |
| return errors | |
| # --------------------------------------------------------------------------- | |
| # MODULE 3 — Prompt Builder | |
| # --------------------------------------------------------------------------- | |
| def build_prompt(description: str, code: str) -> str: | |
| return f"""You are a Python code reviewer. Evaluate whether the code meets the requirements. | |
| Requirements: | |
| {description} | |
| Code: | |
| ```python | |
| {code} | |
| ``` | |
| Reply ONLY with a valid JSON object (no extra text) using exactly these keys: | |
| - "result": "Pass" or "Fail" | |
| - "accuracy": integer 0-100 | |
| - "summary": one sentence explanation | |
| - "issues": numbered list of issues as a single string, e.g. "1. Warning - missing docstring\\n2. Error - off-by-one" | |
| JSON: | |
| """ | |
| # --------------------------------------------------------------------------- | |
| # MODULE 4 — Generation & JSON Parsing | |
| # --------------------------------------------------------------------------- | |
| def _extract_json(text: str) -> dict: | |
| """Try to extract a JSON object from the raw model output.""" | |
| # Find the first { ... } block | |
| match = re.search(r'\{.*?\}', text, re.DOTALL) | |
| if match: | |
| try: | |
| return json.loads(match.group()) | |
| except json.JSONDecodeError: | |
| pass | |
| # Fallback: attempt to parse keys manually | |
| result = "Pass" if re.search(r'"result"\s*:\s*"Pass"', text, re.I) else "Fail" | |
| acc_m = re.search(r'"accuracy"\s*:\s*(\d+)', text) | |
| accuracy = int(acc_m.group(1)) if acc_m else 50 | |
| sum_m = re.search(r'"summary"\s*:\s*"([^"]+)"', text) | |
| summary = sum_m.group(1) if sum_m else "Unable to parse summary from model output." | |
| iss_m = re.search(r'"issues"\s*:\s*"([^"]*)"', text, re.DOTALL) | |
| issues = iss_m.group(1).replace("\\n", "\n") if iss_m else "No issues extracted." | |
| return {"result": result, "accuracy": accuracy, "summary": summary, "issues": issues} | |
| def generate_response(prompt: str) -> dict: | |
| outputs = generator(prompt, do_sample=False, temperature=1.0) | |
| raw_text = outputs[0]["generated_text"] | |
| # Only look at the text appended after the prompt | |
| new_text = raw_text[len(prompt):] | |
| return _extract_json(new_text) | |
| # --------------------------------------------------------------------------- | |
| # Core orchestration | |
| # --------------------------------------------------------------------------- | |
| def format_for_display(raw: dict) -> tuple[str, str, str]: | |
| emoji = "✅" if raw["result"].upper() == "PASS" else "❌" | |
| verdict = f"{emoji} {raw['result']}" | |
| acc_str = f"{raw['accuracy']}%" if raw.get("accuracy", -1) >= 0 else "N/A" | |
| metrics = f"Accuracy: {acc_str}\n\nSummary: {raw['summary']}" | |
| return verdict, metrics, raw.get("issues", "") | |
| def validate_and_evaluate(description: str, code: str): | |
| errors = validate_inputs(description, code) | |
| if errors: | |
| return "", "", "", "\n".join(f"{i+1}. {e}" for i, e in enumerate(errors)) | |
| prompt = build_prompt(description, code) | |
| try: | |
| raw = generate_response(prompt) | |
| except Exception as exc: | |
| return "", "", "", f"❌ Model error: {exc}" | |
| verdict, metrics, issues = format_for_display(raw) | |
| return verdict, metrics, issues, "" | |
| # --------------------------------------------------------------------------- | |
| # MODULE 1 — UI | |
| # --------------------------------------------------------------------------- | |
| CUSTOM_CSS = """ | |
| @import url('https://fonts.googleapis.com/css2?family=Space+Mono:wght@400;700&family=DM+Sans:ital,wght@0,300;0,400;0,500;0,600;1,400&display=swap'); | |
| *, *::before, *::after { box-sizing: border-box; } | |
| body, .gradio-container { | |
| font-family: 'DM Sans', sans-serif !important; | |
| background: #0D0F14 !important; | |
| color: #E8EAF0 !important; | |
| } | |
| .gradio-container { | |
| max-width: 1140px !important; | |
| margin: 0 auto !important; | |
| padding: 32px 24px !important; | |
| } | |
| .eval-header { | |
| display: flex; align-items: center; gap: 16px; | |
| padding: 0 0 28px; border-bottom: 1px solid #2E3140; margin-bottom: 28px; | |
| } | |
| .eval-header .logo { | |
| width: 44px; height: 44px; border-radius: 10px; | |
| background: linear-gradient(135deg, #534AB7 0%, #1D9E75 100%); | |
| display: flex; align-items: center; justify-content: center; | |
| flex-shrink: 0; font-size: 22px; line-height: 1; | |
| } | |
| .eval-header h1 { | |
| font-size: 20px !important; font-weight: 600 !important; | |
| letter-spacing: -0.3px !important; color: #E8EAF0 !important; margin: 0 !important; | |
| } | |
| .eval-header p { font-size: 13px !important; color: #9DA0B0 !important; margin: 2px 0 0 !important; } | |
| .eval-header .model-badge { | |
| margin-left: auto; font-family: 'Space Mono', monospace; font-size: 10px; | |
| background: #1E2028; border: 1px solid #3A3E52; color: #6A6E80; | |
| padding: 4px 10px; border-radius: 4px; letter-spacing: 1.5px; white-space: nowrap; | |
| } | |
| .gradio-textbox textarea, .gradio-code textarea, .gradio-textbox input { | |
| background: #161820 !important; border: 1px solid #2E3140 !important; | |
| border-radius: 10px !important; color: #E8EAF0 !important; | |
| font-family: 'DM Sans', sans-serif !important; font-size: 13.5px !important; | |
| line-height: 1.7 !important; padding: 14px 16px !important; | |
| transition: border-color 0.15s !important; resize: vertical !important; | |
| } | |
| .gradio-textbox textarea:focus, .gradio-code textarea:focus { | |
| border-color: #534AB7 !important; outline: none !important; | |
| box-shadow: 0 0 0 3px rgba(83, 74, 183, 0.15) !important; | |
| } | |
| .gradio-textbox textarea::placeholder { color: #4A4E60 !important; } | |
| #code-input { min-height: 300px; } | |
| #code-input .cm-editor { min-height: 300px; } | |
| .gradio-textbox label span, .gradio-code label span { | |
| font-family: 'DM Sans', sans-serif !important; font-size: 13px !important; | |
| font-weight: 700 !important; color: #E8EAF0 !important; | |
| text-transform: uppercase !important; letter-spacing: 0.8px !important; | |
| } | |
| #eval-btn { | |
| background: #534AB7 !important; border: none !important; color: #fff !important; | |
| font-family: 'DM Sans', sans-serif !important; font-size: 14px !important; | |
| font-weight: 500 !important; padding: 12px 32px !important; border-radius: 8px !important; | |
| cursor: pointer !important; transition: background 0.15s, transform 0.1s !important; | |
| } | |
| #eval-btn:hover { background: #7F77DD !important; transform: translateY(-1px) !important; } | |
| #clear-btn { | |
| background: transparent !important; border: 1px solid #3A3E52 !important; | |
| color: #9DA0B0 !important; font-family: 'DM Sans', sans-serif !important; | |
| font-size: 13px !important; padding: 12px 20px !important; border-radius: 8px !important; | |
| cursor: pointer !important; | |
| } | |
| #clear-btn:hover { border-color: #7F77DD !important; color: #E8EAF0 !important; } | |
| .results-heading { | |
| font-size: 11px !important; font-weight: 500 !important; color: #6A6E80 !important; | |
| text-transform: uppercase !important; letter-spacing: 1px !important; | |
| padding: 0 0 16px !important; border-bottom: 1px solid #2E3140 !important; margin-bottom: 20px !important; | |
| } | |
| #verdict-out textarea, #accuracy-out textarea { | |
| background: #161820 !important; border: 1px solid #2E3140 !important; | |
| border-radius: 10px !important; font-family: 'Space Mono', monospace !important; | |
| font-size: 26px !important; font-weight: 700 !important; text-align: center !important; | |
| padding: 20px !important; color: #E8EAF0 !important; cursor: default !important; | |
| } | |
| #summary-out textarea, #issues-out textarea { | |
| background: #161820 !important; border: 1px solid #2E3140 !important; | |
| border-radius: 10px !important; font-family: 'DM Sans', sans-serif !important; | |
| font-size: 13.5px !important; line-height: 1.7 !important; padding: 16px !important; | |
| color: #C0C3D0 !important; cursor: default !important; | |
| } | |
| #issues-out textarea { min-height: 120px !important; line-height: 1.8 !important; } | |
| #error-out textarea { | |
| background: #1A0E0E !important; border: 1px solid #5a2020 !important; | |
| border-radius: 10px !important; font-family: 'DM Sans', sans-serif !important; | |
| font-size: 13.5px !important; padding: 14px 16px !important; | |
| color: #F0997B !important; cursor: default !important; | |
| } | |
| .divider { height: 1px; background: #2E3140; margin: 20px 0; } | |
| footer { display: none !important; } | |
| ::-webkit-scrollbar { width: 6px; height: 6px; } | |
| ::-webkit-scrollbar-track { background: transparent; } | |
| ::-webkit-scrollbar-thumb { background: #3A3E52; border-radius: 3px; } | |
| ::-webkit-scrollbar-thumb:hover { background: #534AB7; } | |
| """ | |
| HEADER_HTML = """ | |
| <div class="eval-header"> | |
| <div class="logo">🔍</div> | |
| <div> | |
| <h1>Python Code Evaluator</h1> | |
| <p>Powered by GPT-2 (HuggingFace)</p> | |
| </div> | |
| <div class="model-badge">GPT-2 · LOCAL</div> | |
| </div> | |
| """ | |
| RESULTS_HEADING_HTML = """ | |
| <div class="divider"></div> | |
| <div class="results-heading">⬡ Evaluation Results</div> | |
| """ | |
| INPUT_HINT_HTML = """ | |
| <div style="font-size:12px;color:#4A4E60;margin-top:-4px;padding-bottom:4px;"> | |
| Tip — paste your description and code above, then click <strong style="color:#7F77DD">Evaluate</strong>. | |
| Note: GPT-2 is a small model; results are best-effort and may need interpretation. | |
| </div> | |
| """ | |
| def _split_metrics(metrics_str: str): | |
| acc, summary = "", "" | |
| if not metrics_str: | |
| return acc, summary | |
| for line in metrics_str.splitlines(): | |
| if line.startswith("Accuracy:"): | |
| acc = line.replace("Accuracy:", "").strip() | |
| elif line.startswith("Summary:"): | |
| summary = line.replace("Summary:", "").strip() | |
| return acc, summary | |
| def _ui_evaluate(description: str, code: str): | |
| verdict, metrics, issues, error = validate_and_evaluate(description, code) | |
| accuracy, summary = _split_metrics(metrics) | |
| if "PASS" in verdict.upper(): | |
| verdict_display = "✅ PASS" | |
| elif "FAIL" in verdict.upper(): | |
| verdict_display = "❌ FAIL" | |
| else: | |
| verdict_display = verdict or "" | |
| return verdict_display, accuracy, summary, issues, error | |
| with gr.Blocks(title="Python Code Evaluator", css=CUSTOM_CSS) as app: | |
| gr.HTML(HEADER_HTML) | |
| with gr.Row(equal_height=True): | |
| description_input = gr.Textbox( | |
| label="Requirements Description", | |
| placeholder=( | |
| "Describe what the Python code is supposed to do…\n\n" | |
| "Example: Write a function that accepts a list of integers " | |
| "and returns the sum of all even numbers." | |
| ), | |
| lines=10, max_lines=20, elem_id="desc-input", | |
| ) | |
| code_input = gr.Code( | |
| label="Python Code", language="python", | |
| lines=10, max_lines=30, elem_id="code-input", | |
| ) | |
| gr.HTML(INPUT_HINT_HTML) | |
| with gr.Row(): | |
| eval_btn = gr.Button("Evaluate", variant="primary", elem_id="eval-btn", scale=0) | |
| clear_btn = gr.Button("Clear", variant="secondary", elem_id="clear-btn", scale=0) | |
| gr.HTML(RESULTS_HEADING_HTML) | |
| with gr.Row(equal_height=True): | |
| verdict_out = gr.Textbox(label="Verdict", interactive=False, elem_id="verdict-out", scale=1) | |
| accuracy_out = gr.Textbox(label="Accuracy", interactive=False, elem_id="accuracy-out", scale=1) | |
| summary_out = gr.Textbox(label="Summary", interactive=False, lines=3, elem_id="summary-out", scale=2) | |
| issues_out = gr.Textbox(label="Issues Detected", interactive=False, lines=5, elem_id="issues-out") | |
| error_out = gr.Textbox(label="Status / Errors", interactive=False, visible=True, elem_id="error-out") | |
| outputs = [verdict_out, accuracy_out, summary_out, issues_out, error_out] | |
| eval_btn.click(fn=_ui_evaluate, inputs=[description_input, code_input], outputs=outputs) | |
| def _clear(): | |
| return "", "", "", "", "", "" | |
| clear_btn.click( | |
| fn=_clear, inputs=[], | |
| outputs=[description_input, code_input, verdict_out, accuracy_out, summary_out, issues_out], | |
| ) | |
| if __name__ == "__main__": | |
| app.launch() |