Spaces:
Sleeping
Sleeping
File size: 4,270 Bytes
eda351c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 | # openenv.yaml β OpenEnv specification (required by hackathon)
# SecureCodeEnv V2 β Production-Ready Secure Code Generation RL Environment
# Author: Vishal Dhakad (vishaldhakad)
# Meta Γ HuggingFace OpenEnv Hackathon 2026
name: SecureCodeEnv
version: "2.0"
description: >
RL environment for training LLM agents to write production-ready, secure Python code.
9 CWE-grounded tasks across 3 difficulty tiers. 8-dimensional reward system.
Unique features: behavioral adversarial attack grading (unfakeable),
CodeGraph cross-file consistency memory system (novel in RL), multi-language parsing.
author: vishaldhakad
hf_space: vishaldhakad/SecureCodeEnv
server:
host: 0.0.0.0
port: 7860
workers: 2
endpoints:
reset:
method: POST
path: /reset
description: >
Start new episode. Picks task at given difficulty, initialises CodeGraph,
creates Redis-backed session. Returns task, starter code, CodeGraph, session_id.
params:
difficulty: "easy | medium | hard (default: medium)"
session_id: "optional UUID β generated if not provided"
step:
method: POST
path: /step
description: >
Submit agent code. Runs all 8 graders (correctness, behavioral attacks,
static analysis, consistency, performance, documentation, code structure,
supply chain). Updates CodeGraph. Returns weighted reward + per-grader feedback.
body:
code: "Python source code string"
filename: "logical filename for CodeGraph tracking"
task_id: "task identifier from /reset"
session_id: "UUID from /reset"
state:
method: GET
path: /state
description: Read current episode state without advancing it.
params:
session_id: "UUID from /reset"
action_space:
type: text
description: Python (or JS/TS) source code string submitted by the agent
constraints:
max_length: 50000 # 50KB hard limit
min_length: 1
observation_space:
type: structured_json
fields:
- name: total_reward
type: float
range: [0.0, 1.0]
description: Weighted sum of all grader scores
- name: scores
type: dict
description: Per-grader scores (correctness, attack_resist, static_security, etc.)
- name: feedback
type: dict
description: Human-readable feedback per dimension with emoji rating
- name: codegraph
type: dict
description: Full codebase context β conventions, components, imports
- name: done
type: bool
description: True when reward >= 0.90 or step_count >= 5
reward:
type: multi_dimensional
range: [0.0, 1.0]
terminal: 0.90
max_steps: 5
dimensions:
correctness: 0.25 # Does it work including edge cases?
attack_resist: 0.25 # Behavioral adversarial β unfakeable
static_security: 0.15 # bandit + semgrep CWE pattern matching
consistency: 0.15 # CodeGraph cross-file convention adherence
performance: 0.10 # timeit + tracemalloc relative to baseline
documentation: 0.05 # Docstrings + type hints
code_structure: 0.03 # No print(), no bare except, no hardcoded secrets
supply_chain: 0.02 # No typosquatted/malicious imports
tasks:
- id: password_validator
difficulty: easy
cwe: CWE-916
attack_type: weak_password_acceptance
- id: input_sanitizer
difficulty: easy
cwe: CWE-20
attack_type: xss_payload_passthrough
- id: hash_generator
difficulty: easy
cwe: CWE-327
attack_type: shell_invocation_for_hashing
- id: sql_query_builder
difficulty: medium
cwe: CWE-89
attack_type: sql_injection_cursor_spy
- id: file_path_handler
difficulty: medium
cwe: CWE-22
attack_type: path_traversal_open_spy
- id: api_rate_limiter
difficulty: medium
cwe: CWE-307
attack_type: rate_bypass_spoofed_client
- id: file_upload_handler
difficulty: hard
cwe: CWE-434
attack_type: malicious_file_extension
- id: jwt_validator
difficulty: hard
cwe: CWE-347
attack_type: jwt_algorithm_bypass
- id: auth_middleware
difficulty: hard
cwe: CWE-287
attack_type: auth_bypass_timing_shell
runtime:
max_steps_per_episode: 5
max_inference_time_minutes: 20
min_vcpu: 2
min_memory_gb: 8
port: 7860
|