Spaces:
Runtime error
Runtime error
File size: 19,308 Bytes
2a26b7d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 | """
security_engine.py β Phase 1 Enterprise Security Engine
=========================================================
Additive module. Does not modify scanner.py, app.py route signatures,
or any existing data shapes β only adds new computed fields on top of
existing finding lists.
Provides:
compute_security_score(findings) β weighted 0-100 score + risk label
map_compliance(finding) β OWASP + NIST mapping for one finding
enrich_findings_with_compliance(findings) β adds 'owasp'/'nist' to each finding
compute_repo_health(findings, dep_findings) β repository health summary dict
generate_auto_fix(finding) β before/after/explanation/confidence
"""
import re
import logging
logger = logging.getLogger("safeaiscan.security_engine")
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# 1. ADVANCED SECURITY SCORING ENGINE
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
_SEVERITY_WEIGHTS = {
"CRITICAL": 40,
"HIGH": 20,
"MEDIUM": 10,
"LOW": 5,
}
def _risk_label(score: int) -> str:
"""Map a 0-100 score to a human-readable risk level."""
if score >= 90:
return "Excellent"
if score >= 75:
return "Good"
if score >= 50:
return "Moderate"
if score >= 25:
return "High Risk"
return "Critical"
def compute_security_score(findings: list) -> dict:
"""
Weighted security scoring.
Starts at 100 and deducts points per finding based on severity:
CRITICAL = -40, HIGH = -20, MEDIUM = -10, LOW = -5
Score is clamped to [0, 100].
Returns:
{ "security_score": int, "risk_level": str }
"""
score = 100
for f in findings or []:
sev = (f.get("severity") or "LOW").upper()
score -= _SEVERITY_WEIGHTS.get(sev, 5)
score = max(0, min(100, score))
return {
"security_score": score,
"risk_level": _risk_label(score),
}
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# 2. COMPLIANCE MAPPING ENGINE (OWASP Top 10 + NIST 800-53)
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# Mapping rules: match against finding `type`/`title`/`category` (case-insensitive
# substring match). Order matters β first match wins. Falls back to a generic
# "A05 Security Misconfiguration" / "SI-2" mapping if nothing matches.
_COMPLIANCE_RULES: list[tuple[str, str, str]] = [
# (substring to match in finding type/title, OWASP, NIST)
("sql injection", "A03 Injection", "SI-10"),
("f-string sql", "A03 Injection", "SI-10"),
("union select", "A03 Injection", "SI-10"),
("command injection", "A03 Injection", "SI-10"),
("os.popen", "A03 Injection", "SI-10"),
("shell=true", "A03 Injection", "SI-10"),
("eval(", "A03 Injection", "SI-10"),
("exec(", "A03 Injection", "SI-10"),
("deserialize", "A08 Software and Data Integrity Failures", "SI-7"),
("pickle.loads", "A08 Software and Data Integrity Failures", "SI-7"),
("yaml.load", "A08 Software and Data Integrity Failures", "SI-7"),
("xss", "A03 Injection", "SI-10"),
("innerhtml", "A03 Injection", "SI-10"),
("dangerouslysetinnerhtml", "A03 Injection", "SI-10"),
("document.write", "A03 Injection", "SI-10"),
("prototype pollution", "A03 Injection", "SI-10"),
("__proto__", "A03 Injection", "SI-10"),
("path traversal", "A01 Broken Access Control", "AC-3"),
("access control", "A01 Broken Access Control", "AC-3"),
("authoriz", "A01 Broken Access Control", "AC-3"),
# Secrets / keys / tokens / credentials β Cryptographic Failures + SC-13/SC-28
("private key", "A02 Cryptographic Failures", "SC-12"),
("rsa key", "A02 Cryptographic Failures", "SC-12"),
("ssh key", "A02 Cryptographic Failures", "SC-12"),
("aws access key", "A02 Cryptographic Failures", "SC-28"),
("aws secret", "A02 Cryptographic Failures", "SC-28"),
("openai", "A02 Cryptographic Failures", "SC-28"),
("anthropic", "A02 Cryptographic Failures", "SC-28"),
("huggingface", "A02 Cryptographic Failures", "SC-28"),
("github pat", "A02 Cryptographic Failures", "SC-28"),
("github oauth", "A02 Cryptographic Failures", "SC-28"),
("slack bot token", "A02 Cryptographic Failures", "SC-28"),
("sendgrid", "A02 Cryptographic Failures", "SC-28"),
("twilio", "A02 Cryptographic Failures", "SC-28"),
("stripe", "A02 Cryptographic Failures", "SC-28"),
("paypal secret", "A02 Cryptographic Failures", "SC-28"),
("supabase service role", "A02 Cryptographic Failures", "SC-28"),
("database connection", "A02 Cryptographic Failures", "SC-28"),
("jwt secret", "A02 Cryptographic Failures", "SC-13"),
("hardcoded password", "A07 Identification and Authentication Failures", "IA-5"),
("hardcoded token", "A07 Identification and Authentication Failures", "IA-5"),
("hardcoded api key", "A07 Identification and Authentication Failures", "IA-5"),
("high-entropy", "A02 Cryptographic Failures", "SC-28"),
("secrets exposure", "A02 Cryptographic Failures", "SC-28"),
(".env file", "A05 Security Misconfiguration", "CM-6"),
# Crypto weaknesses β A02
("md5", "A02 Cryptographic Failures", "SC-13"),
("sha1", "A02 Cryptographic Failures", "SC-13"),
("sha-1", "A02 Cryptographic Failures", "SC-13"),
("des.new", "A02 Cryptographic Failures", "SC-13"),
("rc4", "A02 Cryptographic Failures", "SC-13"),
("math.random", "A02 Cryptographic Failures", "SC-13"),
("random.random", "A02 Cryptographic Failures", "SC-13"),
# SSRF / network
("ssrf", "A10 Server-Side Request Forgery", "SC-7"),
("curl ", "A10 Server-Side Request Forgery", "SC-7"),
("wget ", "A10 Server-Side Request Forgery", "SC-7"),
("verify=false", "A02 Cryptographic Failures", "SC-8"),
("check_hostname=false", "A02 Cryptographic Failures", "SC-8"),
("http://", "A02 Cryptographic Failures", "SC-8"),
# Misconfiguration / outdated deps β A06
("vulnerable dependency", "A06 Vulnerable and Outdated Components", "SI-2"),
("outdated", "A06 Vulnerable and Outdated Components", "SI-2"),
# Logging / monitoring
("audit", "A09 Security Logging and Monitoring Failures", "AU-2"),
("logging", "A09 Security Logging and Monitoring Failures", "AU-2"),
]
_DEFAULT_OWASP = "A05 Security Misconfiguration"
_DEFAULT_NIST = "SI-2"
def map_compliance(finding: dict) -> dict:
"""
Map a single finding to OWASP Top 10 (2021) and a relevant NIST 800-53
control family. Matching is done on the finding's `type`/`title`/`category`
fields (case-insensitive substring match), falling back to a generic
security misconfiguration mapping.
Returns:
{ "owasp": "A05 Security Misconfiguration", "nist": "SI-2" }
"""
haystack = " ".join(str(finding.get(k, "")) for k in (
"type", "title", "category", "description", "match"
)).lower()
for needle, owasp, nist in _COMPLIANCE_RULES:
if needle in haystack:
return {"owasp": owasp, "nist": nist}
return {"owasp": _DEFAULT_OWASP, "nist": _DEFAULT_NIST}
def enrich_findings_with_compliance(findings: list) -> list:
"""
Return a NEW list of findings, each with 'owasp' and 'nist' keys added.
Does not mutate the input list β original finding dicts are copied.
Safe to call on any list of dicts; unknown shapes get the default mapping.
"""
enriched = []
for f in findings or []:
f2 = dict(f)
f2.update(map_compliance(f))
enriched.append(f2)
return enriched
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# 3. REPOSITORY HEALTH DASHBOARD
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
def compute_repo_health(
findings: list,
dependency_findings: list | None = None,
dependency_count: int = 0,
) -> dict:
"""
Build the repository health summary shown on the dashboard after a
repository scan.
Args:
findings: list of secret/vuln finding dicts (from scanner.py)
dependency_findings: list of dependency vulnerability dicts (optional)
dependency_count: total number of dependencies detected (optional)
Returns:
{
"security_score": int,
"risk_level": str,
"critical_count": int,
"high_count": int,
"medium_count": int,
"low_count": int,
"secret_count": int,
"dependency_count": int,
"outdated_packages": int,
}
"""
findings = findings or []
dependency_findings = dependency_findings or []
counts = {"CRITICAL": 0, "HIGH": 0, "MEDIUM": 0, "LOW": 0}
secret_count = 0
for f in findings:
sev = (f.get("severity") or "LOW").upper()
if sev not in counts:
sev = "LOW"
counts[sev] += 1
# A finding is a "secret" if its type/category mentions a secret-ish term
haystack = (str(f.get("type", "")) + " " + str(f.get("category", "")) + " " + str(f.get("title", ""))).lower()
if any(term in haystack for term in (
"key", "token", "secret", "password", "credential",
"private key", "entropy", "jwt", "connection string"
)):
secret_count += 1
# Fold dependency findings into the severity counts too
outdated = 0
for d in dependency_findings:
sev = (d.get("severity") or "LOW").upper()
if sev not in counts:
sev = "LOW"
counts[sev] += 1
outdated += 1
score_info = compute_security_score(findings + dependency_findings)
return {
"security_score": score_info["security_score"],
"risk_level": score_info["risk_level"],
"critical_count": counts["CRITICAL"],
"high_count": counts["HIGH"],
"medium_count": counts["MEDIUM"],
"low_count": counts["LOW"],
"secret_count": secret_count,
"dependency_count": dependency_count,
"outdated_packages": outdated,
}
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# 4. AI AUTO-FIX ENGINE (rule-based before/after with confidence)
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# Each rule: (regex to find dangerous pattern, replacement template, explanation, confidence)
# `replacement` may use \1 \2 etc. backreferences from the match.
_AUTOFIX_RULES: list[tuple[re.Pattern, str, str, int]] = [
(re.compile(r'\beval\s*\((.+?)\)'),
r'ast.literal_eval(\1)',
"eval() executes arbitrary Python code. ast.literal_eval() only parses literal "
"Python data structures (strings, numbers, tuples, lists, dicts, booleans, None), "
"making it safe for parsing user-supplied data.",
95),
(re.compile(r'\bexec\s*\((.+?)\)'),
r'# exec(\1) β remove dynamic execution; use explicit function dispatch instead',
"exec() runs arbitrary code with full interpreter access. Replace dynamic code "
"execution with an explicit mapping of allowed operations (e.g. a dict of functions).",
80),
(re.compile(r'os\.system\s*\((.+?)\)'),
r'subprocess.run(\1, shell=False, check=True)',
"os.system() invokes a full shell, enabling shell injection via unsanitised input. "
"subprocess.run() with shell=False and an argument list avoids shell interpretation entirely.",
90),
(re.compile(r'subprocess\.call\((.+?),\s*shell\s*=\s*True\)'),
r'subprocess.run(\1, shell=False, check=True)',
"shell=True passes your command through a shell, allowing injection if any part "
"of the command includes untrusted input. Pass arguments as a list with shell=False.",
90),
(re.compile(r'pickle\.loads?\((.+?)\)'),
r'json.loads(\1)',
"pickle.loads() can execute arbitrary code embedded in the pickled data. If the "
"data is simple (dicts, lists, strings, numbers), json.loads() is a safe drop-in "
"replacement with no code-execution risk.",
70),
(re.compile(r'yaml\.load\((.+?)\)'),
r'yaml.safe_load(\1)',
"yaml.load() can instantiate arbitrary Python objects from the YAML document, "
"which can lead to code execution. yaml.safe_load() restricts parsing to basic "
"YAML tags only.",
95),
(re.compile(r'hashlib\.md5\((.+?)\)'),
r'hashlib.sha256(\1)',
"MD5 is cryptographically broken and vulnerable to collision attacks. "
"SHA-256 is a drop-in replacement for non-password hashing needs.",
85),
(re.compile(r'hashlib\.sha1\((.+?)\)'),
r'hashlib.sha256(\1)',
"SHA-1 is deprecated for security-sensitive use due to demonstrated collision "
"attacks. SHA-256 provides a stronger, drop-in replacement.",
85),
(re.compile(r'verify\s*=\s*False'),
r'verify=True',
"Disabling SSL/TLS certificate verification allows man-in-the-middle attacks. "
"Re-enable verification; if using a private CA, pass the CA bundle path instead "
"of disabling verification entirely.",
90),
(re.compile(r'random\.random\(\)'),
r'secrets.token_hex(16)',
"random.random() is a non-cryptographic PRNG and is predictable. For security "
"tokens, session IDs, or password reset codes, use the `secrets` module which "
"is designed for cryptographic use.",
80),
(re.compile(r'Math\.random\(\)'),
r'crypto.getRandomValues(new Uint32Array(1))[0]',
"Math.random() in JavaScript is not cryptographically secure. For tokens or "
"security-relevant randomness, use the Web Crypto API's getRandomValues().",
80),
(re.compile(r'\.innerHTML\s*=\s*(.+)'),
r'.textContent = \1',
"Assigning to innerHTML with untrusted data allows XSS via injected <script> or "
"event-handler attributes. textContent renders the value as plain text, neutralising "
"any HTML/JS payload. If HTML rendering is required, sanitise with DOMPurify first.",
75),
(re.compile(r"f[\"']SELECT.+?\{.+?\}.+?[\"']"),
r'cursor.execute("SELECT ... WHERE id = %s", (value,))',
"f-string interpolation directly into SQL allows SQL injection if any "
"interpolated value comes from user input. Use parameterised queries β the "
"driver escapes values safely.",
85),
]
def generate_auto_fix(finding: dict) -> dict | None:
"""
Given a finding dict (must contain a 'match' or 'description'/'type' field
with the offending code snippet), attempt to generate a before/after fix.
Returns:
{
"before": "<original snippet>",
"after": "<suggested replacement>",
"explanation": "...",
"confidence": 0-100
}
or None if no auto-fix rule matches.
This is a pure, synchronous, rule-based engine β it never calls the network
or the AI model, so it's safe to call on every finding without rate limits.
"""
snippet = finding.get("match") or finding.get("title") or ""
if not snippet:
return None
for pattern, replacement, explanation, confidence in _AUTOFIX_RULES:
m = pattern.search(snippet)
if m:
try:
after = pattern.sub(replacement, snippet, count=1)
except re.error:
after = replacement
return {
"before": snippet.strip(),
"after": after.strip(),
"explanation": explanation,
"confidence": confidence,
}
return None
def enrich_findings_with_autofix(findings: list) -> list:
"""
Return a NEW list of findings, each with an 'auto_fix' key added
(None if no rule matched). Does not mutate input.
"""
enriched = []
for f in findings or []:
f2 = dict(f)
f2["auto_fix"] = generate_auto_fix(f)
enriched.append(f2)
return enriched
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# COMBINED ENRICHMENT (used by app.py and tasks.py)
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
def enrich_findings_full(findings: list) -> list:
"""
Single entry point: adds compliance mapping (owasp/nist) and
auto_fix suggestions to every finding in one pass. Does not mutate input.
"""
findings = findings or []
enriched = []
for f in findings:
f2 = dict(f)
f2.update(map_compliance(f))
f2["auto_fix"] = generate_auto_fix(f)
enriched.append(f2)
return enriched
|