File size: 15,503 Bytes
6cb82c6
 
 
 
 
 
 
295bbd1
1948258
6cb82c6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b6e6a8e
 
 
 
 
6cb82c6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
18b82c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
295bbd1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0d43664
 
 
 
 
 
 
295bbd1
 
 
0d43664
 
 
 
295bbd1
 
 
 
 
 
 
0d43664
 
295bbd1
 
 
 
 
0d43664
 
 
 
 
295bbd1
 
 
 
 
 
 
 
6cb82c6
 
 
 
 
295bbd1
 
 
 
 
 
 
 
 
 
 
 
6cb82c6
 
295bbd1
6cb82c6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
"""
Layer 6 β€” Chain Correlator
Cross-domain finding escalation. No other scanner does this.
Combines findings from different domains to create CRITICAL chain findings.
"""
from __future__ import annotations
import uuid
from pathlib import Path
from typing import List, Dict
from .models import Finding, Severity, Confidence, ScanDomain

# Chain rule: (condition_check_fn, new_title, new_description, new_explanation)
def correlate_chains(findings: List[Finding]) -> List[Finding]:
    """Detect multi-finding chains and return new CRITICAL chain findings."""
    chain_findings: List[Finding] = []

    # Index findings by domain and check_id for fast lookup
    by_domain: Dict[ScanDomain, List[Finding]] = {}
    for f in findings:
        by_domain.setdefault(f.domain, []).append(f)

    secrets = by_domain.get(ScanDomain.SECRETS, [])
    iac = by_domain.get(ScanDomain.IAC, [])
    sast = by_domain.get(ScanDomain.SAST, [])
    config = by_domain.get(ScanDomain.CONFIG, [])
    frontend = by_domain.get(ScanDomain.FRONTEND, [])

    active_secrets = [
        f for f in secrets 
        if not f.is_false_positive 
        and (f.validity == "active" or f.check_id == "COMMITTED_ENV_FILE" or f.confidence == Confidence.HIGH)
    ]
    all_active = [f for f in findings if not f.is_false_positive]

    # Chain 1: Exposed admin route + no auth + hardcoded password
    admin_routes = [f for f in sast if "admin" in (f.file_path or "").lower() or "admin" in f.description.lower()]
    admin_secrets = [f for f in secrets if "admin" in f.title.lower() or "password" in f.title.lower()]
    if admin_routes and admin_secrets:
        chain_findings.append(_make_chain(
            title="CHAIN: Full Admin Takeover β€” Exposed Admin Route + Hardcoded Credentials",
            description="An unauthenticated admin endpoint was found alongside hardcoded admin credentials. An attacker can access full admin functionality with zero authentication.",
            contributing=[admin_routes[0], admin_secrets[0]],
            exploit="An attacker navigates to the admin endpoint and uses the hardcoded credentials found in source code to gain full admin access.",
        ))

    # Chain 2: Live DB secret + DB port exposed in IaC
    db_secrets = [f for f in active_secrets if "db" in f.title.lower() or "connection" in f.title.lower() or "mongodb" in f.title.lower() or "postgres" in f.title.lower()]
    exposed_ports = [f for f in iac if "port" in f.description.lower() or "0.0.0.0" in f.description.lower()]
    if db_secrets and exposed_ports:
        chain_findings.append(_make_chain(
            title="CHAIN: Direct Database Access β€” Exposed Credentials + Open Port",
            description="A live database connection string was found in source code AND the database port is exposed to the internet via infrastructure misconfiguration.",
            contributing=[db_secrets[0], exposed_ports[0]],
            exploit="An attacker uses the leaked connection string with any database client to connect directly to the production database from the internet.",
        ))

    # Chain 3: Client-exposed secret + .env file committed
    client_secrets = [f for f in frontend if "CLIENT_EXPOSED_SECRET" in (f.check_id or "")]
    committed_env = [f for f in secrets if "COMMITTED_ENV_FILE" in (f.check_id or "")]
    if client_secrets and committed_env:
        chain_findings.append(_make_chain(
            title="CHAIN: Payment/Service API Fully Compromised β€” Client Bundle + Git History",
            description="A secret key is exposed both in the browser bundle AND committed to git history. The key is accessible to any website visitor AND to anyone who can view the repository.",
            contributing=[client_secrets[0], committed_env[0]],
            exploit="Any user viewing page source OR anyone with repository access has the live API key. For payment providers like Stripe, this enables fraudulent charges.",
        ))

    # Chain 4: GraphQL introspection + no auth + sensitive data in schema
    graphql = [f for f in config if "GRAPHQL" in (f.check_id or "")]
    no_auth_sast = [f for f in sast if "auth" in f.title.lower() and "missing" in f.title.lower()]
    if graphql and no_auth_sast:
        chain_findings.append(_make_chain(
            title="CHAIN: Full Data Model Exfiltration β€” GraphQL Introspection + No Auth",
            description="GraphQL introspection is enabled on an unauthenticated endpoint. Attackers can enumerate the entire data schema and query any data.",
            contributing=[graphql[0], no_auth_sast[0]],
            exploit="An attacker runs an introspection query to map all types and fields, then crafts queries to dump user data, orders, or any sensitive information without authenticating.",
        ))

    # Chain 5: Live verified secret + no rate limiting
    live_secrets = [f for f in active_secrets]
    no_rate_limit = [f for f in config if "RATE_LIMIT" in (f.check_id or "")]
    if live_secrets and no_rate_limit:
        chain_findings.append(_make_chain(
            title="CHAIN: Credential Stuffing Highway β€” Live API Keys + No Rate Limiting",
            description="Verified live API keys are exposed in the codebase and there is no rate limiting on API endpoints. Attackers can enumerate and use the keys at full speed.",
            contributing=[live_secrets[0], no_rate_limit[0]],
            exploit="Attacker extracts the live API key from the source code/git history and makes unlimited API calls. No rate limiting means no detection or blocking.",
        ))

    # Chain 6: DAST confirmed + static confirmed = escalate
    dast_findings = by_domain.get(ScanDomain.DAST, [])
    for dast_f in dast_findings:
        # Find matching static finding
        for static_f in all_active:
            if static_f.domain == ScanDomain.DAST:
                continue
            if _findings_overlap(dast_f, static_f):
                chain_findings.append(_make_chain(
                    title=f"CHAIN: Runtime + Static Confirmed β€” {dast_f.title}",
                    description="This vulnerability was independently confirmed by both static analysis AND runtime DAST scanning. Exploitability is near-certain.",
                    contributing=[dast_f, static_f],
                    exploit=f"Statically detected and runtime-confirmed: {dast_f.description}",
                ))
                break

    # Chain 7: Prototype Pollution -> Remote Code Execution / Auth Bypass
    proto_pollution = [f for f in sast if "prototype" in f.title.lower() or "pollution" in f.title.lower()]
    unsafe_eval = [f for f in sast if "eval" in f.title.lower() or "exec" in f.title.lower() or "template" in f.title.lower()]
    if proto_pollution and unsafe_eval:
        chain_findings.append(_make_chain(
            title="CHAIN: Critical Prototype Pollution to Remote Code Execution (RCE)",
            description="Prototype pollution combined with dynamic code execution / template generation enables unauthenticated Remote Code Execution on the hosting server.",
            contributing=[proto_pollution[0], unsafe_eval[0]],
            exploit="An attacker pollutes object prototypes (e.g., Object.prototype) to inject dynamic properties that are executed when the server parses template variables or runs dynamic evaluations.",
        ))

    # Chain 8: SSRF -> Internal Cloud Metadata / Port Exposure
    ssrf = [f for f in sast if "ssrf" in f.title.lower() or "request" in f.title.lower() or "fetch" in f.title.lower()]
    ports = [f for f in iac if "port" in f.description.lower() or "internal" in f.description.lower()]
    if ssrf and ports:
        chain_findings.append(_make_chain(
            title="CHAIN: SSRF Server Compromise β€” Internal Network Port Crawling & Cloud Metadata Leaks",
            description="Server-Side Request Forgery vulnerability combined with internal exposed infrastructure ports enables crawling private network assets and stealing temporary IAM credentials.",
            contributing=[ssrf[0], ports[0]],
            exploit="Attacker forces the web server to make requests to internal service endpoints or cloud metadata APIs (e.g. 169.254.169.254) to bypass firewall barriers.",
        ))

    # Chain 9: IDOR -> Database Taint Leakage
    idor = [f for f in sast if "idor" in f.title.lower() or "auth" in f.title.lower() or "permission" in f.title.lower()]
    db_leak = [f for f in sast if "sql" in f.title.lower() or "query" in f.title.lower() or "leak" in f.title.lower()]
    if idor and db_leak:
        chain_findings.append(_make_chain(
            title="CHAIN: Full Tenant Database Exfiltration β€” IDOR Access + Direct DB Query Exposure",
            description="Lack of proper session ownership validation (IDOR) combined with direct database queries allows unauthorized users to query and exfiltrate data from other tenants.",
            contributing=[idor[0], db_leak[0]],
            exploit="An attacker manipulates query indices or resource primary keys in the request parameters, querying direct records of other tenants due to missing permission validations.",
        ))

    # Chain 10: LLM Prompt Injection β†’ Code/Command Execution (CROSS-DOMAIN)
    # This is the most critical chain: attacker injects prompt that causes LLM to output
    # code that gets eval'd/exec'd server-side. Findings can be LLM_VULNS domain (from
    # layer1_llm_vulns.py) or SAST domain (from semgrep ai-agent rules that semgrep
    # tagged as SAST). We match BOTH by domain AND by title/check_id patterns.
    LLM_CHECK_ID_KW = {"prompt", "llm", "langchain", "ai-agent", "generative", "ai-chat", "chatbot", "chatcompletion"}
    IS_LLM_KW_IN_TITLE = {"prompt", "llm", "generative", "ai-chat", "langchain"}
    IS_LLM_KW_IN_CHECK = {"prompt", "llm", "langchain", "ai-agent", "injection", "generative", "output_eval", "tool_execution"}

    llm_injection_findings = [
        f for f in findings
        if not f.is_false_positive and f.severity not in {None, Severity.LOW}
        and (
            f.domain == ScanDomain.LLM_VULNS and "prompt" in f.title.lower()
            or any(kw in f.title.lower() for kw in IS_LLM_KW_IN_TITLE)
            or any(kw in (f.check_id or "").lower() for kw in IS_LLM_KW_IN_CHECK)
        )
    ]
    unsafe_llm_output_sinks = [
        f for f in sast
        if not f.is_false_positive
        and (
            any(kw in f.title.lower() for kw in ["eval", "exec", "template injection", "rce", "command injection"])
            or ("langchain" in (f.check_id or "").lower() and any(kw in f.title.lower() for kw in ["eval", "exec", "template", "command"]))
            or ("llm_output" in (f.check_id or "").lower())
        )
    ]
    if llm_injection_findings and unsafe_llm_output_sinks:
        # Deduplicate by (chain_title, llm_file, sink_file) to avoid NΓ—M duplicates
        # when multiple findings from the same file satisfy both criteria.
        _seen_chains: set = set()

        def _chain_key(title: str, f1: Finding, f2: Finding) -> tuple:
            return (title, f1.file_path, f2.file_path)

        for llm_f in llm_injection_findings:
            for sink_f in unsafe_llm_output_sinks:
                if llm_f.file_path == sink_f.file_path:
                    key = _chain_key("CHAIN: LLM Prompt Injection β†’ Arbitrary Code Execution", llm_f, sink_f)
                    if key in _seen_chains:
                        continue
                    _seen_chains.add(key)
                    chain_findings.append(_make_chain(
                        title="CHAIN: LLM Prompt Injection β†’ Arbitrary Code Execution",
                        description="A prompt injection vulnerability (OWASP LLM01) was found alongside a dynamic code execution sink (eval/exec) in the same file. An attacker injects a malicious prompt that causes the LLM to output code, which is then executed server-side.",
                        contributing=[llm_f, sink_f],
                        exploit="Attacker sends a crafted input that is injected into the LLM prompt. The LLM's response contains malicious JavaScript/payload that reaches an eval() or exec() call, resulting in arbitrary server-side code execution (RCE).",
                    ))
                    break

            # Cross-file chains using taint paths
            if llm_f.file_path and llm_f.taint_path:
                for sink_f in unsafe_llm_output_sinks:
                    if sink_f.file_path and llm_f.file_path != sink_f.file_path:
                        llm_dir = str(Path(llm_f.file_path).parent).replace("\\", "/")
                        sink_dir = str(Path(sink_f.file_path).parent).replace("\\", "/")
                        if llm_dir == sink_dir:
                            key = _chain_key("CHAIN: LLM Prompt Injection β†’ Cross-Module Code Execution", llm_f, sink_f)
                            if key in _seen_chains:
                                continue
                            _seen_chains.add(key)
                            chain_findings.append(_make_chain(
                                title="CHAIN: LLM Prompt Injection β†’ Cross-Module Code Execution",
                                description="A prompt injection in a file using an LLM is chained to a code execution sink in a related module in the same directory. The LLM output can flow to the dangerous sink.",
                                contributing=[llm_f, sink_f],
                                exploit="Attacker injects a prompt that manipulates the LLM output, which is then executed via eval/exec calls in a related module in the same service.",
                            ))
                            break

    return chain_findings

def _findings_overlap(f1: Finding, f2: Finding) -> bool:
    """Check if two findings likely refer to the same vulnerability."""
    if f1.file_path and f2.file_path and f1.file_path == f2.file_path:
        # Same file: check if line numbers are close (within 10 lines)
        if f1.line_number is not None and f2.line_number is not None:
            if abs(f1.line_number - f2.line_number) <= 10:
                return True
        else:
            # Fall back to verifying if they share semantic context via titles
            title_words_1 = set(f1.title.lower().split())
            title_words_2 = set(f2.title.lower().split())
            overlap = title_words_1 & title_words_2 - {"the", "a", "in", "of", "and", "or", "is", "are", "for", "dast", "sast"}
            if len(overlap) >= 2:
                return True

    title_words_1 = set(f1.title.lower().split())
    title_words_2 = set(f2.title.lower().split())
    overlap = title_words_1 & title_words_2 - {"the", "a", "in", "of", "and", "or", "is", "are", "for", "dast", "sast"}
    return len(overlap) >= 3

def _make_chain(title: str, description: str, contributing: List[Finding], exploit: str) -> Finding:
    ids = [f.id for f in contributing]
    return Finding(
        id=uuid.uuid4().hex,
        title=title,
        description=description,
        severity=Severity.CRITICAL,
        confidence=Confidence.HIGH,
        domain=ScanDomain.CHAIN,
        check_id="CHAIN_CORRELATION",
        check_category="chain",
        explanation=exploit,
        suggested_fix="Address each contributing finding. Start with the secret/credential exposure first, then fix the infrastructure and code issues.",
        chained_from=ids,
        confirmed_runtime=any(f.confirmed_runtime for f in contributing),
    )