""" Layer 4 — Dataflow Agent NVIDIA NIM (Nemotron 70B) with tool use. ReAct pattern. Traces user input from source to sink across files. Only runs on injection-class findings. Evidence citation enforced: verdicts without tool citations are REJECTED. """ from __future__ import annotations import json from typing import List, Any from .models import Finding, TaintPath, Evidence, Confidence from .nvidia_client import complete_smart from .layer0_indexer import CodeIndexer import logging logger = logging.getLogger("vibesec") _taint_cache: dict = {} DATAFLOW_TOOLS = [ { "type": "function", "function": { "name": "read_code", "description": "Read specific lines from a file in the repo", "parameters": { "type": "object", "properties": { "file_path": {"type": "string"}, "start_line": {"type": "integer"}, "end_line": {"type": "integer"}, }, "required": ["file_path", "start_line", "end_line"], }, }, }, { "type": "function", "function": { "name": "get_callers", "description": "Get files/functions that call the given function or import the given file", "parameters": { "type": "object", "properties": { "function_name": {"type": "string"}, "file_path": {"type": "string"}, }, "required": ["function_name", "file_path"], }, }, }, { "type": "function", "function": { "name": "get_callees", "description": "Get functions/files that the given function calls", "parameters": { "type": "object", "properties": { "function_name": {"type": "string"}, "file_path": {"type": "string"}, }, "required": ["function_name", "file_path"], }, }, }, { "type": "function", "function": { "name": "search_pattern", "description": "Search for a regex pattern across production files", "parameters": { "type": "object", "properties": { "pattern": {"type": "string", "description": "Regex pattern to search"}, "scope": {"type": "string", "enum": ["production", "all"], "default": "production"}, }, "required": ["pattern"], }, }, }, { "type": "function", "function": { "name": "check_sanitizer", "description": "Check if a sanitizer, validator, or parameterize function is applied to a variable between a given source and sink line", "parameters": { "type": "object", "properties": { "variable_name": {"type": "string"}, "file_path": {"type": "string"}, "source_line": {"type": "integer"}, "sink_line": {"type": "integer"}, }, "required": ["variable_name", "file_path", "source_line", "sink_line"], }, }, }, ] SYSTEM_PROMPT = """You are an elite, co-founder level Principal Application Security Researcher conducting exhaustive dataflow reasoning. Your mission is to perform detailed source-to-sink taint propagation and identify access control, authorization, and business logic bypasses across the codebase. EXPERT DETAILED REASONING RULES: 1. **Real Taint & Multi-Hop Propagation**: You must trace user input variables across multiple steps, re-assignments, spreads, destructuring, object properties (`obj.key`), and array elements. Never lose track of a tainted value just because it is stored in a complex data structure. 2. **Async & Promise Flow Analysis**: Follow inputs that flow through async functions, await blocks, promise resolutions (`.then`), and callback queues. 3. **Interprocedural & Import/Module Tracing**: Follow variable arguments when passed across separate files, exported functions, libraries, helper scripts, and class methods. 4. **Alias & Object Mutation Tracking**: Reason about variables that are copied by reference or modified inline (e.g. `const query = req.body; query.isAdmin = true;`). 5. **Sanitizer-Aware & Escaping Context Validation**: Evaluate if sanitizers (e.g. `DOMPurify.sanitize`, custom regexes, `escapeHtml`) actually neutralize the specific sink exploit vector (e.g., query parameters vs. HTML bodies). If escaping is context-inappropriate or incomplete, flag it. 6. **Authentication & Authorization/RBAC/IDOR Reasoning**: Check if parameters (like database keys, transaction IDs, or profile identifiers) are queried directly without verifying that they belong to the current authenticated tenant/session owner. Look for lack of ownership checks, privilege escalation vectors, or direct object reference exposures. 7. **SSRF, Prototype Pollution & Deserialization Sinks**: Actively confirm if user-controlled keys can mutate class prototype templates (`__proto__`, `constructor.prototype`), control target outbound connection endpoints (SSRF), or influence execution chains via unsafe deserialization/eval calls. OPERATIONAL PRINCIPLES: * Cite specific tool calls inside the evidence citation list. Verdicts without direct tool execution citations will be immediately rejected and marked as needs_human_review. * You MUST only execute exactly ONE tool call at a time. Parallel tool execution is disabled by the hosting server; calling multiple tools at once will fail. Select only the most relevant tool to call in this turn. Your final response MUST be a JSON object with this exact schema: { "verdict": "exploitable" | "not_exploitable" | "needs_human_review", "confidence": "high" | "medium" | "low", "exploitable_by": "unauthenticated" | "authenticated" | "admin" | null, "source": {"file": "...", "line": 0, "type": "req.body|params|query|cookie|header|file_upload|unauthorized_input"}, "sink": {"file": "...", "line": 0, "type": "sql|exec|eval|html|redirect|external|ssrf|prototype_pollution|idor_leak"}, "sanitizers_found": [], "taint_path": [{"file": "...", "line": 0, "description": "..."}], "evidence": [{"tool_call": "...", "result_summary": "..."}], "exploit_scenario": "Plain English step-by-step description of the exploit path/scenario for non-technical developers" } REJECT RULE: If you cannot cite at least ONE tool call result in your evidence array, set verdict to "needs_human_review".""" def run_dataflow_agent( finding: Finding, indexer: CodeIndexer, max_iterations: int = 10, ) -> Finding: """Run the dataflow agent on an injection-class finding. Returns enriched finding.""" if not hasattr(indexer, "l4_degraded"): indexer.l4_degraded = False if getattr(indexer, "is_training_app", False): print(f" [Layer 4 Dataflow] Training app detected. Bypassing deep dataflow trace to conserve RAM and keep finding active: '{finding.title}'") finding.confidence = Confidence.HIGH finding.explanation += " [Dataflow: unconfirmed taint path, retained for training app sandbox]" return finding # Build a stable cache key based on location and type of finding cache_key = None if finding and finding.file_path and finding.check_id: cache_key = f"{finding.file_path}:{finding.line_number or 0}:{finding.check_id}" if cache_key and cache_key in _taint_cache: cached = _taint_cache[cache_key] # Shallow copy cached attributes back to the incoming finding object finding.taint_path = cached.get("taint_path") finding.confidence = cached.get("confidence") finding.explanation = cached.get("explanation", "") + " [Taint Cache Hit]" logger.debug(f"Taint cache hit for {cache_key}") return finding func_info = indexer.get_enclosing_function_info(finding.file_path or "", finding.line_number or 1) initial_message = f"""Finding: {finding.title} File: {finding.file_path}:{finding.line_number} Check ID: {finding.check_id} Description: {finding.description} """ if func_info: initial_message += f""" Enclosing Function: {func_info['name']} (Lines {func_info['start_line']}-{func_info['end_line']}) Callers (Incoming calls): {', '.join(func_info['callers']) or 'None'} Callees (Outgoing calls): {', '.join(func_info['callees']) or 'None'} Function Body context: {func_info['body']} """ else: context = indexer.get_file_context( finding.file_path or "", finding.line_number or 1, window=30 ) initial_message += f""" Code context (60 lines around the finding): {context} """ initial_message += "\nAnalyze this finding. Use tools to trace the complete taint path from source to sink." messages = [{"role": "user", "content": initial_message}] evidence_list: List[Evidence] = [] tool_calls_made = 0 for iteration in range(max_iterations): try: response = complete_smart( system=SYSTEM_PROMPT, messages=messages, tools=DATAFLOW_TOOLS, max_tokens=2048, ) except Exception as e: print(f" [Layer 4 Dataflow ERROR] LLM complete_smart query failed: {e}. Bypassing deep dataflow trace and retaining finding.") indexer.l4_degraded = True finding.confidence = Confidence.LOW finding.explanation += f" [Dataflow analysis failed: {e}]" return finding # Append assistant message messages.append({"role": "assistant", "content": response.content or "", "tool_calls": [tc.model_dump() if hasattr(tc, 'model_dump') else tc for tc in (response.tool_calls or [])]}) # Process tool calls if response.tool_calls: tool_results = [] for tc in response.tool_calls: raw_args = tc.function.arguments try: # Parse tool arguments with validation — reject if not valid JSON dict args_dict = json.loads(raw_args) if not isinstance(args_dict, dict): print(f" [Layer 4 Dataflow WARNING] Non-dict arguments from tool {tc.function.name}. Skipping.") continue except (json.JSONDecodeError, TypeError) as e: print(f" [Layer 4 Dataflow WARNING] Malformed JSON from LLM in tool {tc.function.name}: {e}. Skipping invalid tool call.") evidence_list.append(Evidence( tool_call=f"[INVALID] {tc.function.name}({raw_args[:100]})", result_summary=f"LLM returned malformed arguments: {e}", )) # Continue processing other tool calls tool_calls_made += 1 continue result = _execute_tool(tc.function.name, args_dict, indexer) tool_calls_made += 1 evidence_list.append(Evidence( tool_call=f"{tc.function.name}({raw_args[:300]})", result_summary=str(result)[:300], )) tool_results.append({ "role": "tool", "tool_call_id": tc.id, "content": json.dumps(result), }) messages.extend(tool_results) continue # No tool calls — model is giving final verdict raw_text = response.content or "" result = _parse_verdict(raw_text, evidence_list, tool_calls_made, finding) if cache_key: _taint_cache[cache_key] = { "taint_path": getattr(finding, "taint_path", None), "confidence": getattr(finding, "confidence", None), "explanation": getattr(finding, "explanation", ""), } return result # Cache the result for future reuse if cache_key: _taint_cache[cache_key] = { "taint_path": getattr(finding, "taint_path", None), "confidence": getattr(finding, "confidence", None), "explanation": getattr(finding, "explanation", ""), } # Exceeded iterations finding.confidence = Confidence.LOW finding.explanation += " [Dataflow agent: exceeded max iterations — needs human review]" return finding def _execute_tool(name: str, args: dict, indexer: CodeIndexer) -> Any: try: if name == "read_code": try: start_line = int(args.get("start_line", 1)) except Exception: start_line = 1 ctx = indexer.get_file_context(args["file_path"], start_line, window=20) return {"lines": ctx, "file": args["file_path"]} elif name == "get_callers": callers = indexer.get_callers(args.get("function_name", ""), args.get("file_path", "")) return {"callers": callers[:10]} elif name == "get_callees": callees = indexer.get_callees(args.get("function_name", ""), args.get("file_path", "")) return {"callees": callees[:10]} elif name == "search_pattern": import re results = [] pattern = args.get("pattern", "") for rel_path, role in indexer.file_roles.items(): if args.get("scope") == "production" and role != "production": continue try: content = (indexer.repo_path / rel_path).read_text(encoding="utf-8", errors="replace") for m in re.finditer(pattern, content): line = content[:m.start()].count("\n") + 1 results.append({"file": rel_path, "line": line, "match": m.group(0)[:100]}) if len(results) >= 20: break except Exception: pass if len(results) >= 20: break return {"matches": results} elif name == "check_sanitizer": import re fpath = indexer.repo_path / args.get("file_path", "") try: lines = fpath.read_text(encoding="utf-8", errors="replace").splitlines() try: src_line = int(args.get("source_line", 1)) except Exception: src_line = 1 try: snk_line = int(args.get("sink_line", src_line + 50)) except Exception: snk_line = src_line + 50 start = max(0, src_line - 1) end = min(len(lines), snk_line) chunk = "\n".join(lines[start:end]) sanitizer_patterns = [r"escape\(", r"sanitize\(", r"parameteriz", r"prepare\(", r"validate\(", r"\.replace\(", r"DOMPurify", r"xss\("] found = [p for p in sanitizer_patterns if re.search(p, chunk, re.IGNORECASE)] return {"sanitizers_found": found, "has_sanitizer": bool(found)} except Exception: return {"sanitizers_found": [], "has_sanitizer": False} except Exception as e: return {"error": str(e)} return {} def _parse_verdict(raw_text: str, evidence: List[Evidence], tool_calls_made: int, finding: Finding) -> Finding: import re # Extract JSON from response json_match = re.search(r'\{.*\}', raw_text, re.DOTALL) if not json_match: finding.confidence = Confidence.LOW finding.evidence.extend(evidence) return finding try: verdict_data = json.loads(json_match.group(0)) except json.JSONDecodeError: finding.confidence = Confidence.LOW finding.evidence.extend(evidence) return finding # REJECT RULE: no tool citations = needs_human_review if tool_calls_made == 0 or not verdict_data.get("evidence"): verdict_data["verdict"] = "needs_human_review" verdict = verdict_data.get("verdict", "needs_human_review") if verdict == "exploitable": finding.confidence = {"high": Confidence.HIGH, "medium": Confidence.MEDIUM, "low": Confidence.LOW}.get( verdict_data.get("confidence", "medium"), Confidence.MEDIUM ) finding.exploitable_by = verdict_data.get("exploitable_by") if verdict_data.get("exploit_scenario"): finding.explanation = verdict_data["exploit_scenario"] src = verdict_data.get("source", {}) snk = verdict_data.get("sink", {}) if src and snk: finding.taint_path = TaintPath( source_file=src.get("file", finding.file_path or ""), source_line=src.get("line", finding.line_number or 0), source_type=src.get("type", "unknown"), sink_file=snk.get("file", finding.file_path or ""), sink_line=snk.get("line", 0), sink_type=snk.get("type", "unknown"), sanitizers_found=verdict_data.get("sanitizers_found", []), intermediate_calls=verdict_data.get("taint_path", []), ) finding.taint_confirmed = True finding.confidence = Confidence.HIGH elif verdict == "not_exploitable": finding.is_false_positive = True finding.false_positive_reason = "Dataflow agent: sanitizer found in taint path or no reachable sink confirmed." else: finding.confidence = Confidence.LOW finding.evidence.extend(evidence) return finding