yusufcalisir's picture
deploy: Hugging Face space upload
73ba4f5
Raw History Blame Contribute Delete
24.5 kB
"""AST-Based Dynamic Mutation Testing & Fault Injection Hardening Engine.
Generates real AST mutants across core domain and application services:
1. Relational comparison mutations (==, !=, >, >=, <, <=, in, not in)
2. Logical connector mutations (and <-> or, all <-> any)
3. Boundary threshold scale mutations (Byzantine anomaly bounds)
4. Four-Eyes validation mutations (supervisor approval bypass)
Executes actual test suites against each mutant and records true kill/survival rates.
"""
from __future__ import annotations
import ast
import copy
import logging
import sys
from collections.abc import Callable
from dataclasses import dataclass
from pathlib import Path
from typing import Any, cast
REPO_ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO_ROOT / "backend"))
import numpy as np
logger = logging.getLogger(__name__)
@dataclass
class MutantRecord:
mutant_id: str
target_module: str
lineno: int
mutation_type: str
description: str
status: str # "KILLED" or "SURVIVED"
killed_by: str = ""
class ASTMutator:
"""Dynamically parses Python source code, discovers mutation points, and executes tests."""
def __init__(self) -> None:
self.mutants: list[MutantRecord] = []
def run_all(self) -> list[MutantRecord]:
self.mutants.clear()
self._test_policy_engine_ast_mutants()
self._test_byzantine_defense_mutants()
self._test_four_eyes_mutants()
return self.mutants
def _test_policy_engine_ast_mutants(self) -> None:
"""Injects AST mutations into policy_engine.evaluate_condition and executes real tests."""
from app.application.services import policy_engine
file_path = REPO_ROOT / "backend" / "app" / "application" / "services" / "policy_engine.py"
source = file_path.read_text(encoding="utf-8")
tree = ast.parse(source, filename=str(file_path))
# Find evaluate_condition
eval_func_def = next(
(n for n in tree.body if isinstance(n, ast.FunctionDef) and n.name == "evaluate_condition"),
None,
)
if not eval_func_def:
return
class CandidateCollector(ast.NodeVisitor):
def __init__(self) -> None:
self.candidates: list[tuple[str, ast.AST, int | None, Any]] = []
def visit_Compare(self, node: ast.Compare) -> None:
for i, op in enumerate(node.ops):
self.candidates.append(("Compare", node, i, type(op)))
self.generic_visit(node)
def visit_BoolOp(self, node: ast.BoolOp) -> None:
self.candidates.append(("BoolOp", node, None, type(node.op)))
self.generic_visit(node)
def visit_Call(self, node: ast.Call) -> None:
if isinstance(node.func, ast.Name) and node.func.id in ("all", "any"):
self.candidates.append(("Call", node, None, node.func.id))
self.generic_visit(node)
collector = CandidateCollector()
collector.visit(eval_func_def)
compare_map = {
ast.Eq: ast.NotEq,
ast.NotEq: ast.Eq,
ast.Gt: ast.LtE,
ast.GtE: ast.Lt,
ast.Lt: ast.GtE,
ast.LtE: ast.Gt,
ast.In: ast.NotIn,
ast.NotIn: ast.In,
}
# Targeted test assertion suite for evaluate_condition
def run_policy_tests(eval_fn: Callable[[dict[str, Any], dict[str, Any]], bool]) -> str | None:
"""Runs policy boundary tests against candidate evaluate_condition. Returns failing test name or None."""
# Guard 0: Missing field or operator
if eval_fn({"operator": "=="}, {"amount": 100}):
return "test_missing_field_fails"
if eval_fn({"field": "amount"}, {"amount": 100}):
return "test_missing_operator_fails"
if eval_fn({"operator": "==", "value": 100}, cast("dict[str, Any]", {cast("Any", None): 100})):
return "test_missing_field_with_none_key_fails"
# Test 1: GTE boundary
if not eval_fn({"field": "amount", "operator": ">=", "value": 9000}, {"amount": 9000}):
return "test_gte_exact_boundary"
if eval_fn({"field": "amount", "operator": ">=", "value": 9000}, {"amount": 8999.99}):
return "test_gte_strict_under"
# Test 2: GT boundary
if eval_fn({"field": "amount", "operator": ">", "value": 9000}, {"amount": 9000}):
return "test_gt_exact_boundary"
if not eval_fn({"field": "amount", "operator": ">", "value": 9000}, {"amount": 9000.01}):
return "test_gt_strict_over"
# Test 3: LTE boundary
if not eval_fn({"field": "amount", "operator": "<=", "value": 9000}, {"amount": 9000}):
return "test_lte_exact_boundary"
if eval_fn({"field": "amount", "operator": "<=", "value": 9000}, {"amount": 9000.01}):
return "test_lte_strict_over"
# Test 4: LT boundary
if eval_fn({"field": "amount", "operator": "<", "value": 9000}, {"amount": 9000}):
return "test_lt_exact_boundary"
if not eval_fn({"field": "amount", "operator": "<", "value": 9000}, {"amount": 8999.99}):
return "test_lt_strict_under"
# Test 5: EQ boundary
if not eval_fn({"field": "status", "operator": "==", "value": "active"}, {"status": "ACTIVE"}):
return "test_eq_match"
if eval_fn({"field": "status", "operator": "==", "value": "active"}, {"status": "inactive"}):
return "test_eq_mismatch"
# Test 6: NEQ boundary
if not eval_fn({"field": "status", "operator": "!=", "value": "active"}, {"status": "inactive"}):
return "test_neq_mismatch"
if eval_fn({"field": "status", "operator": "!=", "value": "active"}, {"status": "active"}):
return "test_neq_match"
# Test 7: Logical AND
cond_and = {
"and": [
{"field": "amount", "operator": ">=", "value": 5000},
{"field": "velocity", "operator": ">", "value": 10},
]
}
if not eval_fn(cond_and, {"amount": 5000, "velocity": 12}):
return "test_and_both_true"
if eval_fn(cond_and, {"amount": 5000, "velocity": 8}):
return "test_and_one_false"
# Test 8: Logical OR
cond_or = {
"or": [
{"field": "amount", "operator": ">=", "value": 5000},
{"field": "velocity", "operator": ">", "value": 10},
]
}
if not eval_fn(cond_or, {"amount": 5000, "velocity": 8}):
return "test_or_one_true"
if eval_fn(cond_or, {"amount": 4000, "velocity": 8}):
return "test_or_both_false"
# Test 9: Logical NOT
cond_not = {"not": {"field": "amount", "operator": "<", "value": 1000}}
if eval_fn(cond_not, {"amount": 500}):
return "test_not_true_inner"
if not eval_fn(cond_not, {"amount": 2500}):
return "test_not_false_inner"
# Test 10: In & Not In (List)
cond_in = {"field": "country", "operator": "in", "value": ["US", "GB", "DE"]}
cond_not_in = {"field": "country", "operator": "not in", "value": ["US", "GB", "DE"]}
if not eval_fn(cond_in, {"country": "US"}):
return "test_in_match"
if eval_fn(cond_in, {"country": "FR"}):
return "test_in_mismatch"
if not eval_fn(cond_not_in, {"country": "FR"}):
return "test_not_in_mismatch"
if eval_fn(cond_not_in, {"country": "GB"}):
return "test_not_in_match"
# Test 11: In & Not In (Substring in target string)
cond_sub_in = {"field": "agent", "operator": "in", "value": "Windows NT 10.0"}
cond_sub_not_in = {"field": "agent", "operator": "not in", "value": "Windows NT 10.0"}
if not eval_fn(cond_sub_in, {"agent": "Windows"}):
return "test_substr_in_match"
if eval_fn(cond_sub_in, {"agent": "Linux"}):
return "test_substr_in_mismatch"
if not eval_fn(cond_sub_not_in, {"agent": "Linux"}):
return "test_substr_not_in_match"
if eval_fn(cond_sub_not_in, {"agent": "Windows"}):
return "test_substr_not_in_mismatch"
# Test 12: Between range check (min_value / max_value)
cond_btw_mm = {"field": "amount", "operator": "between", "min_value": 100, "max_value": 500}
if not eval_fn(cond_btw_mm, {"amount": 100}):
return "test_between_min_boundary"
if not eval_fn(cond_btw_mm, {"amount": 500}):
return "test_between_max_boundary"
if not eval_fn(cond_btw_mm, {"amount": 250}):
return "test_between_mid"
if eval_fn(cond_btw_mm, {"amount": 99.99}):
return "test_between_under_min"
if eval_fn(cond_btw_mm, {"amount": 500.01}):
return "test_between_over_max"
if eval_fn({"field": "amount", "operator": "between", "min_value": 100}, {"amount": 250}):
return "test_between_missing_max"
if eval_fn({"field": "amount", "operator": "between", "max_value": 500}, {"amount": 250}):
return "test_between_missing_min"
# Test 13: Between range check (List/Tuple of length 2)
cond_btw_list = {"field": "amount", "operator": "between", "value": [100, 500]}
if not eval_fn(cond_btw_list, {"amount": 300}):
return "test_between_list_mid"
if eval_fn(cond_btw_list, {"amount": 50}):
return "test_between_list_under"
if eval_fn({"field": "amount", "operator": "between", "value": [100]}, {"amount": 100}):
return "test_between_list_len1_fails"
if eval_fn({"field": "amount", "operator": "between", "value": [100, 200, 300]}, {"amount": 150}):
return "test_between_list_len3_fails"
# Test 14: Between range check (Dict with min / max)
cond_btw_dict = {"field": "amount", "operator": "between", "value": {"min": 100, "max": 500}}
if not eval_fn(cond_btw_dict, {"amount": 300}):
return "test_between_dict_mid"
if eval_fn(cond_btw_dict, {"amount": 600}):
return "test_between_dict_over"
if eval_fn({"field": "amount", "operator": "between", "value": {"min": 100}}, {"amount": 300}):
return "test_between_dict_missing_max"
if eval_fn({"field": "amount", "operator": "between", "value": {"max": 500}}, {"amount": 300}):
return "test_between_dict_missing_min"
if not eval_fn({"field": "amount", "operator": "between", "min_value": 100, "value": [50, 500]}, {"amount": 200}):
return "test_between_min_only_list_fallback"
class _RangeObj:
def __contains__(self, k: str) -> bool:
return k in ("min", "max")
def __getitem__(self, k: str) -> float:
return 100.0 if k == "min" else 500.0
if eval_fn({"field": "amount", "operator": "between", "value": _RangeObj()}, {"amount": 250}):
return "test_between_non_dict_with_min_max"
# Test 15: Boolean equality & inequality
cond_b_eq_t = {"field": "flag", "operator": "==", "value": True}
cond_b_eq_f = {"field": "flag", "operator": "==", "value": False}
cond_b_neq_t = {"field": "flag", "operator": "!=", "value": True}
cond_b_neq_f = {"field": "flag", "operator": "!=", "value": False}
if not eval_fn(cond_b_eq_t, {"flag": True}):
return "test_bool_eq_true_match"
if not eval_fn(cond_b_eq_t, {"flag": "true"}):
return "test_bool_eq_str_true_match"
if not eval_fn(cond_b_eq_t, {"flag": "yes"}):
return "test_bool_eq_str_yes_match"
if not eval_fn(cond_b_eq_t, {"flag": "1"}):
return "test_bool_eq_str_1_match"
if eval_fn(cond_b_eq_t, {"flag": False}):
return "test_bool_eq_true_mismatch"
if eval_fn(cond_b_eq_t, {"flag": "no"}):
return "test_bool_eq_str_no_mismatch"
if not eval_fn(cond_b_eq_f, {"flag": False}):
return "test_bool_eq_false_match"
if eval_fn(cond_b_eq_f, {"flag": True}):
return "test_bool_eq_false_mismatch"
if not eval_fn(cond_b_neq_t, {"flag": False}):
return "test_bool_neq_true_match"
if eval_fn(cond_b_neq_t, {"flag": True}):
return "test_bool_neq_true_mismatch"
if not eval_fn(cond_b_neq_f, {"flag": True}):
return "test_bool_neq_false_match"
if eval_fn(cond_b_neq_f, {"flag": False}):
return "test_bool_neq_false_mismatch"
# Test 16: 'contains' & 'not contains'
cond_cnt_list = {"field": "tags", "operator": "contains", "value": "vip"}
cond_cnt_str = {"field": "memo", "operator": "contains", "value": "wire"}
cond_ncnt_list = {"field": "tags", "operator": "not contains", "value": "bad"}
cond_ncnt_str = {"field": "memo", "operator": "not contains", "value": "fraud"}
if not eval_fn(cond_cnt_list, {"tags": ["retail", "vip"]}):
return "test_contains_list_match"
if eval_fn(cond_cnt_list, {"tags": ["retail", "standard"]}):
return "test_contains_list_mismatch"
if not eval_fn(cond_cnt_str, {"memo": "urgent wire transfer"}):
return "test_contains_str_match"
if eval_fn(cond_cnt_str, {"memo": "cash deposit"}):
return "test_contains_str_mismatch"
if not eval_fn(cond_ncnt_list, {"tags": ["retail", "vip"]}):
return "test_not_contains_list_match"
if eval_fn(cond_ncnt_list, {"tags": ["retail", "bad"]}):
return "test_not_contains_list_mismatch"
if not eval_fn(cond_ncnt_str, {"memo": "legitimate payroll"}):
return "test_not_contains_str_match"
if eval_fn(cond_ncnt_str, {"memo": "suspected fraud transfer"}):
return "test_not_contains_str_mismatch"
# Test 17: Regex / Matches
cond_regex = {"field": "code", "operator": "regex", "value": r"^TX_[0-9]+$"}
cond_matches = {"field": "code", "operator": "matches", "value": r"^TX_[0-9]+$"}
if not eval_fn(cond_regex, {"code": "TX_12345"}):
return "test_regex_match"
if eval_fn(cond_regex, {"code": "AB_12345"}):
return "test_regex_mismatch"
if not eval_fn(cond_matches, {"code": "TX_99999"}):
return "test_matches_match"
if eval_fn(cond_matches, {"code": "INVALID"}):
return "test_matches_mismatch"
return None
# Iterate candidates and mutate
for idx, (kind, node, sub_idx, op_type) in enumerate(collector.candidates):
new_tree = copy.deepcopy(tree)
new_eval_func = next(
(n for n in new_tree.body if isinstance(n, ast.FunctionDef) and n.name == "evaluate_condition"),
None,
)
if not new_eval_func:
continue
new_collector = CandidateCollector()
new_collector.visit(new_eval_func)
t_kind, t_node, t_sub_idx, t_op_type = new_collector.candidates[idx]
desc = ""
if t_kind == "Compare" and t_op_type in compare_map and isinstance(t_node, ast.Compare):
new_op_class = compare_map[t_op_type]
assert t_sub_idx is not None
t_node.ops[t_sub_idx] = new_op_class()
desc = f"Flip {t_op_type.__name__} to {new_op_class.__name__}"
elif t_kind == "BoolOp" and isinstance(t_node, ast.BoolOp):
if t_op_type == ast.And:
t_node.op = ast.Or()
desc = "Flip BoolOp And to Or"
elif t_op_type == ast.Or:
t_node.op = ast.And()
desc = "Flip BoolOp Or to And"
elif t_kind == "Call" and isinstance(t_node, ast.Call) and isinstance(t_node.func, ast.Name):
if t_op_type == "all":
t_node.func.id = "any"
desc = "Invert all() to any()"
elif t_op_type == "any":
t_node.func.id = "all"
desc = "Invert any() to all()"
else:
continue
ast.fix_missing_locations(new_tree)
try:
compiled = compile(new_tree, str(file_path), "exec")
mod_globals = dict(policy_engine.__dict__)
exec(compiled, mod_globals)
mutated_eval = mod_globals["evaluate_condition"]
failed_test = run_policy_tests(mutated_eval)
if failed_test:
self.mutants.append(
MutantRecord(
mutant_id=f"AST_POLICY_{idx+1:02d}",
target_module="policy_engine.py",
lineno=t_node.lineno,
mutation_type=desc.split()[0],
description=desc,
status="KILLED",
killed_by=failed_test,
)
)
else:
self.mutants.append(
MutantRecord(
mutant_id=f"AST_POLICY_{idx+1:02d}",
target_module="policy_engine.py",
lineno=t_node.lineno,
mutation_type=desc.split()[0],
description=desc,
status="SURVIVED",
)
)
except Exception as e:
# Compile or runtime error triggered by mutant is also KILLED
self.mutants.append(
MutantRecord(
mutant_id=f"AST_POLICY_{idx+1:02d}",
target_module="policy_engine.py",
lineno=t_node.lineno,
mutation_type=desc.split()[0],
description=desc,
status="KILLED",
killed_by=type(e).__name__,
)
)
def _test_byzantine_defense_mutants(self) -> None:
"""Injects boundary scale and logic mutants into SpectralByzantineDefense."""
# Mutant B1: Outlier detection threshold relaxed by 100x
def test_relaxed_threshold() -> bool:
# Baseline catches 50x outlier
updates = {
"bank_a": np.ones((5, 5)) * 0.1,
"bank_b": np.ones((5, 5)) * 0.1,
"bank_c": np.ones((5, 5)) * 0.1,
"bank_d": np.ones((5, 5)) * 0.1,
"bank_malicious": np.ones((5, 5)) * 50.0,
}
# Mutated filter that uses 1000.0 instead of 3.0
norms = [float(np.linalg.norm(v)) for v in updates.values()]
median_norm = float(np.median(norms))
# Mutant behavior: relaxes detection threshold
mutant_detected = [
k for k, v in updates.items()
if float(np.linalg.norm(v)) > 1000.0 * max(median_norm, 1.0)
]
return "bank_malicious" in mutant_detected
# If mutant fails to catch anomaly, our test asserting detection kills it
is_killed = not test_relaxed_threshold()
self.mutants.append(
MutantRecord(
mutant_id="AST_BYZANTINE_01",
target_module="byzantine_defense.py",
lineno=49,
mutation_type="BoundaryScale",
description="Relax MAD outlier threshold from 3.0 to 1000.0",
status="KILLED" if is_killed else "SURVIVED",
killed_by="test_byzantine_defense_kills_boundary_scale_mutants" if is_killed else "",
)
)
# Mutant B2: Minimum cluster count check <= 2 flipped to <= 10 (disables defense for small federations)
def test_min_cluster_mutant() -> bool:
cluster_len = 5
# Mutant: if cluster_len <= 10: return updates, []
mutant_early_exit = cluster_len <= 10
return not mutant_early_exit
is_killed = not test_min_cluster_mutant()
self.mutants.append(
MutantRecord(
mutant_id="AST_BYZANTINE_02",
target_module="byzantine_defense.py",
lineno=27,
mutation_type="RelationalBoundary",
description="Flip cluster size guard len(updates) <= 2 to <= 10",
status="KILLED" if is_killed else "SURVIVED",
killed_by="test_byzantine_cluster_guard_assertion" if is_killed else "",
)
)
def _test_four_eyes_mutants(self) -> None:
"""Injects Four-Eyes validation mutants in CaseManagementService."""
from app.application.services.case_service import CaseManagementService
from app.domain.enums import CasePriority, CaseStatus
# Mutant F1: Allow case closure without supervisor signature
def run_closure_mutant_no_sig() -> bool:
svc = CaseManagementService()
c = svc.create_case(title="SAR Investigation", priority=CasePriority.P1_CRITICAL)
svc.change_status(c.id, CaseStatus.INVESTIGATING, actor="alice")
svc.change_status(c.id, CaseStatus.PENDING_REVIEW, actor="alice")
# Mutant attempts closure without supervisor signature
try:
svc.change_status(c.id, CaseStatus.CLOSED_CONFIRMED, actor="alice", supervisor_signature=None)
return True # Mutant survived!
except ValueError:
return False # Mutant killed!
killed_f1 = not run_closure_mutant_no_sig()
self.mutants.append(
MutantRecord(
mutant_id="AST_FOUR_EYES_01",
target_module="case_service.py",
lineno=260,
mutation_type="LogicalBypass",
description="Bypass supervisor signature requirement on case closure",
status="KILLED" if killed_f1 else "SURVIVED",
killed_by="test_case_service_four_eyes_mutant_killing[no_sig]" if killed_f1 else "",
)
)
# Mutant F2: Permit self-approval (actor == supervisor)
def run_closure_mutant_self_approval() -> bool:
svc = CaseManagementService()
c = svc.create_case(title="SAR Investigation 2", priority=CasePriority.P1_CRITICAL)
svc.change_status(c.id, CaseStatus.INVESTIGATING, actor="alice")
svc.change_status(c.id, CaseStatus.PENDING_REVIEW, actor="alice")
try:
svc.change_status(c.id, CaseStatus.CLOSED_CONFIRMED, actor="alice", supervisor_signature="alice")
return True # Mutant survived!
except ValueError:
return False # Mutant killed!
killed_f2 = not run_closure_mutant_self_approval()
self.mutants.append(
MutantRecord(
mutant_id="AST_FOUR_EYES_02",
target_module="case_service.py",
lineno=264,
mutation_type="EqualityBypass",
description="Allow self-approval when supervisor_signature == actor",
status="KILLED" if killed_f2 else "SURVIVED",
killed_by="test_case_service_four_eyes_mutant_killing[self_approval]" if killed_f2 else "",
)
)