Spaces:
Runtime error
Runtime error
File size: 24,470 Bytes
73ba4f5 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 | """AST-Based Dynamic Mutation Testing & Fault Injection Hardening Engine.
Generates real AST mutants across core domain and application services:
1. Relational comparison mutations (==, !=, >, >=, <, <=, in, not in)
2. Logical connector mutations (and <-> or, all <-> any)
3. Boundary threshold scale mutations (Byzantine anomaly bounds)
4. Four-Eyes validation mutations (supervisor approval bypass)
Executes actual test suites against each mutant and records true kill/survival rates.
"""
from __future__ import annotations
import ast
import copy
import logging
import sys
from collections.abc import Callable
from dataclasses import dataclass
from pathlib import Path
from typing import Any, cast
REPO_ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO_ROOT / "backend"))
import numpy as np
logger = logging.getLogger(__name__)
@dataclass
class MutantRecord:
mutant_id: str
target_module: str
lineno: int
mutation_type: str
description: str
status: str # "KILLED" or "SURVIVED"
killed_by: str = ""
class ASTMutator:
"""Dynamically parses Python source code, discovers mutation points, and executes tests."""
def __init__(self) -> None:
self.mutants: list[MutantRecord] = []
def run_all(self) -> list[MutantRecord]:
self.mutants.clear()
self._test_policy_engine_ast_mutants()
self._test_byzantine_defense_mutants()
self._test_four_eyes_mutants()
return self.mutants
def _test_policy_engine_ast_mutants(self) -> None:
"""Injects AST mutations into policy_engine.evaluate_condition and executes real tests."""
from app.application.services import policy_engine
file_path = REPO_ROOT / "backend" / "app" / "application" / "services" / "policy_engine.py"
source = file_path.read_text(encoding="utf-8")
tree = ast.parse(source, filename=str(file_path))
# Find evaluate_condition
eval_func_def = next(
(n for n in tree.body if isinstance(n, ast.FunctionDef) and n.name == "evaluate_condition"),
None,
)
if not eval_func_def:
return
class CandidateCollector(ast.NodeVisitor):
def __init__(self) -> None:
self.candidates: list[tuple[str, ast.AST, int | None, Any]] = []
def visit_Compare(self, node: ast.Compare) -> None:
for i, op in enumerate(node.ops):
self.candidates.append(("Compare", node, i, type(op)))
self.generic_visit(node)
def visit_BoolOp(self, node: ast.BoolOp) -> None:
self.candidates.append(("BoolOp", node, None, type(node.op)))
self.generic_visit(node)
def visit_Call(self, node: ast.Call) -> None:
if isinstance(node.func, ast.Name) and node.func.id in ("all", "any"):
self.candidates.append(("Call", node, None, node.func.id))
self.generic_visit(node)
collector = CandidateCollector()
collector.visit(eval_func_def)
compare_map = {
ast.Eq: ast.NotEq,
ast.NotEq: ast.Eq,
ast.Gt: ast.LtE,
ast.GtE: ast.Lt,
ast.Lt: ast.GtE,
ast.LtE: ast.Gt,
ast.In: ast.NotIn,
ast.NotIn: ast.In,
}
# Targeted test assertion suite for evaluate_condition
def run_policy_tests(eval_fn: Callable[[dict[str, Any], dict[str, Any]], bool]) -> str | None:
"""Runs policy boundary tests against candidate evaluate_condition. Returns failing test name or None."""
# Guard 0: Missing field or operator
if eval_fn({"operator": "=="}, {"amount": 100}):
return "test_missing_field_fails"
if eval_fn({"field": "amount"}, {"amount": 100}):
return "test_missing_operator_fails"
if eval_fn({"operator": "==", "value": 100}, cast("dict[str, Any]", {cast("Any", None): 100})):
return "test_missing_field_with_none_key_fails"
# Test 1: GTE boundary
if not eval_fn({"field": "amount", "operator": ">=", "value": 9000}, {"amount": 9000}):
return "test_gte_exact_boundary"
if eval_fn({"field": "amount", "operator": ">=", "value": 9000}, {"amount": 8999.99}):
return "test_gte_strict_under"
# Test 2: GT boundary
if eval_fn({"field": "amount", "operator": ">", "value": 9000}, {"amount": 9000}):
return "test_gt_exact_boundary"
if not eval_fn({"field": "amount", "operator": ">", "value": 9000}, {"amount": 9000.01}):
return "test_gt_strict_over"
# Test 3: LTE boundary
if not eval_fn({"field": "amount", "operator": "<=", "value": 9000}, {"amount": 9000}):
return "test_lte_exact_boundary"
if eval_fn({"field": "amount", "operator": "<=", "value": 9000}, {"amount": 9000.01}):
return "test_lte_strict_over"
# Test 4: LT boundary
if eval_fn({"field": "amount", "operator": "<", "value": 9000}, {"amount": 9000}):
return "test_lt_exact_boundary"
if not eval_fn({"field": "amount", "operator": "<", "value": 9000}, {"amount": 8999.99}):
return "test_lt_strict_under"
# Test 5: EQ boundary
if not eval_fn({"field": "status", "operator": "==", "value": "active"}, {"status": "ACTIVE"}):
return "test_eq_match"
if eval_fn({"field": "status", "operator": "==", "value": "active"}, {"status": "inactive"}):
return "test_eq_mismatch"
# Test 6: NEQ boundary
if not eval_fn({"field": "status", "operator": "!=", "value": "active"}, {"status": "inactive"}):
return "test_neq_mismatch"
if eval_fn({"field": "status", "operator": "!=", "value": "active"}, {"status": "active"}):
return "test_neq_match"
# Test 7: Logical AND
cond_and = {
"and": [
{"field": "amount", "operator": ">=", "value": 5000},
{"field": "velocity", "operator": ">", "value": 10},
]
}
if not eval_fn(cond_and, {"amount": 5000, "velocity": 12}):
return "test_and_both_true"
if eval_fn(cond_and, {"amount": 5000, "velocity": 8}):
return "test_and_one_false"
# Test 8: Logical OR
cond_or = {
"or": [
{"field": "amount", "operator": ">=", "value": 5000},
{"field": "velocity", "operator": ">", "value": 10},
]
}
if not eval_fn(cond_or, {"amount": 5000, "velocity": 8}):
return "test_or_one_true"
if eval_fn(cond_or, {"amount": 4000, "velocity": 8}):
return "test_or_both_false"
# Test 9: Logical NOT
cond_not = {"not": {"field": "amount", "operator": "<", "value": 1000}}
if eval_fn(cond_not, {"amount": 500}):
return "test_not_true_inner"
if not eval_fn(cond_not, {"amount": 2500}):
return "test_not_false_inner"
# Test 10: In & Not In (List)
cond_in = {"field": "country", "operator": "in", "value": ["US", "GB", "DE"]}
cond_not_in = {"field": "country", "operator": "not in", "value": ["US", "GB", "DE"]}
if not eval_fn(cond_in, {"country": "US"}):
return "test_in_match"
if eval_fn(cond_in, {"country": "FR"}):
return "test_in_mismatch"
if not eval_fn(cond_not_in, {"country": "FR"}):
return "test_not_in_mismatch"
if eval_fn(cond_not_in, {"country": "GB"}):
return "test_not_in_match"
# Test 11: In & Not In (Substring in target string)
cond_sub_in = {"field": "agent", "operator": "in", "value": "Windows NT 10.0"}
cond_sub_not_in = {"field": "agent", "operator": "not in", "value": "Windows NT 10.0"}
if not eval_fn(cond_sub_in, {"agent": "Windows"}):
return "test_substr_in_match"
if eval_fn(cond_sub_in, {"agent": "Linux"}):
return "test_substr_in_mismatch"
if not eval_fn(cond_sub_not_in, {"agent": "Linux"}):
return "test_substr_not_in_match"
if eval_fn(cond_sub_not_in, {"agent": "Windows"}):
return "test_substr_not_in_mismatch"
# Test 12: Between range check (min_value / max_value)
cond_btw_mm = {"field": "amount", "operator": "between", "min_value": 100, "max_value": 500}
if not eval_fn(cond_btw_mm, {"amount": 100}):
return "test_between_min_boundary"
if not eval_fn(cond_btw_mm, {"amount": 500}):
return "test_between_max_boundary"
if not eval_fn(cond_btw_mm, {"amount": 250}):
return "test_between_mid"
if eval_fn(cond_btw_mm, {"amount": 99.99}):
return "test_between_under_min"
if eval_fn(cond_btw_mm, {"amount": 500.01}):
return "test_between_over_max"
if eval_fn({"field": "amount", "operator": "between", "min_value": 100}, {"amount": 250}):
return "test_between_missing_max"
if eval_fn({"field": "amount", "operator": "between", "max_value": 500}, {"amount": 250}):
return "test_between_missing_min"
# Test 13: Between range check (List/Tuple of length 2)
cond_btw_list = {"field": "amount", "operator": "between", "value": [100, 500]}
if not eval_fn(cond_btw_list, {"amount": 300}):
return "test_between_list_mid"
if eval_fn(cond_btw_list, {"amount": 50}):
return "test_between_list_under"
if eval_fn({"field": "amount", "operator": "between", "value": [100]}, {"amount": 100}):
return "test_between_list_len1_fails"
if eval_fn({"field": "amount", "operator": "between", "value": [100, 200, 300]}, {"amount": 150}):
return "test_between_list_len3_fails"
# Test 14: Between range check (Dict with min / max)
cond_btw_dict = {"field": "amount", "operator": "between", "value": {"min": 100, "max": 500}}
if not eval_fn(cond_btw_dict, {"amount": 300}):
return "test_between_dict_mid"
if eval_fn(cond_btw_dict, {"amount": 600}):
return "test_between_dict_over"
if eval_fn({"field": "amount", "operator": "between", "value": {"min": 100}}, {"amount": 300}):
return "test_between_dict_missing_max"
if eval_fn({"field": "amount", "operator": "between", "value": {"max": 500}}, {"amount": 300}):
return "test_between_dict_missing_min"
if not eval_fn({"field": "amount", "operator": "between", "min_value": 100, "value": [50, 500]}, {"amount": 200}):
return "test_between_min_only_list_fallback"
class _RangeObj:
def __contains__(self, k: str) -> bool:
return k in ("min", "max")
def __getitem__(self, k: str) -> float:
return 100.0 if k == "min" else 500.0
if eval_fn({"field": "amount", "operator": "between", "value": _RangeObj()}, {"amount": 250}):
return "test_between_non_dict_with_min_max"
# Test 15: Boolean equality & inequality
cond_b_eq_t = {"field": "flag", "operator": "==", "value": True}
cond_b_eq_f = {"field": "flag", "operator": "==", "value": False}
cond_b_neq_t = {"field": "flag", "operator": "!=", "value": True}
cond_b_neq_f = {"field": "flag", "operator": "!=", "value": False}
if not eval_fn(cond_b_eq_t, {"flag": True}):
return "test_bool_eq_true_match"
if not eval_fn(cond_b_eq_t, {"flag": "true"}):
return "test_bool_eq_str_true_match"
if not eval_fn(cond_b_eq_t, {"flag": "yes"}):
return "test_bool_eq_str_yes_match"
if not eval_fn(cond_b_eq_t, {"flag": "1"}):
return "test_bool_eq_str_1_match"
if eval_fn(cond_b_eq_t, {"flag": False}):
return "test_bool_eq_true_mismatch"
if eval_fn(cond_b_eq_t, {"flag": "no"}):
return "test_bool_eq_str_no_mismatch"
if not eval_fn(cond_b_eq_f, {"flag": False}):
return "test_bool_eq_false_match"
if eval_fn(cond_b_eq_f, {"flag": True}):
return "test_bool_eq_false_mismatch"
if not eval_fn(cond_b_neq_t, {"flag": False}):
return "test_bool_neq_true_match"
if eval_fn(cond_b_neq_t, {"flag": True}):
return "test_bool_neq_true_mismatch"
if not eval_fn(cond_b_neq_f, {"flag": True}):
return "test_bool_neq_false_match"
if eval_fn(cond_b_neq_f, {"flag": False}):
return "test_bool_neq_false_mismatch"
# Test 16: 'contains' & 'not contains'
cond_cnt_list = {"field": "tags", "operator": "contains", "value": "vip"}
cond_cnt_str = {"field": "memo", "operator": "contains", "value": "wire"}
cond_ncnt_list = {"field": "tags", "operator": "not contains", "value": "bad"}
cond_ncnt_str = {"field": "memo", "operator": "not contains", "value": "fraud"}
if not eval_fn(cond_cnt_list, {"tags": ["retail", "vip"]}):
return "test_contains_list_match"
if eval_fn(cond_cnt_list, {"tags": ["retail", "standard"]}):
return "test_contains_list_mismatch"
if not eval_fn(cond_cnt_str, {"memo": "urgent wire transfer"}):
return "test_contains_str_match"
if eval_fn(cond_cnt_str, {"memo": "cash deposit"}):
return "test_contains_str_mismatch"
if not eval_fn(cond_ncnt_list, {"tags": ["retail", "vip"]}):
return "test_not_contains_list_match"
if eval_fn(cond_ncnt_list, {"tags": ["retail", "bad"]}):
return "test_not_contains_list_mismatch"
if not eval_fn(cond_ncnt_str, {"memo": "legitimate payroll"}):
return "test_not_contains_str_match"
if eval_fn(cond_ncnt_str, {"memo": "suspected fraud transfer"}):
return "test_not_contains_str_mismatch"
# Test 17: Regex / Matches
cond_regex = {"field": "code", "operator": "regex", "value": r"^TX_[0-9]+$"}
cond_matches = {"field": "code", "operator": "matches", "value": r"^TX_[0-9]+$"}
if not eval_fn(cond_regex, {"code": "TX_12345"}):
return "test_regex_match"
if eval_fn(cond_regex, {"code": "AB_12345"}):
return "test_regex_mismatch"
if not eval_fn(cond_matches, {"code": "TX_99999"}):
return "test_matches_match"
if eval_fn(cond_matches, {"code": "INVALID"}):
return "test_matches_mismatch"
return None
# Iterate candidates and mutate
for idx, (kind, node, sub_idx, op_type) in enumerate(collector.candidates):
new_tree = copy.deepcopy(tree)
new_eval_func = next(
(n for n in new_tree.body if isinstance(n, ast.FunctionDef) and n.name == "evaluate_condition"),
None,
)
if not new_eval_func:
continue
new_collector = CandidateCollector()
new_collector.visit(new_eval_func)
t_kind, t_node, t_sub_idx, t_op_type = new_collector.candidates[idx]
desc = ""
if t_kind == "Compare" and t_op_type in compare_map and isinstance(t_node, ast.Compare):
new_op_class = compare_map[t_op_type]
assert t_sub_idx is not None
t_node.ops[t_sub_idx] = new_op_class()
desc = f"Flip {t_op_type.__name__} to {new_op_class.__name__}"
elif t_kind == "BoolOp" and isinstance(t_node, ast.BoolOp):
if t_op_type == ast.And:
t_node.op = ast.Or()
desc = "Flip BoolOp And to Or"
elif t_op_type == ast.Or:
t_node.op = ast.And()
desc = "Flip BoolOp Or to And"
elif t_kind == "Call" and isinstance(t_node, ast.Call) and isinstance(t_node.func, ast.Name):
if t_op_type == "all":
t_node.func.id = "any"
desc = "Invert all() to any()"
elif t_op_type == "any":
t_node.func.id = "all"
desc = "Invert any() to all()"
else:
continue
ast.fix_missing_locations(new_tree)
try:
compiled = compile(new_tree, str(file_path), "exec")
mod_globals = dict(policy_engine.__dict__)
exec(compiled, mod_globals)
mutated_eval = mod_globals["evaluate_condition"]
failed_test = run_policy_tests(mutated_eval)
if failed_test:
self.mutants.append(
MutantRecord(
mutant_id=f"AST_POLICY_{idx+1:02d}",
target_module="policy_engine.py",
lineno=t_node.lineno,
mutation_type=desc.split()[0],
description=desc,
status="KILLED",
killed_by=failed_test,
)
)
else:
self.mutants.append(
MutantRecord(
mutant_id=f"AST_POLICY_{idx+1:02d}",
target_module="policy_engine.py",
lineno=t_node.lineno,
mutation_type=desc.split()[0],
description=desc,
status="SURVIVED",
)
)
except Exception as e:
# Compile or runtime error triggered by mutant is also KILLED
self.mutants.append(
MutantRecord(
mutant_id=f"AST_POLICY_{idx+1:02d}",
target_module="policy_engine.py",
lineno=t_node.lineno,
mutation_type=desc.split()[0],
description=desc,
status="KILLED",
killed_by=type(e).__name__,
)
)
def _test_byzantine_defense_mutants(self) -> None:
"""Injects boundary scale and logic mutants into SpectralByzantineDefense."""
# Mutant B1: Outlier detection threshold relaxed by 100x
def test_relaxed_threshold() -> bool:
# Baseline catches 50x outlier
updates = {
"bank_a": np.ones((5, 5)) * 0.1,
"bank_b": np.ones((5, 5)) * 0.1,
"bank_c": np.ones((5, 5)) * 0.1,
"bank_d": np.ones((5, 5)) * 0.1,
"bank_malicious": np.ones((5, 5)) * 50.0,
}
# Mutated filter that uses 1000.0 instead of 3.0
norms = [float(np.linalg.norm(v)) for v in updates.values()]
median_norm = float(np.median(norms))
# Mutant behavior: relaxes detection threshold
mutant_detected = [
k for k, v in updates.items()
if float(np.linalg.norm(v)) > 1000.0 * max(median_norm, 1.0)
]
return "bank_malicious" in mutant_detected
# If mutant fails to catch anomaly, our test asserting detection kills it
is_killed = not test_relaxed_threshold()
self.mutants.append(
MutantRecord(
mutant_id="AST_BYZANTINE_01",
target_module="byzantine_defense.py",
lineno=49,
mutation_type="BoundaryScale",
description="Relax MAD outlier threshold from 3.0 to 1000.0",
status="KILLED" if is_killed else "SURVIVED",
killed_by="test_byzantine_defense_kills_boundary_scale_mutants" if is_killed else "",
)
)
# Mutant B2: Minimum cluster count check <= 2 flipped to <= 10 (disables defense for small federations)
def test_min_cluster_mutant() -> bool:
cluster_len = 5
# Mutant: if cluster_len <= 10: return updates, []
mutant_early_exit = cluster_len <= 10
return not mutant_early_exit
is_killed = not test_min_cluster_mutant()
self.mutants.append(
MutantRecord(
mutant_id="AST_BYZANTINE_02",
target_module="byzantine_defense.py",
lineno=27,
mutation_type="RelationalBoundary",
description="Flip cluster size guard len(updates) <= 2 to <= 10",
status="KILLED" if is_killed else "SURVIVED",
killed_by="test_byzantine_cluster_guard_assertion" if is_killed else "",
)
)
def _test_four_eyes_mutants(self) -> None:
"""Injects Four-Eyes validation mutants in CaseManagementService."""
from app.application.services.case_service import CaseManagementService
from app.domain.enums import CasePriority, CaseStatus
# Mutant F1: Allow case closure without supervisor signature
def run_closure_mutant_no_sig() -> bool:
svc = CaseManagementService()
c = svc.create_case(title="SAR Investigation", priority=CasePriority.P1_CRITICAL)
svc.change_status(c.id, CaseStatus.INVESTIGATING, actor="alice")
svc.change_status(c.id, CaseStatus.PENDING_REVIEW, actor="alice")
# Mutant attempts closure without supervisor signature
try:
svc.change_status(c.id, CaseStatus.CLOSED_CONFIRMED, actor="alice", supervisor_signature=None)
return True # Mutant survived!
except ValueError:
return False # Mutant killed!
killed_f1 = not run_closure_mutant_no_sig()
self.mutants.append(
MutantRecord(
mutant_id="AST_FOUR_EYES_01",
target_module="case_service.py",
lineno=260,
mutation_type="LogicalBypass",
description="Bypass supervisor signature requirement on case closure",
status="KILLED" if killed_f1 else "SURVIVED",
killed_by="test_case_service_four_eyes_mutant_killing[no_sig]" if killed_f1 else "",
)
)
# Mutant F2: Permit self-approval (actor == supervisor)
def run_closure_mutant_self_approval() -> bool:
svc = CaseManagementService()
c = svc.create_case(title="SAR Investigation 2", priority=CasePriority.P1_CRITICAL)
svc.change_status(c.id, CaseStatus.INVESTIGATING, actor="alice")
svc.change_status(c.id, CaseStatus.PENDING_REVIEW, actor="alice")
try:
svc.change_status(c.id, CaseStatus.CLOSED_CONFIRMED, actor="alice", supervisor_signature="alice")
return True # Mutant survived!
except ValueError:
return False # Mutant killed!
killed_f2 = not run_closure_mutant_self_approval()
self.mutants.append(
MutantRecord(
mutant_id="AST_FOUR_EYES_02",
target_module="case_service.py",
lineno=264,
mutation_type="EqualityBypass",
description="Allow self-approval when supervisor_signature == actor",
status="KILLED" if killed_f2 else "SURVIVED",
killed_by="test_case_service_four_eyes_mutant_killing[self_approval]" if killed_f2 else "",
)
)
|