ZeroCyber-SLM / app.py
Rootsystem2101's picture
Fix Grafana FPs: skip test-data dirs, respect #nosec, config key names
2d07db5
Raw History Blame Contribute Delete
163 kB
# ZeroCyber-SLM β€” Enterprise Security Scanner (AI Engine: Mistral-7B)
# ─────────────────────────────────────────────────────────
# Patch Jinja2 LRU cache (starlette/gradio compat)
import jinja2.utils as _j2u
_orig_si = _j2u.LRUCache.__setitem__
_orig_gi = _j2u.LRUCache.__getitem__
def _ssi(self, k, v):
try: _orig_si(self, k, v)
except TypeError: pass
def _sgi(self, k):
try: return _orig_gi(self, k)
except TypeError: raise KeyError(k)
def _sg(self, k, d=None):
try: return self[k]
except (KeyError, TypeError): return d
_j2u.LRUCache.__setitem__ = _ssi
_j2u.LRUCache.__getitem__ = _sgi
_j2u.LRUCache.get = _sg
try:
import gradio.networking as _gn
_gn.url_ok = lambda u: True
except Exception:
pass
import gradio as gr
import re, hashlib, ast, json, os, base64
import urllib.request, urllib.parse, urllib.error
from datetime import datetime
# ── Environment-based credentials (set once in HF Space Secrets) ──────────
_ENV_HF_TOKEN = os.environ.get("HF_TOKEN", "")
_ENV_GITHUB_TOKEN = os.environ.get("GITHUB_TOKEN", "")
# ══════════════════════════════════════════════════════════
# OWASP Top 10 (2021) + CWE + CVSS Base Scores
# ══════════════════════════════════════════════════════════
OWASP = {
"SQL Injection": ("A03:2021","Injection", "CWE-89", 9.8),
"Blind SQL Injection": ("A03:2021","Injection", "CWE-89", 9.8),
"NoSQL Injection": ("A03:2021","Injection", "CWE-943", 9.8),
"XSS": ("A03:2021","Injection", "CWE-79", 7.2),
"Stored XSS": ("A03:2021","Injection", "CWE-79", 8.8),
"DOM XSS": ("A03:2021","Injection", "CWE-79", 6.1),
"Command Injection": ("A03:2021","Injection", "CWE-78", 9.8),
"Code Injection": ("A03:2021","Injection", "CWE-94", 9.8),
"LDAP Injection": ("A03:2021","Injection", "CWE-90", 8.8),
"Template Injection": ("A03:2021","Injection", "CWE-1336",9.8),
"Log4Shell": ("A06:2021","Vulnerable Components", "CWE-917", 10.0),
"Path Traversal": ("A01:2021","Broken Access Control", "CWE-22", 7.5),
"File Inclusion": ("A01:2021","Broken Access Control", "CWE-98", 9.8),
"IDOR": ("A01:2021","Broken Access Control", "CWE-639", 8.1),
"Missing Auth": ("A01:2021","Broken Access Control", "CWE-306", 9.1),
"Broken Auth": ("A07:2021","Auth Failures", "CWE-287", 9.8),
"Open Redirect": ("A01:2021","Broken Access Control", "CWE-601", 6.1),
"SSRF": ("A10:2021","SSRF", "CWE-918", 9.8),
"XXE": ("A05:2021","Security Misconfiguration","CWE-611", 9.0),
"CSRF": ("A01:2021","Broken Access Control", "CWE-352", 8.8),
"Insecure File Upload": ("A04:2021","Insecure Design", "CWE-434", 9.8),
"Hardcoded Secret": ("A07:2021","Auth Failures", "CWE-798", 9.1),
"Weak Cryptography": ("A02:2021","Cryptographic Failures", "CWE-327", 7.5),
"Insecure Deserialization": ("A08:2021","Integrity Failures", "CWE-502", 9.8),
"Prototype Pollution": ("A03:2021","Injection", "CWE-1321",8.1),
"ReDoS": ("A05:2021","Security Misconfiguration","CWE-1333",7.5),
"Debug Enabled": ("A05:2021","Security Misconfiguration","CWE-215", 5.3),
"Sensitive Data Exposure": ("A02:2021","Cryptographic Failures", "CWE-200", 7.5),
"Insecure Session": ("A07:2021","Auth Failures", "CWE-384", 8.8),
"Security Misconfiguration":("A05:2021","Security Misconfiguration","CWE-16", 7.5),
"Dependency Vulnerability": ("A06:2021","Vulnerable Components", "CWE-1035",8.8),
"Mass Assignment": ("A04:2021","Insecure Design", "CWE-915", 8.1),
}
CVSS_BASE = {"CRITICAL":9.0, "HIGH":7.0, "MEDIUM":5.0, "LOW":3.0, "INFO":1.0}
SEV_ORDER = ["CRITICAL","HIGH","MEDIUM","LOW","INFO"]
# ══════════════════════════════════════════════════════════
# VULNERABILITY RULES
# Format: (vuln_type, severity, regex, description, fix, confidence)
# confidence: HIGH = very likely real vuln / LOW = needs manual review
# ══════════════════════════════════════════════════════════
RULES = [
# ═══ SQL INJECTION ════════════════════════════════════
("SQL Injection","CRITICAL",
r'(?i)(query|sql|stmt)\s*[+]?=\s*["\'].*?(SELECT|INSERT|UPDATE|DELETE|WHERE|DROP|UNION).*?["\'].*\$',
"SQL query with PHP variable concatenation",
"Use PDO::prepare() with bound parameters. Never build SQL with string concatenation.",
"HIGH"),
("SQL Injection","CRITICAL",
r'(?i)["\'].*(SELECT|INSERT|UPDATE|DELETE|WHERE|FROM)\s+.*["\']\s*\.\s*\$(?!_SESSION)(?!\{)',
"PHP SQL query with direct variable concat (not session)",
"Replace with PDO prepared statements: $stmt = $pdo->prepare('SELECT * FROM users WHERE id = ?'); $stmt->execute([$id]);",
"HIGH"),
("SQL Injection","CRITICAL",
r'(?i)\$\s*(wpdb|db|conn|connection|mysqli|pdo)\s*->\s*(query|get_results|get_row|get_var)\s*\(.*\$(?!_SESSION)',
"ORM/DB query with direct variable (WordPress/PHP)",
"Use $wpdb->prepare() or parameterized queries.",
"HIGH"),
("SQL Injection","CRITICAL",
r'(?i)execute\s*\(\s*["\'].*(%s|%d|\?|\$[0-9]).*["\'],?\s*\)',
"SQL execute with %-format placeholder β€” verify binding",
"Ensure params are passed as separate argument, not formatted into the string.",
"MEDIUM"),
("SQL Injection","CRITICAL",
r'(?i)f["\'][\s]*(SELECT|INSERT|UPDATE|DELETE)\s+\S.*?\{',
"f-string used to build SQL query",
"Never use f-strings or .format() for SQL. Use cursor.execute(sql, params).",
"HIGH"),
("Blind SQL Injection","HIGH",
r'(?i)(?:cursor|db|conn|execute|query)\b[^;{}\n]{0,80}(sleep|benchmark|pg_sleep|waitfor\s+delay)\s*\(',
"Time-based blind SQL injection β€” sleep/delay inside a query execution context",
"Remove time-delay functions from queries; use parameterized queries.",
"MEDIUM"),
("NoSQL Injection","HIGH",
r'(?i)(find|findOne|aggregate)\s*\(\s*\{[^}]*\$_(GET|POST|REQUEST)',
"MongoDB query with unsanitized user input",
"Sanitize inputs; use allowlists; avoid passing raw request data to MongoDB queries.",
"HIGH"),
# ═══ XSS ══════════════════════════════════════════════
# PHP: echo/print with $_GET/$_POST anywhere on the statement
# Pattern uses [^;#\n]* to stay within the statement boundary
# No requirement for [ or ( after $_GET β€” handles $_GET[ 'x' ] with spaces too
# Lazy .*? ensures we find $_GET even with spaces like $_GET[ 'name' ]
# htmlspecialchars/htmlentities detection is handled by _context_confidence (MITIGATIONS dict)
# which checks the surrounding lines including the current line itself.
("XSS","CRITICAL",
r'(?i)(echo|print).*?\$_(GET|POST|REQUEST|COOKIE)',
"PHP output of user-controlled input β€” XSS risk",
"Use htmlspecialchars($val, ENT_QUOTES, 'UTF-8') before any echo of user data.",
"HIGH"),
("XSS","CRITICAL",
r'(?i)<\?=\s*\$_(GET|POST|REQUEST|COOKIE)',
"PHP short echo tag with unsanitized user input",
"Use <?= htmlspecialchars($_GET['x'], ENT_QUOTES) ?> instead.",
"HIGH"),
# JS: innerHTML/outerHTML/document.write with user-controlled data
("DOM XSS","HIGH",
r'(?i)(innerHTML|outerHTML)\s*[+]?=\s*[^;]*(location\.|document\.URL|document\.referrer|window\.name|location\.hash|location\.search|location\.href)',
"DOM XSS: innerHTML assigned from browser location object",
"Use textContent instead of innerHTML. Sanitize with DOMPurify if HTML is needed.",
"HIGH"),
("DOM XSS","HIGH",
r'(?i)document\.write\s*\(.*?(document\.URL|location\.|document\.referrer|window\.name)',
"DOM XSS: document.write with browser location data",
"Avoid document.write entirely. Use safe DOM manipulation APIs.",
"HIGH"),
("XSS","HIGH",
r'(?i)(innerHTML|outerHTML)\s*[+]?=\s*[^;]*(req\.|request\.(body|query|params)|params\[)',
"innerHTML assigned from Express/Node.js request data",
"Use textContent or DOMPurify.sanitize(). Validate inputs server-side.",
"HIGH"),
# Flask/Jinja2 template injection via user data
("XSS","HIGH",
r'(?i)render_template_string\s*\(.*?(request\.(args|form|json|values)|g\.\w+)',
"Flask render_template_string with user-controlled input",
"Use render_template() with a static template file, never render_template_string with user data.",
"HIGH"),
# Stored XSS indicator: saving user input then echoing without escape
("Stored XSS","HIGH",
r'(?i)(mysql_result|mysqli_fetch|PDO.*fetch).*echo(?!.*htmlspecialchars)',
"Fetching DB data and echoing without htmlspecialchars (potential Stored XSS)",
"Always apply htmlspecialchars() to any data fetched from the database before output.",
"MEDIUM"),
# ═══ COMMAND INJECTION ════════════════════════════════
("Command Injection","CRITICAL",
r'(?i)(shell_exec|system|passthru|exec|popen)\s*\([^)]*\$_(GET|POST|REQUEST|COOKIE|SERVER)',
"PHP shell function called directly with user input",
"Never pass user input to shell functions. Use whitelisting and escapeshellarg().",
"HIGH"),
("Command Injection","CRITICAL",
r'(?i)(shell_exec|system|passthru|exec|popen)\s*\([^)]*[\'"][^)]*\.\s*\$(?!_SESSION)',
"PHP shell function with string concatenation (variable appended)",
"Use escapeshellarg() on any variable used in shell commands. Prefer whitelisted commands.",
"HIGH"),
("Command Injection","CRITICAL",
r'(?i)os\.system\s*\([^)]*(%|\.format\s*\(|f["\']|\+\s*\w)',
"Python os.system() with dynamic string β€” command injection risk",
"Use subprocess.run([cmd, arg], shell=False). Never pass user input to os.system().",
"HIGH"),
("Command Injection","HIGH",
r'(?i)subprocess\.(run|call|Popen)\s*\([^,)]+,\s*shell\s*=\s*True',
"subprocess with shell=True β€” allows shell metacharacter injection",
"Use shell=False and pass arguments as a list: subprocess.run(['cmd', user_arg], shell=False).",
"HIGH"),
("Command Injection","HIGH",
r'(?i)subprocess\.(run|call|Popen)\s*\(\s*["\'].*\+',
"subprocess called with string concatenation",
"Always pass args as a list, never as a concatenated string.",
"HIGH"),
# ═══ PATH TRAVERSAL / FILE INCLUSION ══════════════════
("File Inclusion","CRITICAL",
r'(?i)(include|require)(_once)?\s*\(\s*\$_(GET|POST|REQUEST)\s*[\[\(]',
"PHP file inclusion with direct user input β€” Remote/Local File Inclusion (RFI/LFI)",
"Never include files based on user input. Use a strict whitelist of allowed files.",
"HIGH"),
("File Inclusion","CRITICAL",
r'(?i)(include|require)(_once)?\s*\(\s*\$\w+\s*\)',
"PHP file inclusion with variable β€” possible LFI",
"Validate $file against a whitelist of allowed filenames before including.",
"MEDIUM"),
("Path Traversal","HIGH",
r'(?i)(fopen|file_get_contents|readfile|file)\s*\(\s*\$_(GET|POST|REQUEST)',
"PHP file read with direct user-controlled path",
"Validate path with realpath(); ensure it starts with the expected base directory.",
"HIGH"),
("Path Traversal","HIGH",
r'(?i)\bopen\s*\([^,)]*\$_(GET|POST|REQUEST)|open\s*\([^,)]*request\.(args|form|values|json)\[',
"File open with user-controlled path (Python/PHP)",
"Use os.path.realpath() and verify the path starts with the expected base directory.",
"HIGH"),
("Path Traversal","MEDIUM",
r'(?i)\.\./.*\$_(GET|POST|REQUEST)|request\.(args|form|json)\[.*\]\s*\+.*\.\.',
"Directory traversal sequence combined with user input",
"Sanitize user input; resolve canonical path with realpath() before file operations.",
"HIGH"),
# ═══ SSRF ═════════════════════════════════════════════
("SSRF","CRITICAL",
r'(?i)(requests\.get|requests\.post|urllib\.request\.urlopen|curl_exec|file_get_contents|fetch)\s*\([^)]*\$_(GET|POST|REQUEST)',
"HTTP request made to user-controlled URL β€” Server-Side Request Forgery",
"Validate URL against an allowlist. Block private IPs (10.x, 172.16-31.x, 192.168.x, 127.x, 169.254.x).",
"HIGH"),
("SSRF","CRITICAL",
r'(?i)(requests\.get|requests\.post|httpx|aiohttp)\s*\([^)]*(?:url|target|host|endpoint)\s*=\s*(request\.|req\.|params\[|args\[)',
"HTTP request with user-controlled URL parameter (Python)",
"Implement URL allowlisting. Use a safe HTTP library wrapper that enforces the allowlist.",
"HIGH"),
("SSRF","CRITICAL",
r'(?i)169\.254\.169\.254|metadata\.google\.internal|169\.254\.170\.2',
"Cloud metadata service endpoint hardcoded or accessed",
"Block access to metadata endpoints at the network level. Use IMDSv2 with token requirement.",
"HIGH"),
("SSRF","HIGH",
r'(?i)(file|gopher|dict|ftp)://[^"\'\s]*\$',
"Dangerous URL scheme with user-controlled value",
"Only allow http/https schemes. Validate and parse URL before making any request.",
"HIGH"),
# ═══ HARDCODED SECRETS ════════════════════════════════
("Hardcoded Secret","CRITICAL",
r'(?i)\b(password|passwd|pwd|pass)\s*=\s*["\'][^"\']{4,}["\'](?!\s*#.*placeholder)',
"Hardcoded password string",
"Load passwords from environment variables: os.environ['DB_PASSWORD'] or a secrets vault.",
"HIGH"),
("Hardcoded Secret","CRITICAL",
r'(?i)(secret_key|secret|api_key|apikey|api[-_]secret)\s*=\s*["\'][^"\']{8,}["\']',
"Hardcoded API key or secret",
"Use environment variables or a secrets manager (AWS Secrets Manager, HashiCorp Vault).",
"HIGH"),
("Hardcoded Secret","CRITICAL",
r'(?i)(aws_access_key_id|aws_secret_access_key)\s*=\s*["\'][A-Za-z0-9/+]{16,}["\']',
"Hardcoded AWS credentials",
"Use IAM roles (EC2/Lambda) or AWS Secrets Manager. Rotate the key immediately.",
"HIGH"),
("Hardcoded Secret","CRITICAL",
r'AKIA[0-9A-Z]{16}',
"AWS Access Key ID pattern detected in source code",
"Rotate this key immediately. Use IAM roles instead of hardcoded credentials.",
"HIGH"),
("Hardcoded Secret","HIGH",
r'(?i)-----BEGIN (RSA|EC|OPENSSH|DSA|PGP) PRIVATE KEY',
"Private cryptographic key embedded in source code",
"Remove key from code immediately. Store in secure key management. Add to .gitignore.",
"HIGH"),
("Hardcoded Secret","HIGH",
r'(?i)(token|auth_token|access_token|bearer)\s*=\s*["\'][A-Za-z0-9._\-]{20,}["\']',
"Hardcoded authentication token",
"Load tokens from environment variables. Rotate the exposed token immediately.",
"HIGH"),
("Hardcoded Secret","HIGH",
r'(?i)(database_url|db_url|connection_string|dsn)\s*=\s*["\'].*://.+:.+@',
"Database connection string with credentials embedded",
"Use environment variables for connection strings. Never commit credentials to Git.",
"HIGH"),
("Hardcoded Secret","MEDIUM",
r'(?i)(private_key|client_secret|consumer_secret|webhook_secret)\s*=\s*["\'][^"\']{12,}["\']',
"Potential hardcoded cryptographic or OAuth secret",
"Move to environment variables or secrets management system.",
"MEDIUM"),
# ═══ WEAK CRYPTOGRAPHY ════════════════════════════════
("Weak Cryptography","HIGH",
r'(?i)\b(md5|sha1)\s*\([^)]*\$(?!_SESSION)',
"MD5 or SHA-1 used with user data or passwords β€” cryptographically broken",
"Use password_hash() (PHP) or bcrypt/argon2 (Python/Node) for passwords. SHA-256+ for hashing.",
"HIGH"),
("Weak Cryptography","HIGH",
r'(?i)hashlib\.(md5|sha1)\s*\(',
"Python hashlib MD5/SHA1 β€” weak algorithm",
"Use hashlib.sha256() or hashlib.sha3_256(). For passwords use bcrypt or argon2-cffi.",
"HIGH"),
("Weak Cryptography","HIGH",
r'(?i)\b(DES|3DES|RC4|RC2|Blowfish)\s*[\.(]',
"Weak or broken cipher algorithm",
"Use AES-256-GCM or ChaCha20-Poly1305. Never use DES, RC4, or RC2 in new code.",
"HIGH"),
("Weak Cryptography","HIGH",
r'(?i)Cipher\.getInstance\s*\(\s*["\']AES(/ECB|/CBC)["\']',
"Java AES with ECB mode (no IV) or CBC without integrity check",
"Use AES/GCM/NoPadding which provides authenticated encryption.",
"HIGH"),
("Weak Cryptography","MEDIUM",
r'(?i)(?<!\bnum?p?y?\b)(?<!\bnp\b)(?<!\bmath\b)\brandom\.(random|randint|uniform|choice|shuffle)\s*\('
r'(?=[^)]*(?:token|secret|key|password|session|nonce|salt|csrf|auth|otp|pin))',
"Python random module used for security-sensitive value β€” not cryptographically secure",
"Use the secrets module for tokens, session IDs, or any security-sensitive randomness.",
"MEDIUM"),
("Weak Cryptography","MEDIUM",
r'(?i)mt_rand\s*\(',
"PHP mt_rand() β€” not cryptographically secure",
"Use random_bytes() or random_int() for security-sensitive values.",
"MEDIUM"),
("Insecure Session","HIGH",
r'(?i)session\.permanent\s*=\s*True|setcookie\s*\([^)]*httponly\s*=\s*false|setcookie\s*\([^)]*secure\s*=\s*false',
"Insecure session or cookie configuration",
"Set HttpOnly=True, Secure=True, SameSite=Strict on all session cookies.",
"HIGH"),
("Insecure Session","MEDIUM",
r'(?i)session_start\s*\(\s*\).*session_id\s*\(\s*\$_(GET|POST|REQUEST)',
"PHP session ID taken from user input β€” session fixation attack",
"Never set session ID from user input. Call session_regenerate_id(true) after login.",
"HIGH"),
# ═══ INSECURE DESERIALIZATION ════════════════════════
("Insecure Deserialization","CRITICAL",
r'(?i)pickle\.(loads?)\s*\(',
"Python pickle deserialization β€” arbitrary code execution if data is untrusted",
"Never deserialize untrusted data with pickle. Use JSON or a safe alternative.",
"HIGH"),
("Insecure Deserialization","CRITICAL",
r'(?i)yaml\.load\s*\([^,)]+\)',
"PyYAML yaml.load() without safe Loader β€” code execution via YAML tags",
"Replace with yaml.safe_load(). Never use yaml.load() on untrusted input.",
"HIGH"),
("Insecure Deserialization","CRITICAL",
r'(?i)unserialize\s*\(\s*\$_(GET|POST|REQUEST|COOKIE)',
"PHP unserialize() with user input β€” object injection / code execution",
"Never unserialize user-controlled data. Use JSON: json_decode() instead.",
"HIGH"),
("Insecure Deserialization","HIGH",
r'(?i)(ObjectInputStream|readObject|XMLDecoder)\s*\(',
"Java deserialization β€” potential remote code execution",
"Use safe deserialization libraries. Implement object deserialization filters (JEP 290).",
"MEDIUM"),
("Insecure Deserialization","MEDIUM",
r'(?i)marshal\.(loads?)\s*\(',
"Python marshal deserialization β€” unsafe with untrusted data",
"Use JSON for data exchange. Avoid marshal for any data from external sources.",
"MEDIUM"),
# ═══ CODE INJECTION / EVAL ════════════════════════════
("Code Injection","CRITICAL",
r'(?i)\beval\s*\([^)]*\$_(GET|POST|REQUEST|COOKIE)',
"PHP eval() with direct user input β€” remote code execution",
"Remove eval() entirely. No legitimate use case justifies eval() with user data.",
"HIGH"),
("Code Injection","CRITICAL",
r'(?i)\beval\s*\([^)]*(?:request\.(args|form|json|data)|req\.body|req\.query)',
"eval() with Express/Flask request data β€” remote code execution",
"Remove eval(). Use safe parsers or JSON.parse() for data processing.",
"HIGH"),
("Code Injection","HIGH",
r'(?i)\beval\s*\([^)]*\$(?!_SESSION)\w+',
"PHP eval() with variable β€” verify input is not user-controlled",
"Avoid eval() entirely. If unavoidable, ensure the variable cannot contain user data.",
"MEDIUM"),
# ═══ TEMPLATE INJECTION ═══════════════════════════════
("Template Injection","CRITICAL",
r'(?i)Environment\s*\([^)]*\)\s*\.?(from_string|get_template)\s*\([^)]*(?:request|input|param|\$_(GET|POST))',
"Jinja2/Twig template created from user input β€” SSTI",
"Use static template files. Never pass user input to template constructors.",
"HIGH"),
("Template Injection","HIGH",
r'(?i)(smarty|twig|blade).*\$_(GET|POST|REQUEST)',
"PHP template engine with direct user input",
"Escape all user data before passing to templates. Use auto-escaping.",
"MEDIUM"),
# ═══ XXE ══════════════════════════════════════════════
("XXE","HIGH",
r'(?i)(xml\.etree|lxml|minidom|SAXParser|DocumentBuilder|XMLReader).*parse',
"XML parsing detected β€” verify external entity protection",
"Disable external entities. Python: defusedxml library. Java: factory.setFeature(DISALLOW_DOCTYPE_DECL, true).",
"LOW"),
("XXE","CRITICAL",
r'(?i)<!ENTITY\s+\w+\s+SYSTEM\s+["\']',
"XML external entity (XXE) definition detected",
"Disable DTD processing entirely. Never allow external entities in XML parsers.",
"HIGH"),
("XXE","HIGH",
r'(?i)libxml_disable_entity_loader\s*\(\s*(false|0)\s*\)',
"PHP libxml external entity loading explicitly enabled",
"Set libxml_disable_entity_loader(true) before parsing any XML.",
"HIGH"),
# ═══ OPEN REDIRECT ════════════════════════════════════
("Open Redirect","HIGH",
r'(?i)header\s*\(\s*["\']location\s*:\s*["\']?\s*\.\s*\$_(GET|POST|REQUEST)',
"PHP header() redirect with user-controlled destination",
"Validate redirect against a whitelist of allowed URLs. Never redirect to arbitrary user input.",
"HIGH"),
("Open Redirect","HIGH",
r'(?i)(redirect|location\.href|location\.replace)\s*[=(]\s*[^;]*(req\.(query|body|params)|request\.(args|form))',
"Open redirect with user-controlled URL (JS/Python)",
"Implement a redirect allowlist. Use relative URLs only where possible.",
"HIGH"),
("Open Redirect","MEDIUM",
r'(?i)header\s*\(\s*["\']location\s*:.*\$\w+',
"PHP header() redirect with variable β€” verify variable is not user-controlled",
"Ensure redirect destination is validated against an allowlist.",
"MEDIUM"),
# ═══ INSECURE FILE UPLOAD ═════════════════════════════
("Insecure File Upload","CRITICAL",
r'(?i)move_uploaded_file\s*\([^,]+,\s*[^)]*\$_(GET|POST|REQUEST|FILES)',
"PHP file upload with user-controlled destination path",
"Use a fixed upload directory. Rename files randomly. Validate MIME type server-side.",
"HIGH"),
("Insecure File Upload","HIGH",
r'(?i)move_uploaded_file|$_FILES\[.*\]\[.*tmp_name',
"PHP file upload handling β€” verify extension and MIME type validation",
"Whitelist allowed extensions. Validate MIME with finfo_file(). Store outside web root.",
"MEDIUM"),
("Insecure File Upload","HIGH",
r'(?i)(multer|formidable|multiparty|busboy).*(?!\.limits)',
"Node.js file upload without visible size limits",
"Set file size limits and allowed MIME types. Rename uploaded files. Store outside web root.",
"LOW"),
# ═══ CSRF ═════════════════════════════════════════════
("CSRF","HIGH",
r'(?i)<form[^>]+method\s*=\s*["\']post["\'][^>]*>(?!.*csrf|.*token|.*nonce)',
"HTML POST form without visible CSRF token",
"Add CSRF token to all state-changing forms. Use SameSite=Strict cookies.",
"MEDIUM"),
("CSRF","MEDIUM",
r'(?i)csrf_exempt\s*\(',
"CSRF protection explicitly disabled (Django csrf_exempt)",
"Remove csrf_exempt unless absolutely necessary. Implement alternative CSRF protection.",
"HIGH"),
# ═══ PROTOTYPE POLLUTION (JS) ═════════════════════════
("Prototype Pollution","HIGH",
r'(?i)Object\.assign\s*\(\s*\{?\s*\}\s*,\s*(req\.|request\.|body|params|query)\w*\)',
"Object.assign with entire request object β€” prototype pollution risk",
"Use a safe deep merge library. Never merge user input into {} without sanitization.",
"HIGH"),
("Prototype Pollution","HIGH",
r'(?i)(\w+)\[(\w+)\]\[(\w+)\]\s*=\s*(req\.|request\.)',
"Bracket notation assignment with user-controlled key β€” prototype pollution",
"Validate keys against an allowlist. Block __proto__, constructor, prototype keys.",
"MEDIUM"),
# ═══ LOG4SHELL ════════════════════════════════════════
("Log4Shell","CRITICAL",
r'(?i)\$\{jndi\s*:\s*(ldap|rmi|dns|iiop|corba|nds|http)s?\s*://',
"Log4Shell JNDI injection string detected",
"Upgrade Log4j to 2.17.1+. Set log4j2.formatMsgNoLookups=true immediately.",
"HIGH"),
# ═══ LDAP INJECTION ═══════════════════════════════════
("LDAP Injection","HIGH",
r'(?i)(ldap_search|ldap_bind|ldap_add|ldap_modify)\s*\([^)]*\$_(GET|POST|REQUEST)',
"PHP LDAP function with direct user input",
"Escape LDAP special characters with ldap_escape(). Use parameterized LDAP queries.",
"HIGH"),
# ═══ MASS ASSIGNMENT ══════════════════════════════════
("Mass Assignment","HIGH",
r'(?i)(User|Model|Record)\s*\.\s*(create|update|new)\s*\(\s*params\b|\.update_attributes\s*\(\s*params\b',
"Rails mass assignment with unfiltered params β€” overwrites protected attributes",
"Use strong parameters: params.require(:model).permit(:field1, :field2).",
"HIGH"),
# ═══ DEBUG / INFO EXPOSURE ════════════════════════════
("Debug Enabled","HIGH",
r'(?i)\bDEBUG\s*=\s*True\b|app\.debug\s*=\s*True\b|WP_DEBUG.*true',
"Debug mode enabled β€” exposes stack traces and internal details in production",
"Set DEBUG=False. Load from environment: DEBUG = os.environ.get('DEBUG', 'False') == 'True'.",
"HIGH"),
("Sensitive Data Exposure","HIGH",
r'(?i)(traceback\.print_exc|traceback\.format_exc|print_r\(\s*\$_(SERVER|ENV|GET|POST))',
"Internal error details or server variables exposed to output",
"Log errors server-side only. Return generic error messages to users.",
"HIGH"),
("Debug Enabled","MEDIUM",
r'(?i)(console\.log|print|var_dump|debug)\s*\([^)]*\b(password|token|secret|api_key|apikey|auth_key|private_key|secretkey)\b',
"Sensitive value logged to console/output",
"Remove sensitive data from logs. Use structured logging with field filtering.",
"HIGH"),
# ═══ DEPENDENCY VULNERABILITIES ═══════════════════════
("Dependency Vulnerability","MEDIUM",
r'(?i)(require|import)\s*["\']log4j|require\s*["\']node-serialize|require\s*["\']st\b',
"Import of known vulnerable library detected",
"Update to a patched version. Run npm audit / pip-audit / Dependabot regularly.",
"HIGH"),
("Dependency Vulnerability","MEDIUM",
r'(?i)eval\s*\(\s*require\s*\(["\']node-serialize',
"node-serialize deserialization RCE (CVE-2017-5941)",
"Remove node-serialize. Use JSON.parse() instead.",
"HIGH"),
# ═══ REDOS ════════════════════════════════════════════
("ReDoS","MEDIUM",
r'(?i)re\.(match|search|fullmatch|findall)\s*\(\s*["\'].*(\+\*|\*\+|\(\w+\+\)+\*|\{[0-9]+,\s*\}.*(\*|\+))',
"Potentially catastrophic backtracking regex (ReDoS)",
"Simplify the regex. Use a timeout wrapper or switch to the re2 library.",
"MEDIUM"),
# ═══ LOGIC FLAWS & SECURITY DESIGN ═══════════════════
("Broken Auth","CRITICAL",
r'(?i)\.decode\s*\([^)]*algorithms\s*=\s*\[.*none.*\]',
"JWT decoded with algorithm 'none' β€” authentication bypass possible",
"Always specify allowed algorithms explicitly. Never accept 'none': algorithms=['HS256']",
"HIGH"),
("Broken Auth","HIGH",
r'(?i)\b(if|return|assert)\s+\w+\s*==\s*\w*(token|password|secret|hash|key)\w*\s',
"Non-constant-time comparison for secret value β€” timing attack risk",
"Use hmac.compare_digest() (Python) or hash_equals() (PHP) for all secret comparisons.",
"MEDIUM"),
("Security Misconfiguration","HIGH",
r'(?i)Access-Control-Allow-Origin["\']?\s*[,:]\s*["\']?\*["\']?',
"CORS wildcard (*) β€” any origin can read API responses",
"Restrict to specific origins: 'Access-Control-Allow-Origin: https://yourdomain.com'",
"HIGH"),
("Security Misconfiguration","HIGH",
r'(?i)(cors|CORS)\s*\(\s*\{[^}]*origin\s*:\s*["\']?\*["\']?',
"Express/Node.js CORS configured with wildcard origin",
"Specify allowed origins explicitly: cors({ origin: ['https://yourdomain.com'] })",
"HIGH"),
("Sensitive Data Exposure","HIGH",
r'(?i)(console\.log|logger\.(debug|info)|print|logging\.(debug|info))\s*\([^)]*?(password|passwd|token|secret|api_key|credit_card|ssn|cvv)',
"Sensitive data (password/token/key) passed to logger/console",
"Never log sensitive data. Mask values: log('auth', {'user': user_id}) not the password.",
"HIGH"),
("Security Misconfiguration","HIGH",
r'(?i)(chmod|os\.chmod)\s*\([^,]+,\s*(0o?777|0o?776|0o?666)',
"World-writable file permissions set (777/666) β€” privilege escalation risk",
"Use 0o644 for files, 0o755 for directories. Never use 777 in production.",
"HIGH"),
("IDOR","HIGH",
r'(?i)(findById|find_by_id|get_object_or_404|Model\.objects\.get|findOne)\s*\([^)]*\b(id|pk|user_id)\s*[=:]\s*(req\.|request\.|params\.|args\.|\$_(GET|POST))',
"Database object fetched by user-supplied ID without ownership verification β€” IDOR",
"After fetching, verify obj.owner_id == current_user.id before returning data.",
"HIGH"),
("Missing Auth","HIGH",
r'(?i)@(app|router|blueprint)\.(get|post|put|delete|patch)\s*\(["\'][^"\']*?(admin|manage|dashboard|config|settings|delete|ban|reset)[^"\']*["\'](?!.*login_required|.*auth)',
"Admin/sensitive route without visible auth decorator",
"Add @login_required or @admin_required decorator to all admin routes.",
"MEDIUM"),
("Weak Cryptography","HIGH",
r'(?i)\bsecrets\b.*?=.*?\brandom\b|\btoken\b.*?=.*?\brandom\.(randint|randrange|getrandbits|choice)',
"Python random module used for security token β€” cryptographically weak",
"Use secrets.token_hex(32) or secrets.token_urlsafe(32) for security tokens.",
"HIGH"),
("Security Misconfiguration","MEDIUM",
r'(?i)verify\s*=\s*False\s*\)|ssl_verify\s*=\s*False|check_hostname\s*=\s*False',
"SSL/TLS certificate verification disabled β€” man-in-the-middle attack risk",
"Never disable SSL verification in production. Fix the certificate instead.",
"HIGH"),
("Security Misconfiguration","MEDIUM",
r'(?i)http://(?!localhost|127\.0\.0\.1|0\.0\.0\.0)[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}.*?(api|auth|login|payment|token)',
"Plain HTTP (not HTTPS) used for sensitive API endpoint",
"Always use HTTPS for authentication, payment, and API endpoints.",
"HIGH"),
# ═══ SPECIFIC SECRET / API KEY PATTERNS ═══════════════
("Hardcoded Secret","CRITICAL",
r'ghp_[A-Za-z0-9]{36}',
"GitHub Personal Access Token (ghp_) in source code",
"Revoke immediately at github.com/settings/tokens. Use environment variables or GitHub Secrets.",
"HIGH"),
("Hardcoded Secret","CRITICAL",
r'gho_[A-Za-z0-9]{36}',
"GitHub OAuth Token (gho_) in source code",
"Revoke immediately at github.com/settings/tokens. Store tokens outside source code.",
"HIGH"),
("Hardcoded Secret","CRITICAL",
r'sk_live_[A-Za-z0-9]{24,}',
"Stripe Live Secret Key in source code β€” immediate financial risk",
"Revoke at dashboard.stripe.com/apikeys immediately. Use sk_test_ keys for development only.",
"HIGH"),
("Hardcoded Secret","CRITICAL",
r'AIza[0-9A-Za-z\-_]{35}',
"Google API Key detected in source code",
"Restrict key usage at console.cloud.google.com. Store in environment variables.",
"HIGH"),
("Hardcoded Secret","HIGH",
r'xox[baprs]-[A-Za-z0-9-]{10,50}',
"Slack API Token (xoxb/xoxp/xoxa) in source code",
"Revoke at api.slack.com/apps. Use environment variables. Enable token rotation.",
"HIGH"),
("Hardcoded Secret","HIGH",
r'SG\.[A-Za-z0-9\-_]{22}\.[A-Za-z0-9\-_]{43}',
"SendGrid API Key in source code",
"Revoke at app.sendgrid.com/settings/api_keys. Use environment variable SENDGRID_API_KEY.",
"HIGH"),
("Hardcoded Secret","HIGH",
r'AC[a-f0-9]{32}',
"Twilio Account SID pattern detected β€” possible credential exposure",
"Verify this is not a real SID. Store credentials in environment variables.",
"MEDIUM"),
("Hardcoded Secret","HIGH",
r'(?i)(FIREBASE_(?:API_KEY|SECRET|TOKEN)|firebaseConfig\s*=\s*\{[^}]*apiKey\s*:\s*["\'])',
"Firebase credentials or API key in source code",
"Use Firebase App Check. Restrict API keys. Never embed service account credentials in client code.",
"HIGH"),
("Hardcoded Secret","HIGH",
r'-----BEGIN CERTIFICATE-----',
"Certificate embedded in source code",
"Store certificates in secure key management or deployment secrets. Remove from version control.",
"MEDIUM"),
# ═══ JAVA ENTERPRISE RULES ════════════════════════════
("SQL Injection","CRITICAL",
r'(?i)String\s+\w*[Ss][Qq][Ll]\w*\s*=\s*["\'].*\+\s*\w+|"SELECT.*"\s*\+\s*\w+|"INSERT.*"\s*\+\s*\w+|"UPDATE.*"\s*\+\s*\w+',
"Java SQL query built with string concatenation β€” SQL injection",
"Use PreparedStatement: PreparedStatement ps = conn.prepareStatement('SELECT * FROM t WHERE id=?'); ps.setInt(1, id);",
"HIGH"),
("SQL Injection","CRITICAL",
r'(?i)String\.format\s*\(\s*["\'].*?(SELECT|INSERT|UPDATE|DELETE|WHERE)',
"Java String.format() used to build SQL query β€” SQL injection",
"Replace with PreparedStatement or use JPA/Spring Data repository methods.",
"HIGH"),
("Command Injection","CRITICAL",
r'(?i)Runtime\.getRuntime\s*\(\s*\)\s*\.exec\s*\([^)]*\+',
"Java Runtime.exec() with concatenated string β€” command injection",
"Use ProcessBuilder with a string array: new ProcessBuilder(cmd, arg1, arg2). Never concatenate user data.",
"HIGH"),
("XXE","CRITICAL",
r'(?i)DocumentBuilderFactory\.newInstance\s*\(\s*\)',
"Java DocumentBuilderFactory without XXE protection β€” verify setFeature() calls",
"Add: factory.setFeature('http://apache.org/xml/features/disallow-doctype-decl', true)",
"MEDIUM"),
("Security Misconfiguration","CRITICAL",
r'(?i)management\.endpoints\.web\.exposure\.include\s*=\s*["\']?\*["\']?',
"Spring Boot Actuator exposes ALL endpoints β€” env/heapdump/shutdown accessible to attackers",
"Restrict: management.endpoints.web.exposure.include=health,info",
"HIGH"),
("Insecure Deserialization","CRITICAL",
r'(?i)(ObjectInputStream|XMLDecoder|XStream|Kryo)\s+\w+\s*=\s*new\s+(ObjectInputStream|XMLDecoder|XStream|Kryo)',
"Java deserialization library instantiated β€” verify input is trusted",
"Use serialization filters (JEP 290). Never deserialize untrusted data. Prefer JSON.",
"MEDIUM"),
# ═══ GO LANGUAGE RULES ════════════════════════════════
("SQL Injection","CRITICAL",
r'(?i)fmt\.Sprintf\s*\(\s*["\'].*?(SELECT|INSERT|UPDATE|DELETE|WHERE)',
"Go fmt.Sprintf used to build SQL query β€” SQL injection risk",
"Use db.Query('SELECT * FROM t WHERE id = ?', id) with parameterized queries.",
"HIGH"),
("SQL Injection","HIGH",
r'(?i)db\.(Query|Exec|QueryRow)\s*\(\s*fmt\.(Sprintf|Errorf)',
"Go database query with fmt.Sprintf β€” SQL injection risk",
"Pass the query string directly with parameters as additional arguments to db.Query().",
"HIGH"),
("Command Injection","HIGH",
r'(?i)exec\.Command\s*\([^)]*\+',
"Go exec.Command with string concatenation β€” command injection risk",
"Never concatenate user input. Pass arguments as separate strings: exec.Command('cmd', arg1, arg2)",
"HIGH"),
("Path Traversal","HIGH",
r'(?i)(?:os\.Open|os\.ReadFile|ioutil\.ReadFile|http\.ServeFile)\s*\([^)]*\+',
"Go file operation with concatenated path β€” path traversal risk",
"Use filepath.Clean() and verify the result starts with the expected base directory.",
"MEDIUM"),
# ═══ C# / ASP.NET RULES ═══════════════════════════════
("SQL Injection","CRITICAL",
r'(?i)new\s+SqlCommand\s*\(\s*["\'][^"\']*\+|SqlCommand\s*\([^)]*string\.(Format|Concat)',
"C# SqlCommand built with string concatenation β€” SQL injection",
"Use parameterized queries: cmd.Parameters.AddWithValue('@param', value). Use EF Core or Dapper.",
"HIGH"),
("Command Injection","HIGH",
r'(?i)Process\.Start\s*\([^)]*\+|new\s+ProcessStartInfo\s*\(\s*[^)]*\+',
"C# Process.Start with concatenated string β€” command injection",
"Whitelist allowed commands. Never pass user input directly to Process.Start().",
"HIGH"),
("Security Misconfiguration","HIGH",
r'(?i)<\s*customErrors\s+mode\s*=\s*["\']Off["\']',
"ASP.NET customErrors mode='Off' β€” detailed error pages shown to users in production",
"Set mode='On' or mode='RemoteOnly' in production to hide stack traces from users.",
"HIGH"),
("Insecure Deserialization","CRITICAL",
r'(?i)(BinaryFormatter|SoapFormatter|NetDataContractSerializer|LosFormatter)\s*\(\s*\)',
"C# BinaryFormatter/SoapFormatter detected β€” insecure deserialization (deprecated)",
"BinaryFormatter is disabled in .NET 5+. Use System.Text.Json or MessagePack instead.",
"HIGH"),
# ═══ RUBY / RAILS RULES ═══════════════════════════════
("SQL Injection","CRITICAL",
r'(?i)(find_by_sql|execute|connection\.execute)\s*\(\s*["\'][^"\']*#\{|where\s*\(\s*["\'][^"\']*#\{',
"Ruby ActiveRecord raw SQL with string interpolation #{} β€” SQL injection",
"Use parameterized form: User.where('name = ?', name) or User.where(name: name).",
"HIGH"),
("Code Injection","CRITICAL",
r'(?i)\beval\s*\([^)]*params\b|\beval\s*\([^)]*request\.',
"Ruby eval() with request parameters β€” remote code execution",
"Remove eval() entirely. Never evaluate user-controlled strings in Ruby.",
"HIGH"),
("Mass Assignment","HIGH",
r'(?i)\.update\s*\(\s*params\s*\[\s*:\w+\s*\]\s*\)(?!\.permit)',
"Rails mass assignment without strong parameters (.permit not called)",
"Use strong parameters: params.require(:model).permit(:field1, :field2)",
"HIGH"),
# ═══ JWT / AUTH IMPROVEMENTS ══════════════════════════
("Broken Auth","HIGH",
r'(?i)jwt\.sign\s*\([^,]+,\s*["\'][^"\']{1,15}["\']',
"JWT signed with a very short secret (< 16 chars) β€” brute-forceable",
"Use a cryptographically random 256-bit secret. Generate with: openssl rand -hex 32",
"HIGH"),
("Broken Auth","HIGH",
r'(?i)jwt\.encode\s*\([^)]*,\s*["\'][^"\']{1,15}["\']',
"Python JWT encoded with short secret key β€” brute-force risk",
"Use a long random secret: SECRET = secrets.token_hex(32)",
"HIGH"),
("Broken Auth","HIGH",
r'(?i)(session|cookie)\[[\'"]:?user_?id[\'"]?\]\s*=\s*params\[|session\[:user\]\s*=\s*\w+(?!\.authenticated)',
"Session user ID set from request params without explicit auth verification",
"Verify authentication before setting session. Use proper auth framework methods.",
"MEDIUM"),
# ═══ PHP ADDITIONAL ═══════════════════════════════════
("SQL Injection","CRITICAL",
r'(?i)mysql_query\s*\([^)]*\$_(GET|POST|REQUEST|COOKIE)',
"Deprecated mysql_query() with user input β€” SQL injection + obsolete API (removed PHP 7)",
"Migrate to PDO: $stmt = $pdo->prepare('SELECT * FROM t WHERE id=?'); $stmt->execute([$id]);",
"HIGH"),
("Command Injection","CRITICAL",
r'(?i)preg_replace\s*\([^,]*e\b[^,]*,[^,]+,\s*\$_(GET|POST|REQUEST)',
"PHP preg_replace with /e modifier evaluates replacement as PHP code β€” RCE",
"Remove the /e modifier. Use preg_replace_callback() instead.",
"HIGH"),
("Insecure File Upload","CRITICAL",
r'(?i)\$_(FILES)\s*\[.*\]\s*\[[\'"](name|type)[\'"]]\](?!.*in_array|.*whitelist|.*allowlist)',
"PHP file upload using $_FILES['name'] or ['type'] without extension whitelist",
"Never trust the client-supplied filename or MIME type. Validate extension server-side with finfo_file().",
"HIGH"),
# ═══ JAVASCRIPT / TYPESCRIPT ADDITIONAL ═══════════════
("Code Injection","CRITICAL",
r'(?i)new\s+Function\s*\([^)]*(?:req\.|request\.|params\[|query\[|body\.)',
"JavaScript new Function() with user-controlled input β€” code injection",
"Never create functions from user-supplied strings. Use JSON.parse() for data.",
"HIGH"),
("Prototype Pollution","HIGH",
r'(?i)_\.merge\s*\(\s*\{?\s*\}?\s*,.*(?:req\.|body|params|query)',
"Lodash _.merge with user-controlled object β€” prototype pollution",
"Use _.mergeWith() with a customizer that rejects __proto__ keys, or use safe alternatives.",
"HIGH"),
("Prototype Pollution","HIGH",
r'(?i)JSON\.parse\s*\([^)]*(?:req\.|request\.|body|params)',
"JSON.parse with user input β€” prototype pollution if result is merged into an object",
"Validate parsed JSON structure before merging. Block __proto__ and constructor keys.",
"LOW"),
("XSS","HIGH",
r'(?i)\$\s*\(\s*(?:location\.|document\.URL|document\.referrer|window\.name)',
"jQuery selector using browser location data β€” DOM XSS risk",
"Never pass browser location data to jQuery $(). Use textContent or DOMPurify.",
"HIGH"),
("XSS","HIGH",
r'(?i)dangerouslySetInnerHTML\s*=\s*\{\s*\{',
"React dangerouslySetInnerHTML used β€” potential XSS if content is user-controlled",
"Sanitize content with DOMPurify.sanitize() before using dangerouslySetInnerHTML.",
"MEDIUM"),
# ═══ NODE.JS / TYPESCRIPT SPECIFIC ════════════════════
# Knex / DB driver raw SQL with template literal interpolation
("SQL Injection","CRITICAL",
r'(?i)\.(raw|whereRaw|havingRaw|joinRaw|orderByRaw)\s*\(`[^`]*\$\{',
"Knex/DB .raw() with template literal interpolation ${} β€” SQL injection",
"Use parameterized placeholders: .raw('SELECT ?? WHERE id = ?', [table, id]) β€” never ${variable}.",
"HIGH"),
# TypeORM .where() with string interpolation or concatenation
("SQL Injection","HIGH",
r'(?i)\.(where|andWhere|orWhere)\s*\(`[^`]*\$\{|\.(where|andWhere|orWhere)\s*\(["\'][^"\']*["\']\s*\+',
"TypeORM .where() with template literal interpolation or string concat β€” SQL injection",
"Use parameterized form: .where('col = :val', { val: userInput })",
"HIGH"),
# Prisma $queryRaw with manual interpolation or $queryRawUnsafe
("SQL Injection","CRITICAL",
r'(?i)\.\$queryRawUnsafe\s*\(|prisma\.\$executeRawUnsafe\s*\(',
"Prisma $queryRawUnsafe/$executeRawUnsafe β€” bypasses parameterization, SQL injection risk",
"Use prisma.$queryRaw tagged template literal which auto-parameterizes inputs.",
"HIGH"),
# Vue v-html directive
("XSS","HIGH",
r'v-html\s*=\s*["\']?\s*\w',
"Vue v-html directive β€” XSS if content is user-controlled",
"Sanitize with DOMPurify.sanitize() before v-html. Use {{ text }} for plain text output.",
"MEDIUM"),
# Node.js child_process with template literal containing variable
("Command Injection","CRITICAL",
r'(?i)(execSync?|spawnSync?|execFileSync?)\s*\(`[^`]*\$\{',
"Node.js child_process with template literal ${} β€” command injection",
"Never use template literals in exec/execSync. Use spawn() with argument array.",
"HIGH"),
("Command Injection","HIGH",
r'(?i)(?:exec|execSync)\s*\([^)]*\+\s*\w+',
"Node.js exec() with string concatenation β€” command injection",
"Use spawn() with an argument array. Never concatenate user input into shell commands.",
"HIGH"),
# Express res.sendFile / res.download with user input
("Path Traversal","HIGH",
r'(?i)res\.(sendFile|sendfile|download)\s*\([^)]*(?:req\.(params|query|body)|params\[|query\[|body\.)',
"Express res.sendFile/download with user-controlled path β€” path traversal",
"Use path.resolve() and verify the path starts with your static directory.",
"HIGH"),
# Node.js fs operations with user-controlled path
("Path Traversal","HIGH",
r'(?i)fs\.(readFile|writeFile|appendFile|createReadStream|createWriteStream|readFileSync|writeFileSync|unlink)\s*\([^,)]*(?:req\.(params|query|body)|params\[|query\[)',
"Node.js fs operation with user-controlled path β€” path traversal",
"Validate: const safe = path.resolve(BASE, userPath); if (!safe.startsWith(BASE)) throw Error('Forbidden');",
"HIGH"),
# Axios SSRF in TypeScript/Node.js
("SSRF","CRITICAL",
r'(?i)axios\.(get|post|put|delete|request|head)\s*\([^)]*(?:req\.(body|query|params)\.|params\[|query\[|body\.)',
"Axios HTTP request with user-controlled URL β€” SSRF risk",
"Validate URL against an allowlist. Block 169.254.x.x, 10.x.x.x, 172.16-31.x.x, 192.168.x.x.",
"HIGH"),
# Node.js fetch() SSRF
("SSRF","CRITICAL",
r'(?i)\bfetch\s*\(\s*(?:req\.(body|query|params)\.|params\[|query\[|body\.|\w+url\w*)',
"Node.js fetch() with potentially user-controlled URL β€” SSRF risk",
"Validate and allowlist URLs before calling fetch().",
"MEDIUM"),
# Hardcoded JWT secret fallback
("Broken Auth","HIGH",
r'(?i)(?:jwt\.verify|jwt\.sign)\s*\([^)]*process\.env\.\w+\s*\|\|\s*["\']',
"JWT using fallback hardcoded secret when env var is missing",
"Never fall back to a hardcoded secret. Fail fast if the env var is missing.",
"HIGH"),
# eval() in TypeScript with request data
("Code Injection","CRITICAL",
r'(?i)\beval\s*\([^)]*(?:req\.(body|query|params)|params\[|query\[|body\.)',
"eval() with Express request data β€” remote code execution",
"Remove eval() entirely. Use JSON.parse() or a safe expression evaluator.",
"HIGH"),
# new Function() with user input
("Code Injection","CRITICAL",
r'(?i)new\s+Function\s*\([^)]*(?:req\.(body|query|params)|params\[|query\[|body\.)',
"new Function() with user-controlled input β€” code injection equivalent to eval()",
"Never create functions from user-supplied strings.",
"HIGH"),
# TypeScript/Node.js hardcoded secret patterns
("Hardcoded Secret","CRITICAL",
r'(?i)(?:jwtSecret|JWT_SECRET|jwtKey|tokenSecret)\s*[=:]\s*["\'][^"\']{6,}["\'](?!\s*\|\|)',
"Hardcoded JWT secret in TypeScript/Node.js source",
"Load from environment: process.env.JWT_SECRET. Minimum 256-bit random value.",
"HIGH"),
# Missing authorization check on route (TypeScript/Express)
("Missing Auth","HIGH",
r'(?i)(?:router|app)\.(get|post|put|delete|patch)\s*\(["\'][^"\']*(?:admin|manage|dashboard|delete|ban|reset|config)[^"\']*["\'](?!.*(?:auth|verify|guard|middleware|isAdmin|requireAuth))',
"Express route for admin/sensitive path without visible auth middleware",
"Add authentication middleware: router.use('/admin', authMiddleware); or use Guards.",
"MEDIUM"),
]
SCANNABLE_EXT = {
".py",".php",".js",".ts",".jsx",".tsx",".java",".rb",
".go",".cs",".sh",".bash",".yml",".yaml",".env",
".cfg",".conf",".config",".xml",".html",".htm",
".jsp",".asp",".aspx",".pl",".vue",".svelte",
}
SKIP_DIRS = {
"node_modules","vendor","dist","build",".git","__pycache__",
"bower_components","venv",".venv","coverage",
"fixtures","static","assets","docs","documentation",
".next",".nuxt","storybook-static","public","i18n","locales",
"examples","example","demo","demos","sample","samples","playground",
}
# Unit-test files β†’ always skipped (mock credentials cause massive FP)
# Integration test dirs (test/, tests/) are still scanned but with
# severity reduction for Hardcoded Secret findings.
SKIP_FILE_SUFFIXES = frozenset({
".spec.ts", ".spec.js", ".spec.tsx", ".spec.jsx",
".test.ts", ".test.js", ".test.tsx", ".test.jsx",
".spec.py", ".test.py", ".spec.rb", ".test.rb",
".spec.java", ".test.java",
})
# Directories that are pure test infrastructure (no real code)
SKIP_TEST_INFRA_DIRS = frozenset({
"cypress", "__tests__", "__mocks__", "mocks", "stubs",
"e2e", "jest", "jasmine", "storetest", "testdata", "testutil",
"fakestore", "mockstore", "teststore",
"test-data", "test_data", "test-fixtures", "test_fixtures",
"testfixtures", "snapshots", "__snapshots__",
})
def _is_test_context(path):
"""True when path belongs to a test/integration dir (not a unit-test file)."""
p = path.lower()
return (
"/test/" in p or "/tests/" in p or "/spec/" in p
or "/test-data/" in p or "/testdata/" in p
or "/test-fixtures/" in p or "/test_fixtures/" in p
or p.startswith("test/") or p.startswith("tests/")
)
PRIORITY_PATH_KEYWORDS = (
"auth","login","admin","sql","db","database","query","user","account",
"upload","exec","shell","cmd","password","secret","token","config",
"include","api","controller","router","middleware","session","crypto",
"vuln","payment","checkout","profile","register","forgot","reset",
"service","handler","guard","model","resolver","gateway","interceptor",
"permission","role","webhook","oauth","jwt","crypt","hash","sanitize",
)
MIN_FILE_BYTES = 500 # skip empty re-export stubs
MAX_FILE_BYTES = 300 * 1024 # 300 KB
DEFAULT_MAX = 300
HARD_MAX = 500
# ══════════════════════════════════════════════════════════
# PYTHON AST DEEP SCANNER
# ══════════════════════════════════════════════════════════
class PythonASTScanner(ast.NodeVisitor):
def __init__(self):
self.findings, self.lines = [], []
def _add(self, node, vuln, sev, desc, fix, conf="HIGH"):
ln = getattr(node, "lineno", 0)
snippet = self.lines[ln-1].strip()[:150] if 0 < ln <= len(self.lines) else ""
self.findings.append({
"vuln":vuln,"severity":sev,"line":ln,
"snippet":snippet,"desc":desc,"fix":fix,
"confidence":conf,"source":"AST",
})
def scan(self, src):
self.findings, self.lines = [], src.splitlines()
try:
self.visit(ast.parse(src))
except SyntaxError:
pass
return self.findings
def visit_Call(self, node):
# eval/exec/compile with non-constant
if isinstance(node.func, ast.Name) and node.func.id in ("eval","exec","compile"):
if node.args and not isinstance(node.args[0], ast.Constant):
self._add(node,"Code Injection","CRITICAL",
f"{node.func.id}() called with non-constant argument",
f"Remove {node.func.id}(). Use safe parsers or whitelisted operations.")
if isinstance(node.func, ast.Attribute):
mod = node.func.value
attr = node.func.attr
mod_name = getattr(mod,"id","") if isinstance(mod, ast.Name) else ""
# pickle.loads
if attr in ("loads","load") and mod_name == "pickle":
self._add(node,"Insecure Deserialization","CRITICAL",
"pickle.loads() deserializes arbitrary objects β€” code execution risk",
"Use json.loads() for data exchange. Never unpickle untrusted data.")
# yaml.load without safe loader
if attr == "load" and mod_name == "yaml":
has_safe = any(
("Safe" in getattr(kw.value,"attr","") or "Safe" in getattr(kw.value,"id",""))
for kw in node.keywords if kw.arg == "Loader"
)
if not has_safe:
self._add(node,"Insecure Deserialization","CRITICAL",
"yaml.load() without SafeLoader β€” YAML tags execute arbitrary Python",
"Use yaml.safe_load() instead of yaml.load().")
# subprocess shell=True
if attr in ("run","call","Popen") and mod_name == "subprocess":
for kw in node.keywords:
if kw.arg == "shell" and isinstance(kw.value, ast.Constant) and kw.value.value:
self._add(node,"Command Injection","HIGH",
"subprocess called with shell=True β€” allows shell metacharacter injection",
"Use shell=False and pass arguments as a list.")
# os.system with non-constant
if attr == "system" and mod_name == "os":
if node.args and not isinstance(node.args[0], ast.Constant):
self._add(node,"Command Injection","CRITICAL",
"os.system() with dynamic argument β€” command injection risk",
"Use subprocess.run([cmd, arg], shell=False).")
# hashlib weak algorithms
if attr == "new" and mod_name == "hashlib":
algo = getattr(node.args[0],"value","") if node.args else ""
if isinstance(algo,str) and algo.lower() in ("md5","sha1","sha"):
self._add(node,"Weak Cryptography","HIGH",
f"hashlib.new('{algo}') β€” cryptographically broken algorithm",
"Use hashlib.sha256() or sha3_256(). For passwords use bcrypt/argon2.")
# marshal.loads
if attr in ("loads","load") and mod_name == "marshal":
self._add(node,"Insecure Deserialization","MEDIUM",
"marshal.loads() is unsafe with untrusted data",
"Use JSON for data exchange. Never use marshal with external data.")
self.generic_visit(node)
def visit_Import(self, node):
risky = {"pickle":"CRITICAL","marshal":"MEDIUM","shelve":"LOW"}
for alias in node.names:
if alias.name in risky:
self._add(node,"Insecure Deserialization", risky[alias.name],
f"'{alias.name}' imported β€” ensure only trusted data is deserialized",
"Prefer JSON or safer serialization formats.",
"LOW")
self.generic_visit(node)
# ══════════════════════════════════════════════════════════
# TAINT TRACKER β€” Multi-line Data Flow Analysis
# Tracks user-controlled variables from source to sink
# across multiple lines (catches what single-line regex misses)
# ══════════════════════════════════════════════════════════
class TaintTracker:
"""
Two-pass taint analysis:
Pass 1 β€” identify variables assigned from user input (sources)
Pass 2 β€” check if tainted variables reach dangerous functions (sinks)
"""
PHP_SOURCES = re.compile(
r'\$(\w+)\s*=\s*'
r'(?:'
r'\$_(GET|POST|REQUEST|COOKIE|FILES|SERVER)\s*\[|'
r'(?:trim|stripslashes|strip_tags|intval|addslashes)\s*\(\s*\$_(GET|POST|REQUEST|COOKIE)\s*\[|'
r'(?:filter_input|filter_var)\s*\([^)]*INPUT_(?:GET|POST|COOKIE)'
r')',
re.IGNORECASE
)
PY_SOURCES = re.compile(
r'(\w+)\s*=\s*'
r'(?:'
r'request\.(?:args|form|values|json|data|cookies|headers)\s*[\.\[]|'
r'request\.(?:args|form|values|json|data|cookies)\.get\s*\(|'
r'req\.(?:query|body|params)\s*[\.\[]|'
r'flask\.request\.|'
r'bottle\.request\.|'
r'input\s*\('
r')',
re.IGNORECASE
)
SINKS = {
"Command Injection": [
r'(?i)(shell_exec|system|exec|passthru|popen|proc_open)\s*\([^;{{]*\${V}\b',
r'(?i)(os\.system|subprocess\.run|subprocess\.call|os\.popen)\s*\([^)]*\b{V}\b',
r'`[^`]*\${V}\b',
],
"SQL Injection": [
r'(?i)["\'][^"\']*["\'\s]\s*\.\s*\${V}\b',
r'(?i)(query|execute|mysql_query|mysqli_query|pg_query)\s*\([^;]*\${V}\b',
r'(?i)f["\'].*?(SELECT|INSERT|UPDATE|DELETE|WHERE).*\b{V}\b',
r'(?i)["\'].*?(SELECT|INSERT|UPDATE|DELETE|WHERE).*["\'].*\+.*\b{V}\b',
],
"XSS": [
r'(?i)(echo|print)\s+[^;]*\${V}\b(?!.*htmlspecialchars)(?!.*htmlentities)',
r'(?i)res\.(?:send|write|end)\s*\([^)]*\b{V}\b',
],
"File Inclusion": [
r'(?i)(include|require)(_once)?\s*\([^;]*\${V}\b',
],
"Path Traversal": [
r'(?i)(fopen|file_get_contents|readfile|unlink|rename|copy)\s*\([^;]*\${V}\b',
r'(?i)\bopen\s*\([^)]*\b{V}\b',
],
"SSRF": [
r'(?i)(curl_exec|file_get_contents|fsockopen|requests\.get|requests\.post|urllib\.request\.urlopen)\s*\([^;)]*\b{V}\b',
],
"Code Injection": [
r'(?i)\b(eval|exec|compile|assert)\s*\([^)]*\b{V}\b',
],
}
def scan(self, content, filename, lang="php"):
ext = os.path.splitext(filename)[1].lower()
lines = content.splitlines()
tainted = {} # var_name -> (source_label, source_lineno)
# ── Pass 1: Find sources ─────────────────────────────
source_re = self.PHP_SOURCES if ext == ".php" else self.PY_SOURCES
for lineno, line in enumerate(lines, 1):
m = source_re.search(line)
if m:
var_name = m.group(1)
# Label: first non-None group after group(1)
label = next((g for g in m.groups()[1:] if g), "USER_INPUT")
tainted[var_name] = (label, lineno)
if not tainted:
return []
# ── Pass 2: Find sinks ───────────────────────────────
findings = []
seen = set()
for var_name, (source_label, src_line) in tainted.items():
for vuln_type, patterns in self.SINKS.items():
for lineno, line in enumerate(lines, 1):
if lineno == src_line:
continue
for pat_tmpl in patterns:
# Replace {V} or $\{V\} placeholder with the actual var name
pat = pat_tmpl.replace("{V}", re.escape(var_name))
try:
if re.search(pat, line):
key = (filename, lineno, vuln_type, var_name)
if key in seen:
continue
seen.add(key)
findings.append({
"vuln": vuln_type,
"severity": "CRITICAL",
"line": lineno,
"snippet": line.strip()[:150],
"desc": (
f"[TaintTrack] ${var_name} from "
f"$_{source_label} (line {src_line}) "
f"reaches {vuln_type} sink unsanitized"
),
"fix": (
"Sanitize before use: "
"escapeshellarg() / htmlspecialchars() / "
"prepared statements / realpath() check."
),
"confidence": "HIGH",
"source": "TaintTrack",
"file": filename,
"note": f"Variable ${var_name} originates from user input at line {src_line}.",
})
break
except re.error:
pass
return findings
# ══════════════════════════════════════════════════════════
# GO TAINT TRACKER
# Multi-line taint analysis for Go (net/http, Echo, Gin, Chi)
# ══════════════════════════════════════════════════════════
class GoTaintTracker:
SOURCES = re.compile(
r'(\w+)\s*:?=\s*'
r'(?:r\.(?:URL\.Query\(\)\.Get|FormValue|PostFormValue|'
r'Header\.Get|PathValue|Cookie)\s*\(|'
r'c\.(?:QueryParam|FormValue|Param|GetHeader|PathParam)\s*\(|'
r'ctx\.(?:QueryParam|FormValue|Param)\s*\(|'
r'chi\.URLParam\s*\(|'
r'mux\.Vars\s*\([^)]+\)\s*\[|'
r'vars\s*\[)',
re.IGNORECASE,
)
SINKS = {
"SQL Injection": [
r'(?i)(?:db|tx|conn)\s*\.\s*(?:Query|Exec|QueryRow|QueryContext|ExecContext)\s*\([^,)]*\b{V}\b',
r'(?i)fmt\.Sprintf\s*\(["\'][^"\']*(?:SELECT|INSERT|UPDATE|DELETE|WHERE)[^"\']*["\'],\s*[^)]*\b{V}\b',
r'(?i)["\'][^"\']*(?:SELECT|INSERT|UPDATE|DELETE|WHERE)[^"\']*["\']\s*\+\s*\b{V}\b',
],
"Command Injection": [
r'(?i)exec\.Command\s*\([^)]*\b{V}\b',
r'(?i)exec\.CommandContext\s*\([^)]*\b{V}\b',
r'(?i)os\.StartProcess\s*\([^)]*\b{V}\b',
],
"Path Traversal": [
r'(?i)os\.(?:Open|Create|ReadFile|WriteFile|Stat|Remove)\s*\([^)]*\b{V}\b',
r'(?i)filepath\.(?:Join|Clean|Abs)\s*\([^)]*\b{V}\b',
r'(?i)http\.ServeFile\s*\([^)]*\b{V}\b',
],
"SSRF": [
r'(?i)http\.(?:Get|Post|NewRequest)\s*\([^)]*\b{V}\b',
r'(?i)client\.(?:Get|Post|Do)\s*\([^)]*\b{V}\b',
],
"XSS": [
r'(?i)(?:fmt\.Fprint|fmt\.Fprintf|w\.Write)\s*\([^)]*\b{V}\b',
r'(?i)(?:c\.String|c\.HTML|ctx\.String|ctx\.HTML)\s*\([^)]*\b{V}\b',
],
}
def scan(self, content: str, filename: str) -> list:
lines = content.splitlines()
tainted = {}
findings, seen = [], set()
# Pass 1 β€” sources
for lineno, line in enumerate(lines, 1):
m = self.SOURCES.search(line)
if m:
tainted[m.group(1)] = lineno
if not tainted:
return []
# Pass 2 β€” sinks
for var_name, src_line in tainted.items():
for vuln_type, patterns in self.SINKS.items():
for lineno, line in enumerate(lines, 1):
if lineno == src_line:
continue
for pat_tmpl in patterns:
pat = pat_tmpl.replace("{V}", re.escape(var_name))
try:
if re.search(pat, line):
key = (filename, lineno, vuln_type, var_name)
if key in seen:
continue
seen.add(key)
suppressed, _ = _is_suppressed(vuln_type, line)
if suppressed:
continue
findings.append({
"vuln": vuln_type,
"severity": "CRITICAL",
"line": lineno,
"snippet": line.strip()[:150],
"desc": (
f"[GoTaint] `{var_name}` from HTTP request "
f"(line {src_line}) reaches {vuln_type} sink unsanitized."
),
"fix": (
"Validate and sanitize before use. "
"Use parameterized queries (db.Query(sql, args...)), "
"filepath.Clean() + prefix check, or allow-list validation."
),
"confidence": "HIGH",
"source": "GoTaint",
"file": filename,
"note": (
f"`{var_name}` originates from user-controlled HTTP input "
f"at line {src_line}."
),
})
break
except re.error:
pass
return findings
# ══════════════════════════════════════════════════════════
# JAVA TAINT TRACKER
# Multi-line taint analysis for Java/Spring/Jakarta EE
# ══════════════════════════════════════════════════════════
class JavaTaintTracker:
SOURCES = re.compile(
r'(?:String\s+)?(\w+)\s*=\s*'
r'(?:request\.(?:getParameter|getHeader|getAttribute|getQueryString)\s*\(|'
r'httpRequest\.(?:getParameter|getHeader)\s*\()',
re.IGNORECASE,
)
# Also catch Spring @RequestParam / @PathVariable annotations
ANNOTATION_SRC = re.compile(
r'@(?:RequestParam|PathVariable|RequestBody|RequestHeader)\b[^)]*\)\s+'
r'(?:String|int|long|Object)\s+(\w+)',
re.IGNORECASE,
)
SINKS = {
"SQL Injection": [
r'(?i)(?:statement|stmt|preparedStatement|ps)\s*\.'
r'(?:execute|executeQuery|executeUpdate)\s*\([^)]*\b{V}\b',
r'(?i)["\'][^"\']*(?:SELECT|INSERT|UPDATE|DELETE|WHERE)[^"\']*["\']\s*\+\s*\b{V}\b',
r'(?i)(?:createQuery|createNativeQuery)\s*\([^)]*\b{V}\b',
],
"Command Injection": [
r'(?i)(?:Runtime\.getRuntime\(\)\.exec|ProcessBuilder)\s*\([^)]*\b{V}\b',
],
"Path Traversal": [
r'(?i)new\s+File\s*\([^)]*\b{V}\b',
r'(?i)Paths\.get\s*\([^)]*\b{V}\b',
r'(?i)new\s+FileInputStream\s*\([^)]*\b{V}\b',
],
"SSRF": [
r'(?i)new\s+URL\s*\([^)]*\b{V}\b',
r'(?i)(?:RestTemplate|HttpClient|WebClient)\b[^;]*\b{V}\b',
],
"XSS": [
r'(?i)(?:response\.getWriter\(\)\.(?:print|write)|out\.print)\s*\([^)]*\b{V}\b',
r'(?i)model\.addAttribute\s*\([^,]+,\s*\b{V}\b',
],
}
def scan(self, content: str, filename: str) -> list:
lines = content.splitlines()
tainted = {}
findings, seen = [], set()
for lineno, line in enumerate(lines, 1):
for pattern in (self.SOURCES, self.ANNOTATION_SRC):
m = pattern.search(line)
if m:
tainted[m.group(1)] = lineno
if not tainted:
return []
for var_name, src_line in tainted.items():
for vuln_type, patterns in self.SINKS.items():
for lineno, line in enumerate(lines, 1):
if lineno == src_line:
continue
for pat_tmpl in patterns:
pat = pat_tmpl.replace("{V}", re.escape(var_name))
try:
if re.search(pat, line):
key = (filename, lineno, vuln_type, var_name)
if key in seen:
continue
seen.add(key)
suppressed, _ = _is_suppressed(vuln_type, line)
if suppressed:
continue
findings.append({
"vuln": vuln_type,
"severity": "CRITICAL",
"line": lineno,
"snippet": line.strip()[:150],
"desc": (
f"[JavaTaint] `{var_name}` from HTTP request "
f"(line {src_line}) reaches {vuln_type} sink unsanitized."
),
"fix": (
"Use PreparedStatement with parameterized queries, "
"ESAPI for encoding, or validated allow-lists."
),
"confidence": "HIGH",
"source": "JavaTaint",
"file": filename,
"note": (
f"`{var_name}` originates from HTTP user input "
f"at line {src_line}."
),
})
break
except re.error:
pass
return findings
# ══════════════════════════════════════════════════════════
# CROSS-FILE TAINT TRACKER
# Tracks user-controlled values exported from one file and
# consumed as SQL/cmd sinks in another file.
# ══════════════════════════════════════════════════════════
class CrossFileTaintTracker:
"""
Two-phase cross-file taint analysis:
Phase 1 β€” scan every file for exported tainted symbols
(functions / variables that return / assign user input).
Phase 2 β€” scan every file for sinks that use those symbols.
Only raises findings when source and sink are in *different* files,
so it adds signal that single-file TaintTracker misses.
"""
# ── Sources: user-controlled values assigned to a named symbol ──
_SOURCE_RE = re.compile(
r'(?:'
# JS/TS: export function/const/let/var that contains req.body / req.query etc.
r'(?:export\s+(?:default\s+)?(?:function|const|let|var|async function)\s+(\w+))'
r'|'
# Python: def func_name(...): with request.args / request.form in body
r'(?:^def\s+(\w+)\s*\()'
r'|'
# PHP: function name
r'(?:^function\s+(\w+)\s*\()'
r')',
re.MULTILINE,
)
_TAINTED_BODY_RE = re.compile(
r'\b(req\.(body|query|params|headers?|cookies?)|'
r'request\.(args|form|values|json|data|cookies)|'
r'\$_(GET|POST|REQUEST|COOKIE)|'
r'ctx\.(params|query|body)|'
r'event\.(body|queryString)|'
r'flask\.request|'
r'input\()\b',
re.IGNORECASE,
)
# ── Sinks: dangerous function calls that take a symbol ──
_SINK_PATTERNS = [
("SQL Injection", re.compile(r'(?i)(\.query|\.execute|\.raw|whereRaw|knex\.raw)\s*\(.*\b{V}\b')),
("SQL Injection", re.compile(r'(?i)["\'][^"\']*(?:SELECT|INSERT|UPDATE|DELETE)[^"\']*["\'].*\+.*\b{V}\b')),
("Command Injection",re.compile(r'(?i)(exec|spawn|system|os\.popen|subprocess)\s*\(.*\b{V}\b')),
("SSRF", re.compile(r'(?i)(fetch|axios|requests\.get|urllib|http\.get)\s*\(.*\b{V}\b')),
("Path Traversal", re.compile(r'(?i)(open|readFile|fs\.read|file_get_contents)\s*\(.*\b{V}\b')),
("XSS", re.compile(r'(?i)(innerHTML|dangerouslySetInnerHTML|res\.send|res\.write|echo)\s*.*\b{V}\b')),
]
def _extract_tainted_symbols(self, content: str, filename: str) -> set:
"""Return set of symbol names that touch user input in this file."""
tainted = set()
lines = content.splitlines()
# Quick pre-check: does this file touch user input at all?
if not self._TAINTED_BODY_RE.search(content):
return tainted
# Find function/const declarations; check if body (next 30 lines) has sources
for m in self._SOURCE_RE.finditer(content):
sym = next((g for g in m.groups() if g), None)
if not sym:
continue
start = content[:m.start()].count('\n')
body_lines = lines[start: start + 30]
if any(self._TAINTED_BODY_RE.search(ln) for ln in body_lines):
tainted.add(sym)
# Also track module-level variable assignments
var_re = re.compile(
r'^(?:const|let|var|export const|export let)?\s*(\w+)\s*='
r'.*(?:req\.|request\.|ctx\.|\$_(?:GET|POST|REQUEST))',
re.MULTILINE | re.IGNORECASE,
)
for m in var_re.finditer(content):
tainted.add(m.group(1))
return tainted
# Common JS/Python identifiers that are NOT user-controlled data.
# Matching these would generate massive FP noise.
_SKIP_SYMS = frozenset({
# Error / callback conventions
"error", "err", "e", "ex", "exception", "cause",
"cb", "callback", "next", "done", "resolve", "reject",
# Generic names
"result", "results", "value", "values", "val", "item", "items",
"data", "payload", "response", "res", "reply",
"fn", "func", "handler", "middleware", "wrapper",
"app", "router", "server", "client", "db", "conn",
"self", "ctx", "context", "scope", "opts", "options", "config",
"i", "j", "k", "n", "x", "y", "t", "s", "p", "c",
"id", "key", "name", "type", "kind", "mode", "flag",
"msg", "message", "text", "str", "buf", "buffer",
"file", "path", "url", "uri", "link", "src",
# Framework internals
"req", "request", "body", "query", "params", # these are sources themselves
"schema", "model", "table", "column", "field",
})
def analyze(self, file_contents: dict) -> list:
"""
file_contents: {filename: source_code_string}
Returns list of finding dicts.
"""
# Phase 1: collect tainted symbols per file
tainted_by_file: dict[str, set] = {}
for fname, code in file_contents.items():
syms = self._extract_tainted_symbols(code, fname)
if syms:
tainted_by_file[fname] = syms
if not tainted_by_file:
return []
# Build global tainted set (symbol β†’ source file)
# Filter out: common ambiguous names, single-char vars, reserved words
global_tainted: dict[str, str] = {}
for fname, syms in tainted_by_file.items():
for sym in syms:
if sym.lower() in self._SKIP_SYMS:
continue
if len(sym) <= 2: # too generic: i, id, fn, ...
continue
global_tainted[sym] = fname
# Phase 2: scan all files for sinks using those symbols
findings = []
seen = set()
for fname, code in file_contents.items():
lines = code.splitlines()
for sym, source_file in global_tainted.items():
if source_file == fname:
continue # same-file flow is handled by TaintTracker
for vuln_type, sink_re_template in self._SINK_PATTERNS:
# Substitute {V} placeholder with actual symbol name
pattern_str = sink_re_template.pattern.replace("{V}", re.escape(sym))
try:
pat = re.compile(pattern_str, re.IGNORECASE)
except re.error:
continue
for lineno, line in enumerate(lines, 1):
if pat.search(line):
key = (fname, lineno, vuln_type, sym)
if key in seen:
continue
seen.add(key)
# Skip if FP suppression triggers
suppressed, _ = _is_suppressed(vuln_type, line)
if suppressed:
continue
findings.append({
"vuln": vuln_type,
"severity": "HIGH",
"line": lineno,
"snippet": line.strip()[:150],
"desc": (
f"[CrossFileTaint] `{sym}` originates from user input "
f"in `{os.path.basename(source_file)}` and reaches "
f"{vuln_type} sink here without sanitization."
),
"fix": (
"Validate and sanitize the value before passing it "
"across module boundaries. Use parameterized queries, "
"escapeshellarg(), or allow-list validation."
),
"confidence": "MEDIUM",
"source": "CrossFileTaint",
"file": fname,
"note": (
f"Taint source: `{os.path.basename(source_file)}`. "
"Cross-file data-flow β€” single-file scanners miss this."
),
})
return findings
# ══════════════════════════════════════════════════════════
# DEPENDENCY SCANNER β€” CVE Detection via OSV.dev API
# Supports: requirements.txt, package.json, composer.json,
# Gemfile.lock, go.mod, Pipfile.lock, pom.xml
# ══════════════════════════════════════════════════════════
class DependencyScanner:
OSV_API = "https://api.osv.dev/v1/query"
BATCH_API = "https://api.osv.dev/v1/querybatch"
ECOSYSTEMS = {
"requirements.txt": "PyPI",
"Pipfile": "PyPI",
"Pipfile.lock": "PyPI",
"setup.cfg": "PyPI",
"pyproject.toml": "PyPI",
"package.json": "npm",
"package-lock.json": "npm",
"yarn.lock": "npm",
"composer.json": "Packagist",
"composer.lock": "Packagist",
"Gemfile": "RubyGems",
"Gemfile.lock": "RubyGems",
"go.mod": "Go",
"go.sum": "Go",
"pom.xml": "Maven",
"build.gradle": "Maven",
}
# ── Parsers ──────────────────────────────────────────────
def _parse_requirements(self, content):
pkgs = []
for line in content.splitlines():
line = line.strip()
if not line or line.startswith(("#", "-", "http")):
continue
m = re.match(r'^([A-Za-z0-9_.-]+)\s*[=~><!]+\s*([0-9][A-Za-z0-9._*-]*)', line)
if m:
pkgs.append((m.group(1), m.group(2)))
return pkgs
def _parse_package_json(self, content):
try:
data = json.loads(content)
pkgs = []
for section in ("dependencies", "devDependencies", "peerDependencies"):
for pkg, ver in (data.get(section) or {}).items():
ver = re.sub(r'^[\^~>=< ]', '', str(ver)).split(" ")[0].strip()
if re.match(r'^[0-9]', ver):
pkgs.append((pkg, ver))
return pkgs
except Exception:
return []
def _parse_composer(self, content):
try:
data = json.loads(content)
pkgs = []
for section in ("require", "require-dev"):
for pkg, ver in (data.get(section) or {}).items():
if pkg == "php":
continue
ver = re.sub(r'^[\^~>=< v]', '', str(ver)).split(" ")[0].strip()
if re.match(r'^[0-9]', ver):
pkgs.append((pkg, ver))
return pkgs
except Exception:
return []
def _parse_gemfile_lock(self, content):
pkgs = []
in_specs = False
for line in content.splitlines():
if "GEM" in line:
in_specs = True
if in_specs:
m = re.match(r'\s{4}([a-z][a-z0-9_-]*)\s+\(([0-9][^)]*)\)', line)
if m:
pkgs.append((m.group(1), m.group(2)))
return pkgs
def _parse_go_mod(self, content):
pkgs = []
for line in content.splitlines():
m = re.match(r'\s*(require\s+)?([a-z][a-z0-9./\-]+)\s+v([0-9][^\s]*)', line)
if m:
pkgs.append((m.group(2), m.group(3)))
return pkgs
def _parse_file(self, filename, content):
fname = os.path.basename(filename)
if fname in ("requirements.txt", "Pipfile", "setup.cfg"):
return self._parse_requirements(content)
if fname == "package.json":
return self._parse_package_json(content)
if fname in ("composer.json", "composer.lock"):
return self._parse_composer(content)
if fname == "Gemfile.lock":
return self._parse_gemfile_lock(content)
if fname in ("go.mod", "go.sum"):
return self._parse_go_mod(content)
return []
# ── OSV Query ────────────────────────────────────────────
def _query_osv_batch(self, queries):
"""Query OSV.dev batch API. queries = list of {package, version, ecosystem}"""
if not queries:
return []
try:
body = json.dumps({"queries": [
{"version": q["version"],
"package": {"name": q["package"], "ecosystem": q["ecosystem"]}}
for q in queries
]}).encode("utf-8")
req = urllib.request.Request(
self.BATCH_API, data=body,
headers={"Content-Type": "application/json"}, method="POST"
)
with urllib.request.urlopen(req, timeout=20) as resp:
data = json.loads(resp.read().decode())
return data.get("results", [])
except Exception:
return []
def _severity_from_osv(self, vuln):
"""Extract CVSS severity from OSV vuln object."""
for sev in vuln.get("severity", []):
score_str = sev.get("score", "")
m = re.search(r'(\d+\.?\d*)', score_str)
if m:
s = float(m.group(1))
if s >= 9.0: return "CRITICAL", s
if s >= 7.0: return "HIGH", s
if s >= 4.0: return "MEDIUM", s
return "LOW", s
return "HIGH", 7.0 # default if no CVSS
def scan(self, dep_files):
"""
dep_files: list of (filename, content) tuples
Returns: list of finding dicts
"""
# Build query list
queries, meta = [], []
for fname, content in dep_files:
eco = self.ECOSYSTEMS.get(os.path.basename(fname), "PyPI")
pkgs = self._parse_file(fname, content)
for pkg, ver in pkgs[:40]: # max 40 per file
if not ver or "*" in ver:
continue
queries.append({"package": pkg, "version": ver, "ecosystem": eco})
meta.append({"file": fname, "package": pkg, "version": ver})
if not queries:
return []
# Batch query (max 1000 per call, we stay well under)
results = self._query_osv_batch(queries[:200])
findings = []
for i, result in enumerate(results):
vulns = result.get("vulns", [])
if not vulns or i >= len(meta):
continue
m = meta[i]
for v in vulns[:3]: # max 3 CVEs per package
vuln_id = v.get("id", "UNKNOWN")
aliases = v.get("aliases", [])
cve = next((a for a in aliases if a.startswith("CVE-")), vuln_id)
summary = v.get("summary", v.get("details", "No description"))[:200]
sev, score = self._severity_from_osv(v)
# Fixed version
fixed_in = []
for affected in v.get("affected", []):
for rng in affected.get("ranges", []):
for evt in rng.get("events", []):
if "fixed" in evt:
fixed_in.append(evt["fixed"])
fix_str = (f"Upgrade to version {', '.join(fixed_in[:2])}"
if fixed_in else "Upgrade to latest version")
findings.append({
"vuln": "Dependency Vulnerability",
"severity": sev,
"line": 0,
"snippet": f"{m['package']}=={m['version']}",
"desc": f"{cve}: {summary}",
"fix": fix_str,
"confidence": "HIGH",
"source": "OSV",
"file": m["file"],
"note": f"Known vulnerability in {m['package']} {m['version']}. CVE: {cve}",
"cve": cve,
})
return findings
# ══════════════════════════════════════════════════════════
# CORE FILE SCANNER
# ══════════════════════════════════════════════════════════
# JS/TS string + comment stripper
# Replaces string literals and inline // comments with whitespace so that
# regex rules are never triggered by content that lives inside strings.
_JS_EXTS = frozenset({".js", ".ts", ".jsx", ".tsx", ".mjs", ".cjs"})
def _strip_js_strings(line: str) -> str:
"""
Replace string/template literal content and // comments with spaces.
This prevents rules from firing on eval(...) that is merely a string value,
e.g. var xss = 'javascript:eval(...)';
"""
result = []
i, n = 0, len(line)
while i < n:
c = line[i]
# // inline comment β€” blank the rest
if c == '/' and i + 1 < n and line[i + 1] == '/':
result.extend([' '] * (n - i))
break
# string literal: single, double, or template
if c in ('"', "'", '`'):
quote = c
result.append(' ') # replace opening quote
i += 1
while i < n:
ch = line[i]
if ch == '\\' and i + 1 < n:
result.append(' ') # escape sequence β€” blank both chars
result.append(' ')
i += 2
continue
if ch == quote:
result.append(' ') # replace closing quote
i += 1
break
result.append(' ') # replace string content
i += 1
continue
result.append(c)
i += 1
return ''.join(result)
def _is_comment(line, ext):
s = line.strip()
if ext == ".py" and s.startswith("#"): return True
if ext in (".js",".ts",".java",".cs",".go") and s.startswith("//"): return True
if ext == ".php" and (s.startswith("//") or s.startswith("#")): return True
return False
# ── Context-aware mitigation detection ──────────────────────────────────────
# When a sanitization/protection function is found near a vulnerability finding,
# confidence is downgraded to LOW (still reported for manual review, not silenced).
CONTEXT_WIN = 8 # lines to check before and after the finding
MITIGATIONS = {
"Command Injection": [
r"escapeshellarg\s*\(",
r"escapeshellcmd\s*\(",
r"is_numeric\s*\(",
r"preg_match\s*\(['\"][^'\"]*\^.*\$['\"]", # strict regex validation
],
"SQL Injection": [
r"->prepare\s*\(",
r"PDO\s*::",
r"bindParam\s*\(",
r"bindValue\s*\(",
r"\$wpdb\s*->\s*prepare\s*\(",
r"pg_query_params\s*\(",
],
"XSS": [
r"htmlspecialchars\s*\(",
r"htmlentities\s*\(",
r"strip_tags\s*\(",
r"ENT_QUOTES",
r"sanitize_text_field\s*\(",
r"esc_html\s*\(",
],
"File Inclusion": [
r"in_array\s*\(",
r"array_search\s*\(",
r"\[\s*['\"].*['\"]\s*,\s*['\"]", # allowlist array
],
"Path Traversal": [
r"realpath\s*\(",
r"basename\s*\(",
r"str_replace\s*\([^,]*\.\./",
],
"Open Redirect": [
r"filter_var\s*\(",
r"parse_url\s*\(",
r"in_array\s*\(",
],
"Insecure Deserialization": [
r"json_decode\s*\(",
r"json_loads\s*\(",
],
}
def _context_confidence(lines, lineno, vuln_type):
"""
Check if a known mitigation/sanitization pattern exists within CONTEXT_WIN
lines of the finding. If so, return 'LOW' β€” the code may already be protected.
Returns None if no mitigation detected (confidence stays as defined in the rule).
"""
pats = MITIGATIONS.get(vuln_type, [])
if not pats:
return None
start = max(0, lineno - CONTEXT_WIN - 1)
end = min(len(lines), lineno + CONTEXT_WIN)
ctx = "\n".join(lines[start:end])
if any(re.search(p, ctx, re.IGNORECASE) for p in pats):
return "LOW"
return None
_SQL_KEYWORDS = re.compile(r'\b(SELECT|INSERT|UPDATE|DELETE|FROM|WHERE|JOIN)\b', re.IGNORECASE)
_TEMPLATE_VAR = re.compile(r'[$][{][^}]+[}]') # JS/TS template literals: ${var}
_FSTRING_VAR = re.compile(r'[{][a-zA-Z_]\w*[^}]*[}]') # Python f-strings: {var}
# ══════════════════════════════════════════════════════════
# PERMANENT FALSE-POSITIVE SUPPRESSION LAYER
# Applied per-line before any finding is emitted.
# Each entry: (vuln_type_or_None, compiled_regex, reason)
# vuln_type=None β†’ applies to ALL vulnerability types
# ══════════════════════════════════════════════════════════
_FP_SUPPRESSIONS = [
# ── SQL: dynamic ? placeholder builder ─────────────────
# emails.map(() => '?') / ids.map((_) => '?')
# Generates "?,?,?" bound values β€” never injects user data
(None,
re.compile(r'\$\{[^}]+\.map\s*\(\s*[^)]*\)\s*\.?(join\s*\(|map\s*\()?[^}]*\?[^}]*\}', re.IGNORECASE),
"Dynamic ? placeholder builder β€” values are bound, not interpolated"),
# ── SQL: ternary returning SQL keyword literals only ────
# ${isUnique ? 'UNIQUE ' : ''} / ${desc ? 'DESC' : 'ASC'}
(None,
re.compile(r'\$\{\s*\w+\s*\?\s*[\'"][A-Z_\s]*[\'"]\s*:\s*[\'"][A-Z_\s]*[\'"]\s*\}', re.IGNORECASE),
"Ternary returning fixed SQL keyword literals β€” not user-controlled"),
# ── SSRF: file:// scheme used for module/script loading ─
# file://${entrypointPath} β€” local filesystem, not HTTP
("SSRF",
re.compile(r'file://\$\{', re.IGNORECASE),
"file:// URL scheme for local module loading β€” not an HTTP SSRF vector"),
# ── File Upload: type-only declarations, not handlers ───
# import type { BusboyFileStream } / : BusboyFileStream
("Insecure File Upload",
re.compile(r'(?:import\s+type\b|:\s*(?:BusboyFileStream|Readable|Stream|FileStream)\b)', re.IGNORECASE),
"Type-only reference β€” no actual upload handler present on this line"),
# ── SQL: CREATE INDEX with boolean flag only ────────────
# CREATE ${flag ? 'UNIQUE ' : ''}INDEX β€” flag is internal boolean
(None,
re.compile(r'CREATE\s+\$\{\s*\w+\s*\?\s*[\'"]UNIQUE\s*[\'"]\s*:\s*[\'"][\'"]', re.IGNORECASE),
"CREATE INDEX with internal boolean flag β€” not user-controlled"),
# ── Open Redirect: res.redirect() with a SAFE internal URL ─
# res.redirect(303, `./callback?${...}`) β€” relative URL, safe
("Open Redirect",
re.compile(r'res\.redirect\s*\(\s*\d+\s*,\s*[\'"`]\./|res\.redirect\s*\(\s*[\'"`]\./', re.IGNORECASE),
"Redirect target is a relative path β€” not user-controlled"),
# ── Test assertion methods β€” value is checked, not used ─────────────────
# assertOutput / assertEqual / assertIn etc. contain strings being *verified*,
# not secrets or SQL being *executed*. Applies to all vuln types.
(None,
re.compile(r'\b(assertOutput|assertContains|assertIn|assertNotIn|assertEqual|'
r'assertRaises|assertLogs|assertWarns|assertRegex|assertIn\b)\s*\(', re.IGNORECASE),
"Value is inside a test assertion method β€” not executed, only verified"),
# ── #nosec / nolint annotation β€” developer explicitly suppressed ──────
# GoSec #nosec G101, golangci nolint, NOSONAR β€” trusted developer decision
(None,
re.compile(r'(?:#nosec|//\s*nosec|//\s*nolint|NOSONAR|//\s*#nosec)', re.IGNORECASE),
"Developer #nosec / nolint annotation β€” explicitly marked as safe"),
# ── Debug Enabled: Go structured logger β€” keyword in message string ───
# logger.Debug("ID token expired..") β€” "token" is in the log message, not a leaked value
# s.logger.Debug("Failed to unmarshal secret value..") β€” "secret" describes the topic
("Debug Enabled",
re.compile(r'\.\s*(?:Debug|Info|Warn|Error|Log)\s*\(\s*["\'][^"\']*'
r'(?:token|secret|password|key|auth)[^"\']*["\']', re.IGNORECASE),
"Sensitive keyword is inside log message string β€” describing the topic, not leaking a value"),
# ── Hardcoded Secret: value is a dotted config path β€” NOT a secret ────
# ClientAPIKey = "auth.client.api-key"
# forceUseGraphAPIKey = "force_use_graph_api"
# Config paths contain dots/hyphens and are all-lowercase identifiers
("Hardcoded Secret",
re.compile(r'=\s*["\'][a-z][a-z0-9]*(?:[.\-_][a-z0-9]+){2,}["\']'),
"Value is a dotted config path identifier β€” not an actual credential"),
# ── Hardcoded Secret: variable naming a config field (Key/Name suffix) ──
# const tokenKey = "auth_token" β€” names a config field, not a token
("Hardcoded Secret",
re.compile(r'(?:Key|Name|Field|Column|Label|Type|Kind|Mode|Setting)\s*'
r'(?:=|:=)\s*["\'][a-z_][a-z0-9_.\-]+["\']'),
"Variable names a config field/column β€” not an actual credential value"),
]
def _is_suppressed(vuln: str, line: str) -> tuple:
for fp_vuln, fp_re, reason in _FP_SUPPRESSIONS:
if fp_vuln and fp_vuln != vuln:
continue
if fp_re.search(line):
return True, reason
return False, ""
# ══════════════════════════════════════════════════════════
# SMART TAINT ENGINE
# Per-file data-flow analysis:
# Phase 1 β€” build {variable: "user"} map from HTTP sources
# Phase 2 β€” at each rule match, check if variables reaching
# the dangerous sink are user-controlled
#
# Default: if no user-input source found in file β†’ LOW confidence
# Only CRITICAL/HIGH when we can prove user data reaches the sink.
# ══════════════════════════════════════════════════════════
class SmartTaintEngine:
# Vulnerability types that REQUIRE user input proof to be HIGH/CRITICAL
INJECTION_VULNS = frozenset({
"SQL Injection", "Blind SQL Injection", "NoSQL Injection",
"Command Injection", "Code Injection",
"XSS", "Stored XSS", "DOM XSS",
"SSRF", "Path Traversal", "File Inclusion",
"LDAP Injection", "Template Injection",
})
# ── HTTP source patterns per language ────────────────────────────────
_SRC = [
# Go: net/http, Echo, Gin, Chi, Gorilla/mux
(frozenset({".go"}), re.compile(
r'(\w+)\s*:?=\s*'
r'(?:r\.(?:URL\.Query\(\)\.Get|FormValue|PostFormValue|'
r'Header\.Get|PathValue|URL\.Path|Cookie)\s*\(|'
r'c\.(?:QueryParam|FormValue|Param|GetHeader|QueryString|'
r'PathParam|Bind)\s*\(|'
r'ctx\.(?:QueryParam|FormValue|Param|Value)\s*\(|'
r'chi\.URLParam\s*\(|'
r'mux\.Vars\s*\([^)]+\)\s*\[|'
r'vars\s*\[)',
re.IGNORECASE,
)),
# JS / TS
(frozenset({".js", ".ts", ".jsx", ".tsx", ".mjs", ".cjs"}), re.compile(
r'(?:'
r'(?:const|let|var)\s+(\w+)\s*=\s*(?:req|request|ctx)\.'
r'(?:body|query|params|headers?|cookies?)|'
r'const\s*\{\s*(\w+)[^}]*\}\s*=\s*(?:req|request|ctx)\.'
r'(?:body|query|params)|'
r'(\w+)\s*=\s*(?:req|request|ctx)\.'
r'(?:body|query|params|headers?|cookies?)'
r')',
re.IGNORECASE,
)),
# Python: Flask, Django, FastAPI, aiohttp
(frozenset({".py"}), re.compile(
r'(\w+)\s*=\s*'
r'(?:request\.(?:args|form|values|json|data|cookies|headers|files)'
r'(?:\s*[\.\[]|\s*\.get\s*\()|'
r'flask\.request\.|'
r'await\s+request\.(?:json|body|form)\s*\(\)|'
r'input\s*\()',
re.IGNORECASE,
)),
# PHP
(frozenset({".php"}), re.compile(
r'\$(\w+)\s*=\s*\$_(?:GET|POST|REQUEST|COOKIE|FILES)\s*\[',
re.IGNORECASE,
)),
# Java / Kotlin
(frozenset({".java", ".kt"}), re.compile(
r'(?:String\s+)?(\w+)\s*=\s*'
r'(?:request\.(?:getParameter|getAttribute|getHeader|'
r'getQueryString)\s*\(|'
r'@(?:RequestParam|PathVariable|RequestBody)\b)',
re.IGNORECASE,
)),
# Ruby
(frozenset({".rb"}), re.compile(
r'(\w+)\s*=\s*params\s*[\[:]',
re.IGNORECASE,
)),
# C# / ASP.NET
(frozenset({".cs"}), re.compile(
r'(\w+)\s*=\s*'
r'(?:Request\.(?:Query|Form|Params|Headers|Cookies)\s*[\[\."]|'
r'\[From(?:Query|Body|Route|Header)\])',
re.IGNORECASE,
)),
]
# Simple assignment propagation: dst = src
_ASSIGN_RE = re.compile(
r'(?:(?:const|let|var|string|String|int|bool)\s+)?'
r'(\w{2,})\s*:?=\s*'
r'(?:[^\n=]{0,40}\b(\w{2,})\b)',
)
# Identifiers to always ignore (keywords, builtins, SQL verbs)
_IGNORE = frozenset({
"nil","null","true","false","True","False","None",
"err","error","ok","e","n","i","j","k","s","v","t","f","b","c","p","r",
"SELECT","INSERT","UPDATE","DELETE","FROM","WHERE","JOIN","AND","OR",
"NOT","IN","AS","ON","SET","VALUES","LIMIT","OFFSET","ORDER","BY",
"GROUP","HAVING","INNER","OUTER","LEFT","RIGHT","CROSS","INTO",
"fmt","sql","db","tx","log","http","time","strings","strconv",
"append","len","cap","make","new","var","const","func","return",
"if","else","for","range","switch","case","default","break","continue",
"self","cls","this","super","import","from","raise","pass","with","as",
"def","class","lambda","yield","async","await","try","except","finally",
"String","Integer","Boolean","Object","List","Map","Array","Dict",
"req","res","request","response","ctx","context","next","app","router",
"model","schema","db","conn","cursor","session","query","result",
})
def build_taint_map(self, content: str, ext: str) -> dict:
"""Returns {var_name: 'user'} for variables traceable to HTTP user input."""
taint: dict[str, str] = {}
lines = content.splitlines()
src_re = None
for exts, pattern in self._SRC:
if ext in exts:
src_re = pattern
break
if src_re is None:
return taint
# Pass 1 β€” direct sources
for line in lines:
m = src_re.search(line)
if m:
var = next((g for g in m.groups() if g), None)
if var and var not in self._IGNORE:
taint[var] = "user"
# Pass 2 — propagation (two rounds to catch a→b→c chains)
for _ in range(2):
for line in lines:
for m in self._ASSIGN_RE.finditer(line):
dst, src = m.group(1), m.group(2)
if (src and dst
and taint.get(src) == "user"
and dst not in taint
and dst not in self._IGNORE):
taint[dst] = "user"
return taint
def validate(self, line: str, vuln: str, taint: dict,
original_conf: str) -> tuple:
"""
Returns (confidence, note).
Rules:
1. Non-injection vuln β†’ unchanged (Hardcoded Secret, CSRF, etc.)
2. Language not tracked (empty taint, no sources) β†’ LOW
3. User-tainted var found in line β†’ HIGH (confirmed)
4. Variables present but none user-tainted β†’ LOW (likely internal)
"""
if vuln not in self.INJECTION_VULNS:
return original_conf, ""
# Extract candidate identifiers from the matched line
tokens = {
tok for tok in re.findall(r'\b([a-zA-Z_]\w*)\b', line)
if tok not in self._IGNORE
and len(tok) >= 2
and not tok.isupper() # skip SQL KEYWORDS
and not tok[0].isupper() # skip ClassName references
}
# No user sources in file at all β†’ file likely doesn't handle HTTP
if not taint:
return "LOW", (
"No HTTP user-input sources found in this file β€” "
"variable is likely internal/framework-controlled."
)
user_vars = tokens & taint.keys()
if user_vars:
return "HIGH", (
f"User-controlled variable(s) `{'`, `'.join(sorted(user_vars))}` "
"confirmed reaching this sink."
)
return "LOW", (
"No user-controlled variables detected at this sink β€” "
"data appears internal or framework-generated."
)
_smart_taint = SmartTaintEngine()
def scan_file_content(content, filename):
findings = []
ext = os.path.splitext(filename)[1].lower()
lines = content.splitlines()
test_ctx = _is_test_context(filename)
is_js = ext in _JS_EXTS
# ── SmartTaintEngine: build per-file user-input variable map ──────────
file_taint = _smart_taint.build_taint_map(content, ext)
for lineno, line in enumerate(lines, 1):
if _is_comment(line, ext):
continue
# JS/TS: strip string literals so regex never fires on string content
scan_line = _strip_js_strings(line) if is_js else line
for vuln, sev, pattern, desc, fix, conf in RULES:
try:
if not re.search(pattern, scan_line):
continue
# ── Permanent FP suppression ─────────────────────────────
suppressed, _ = _is_suppressed(vuln, line)
if suppressed:
continue
actual_vuln = vuln
actual_sev = sev
# ── Reclassify SQL-shaped Hardcoded Secret ───────────────
if vuln == "Hardcoded Secret" and _SQL_KEYWORDS.search(line):
if _TEMPLATE_VAR.search(line) or _FSTRING_VAR.search(line):
actual_vuln = "SQL Injection"
actual_sev = "CRITICAL"
conf = "HIGH"
# ── Test-context: secrets β†’ INFO ─────────────────────────
test_note = ""
_is_sample = re.search(
r'(?i)(sample|fixture|seed|fake|mock|dummy|stub|sampledata)',
filename
)
if (test_ctx or _is_sample) and actual_vuln == "Hardcoded Secret":
actual_sev = "INFO"
conf = "LOW"
test_note = "Test/sample context β€” credential is likely mock/fixture data."
# ── Nearby mitigation check ──────────────────────────────
ctx_conf = _context_confidence(lines, lineno, actual_vuln)
final_conf = ctx_conf if ctx_conf else conf
# ── SmartTaintEngine: data-flow validation ───────────────
smart_conf, smart_note = _smart_taint.validate(
line, actual_vuln, file_taint, final_conf
)
final_conf = smart_conf
# ── Confidence gate: suppress unconfirmed injection ──────
# SmartTaintEngine couldn't confirm user input at the sink β†’
# demote to INFO so it never inflates the security grade.
# This is the permanent architectural fix: only confirmed
# injection findings appear as actionable.
if final_conf == "LOW" and actual_vuln in SmartTaintEngine.INJECTION_VULNS:
actual_sev = "INFO"
note = test_note or smart_note or (
"Possible mitigation detected nearby β€” manual review recommended."
if ctx_conf == "LOW" else ""
)
findings.append({
"vuln": actual_vuln, "severity": actual_sev, "line": lineno,
"snippet": line.strip()[:150], "desc": desc,
"fix": fix, "confidence": final_conf,
"note": note, "source": "Regex", "file": filename,
})
break
except re.error:
pass
# Deep Python AST scan
if ext == ".py":
_ast_test = _is_test_context(filename)
_ast_sample = bool(re.search(
r'(?i)(sample|fixture|seed|fake|mock|dummy|stub|sampledata)',
filename
))
for f in PythonASTScanner().scan(content):
f["file"] = filename
f.setdefault("note", "")
# ── Apply test/sample context to AST findings ──────────
if _ast_test or _ast_sample:
if f.get("confidence") == "LOW":
f["severity"] = "INFO"
f["note"] = "In test/sample context with low confidence β€” informational only"
elif f["severity"] == "CRITICAL":
f["severity"] = "MEDIUM"
f["note"] = "Downgraded β€” found in test/sample context"
elif f["severity"] == "HIGH":
f["severity"] = "LOW"
f["note"] = "Downgraded β€” found in test/sample context"
findings.append(f)
# Multi-line taint analysis β€” per language
_ml_test = _is_test_context(filename)
def _apply_test_gate(f):
"""Downgrade taint-tracker findings in test context."""
if _ml_test:
if f["severity"] == "CRITICAL":
f["severity"] = "LOW"
f["note"] = f.get("note","") + " [test context β€” downgraded]"
elif f["severity"] == "HIGH":
f["severity"] = "INFO"
f["note"] = f.get("note","") + " [test context β€” informational]"
return f
if ext in (".php", ".py"):
for f in TaintTracker().scan(content, filename):
findings.append(_apply_test_gate(f))
if ext == ".go":
for f in GoTaintTracker().scan(content, filename):
findings.append(_apply_test_gate(f))
if ext in (".java", ".kt"):
for f in JavaTaintTracker().scan(content, filename):
findings.append(_apply_test_gate(f))
# Deduplicate: same file + line + vuln type
seen, unique = set(), []
for f in findings:
key = (f.get("file",""), f.get("line",0), f.get("vuln",""))
if key not in seen:
seen.add(key)
unique.append(f)
return unique
# ══════════════════════════════════════════════════════════
# CONTEXT-AWARE CONFIDENCE ENGINE (built-in, no API needed)
# Adjusts confidence based on taint source proximity:
# - HIGH β†’ variable looks like real user input
# - MEDIUM β†’ unclear / internal-looking
# - LOW β†’ binding array / internal function / test context
# ══════════════════════════════════════════════════════════
# Patterns that strongly suggest user-controlled input
_USER_INPUT_RE = re.compile(
r'\b(req\.(body|query|params|headers?|cookies?|files?)|'
r'request\.(body|query|params|data|form|json|args)|'
r'params\[|query\[|body\.|'
r'\$_(GET|POST|REQUEST|COOKIE|FILES)|'
r'input\(|sys\.argv|os\.environ|getenv\(|'
r'flask\.request|django\.request|'
r'ctx\.params|event\.|payload\.)\b',
re.IGNORECASE,
)
# Patterns that suggest internal / framework-generated values
_INTERNAL_RE = re.compile(
r'\b(schema\.|metadata\.|model\.|builder\.|'
r'fieldLowerFn|caseFragments|scalarAttributes|'
r'invJoinColumn|joinColumn|columnName|tableName|'
r'table\.name|column\.name|attr\.name)\b',
re.IGNORECASE,
)
# Pattern: variable appears only inside a bindings array, not the SQL string
# e.g. .raw(`SQL ?`, [col, `${value}`]) ← value is bound safely
_BINDING_ARRAY_RE = re.compile(
r'\.(raw|whereRaw|havingRaw)\s*\(`[^`]*`\s*,\s*\[[^\]]*\$\{',
re.IGNORECASE,
)
# _taint_confidence removed β€” replaced by SmartTaintEngine.validate()
# ══════════════════════════════════════════════════════════
# GITHUB API
# ══════════════════════════════════════════════════════════
# ══════════════════════════════════════════════════════════
# ZEROCYBER-SLM AI ENGINE (Mistral-7B Β· HF Inference API)
# ══════════════════════════════════════════════════════════
HF_MODEL_ID = "mistralai/Mistral-7B-Instruct-v0.3"
HF_MODEL_URL = (
"https://router.huggingface.co/hf-inference/models/"
+ HF_MODEL_ID
+ "/v1/chat/completions"
)
def _hf_chat(messages, hf_token, max_tokens=400, temperature=0.1):
"""
Shared HuggingFace chat-completions call (OpenAI-compatible format).
Returns the assistant reply text, or '' on failure.
"""
try:
body = json.dumps({
"model": HF_MODEL_ID,
"messages": messages,
"max_tokens": max_tokens,
"temperature": temperature,
}).encode("utf-8")
req = urllib.request.Request(
HF_MODEL_URL, data=body,
headers={
"Authorization": f"Bearer {hf_token.strip()}",
"Content-Type": "application/json",
"User-Agent": "ZeroCyber-SLM",
}
)
with urllib.request.urlopen(req, timeout=90) as r:
result = json.loads(r.read().decode())
return result["choices"][0]["message"]["content"]
except Exception:
return ""
def _hf_call(prompt, hf_token, max_tokens=400, temperature=0.1):
"""Backward-compat wrapper β€” delegates to _hf_chat."""
if not hf_token or not hf_token.strip():
return ""
return _hf_chat(
[{"role": "user", "content": prompt}],
hf_token, max_tokens=max_tokens, temperature=temperature,
)
def zerocyber_ai_analyze(findings, repo_name, hf_token):
"""AI-powered deep analysis of top findings using ZeroCyber-SLM AI Engine."""
if not hf_token or not hf_token.strip():
return ""
top = [f for f in findings if f.get("severity") in ("CRITICAL", "HIGH")][:5]
if not top:
top = findings[:3]
if not top:
return ""
lines = ""
for i, f in enumerate(top, 1):
lines += (f"{i}. [{f.get('severity')}] {f.get('vuln')} β€” "
f"{f.get('file')} line {f.get('line')}\n"
f" Code: {f.get('snippet','')[:200]}\n\n")
messages = [
{"role": "system",
"content": "You are ZeroCyber-SLM, an expert application security engineer."},
{"role": "user",
"content": (
f"Analyze these security vulnerabilities found in the repository '{repo_name}' "
f"and provide a concise professional security assessment.\n\n"
f"Findings:\n{lines}\n"
f"For each finding provide:\n"
f"1. Attack scenario β€” how an attacker would exploit this in the real world\n"
f"2. Business impact if exploited\n"
f"3. Specific remediation with a corrected code example\n\n"
f"Format your response as markdown. Be technical, precise, and concise."
)},
]
try:
body = json.dumps({
"model": HF_MODEL_ID,
"messages": messages,
"max_tokens": 900,
"temperature": 0.2,
}).encode("utf-8")
req = urllib.request.Request(
HF_MODEL_URL, data=body,
headers={
"Authorization": f"Bearer {hf_token.strip()}",
"Content-Type": "application/json",
"User-Agent": "ZeroCyber-SLM",
}
)
with urllib.request.urlopen(req, timeout=90) as r:
result = json.loads(r.read().decode())
return result["choices"][0]["message"]["content"]
except urllib.error.HTTPError as e:
err_body = e.read().decode()
if "loading" in err_body.lower():
return "*ZeroCyber-SLM AI Engine is loading β€” please retry in 30 seconds.*"
return f"*AI Engine unavailable (HTTP {e.code}): {err_body[:200]}*"
except Exception as e:
return f"*AI Engine error: {e}*"
def zerocyber_ai_verify(findings, repo_name, hf_token):
"""
Adversarial Verifier β€” inspired by 07_verification_agent pattern.
For each CRITICAL/HIGH finding, returns verdict:
CONFIRMED | LIKELY | FALSE_POSITIVE
Returns dict keyed by "file:line".
"""
if not hf_token or not hf_token.strip():
return {}
candidates = [f for f in findings if f.get("severity") in ("CRITICAL", "HIGH")][:8]
if not candidates:
return {}
lines = ""
for i, f in enumerate(candidates, 1):
lines += (
f"FINDING_{i}: [{f.get('severity')}] {f.get('vuln')} "
f"in {f.get('file')} line {f.get('line')}\n"
f" Code: {f.get('snippet','')[:160]}\n\n"
)
prompt = (
f"<s>[INST] You are ZeroCyber-SLM Adversarial Verifier. "
f"Your job is to challenge each finding in '{repo_name}' β€” try to DISPROVE it.\n\n"
f"Verdicts:\n"
f"- CONFIRMED: user-controlled input reaches dangerous sink, no sanitization visible\n"
f"- LIKELY: appears vulnerable but context is incomplete\n"
f"- FALSE_POSITIVE: constant/enum/test data, or sanitized before use\n\n"
f"Findings:\n{lines}"
f"Reply ONLY in this exact format (one line per finding):\n"
f"FINDING_1: CONFIRMED - reason max 12 words\n"
f"FINDING_2: LIKELY - reason max 12 words\n"
f"FINDING_3: FALSE_POSITIVE - reason max 12 words\n"
f"[/INST]"
)
text = _hf_call(prompt, hf_token, max_tokens=350, temperature=0.1)
verdicts = {}
pattern = re.compile(
r'FINDING_(\d+):\s*(CONFIRMED|LIKELY|FALSE_POSITIVE)\s*[-–]\s*(.+)',
re.IGNORECASE
)
for m in pattern.finditer(text):
idx = int(m.group(1)) - 1
if 0 <= idx < len(candidates):
f = candidates[idx]
key = f"{f.get('file')}:{f.get('line')}"
verdicts[key] = {
"verdict": m.group(2).upper(),
"reason": m.group(3).strip()
}
return verdicts
def zerocyber_ai_executive_summary(findings, repo_name, hf_token):
"""
Executive Narrative β€” inspired by 22_away_summary pattern.
2-3 sentences: business risk, top threats, urgency.
"""
if not hf_token or not hf_token.strip() or not findings:
return ""
counts = {}
for f in findings:
counts[f.get("severity", "INFO")] = counts.get(f.get("severity", "INFO"), 0) + 1
top_types = list(dict.fromkeys(
f.get("vuln", "") for f in findings
if f.get("severity") in ("CRITICAL", "HIGH")
))[:4]
prompt = (
f"<s>[INST] You are ZeroCyber-SLM. Write a 2-3 sentence executive security narrative "
f"for the repository '{repo_name}'.\n"
f"Statistics: {counts.get('CRITICAL',0)} Critical Β· {counts.get('HIGH',0)} High Β· "
f"{counts.get('MEDIUM',0)} Medium Β· {counts.get('LOW',0)} Low\n"
f"Top risks: {', '.join(top_types) or 'various'}\n\n"
f"Rules: focus on business risk and urgency, be direct, no bullet points, "
f"do NOT start with 'The repository', professional tone.\n"
f"Output ONLY the 2-3 sentences. [/INST]"
)
return _hf_call(prompt, hf_token, max_tokens=120, temperature=0.3).strip()
GITHUB_API = "https://api.github.com"
_HDRS = {"Accept":"application/vnd.github+json","User-Agent":"ZeroCyber-SLM"}
def github_req(url, token=None):
try:
hdrs = dict(_HDRS)
if token: hdrs["Authorization"] = f"token {token}"
req = urllib.request.Request(url, headers=hdrs)
with urllib.request.urlopen(req, timeout=20) as r:
return json.loads(r.read().decode()), None
except urllib.error.HTTPError as e:
msgs = {403:"GitHub rate limit β€” add a token for 5 000 req/hr.",
404:"Repo not found or is private.",401:"Invalid GitHub token."}
return None, msgs.get(e.code, f"HTTP {e.code}: {e.reason}")
except Exception as e:
return None, str(e)
def parse_repo_url(url):
url = url.strip().rstrip("/")
if re.match(r"^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$", url):
p = url.split("/"); return p[0], p[1], None
m = re.search(r"github\.com/([A-Za-z0-9_.-]+)/([A-Za-z0-9_.-]+)", url)
if m: return m.group(1), m.group(2), None
return None, None, "Invalid URL. Use: owner/repo or https://github.com/owner/repo"
# Matches: 2faSpec.ts *.spec.ts *.test.js *.test.api.ts (any multi-extension pattern)
_UNIT_TEST_FILE_RE = re.compile(
r'(spec|test)\.[a-z]' # *Spec.ts *Test.js *spec.ts
r'|\.(spec|test)\.', # *.spec.ts *.test.api.ts (embedded between dots)
re.IGNORECASE
)
def _should_skip(path):
p = path.lower()
parts = p.split("/")
# Skip by blacklisted directory name
if any(seg in SKIP_DIRS for seg in parts):
return True
# Skip pure test-infrastructure directories
if any(seg in SKIP_TEST_INFRA_DIRS for seg in parts):
return True
# Skip unit-test / spec files β€” they contain mock credentials (FP factory)
# Handles: *.spec.ts *.test.js *Spec.ts *Test.ts (case-insensitive)
fname = parts[-1] if parts else ""
if _UNIT_TEST_FILE_RE.search(fname):
return True
return False
def _sort_key(item):
path = item.get("path","").lower()
ext = os.path.splitext(path)[1]
ext_score = 0 if ext in SCANNABLE_EXT else 2
path_score = 0 if any(k in path for k in PRIORITY_PATH_KEYWORDS) else 1
# Negative size = larger files first (security logic lives in larger files,
# not in 30-byte re-export stubs)
return (ext_score, path_score, -item.get("size", 0))
# ══════════════════════════════════════════════════════════
# REPORT GENERATOR
# ══════════════════════════════════════════════════════════
def _top_sev(findings):
for s in SEV_ORDER:
if any(f.get("severity")==s for f in findings): return s
return "CLEAN"
def generate_report(all_findings, repo_name, files_scanned, total_files,
file_results=None, repo_info=None, low_coverage=False,
ai_analysis="", ai_verify_results=None, ai_exec_summary=""):
ts = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
scan_id = hashlib.sha256(f"{repo_name}{ts}".encode()).hexdigest()[:14].upper()
counts = {}
for f in all_findings:
counts[f["severity"]] = counts.get(f["severity"],0) + 1
total_issues = sum(1 for f in all_findings if f.get("severity") != "INFO")
risk = min(10.0,
counts.get("CRITICAL",0)*2.5 + counts.get("HIGH",0)*1.5 +
counts.get("MEDIUM",0)*0.8 + counts.get("LOW",0)*0.3)
overall = ("CRITICAL" if counts.get("CRITICAL",0) else
"HIGH" if counts.get("HIGH",0) else
"MEDIUM" if counts.get("MEDIUM",0) else
"LOW" if counts.get("LOW",0) else "CLEAN")
# Security Grade (A-F)
if risk == 0: grade, grade_note = "A+", "No issues detected"
elif risk < 1.5: grade, grade_note = "A", "Minor informational findings"
elif risk < 3.0: grade, grade_note = "B", "Low-severity issues present"
elif risk < 5.0: grade, grade_note = "C", "Moderate security concerns"
elif risk < 7.0: grade, grade_note = "D", "Significant vulnerabilities found"
else: grade, grade_note = "F", "Critical issues requiring immediate action"
# ── Header ──────────────────────────────────────────
ai_vr = ai_verify_results or {}
r = f"# ZeroCyber-SLM β€” Security Assessment Report\n\n"
# ── AI Executive Narrative (away-summary pattern) ────
if ai_exec_summary:
r += f"> **ZeroCyber-SLM AI:** {ai_exec_summary}\n\n"
r += f"| | |\n|---|---|\n"
r += f"| **Target** | `{repo_name}` |\n"
r += f"| **Scan ID** | `{scan_id}` |\n"
r += f"| **Date** | {ts} |\n"
r += f"| **Files Scanned** | {files_scanned} of {total_files} total |\n"
r += f"| **Engine** | ZeroCyber-SLM Β· {len(RULES)} rules Β· Python AST + TaintTrack + OSV + ZeroCyber-SLM AI Engine |\n"
r += f"| **Standards** | OWASP Top 10 (2021) Β· CWE Β· CVSS v3.1 Β· SANS Top 25 |\n\n"
if not all_findings:
if low_coverage:
r += "## ⚠️ Inconclusive β€” Low Scan Coverage\n\n"
r += "No vulnerabilities found in the scanned files, **but coverage was too low for a reliable conclusion.**\n\n"
r += "**Action Required:** Add a GitHub Token to scan more files and get a trustworthy result.\n\n"
else:
r += "## Security Grade: A+ β€” No Vulnerabilities Detected\n\n"
r += "All scanned files passed security analysis. No known vulnerability patterns found.\n\n"
r += "> Continue monitoring with every code change. Set up automated SAST in CI/CD.\n\n"
r += f"---\n*ZeroCyber-SLM | {len(RULES)} rules | OWASP Top 10 | CWE | CVSS v3.1*"
return r
# ── Executive Summary ────────────────────────────────
r += "---\n\n## Executive Summary\n\n"
r += f"| Metric | Value |\n|---|---|\n"
r += f"| **Security Grade** | **{grade}** β€” {grade_note} |\n"
r += f"| **Overall Risk Rating** | **{overall}** |\n"
r += f"| **CVSS-like Risk Score** | **{round(risk,1)} / 10.0** |\n"
info_count = counts.get('INFO', 0)
r += f"| **Actionable Findings** | **{total_issues}** |\n"
r += f"| Critical | {counts.get('CRITICAL',0)} |\n"
r += f"| High | {counts.get('HIGH',0)} |\n"
r += f"| Medium | {counts.get('MEDIUM',0)} |\n"
r += f"| Low | {counts.get('LOW',0)} |\n"
if info_count:
r += f"| Info (unconfirmed) | {info_count} |\n"
r += "\n"
# Risk bar
bar_total = 30
if total_issues > 0:
crit_w = int(bar_total * counts.get("CRITICAL",0) / total_issues)
high_w = int(bar_total * counts.get("HIGH",0) / total_issues)
med_w = bar_total - crit_w - high_w
r += f"**Risk Distribution:** "
r += f"`{'C'*crit_w}{'H'*high_w}{'M'*med_w}` "
r += f"(C={counts.get('CRITICAL',0)} H={counts.get('HIGH',0)} M={counts.get('MEDIUM',0)})\n\n"
r += "---\n\n"
# ── OWASP Breakdown ──────────────────────────────────
owasp_counts = {}
for f in all_findings:
v = f.get("vuln","")
if v in OWASP:
cat = OWASP[v][0]
owasp_counts[cat] = owasp_counts.get(cat,0)+1
if owasp_counts:
r += "## OWASP Top 10 (2021) Impact\n\n"
r += "| OWASP Category | Findings |\n|---|---|\n"
for cat, cnt in sorted(owasp_counts.items(), key=lambda x:-x[1]):
r += f"| {cat} | {cnt} |\n"
r += "\n---\n\n"
# ── AI Verification Summary (adversarial verifier pattern) ──
if ai_vr:
v_counts = {"CONFIRMED": 0, "LIKELY": 0, "FALSE_POSITIVE": 0}
for v in ai_vr.values():
v_counts[v["verdict"]] = v_counts.get(v["verdict"], 0) + 1
unverified = len([f for f in all_findings
if f.get("severity") in ("CRITICAL","HIGH")
and f"{f.get('file')}:{f.get('line')}" not in ai_vr])
r += "## AI Verification Results\n\n"
r += "> *ZeroCyber-SLM Adversarial Verifier challenged each CRITICAL/HIGH finding*\n\n"
r += "| Verdict | Count | Meaning |\n|---|---|---|\n"
r += f"| βœ… CONFIRMED | {v_counts['CONFIRMED']} | Real vulnerability β€” exploit path verified |\n"
r += f"| ⚠️ LIKELY | {v_counts['LIKELY']} | Probable vulnerability β€” manual review advised |\n"
r += f"| ❌ FALSE POSITIVE | {v_counts['FALSE_POSITIVE']} | Ruled out β€” constant, enum, or sanitized |\n"
r += f"| πŸ” UNVERIFIED | {unverified} | Not yet AI-reviewed |\n"
r += "\n---\n\n"
# ── Files Report ─────────────────────────────────────
if file_results:
vuln_files = [fr for fr in file_results if fr["findings"]]
clean_files = len(file_results) - len(vuln_files)
r += f"## Files Report ({len(vuln_files)} vulnerable Β· {clean_files} clean)\n\n"
r += "| # | File | Risk | Findings | C | H | M | L |\n"
r += "|---|---|---|---|---|---|---|---|\n"
for idx, fr in enumerate(file_results, 1):
fc = {}
for fnd in fr["findings"]: fc[fnd["severity"]] = fc.get(fnd["severity"],0)+1
sev = fr["highest_sev"]
nm = fr["file"]
if len(nm) > 55: nm = "..."+nm[-52:]
total_f = len(fr["findings"])
r += (f"| {idx} | `{nm}` | {sev} | {total_f} | "
f"{fc.get('CRITICAL',0)} | {fc.get('HIGH',0)} | "
f"{fc.get('MEDIUM',0)} | {fc.get('LOW',0)} |\n")
r += "\n---\n\n"
# ── Detailed Findings ────────────────────────────────
VERDICT_BADGE = {
"CONFIRMED": "βœ… CONFIRMED",
"LIKELY": "⚠️ LIKELY",
"FALSE_POSITIVE":"❌ FALSE POSITIVE",
}
def _verdict_order(f):
key = f"{f.get('file')}:{f.get('line')}"
v = ai_vr.get(key, {}).get("verdict", "UNVERIFIED")
return {"CONFIRMED": 0, "LIKELY": 1, "UNVERIFIED": 2, "FALSE_POSITIVE": 3}.get(v, 2)
r += "## Detailed Findings\n\n"
for sev in SEV_ORDER:
sev_findings = [f for f in all_findings if f.get("severity") == sev]
if not sev_findings: continue
# Sort: CONFIRMED first within each severity
sev_findings.sort(key=_verdict_order)
if sev == "INFO":
r += f"### ℹ️ INFO β€” {len(sev_findings)} Unconfirmed / Informational Finding(s)\n\n"
r += "> *These findings were not confirmed by taint analysis. "
r += "They may be false positives and are shown for completeness only.*\n\n"
else:
r += f"### {sev} β€” {len(sev_findings)} Finding(s)\n\n"
for i, f in enumerate(sev_findings, 1):
v = f.get("vuln", "Unknown")
owasp_d = OWASP.get(v, ("N/A","N/A","N/A",0.0))
cvss = owasp_d[3] if len(owasp_d) > 3 else CVSS_BASE.get(sev, 0)
cwe_num = owasp_d[2].replace("CWE-","")
conf = f.get("confidence","MEDIUM")
fkey = f"{f.get('file')}:{f.get('line')}"
vdata = ai_vr.get(fkey, {})
verdict = vdata.get("verdict", "")
badge = VERDICT_BADGE.get(verdict, "πŸ” UNVERIFIED") if verdict else ""
r += f"#### {i}. {v}"
if badge:
r += f" &nbsp; {badge}"
r += "\n\n"
r += f"| Field | Detail |\n|---|---|\n"
r += f"| **Severity** | {sev} |\n"
r += f"| **CVSS Score** | {cvss} |\n"
r += f"| **Confidence** | {conf} |\n"
if badge:
r += f"| **AI Verdict** | {badge} |\n"
if vdata.get("reason"):
r += f"| **Verdict Reason** | {vdata['reason']} |\n"
r += f"| **File** | `{f.get('file','N/A')}` |\n"
r += f"| **Line** | {f.get('line','?')} |\n"
r += f"| **OWASP** | {owasp_d[0]} β€” {owasp_d[1]} |\n"
r += f"| **CWE** | {owasp_d[2]} |\n"
r += f"| **Detection** | {f.get('source','Regex')} |\n\n"
if f.get("note"):
r += f"> **Note:** {f['note']}\n\n"
r += f"**Description:** {f.get('desc','N/A')}\n\n"
r += f"**Vulnerable Code:**\n```\n{f.get('snippet','')}...\n```\n\n"
r += f"**Remediation:** {f.get('fix','N/A')}\n\n"
r += "**References:**\n"
if cwe_num.isdigit():
r += f"- CWE-{cwe_num}: https://cwe.mitre.org/data/definitions/{cwe_num}.html\n"
r += f"- OWASP: https://owasp.org/Top10/\n"
r += "\n---\n\n"
# ── ZeroCyber-SLM AI Analysis ────────────────────────
if ai_analysis and ai_analysis.strip():
r += "## ZeroCyber-SLM AI Analysis\n\n"
r += "> *Powered by ZeroCyber-SLM AI Engine β€” deep vulnerability analysis of top findings*\n\n"
r += ai_analysis.strip() + "\n\n"
r += "---\n\n"
# ── Remediation Roadmap ──────────────────────────────
r += "## Remediation Roadmap\n\n"
if counts.get("CRITICAL",0):
r += "### Immediate Actions (Before Next Deployment)\n\n"
r += "1. **Fix all CRITICAL vulnerabilities** β€” these are exploitable by attackers today\n"
r += "2. **Rotate all exposed credentials** β€” assume any hardcoded secrets are compromised\n"
r += "3. **Disable debug mode** in all production environments\n"
r += "4. **Patch or remove** any components with known CVEs\n\n"
if counts.get("HIGH",0):
r += "### Short-Term (Within 1 Sprint)\n\n"
r += "1. Address all HIGH severity findings\n"
r += "2. Implement input validation and output encoding throughout\n"
r += "3. Adopt parameterized queries for all database interactions\n"
r += "4. Implement Content Security Policy (CSP) headers\n\n"
r += "### Long-Term (Security Program)\n\n"
r += "1. Integrate SAST into CI/CD pipeline (fail builds on CRITICAL/HIGH)\n"
r += "2. Implement dependency scanning (Dependabot, Snyk, pip-audit)\n"
r += "3. Schedule quarterly security code reviews\n"
r += "4. Adopt OWASP SAMM security maturity model\n"
r += "5. Train developers on OWASP Top 10 and secure coding practices\n\n"
r += "---\n\n"
r += "*This report was generated by ZeroCyber-SLM.* \n"
r += f"*{len(RULES)} detection rules | OWASP Top 10 (2021) | CWE | CVSS v3.1 | Python AST | TaintTrack | GoTaint | JavaTaint | CrossFileTaint | SmartTaintEngine | OSV | ZeroCyber-SLM AI Engine*"
return r
def save_report(text, label="report"):
import tempfile
safe = re.sub(r"[^a-zA-Z0-9_-]","_", label)[:40]
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
path = os.path.join(tempfile.gettempdir(), f"ZeroCyber_{safe}_{ts}.md")
with open(path,"w",encoding="utf-8") as fh:
fh.write(text)
return path
def save_report_pdf(text, label="report"):
"""Convert the Markdown report to a clean PDF using fpdf2."""
import tempfile
try:
from fpdf import FPDF
except ImportError:
return None # fpdf2 not installed β€” silently skip
safe = re.sub(r"[^a-zA-Z0-9_-]", "_", label)[:40]
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
path = os.path.join(tempfile.gettempdir(), f"ZeroCyber_{safe}_{ts}.pdf")
pdf = FPDF()
pdf.set_auto_page_break(auto=True, margin=15)
pdf.add_page()
# ── Title ────────────────────────────────────────────
pdf.set_font("Helvetica", "B", 16)
pdf.set_text_color(15, 23, 42) # slate-900
pdf.cell(0, 10, "ZeroCyber-SLM Security Report", ln=True, align="C")
pdf.set_font("Helvetica", "", 9)
pdf.set_text_color(100, 116, 139) # slate-500
pdf.cell(0, 6, f"Generated: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}",
ln=True, align="C")
pdf.ln(4)
# ── Body: parse Markdown lines ───────────────────────
for raw_line in text.splitlines():
line = raw_line.strip()
# Strip Markdown formatting for clean PDF output
line = re.sub(r'\*\*([^*]+)\*\*', r'\1', line) # bold
line = re.sub(r'\*([^*]+)\*', r'\1', line) # italic
line = re.sub(r'`([^`]+)`', r'\1', line) # inline code
line = re.sub(r'\[([^\]]+)\]\([^)]+\)', r'\1', line) # links
# Replace non-latin characters fpdf can't render with '?'
line = line.encode("latin-1", errors="replace").decode("latin-1")
if line.startswith("# "):
pdf.set_font("Helvetica", "B", 14)
pdf.set_text_color(15, 23, 42)
pdf.ln(3)
pdf.multi_cell(0, 7, line[2:])
pdf.ln(1)
elif line.startswith("## "):
pdf.set_font("Helvetica", "B", 12)
pdf.set_text_color(30, 64, 175) # blue-800
pdf.ln(2)
pdf.multi_cell(0, 6, line[3:])
pdf.ln(1)
elif line.startswith("### "):
pdf.set_font("Helvetica", "B", 10)
pdf.set_text_color(55, 65, 81)
pdf.ln(1)
pdf.multi_cell(0, 5, line[4:])
elif line.startswith("---"):
pdf.set_draw_color(203, 213, 225)
pdf.ln(1)
pdf.line(pdf.get_x(), pdf.get_y(), pdf.get_x() + 190, pdf.get_y())
pdf.ln(2)
elif line.startswith("| "):
pdf.set_font("Courier", "", 8)
pdf.set_text_color(30, 30, 30)
pdf.multi_cell(0, 5, line)
elif line.startswith("- ") or line.startswith("* "):
pdf.set_font("Helvetica", "", 9)
pdf.set_text_color(30, 30, 30)
pdf.multi_cell(0, 5, " " + line)
elif line.startswith(">"):
pdf.set_font("Helvetica", "I", 9)
pdf.set_text_color(100, 116, 139)
pdf.multi_cell(0, 5, " " + line.lstrip("> ").strip())
elif line == "":
pdf.ln(2)
else:
pdf.set_font("Helvetica", "", 9)
pdf.set_text_color(30, 30, 30)
pdf.multi_cell(0, 5, line)
pdf.output(path)
return path
# ══════════════════════════════════════════════════════════
# GRADIO HANDLER FUNCTIONS
# ══════════════════════════════════════════════════════════
def scan_github_repo(repo_url, max_files_str, progress=gr.Progress()):
github_token = _ENV_GITHUB_TOKEN
hf_token = _ENV_HF_TOKEN
if not repo_url.strip():
return "Please enter a GitHub repository URL.", None, None
try:
max_files = max(10, min(HARD_MAX, int(max_files_str)))
except Exception:
max_files = DEFAULT_MAX
owner, repo, err = parse_repo_url(repo_url)
if err:
return f"Error: {err}", None, None
token = github_token.strip() or None
progress(0.03, desc="Connecting to GitHub API...")
info, err = github_req(f"{GITHUB_API}/repos/{owner}/{repo}", token)
if err:
return f"Error: {err}", None, None
branch = info.get("default_branch","main")
repo_name = f"{owner}/{repo}"
stars = info.get("stargazers_count",0)
language = info.get("language","Unknown")
size_kb = info.get("size",0)
description = info.get("description","") or ""
is_private = info.get("private", False)
license_n = (info.get("license") or {}).get("name","Unknown")
open_issues = info.get("open_issues_count",0)
progress(0.08, desc="Fetching repository file tree...")
tree_data, err = github_req(
f"{GITHUB_API}/repos/{owner}/{repo}/git/trees/{branch}?recursive=1", token)
if err:
return f"Error: {err}", None, None
if tree_data.get("truncated"):
progress(0.09, desc="Large repo β€” tree truncated by GitHub. Scanning available files...")
tree = tree_data.get("tree",[])
scannable = [
it for it in tree
if it.get("type") == "blob"
and os.path.splitext(it.get("path",""))[1].lower() in SCANNABLE_EXT
and MIN_FILE_BYTES <= it.get("size",0) <= MAX_FILE_BYTES
and not _should_skip(it.get("path",""))
]
total_files = len(scannable)
if total_files == 0:
return f"No scannable source files found in `{repo_name}`.\nCheck that the repo is public and contains source code.", None, None
scannable.sort(key=_sort_key)
batch = scannable[:max_files]
all_findings, file_results, files_scanned, fetch_errors = [], [], 0, 0
_xfile_corpus: dict[str, str] = {} # path β†’ content (for cross-file taint)
for i, item in enumerate(batch):
path = item.get("path","")
progress(0.1 + 0.82*(i/len(batch)), desc=f"[{i+1}/{len(batch)}] {path}")
url = f"{GITHUB_API}/repos/{owner}/{repo}/contents/{urllib.parse.quote(path)}"
data, err = github_req(url, token)
if err or not data or data.get("encoding") != "base64":
fetch_errors += 1
continue
try:
content = base64.b64decode(data["content"]).decode("utf-8", errors="replace")
except Exception:
fetch_errors += 1
continue
findings = scan_file_content(content, path)
all_findings.extend(findings)
file_results.append({
"file": path,
"findings": findings,
"highest_sev": _top_sev(findings),
})
files_scanned += 1
_xfile_corpus[path] = content # collect for cross-file analysis
# ── Cross-file taint tracking ────────────────────────
progress(0.87, desc="Running cross-file taint analysis...")
xft_findings = CrossFileTaintTracker().analyze(_xfile_corpus)
if xft_findings:
all_findings.extend(xft_findings)
# Group cross-file findings into a virtual file entry
file_results.append({
"file": "CROSS-FILE TAINT",
"findings": xft_findings,
"highest_sev": _top_sev(xft_findings),
})
# ── Dependency scanning ──────────────────────────────
progress(0.88, desc="Scanning dependencies for known CVEs...")
dep_scanner = DependencyScanner()
dep_file_names = list(DependencyScanner.ECOSYSTEMS.keys())
dep_files_found = []
for dep_fname in dep_file_names:
url = f"{GITHUB_API}/repos/{owner}/{repo}/contents/{urllib.parse.quote(dep_fname)}"
data, err = github_req(url, token)
if err or not data or data.get("encoding") != "base64":
continue
try:
dep_content = base64.b64decode(data["content"]).decode("utf-8", errors="replace")
dep_files_found.append((dep_fname, dep_content))
except Exception:
continue
dep_findings = dep_scanner.scan(dep_files_found)
all_findings.extend(dep_findings)
if dep_findings:
file_results.append({
"file": "DEPENDENCIES",
"findings": dep_findings,
"highest_sev": _top_sev(dep_findings),
})
progress(0.88, desc="ZeroCyber-SLM AI Engine β€” adversarial verification...")
ai_verify_results = zerocyber_ai_verify(all_findings, repo_name, hf_token)
progress(0.92, desc="ZeroCyber-SLM AI Engine β€” deep analysis & executive summary...")
ai_analysis = zerocyber_ai_analyze(all_findings, repo_name, hf_token)
ai_exec_summary = zerocyber_ai_executive_summary(all_findings, repo_name, hf_token)
progress(0.95, desc="Generating professional report...")
# ── Coverage analysis ────────────────────────────────────
attempted = len(batch)
coverage_pct = round(100 * files_scanned / total_files, 1) if total_files else 0
rate_limit_hit = fetch_errors > 0 and files_scanned < attempted * 0.7
coverage_warn = ""
if rate_limit_hit:
coverage_warn = (
f"\n> ⚠️ **Rate Limit Warning:** {fetch_errors} files could not be fetched "
f"(GitHub API limit reached). "
f"**Add a GitHub Token** to raise the limit from 60 β†’ 5,000 req/hr and get complete results.\n"
)
elif files_scanned < 20 and total_files > 100:
coverage_warn = (
f"\n> ⚠️ **Low Coverage:** Only {files_scanned} files scanned from {total_files} available. "
f"Results may not represent the full security posture of this repository.\n"
)
repo_header = (
f"## Repository Overview\n\n"
f"| Field | Value |\n|---|---|\n"
f"| **Repository** | [{repo_name}](https://github.com/{repo_name}) |\n"
f"| **Description** | {description[:120] or 'N/A'} |\n"
f"| **Stars** | {stars:,} |\n"
f"| **Primary Language** | {language} |\n"
f"| **Size** | {size_kb:,} KB |\n"
f"| **Branch** | `{branch}` |\n"
f"| **License** | {license_n} |\n"
f"| **Open Issues** | {open_issues:,} |\n"
f"| **Visibility** | {'Private' if is_private else 'Public'} |\n"
f"| **Scan Coverage** | {files_scanned} of {attempted} attempted Β· {coverage_pct}% of {total_files} total |\n\n"
f"{coverage_warn}"
f"---\n\n"
)
report_md = repo_header + generate_report(
all_findings, repo_name, files_scanned, total_files, file_results, info,
low_coverage=rate_limit_hit or (files_scanned < 20 and total_files > 100),
ai_analysis=ai_analysis,
ai_verify_results=ai_verify_results,
ai_exec_summary=ai_exec_summary,
)
progress(1.0, desc="Scan complete.")
label = repo_name.replace("/","_")
return report_md, save_report(report_md, label), save_report_pdf(report_md, label)
def scan_uploaded_file(file_obj):
hf_token = _ENV_HF_TOKEN
if file_obj is None:
return "Please upload a file.", None, None
try:
with open(file_obj.name,"r",encoding="utf-8",errors="replace") as fh:
content = fh.read()
except Exception as e:
return f"Could not read file: {e}", None, None
filename = os.path.basename(file_obj.name)
ext = os.path.splitext(filename)[1].lower()
if ext not in SCANNABLE_EXT:
return f"File type `{ext}` not supported. Upload a source code file.", None, None
findings = scan_file_content(content, filename)
ai_verify_results = zerocyber_ai_verify(findings, filename, hf_token)
ai_analysis = zerocyber_ai_analyze(findings, filename, hf_token)
ai_exec_summary = zerocyber_ai_executive_summary(findings, filename, hf_token)
fr = [{"file":filename,"findings":findings,"highest_sev":_top_sev(findings)}]
report_md = generate_report(findings, filename, 1, 1, fr,
ai_analysis=ai_analysis,
ai_verify_results=ai_verify_results,
ai_exec_summary=ai_exec_summary)
return report_md, save_report(report_md, filename), save_report_pdf(report_md, filename)
def scan_code_snippet(code, lang_hint):
hf_token = _ENV_HF_TOKEN
if not code.strip():
return "Please paste some code to analyze.", None, None
ext_map = {
"Python":".py","JavaScript":".js","TypeScript":".ts","PHP":".php",
"Java":".java","Go":".go","Ruby":".rb","C#":".cs","Shell/Bash":".sh",
"HTML":".html","YAML":".yml",
}
ext = ext_map.get(lang_hint, ".py")
fname = f"snippet{ext}"
findings = scan_file_content(code, fname)
label = f"Code Snippet ({lang_hint})"
ai_verify_results = zerocyber_ai_verify(findings, label, hf_token)
ai_analysis = zerocyber_ai_analyze(findings, label, hf_token)
ai_exec_summary = zerocyber_ai_executive_summary(findings, label, hf_token)
fr = [{"file":fname,"findings":findings,"highest_sev":_top_sev(findings)}]
report_md = generate_report(findings, label, 1, 1, fr,
ai_analysis=ai_analysis,
ai_verify_results=ai_verify_results,
ai_exec_summary=ai_exec_summary)
snippet_label = f"snippet_{lang_hint}"
return report_md, save_report(report_md, snippet_label), save_report_pdf(report_md, snippet_label)
# ══════════════════════════════════════════════════════════
# DEMO CODE
# ══════════════════════════════════════════════════════════
DEMO_CODE = '''import pickle, os, subprocess, yaml
import hashlib, random
# CWE-798: Hardcoded credentials
DB_PASSWORD = "SuperSecret123!"
API_KEY = "sk-prod-abc123xyz789secretkey"
def login(username, password):
# CWE-89: SQL Injection
query = "SELECT * FROM users WHERE user='" + username + "' AND pass='" + password + "'"
cursor.execute(query)
# CWE-327: Weak cryptography
hashed = hashlib.md5(password.encode()).hexdigest()
def run_report(user_cmd):
# CWE-78: Command Injection
os.system("generate_report " + user_cmd)
subprocess.run(user_cmd, shell=True)
def load_user_data(raw_bytes):
# CWE-502: Insecure Deserialization
return pickle.loads(raw_bytes)
def load_config(stream):
# CWE-502: Unsafe YAML
return yaml.load(stream)
def fetch_url(url):
# CWE-918: SSRF
import requests
return requests.get(url).text
def get_file(path):
# CWE-22: Path Traversal
with open(path) as f:
return f.read()
def make_token():
# CWE-330: Insecure random
return str(random.randint(100000, 999999))
'''
# ══════════════════════════════════════════════════════════
# GRADIO UI
# ══════════════════════════════════════════════════════════
with gr.Blocks(title="ZeroCyber-SLM β€” Enterprise Security Scanner") as demo:
gr.HTML("""
<div style="text-align:center;padding:30px 0 12px;background:linear-gradient(135deg,#0f172a,#1e3a5f);border-radius:12px;margin-bottom:16px">
<h1 style="font-size:2rem;font-weight:900;color:#60a5fa;margin:0">
ZeroCyber-SLM
</h1>
<p style="color:#94a3b8;margin:8px 0 0;font-size:.95rem">
Enterprise-Grade Security Scanner &nbsp;|&nbsp; GitHub Repos &nbsp;|&nbsp; Code Files &nbsp;|&nbsp; OWASP Top 10 &nbsp;|&nbsp; CVSS v3.1
</p>
<div style="margin-top:14px;display:flex;justify-content:center;gap:8px;flex-wrap:wrap">
<span style="background:#dc2626;color:#fff;padding:4px 14px;border-radius:20px;font-size:.82rem;font-weight:700">
100+ Rules
</span>
<span style="background:#7c3aed;color:#fff;padding:4px 14px;border-radius:20px;font-size:.82rem;font-weight:700">
Python AST Engine
</span>
<span style="background:#065f46;color:#fff;padding:4px 14px;border-radius:20px;font-size:.82rem;font-weight:700">
OWASP Β· CWE Β· CVSS v3.1
</span>
<span style="background:#1e40af;color:#fff;padding:4px 14px;border-radius:20px;font-size:.82rem;font-weight:700">
Multi-Language
</span>
<span style="background:#92400e;color:#fff;padding:4px 14px;border-radius:20px;font-size:.82rem;font-weight:700">
ZeroCyber-SLM AI Engine
</span>
</div>
</div>
""")
with gr.Tabs():
# ── Tab 1: GitHub Scanner ──────────────────────
with gr.Tab("GitHub Repository Scanner"):
gr.Markdown(
"### Scan any public GitHub repository\n"
"Supports Python Β· JavaScript Β· TypeScript Β· PHP Β· Java Β· Go Β· Ruby Β· C# Β· Shell Β· YAML Β· HTML Β· JSP"
)
with gr.Row():
repo_in = gr.Textbox(
label="GitHub Repository URL",
placeholder="owner/repo or https://github.com/owner/repo",
scale=3,
)
max_in = gr.Textbox(label="Max Files to Scan", value=str(DEFAULT_MAX), scale=1)
gr.Examples(
examples=[
["digininja/DVWA", "100"],
["WebGoat/WebGoat", "100"],
["OWASP/juice-shop", "100"],
["adeyosemanputra/pygoat", "100"],
["vulhub/vulhub", "80"],
],
inputs=[repo_in, max_in],
label="Intentionally Vulnerable Repos (ideal for testing)"
)
scan_btn = gr.Button("Scan Repository", variant="primary", size="lg")
repo_out = gr.Markdown()
with gr.Row():
repo_dl = gr.File(label="Download Report (.md)", interactive=False)
repo_dl_pdf = gr.File(label="Download Report (.pdf)", interactive=False)
scan_btn.click(fn=scan_github_repo,
inputs=[repo_in, max_in],
outputs=[repo_out, repo_dl, repo_dl_pdf])
# ── Tab 2: File Scanner ────────────────────────
with gr.Tab("File Scanner"):
gr.Markdown(
"### Upload a source code file for deep security analysis\n"
"`.py` `.js` `.ts` `.php` `.java` `.go` `.rb` `.cs` `.sh` `.yml` `.xml` `.html` `.jsp` `.vue`"
)
file_in = gr.File(label="Upload Source Code File",
file_types=list(SCANNABLE_EXT))
file_btn = gr.Button("Analyze File", variant="primary", size="lg")
file_out = gr.Markdown()
with gr.Row():
file_dl = gr.File(label="Download Report (.md)", interactive=False)
file_dl_pdf = gr.File(label="Download Report (.pdf)", interactive=False)
file_btn.click(fn=scan_uploaded_file,
inputs=[file_in],
outputs=[file_out, file_dl, file_dl_pdf])
# ── Tab 3: Code Snippet ────────────────────────
with gr.Tab("Code Snippet Analyzer"):
gr.Markdown("### Paste code for instant security analysis")
lang_in = gr.Dropdown(
choices=["Python","JavaScript","TypeScript","PHP","Java",
"Go","Ruby","C#","Shell/Bash","HTML","YAML"],
value="Python", label="Language",
)
code_in = gr.Code(label="Paste code here", language="python",
value=DEMO_CODE, lines=22)
snippet_btn = gr.Button("Analyze Code", variant="primary", size="lg")
snippet_out = gr.Markdown()
with gr.Row():
snippet_dl = gr.File(label="Download Report (.md)", interactive=False)
snippet_dl_pdf = gr.File(label="Download Report (.pdf)", interactive=False)
snippet_btn.click(fn=scan_code_snippet,
inputs=[code_in, lang_in],
outputs=[snippet_out, snippet_dl, snippet_dl_pdf])
# ── Tab 4: About ───────────────────────────────
with gr.Tab("About"):
gr.Markdown(f"""
## ZeroCyber-SLM β€” Enterprise Security Scanner
**{len(RULES)} detection rules | ZeroCyber-SLM AI Engine | Python AST semantic analysis | OWASP Top 10 (2021) | CWE | CVSS v3.1**
### Competitive Analysis
| Feature | **ZeroCyber-SLM** | Snyk | SonarCloud | Semgrep | Checkmarx |
|---|:---:|:---:|:---:|:---:|:---:|
| 100% Offline | YES | NO | NO | NO | NO |
| GitHub Repo Scan | YES | YES | YES | YES | YES |
| Python AST Analysis | YES | NO | YES | YES | YES |
| Taint Tracking (Data Flow) | YES | YES | YES | YES | YES |
| Dependency CVE Scanning | YES | YES | YES | YES | YES |
| Logic Flaw Detection | YES | NO | NO | YES | NO |
| OWASP Top 10 (2021) | YES | YES | YES | YES | YES |
| CWE Mapping | YES | YES | YES | YES | YES |
| CVSS v3.1 Scores | YES | YES | NO | NO | YES |
| Per-File Report | YES | YES | YES | YES | YES |
| Fix Recommendations | YES | YES | YES | YES | YES |
| Confidence Levels | YES | YES | NO | YES | YES |
| Zero Data Exfiltration | YES | NO | NO | NO | NO |
| Free & Unlimited | YES | Limited | Limited | Limited | NO |
| AI-Powered Analysis | YES | NO | NO | NO | NO |
### Detection Coverage ({len(RULES)} Rules)
**Injection:** SQL Injection (Python/PHP/Java/Go/C#/Ruby) Β· Blind SQLi Β· NoSQL Injection Β·
XSS Β· Stored XSS Β· DOM XSS Β· React dangerouslySetInnerHTML Β·
Command Injection Β· Code Injection Β· LDAP Injection Β· Template Injection Β· Log4Shell
**Access Control:** Path Traversal Β· File Inclusion (LFI/RFI) Β· IDOR Β· Missing Auth Β·
Open Redirect Β· CSRF Β· Mass Assignment (Rails/Django)
**Cryptography & Auth:** Hardcoded Secrets Β· GitHub/Stripe/Google/Slack/SendGrid API Keys Β·
AWS Credentials Β· Private Keys Β· Weak Crypto (MD5/SHA1/DES/RC4) Β·
Insecure Random Β· Insecure Session Management Β· JWT Weak Secret
**Language-Specific:** Spring Boot Actuator misconfiguration Β· C# BinaryFormatter Β·
Java DocumentBuilderFactory XXE Β· Go exec.Command injection Β· Ruby eval() Β·
PHP preg_replace /e modifier Β· ASP.NET customErrors
**Other:** SSRF Β· XXE Β· Insecure Deserialization Β· Insecure File Upload Β· Debug Enabled Β·
Sensitive Data Exposure Β· Prototype Pollution Β· ReDoS Β· Dependency Vulnerabilities (OSV CVE)
### Security Analysis Engines
- **ZeroCyber-SLM AI Engine** β€” AI deep analysis: attack scenarios, business impact, remediation code
- **Regex Engine** β€” {len(RULES)} rules across all supported languages
- **Python AST Engine** β€” semantic analysis beyond regex (pickle, yaml.load, subprocess, hashlib)
- **TaintTracker** β€” multi-line data flow: user input β†’ dangerous sink (PHP + Python)
- **OSV Dependency Scanner** β€” CVE lookup via osv.dev API (16 manifest formats)
- **Context-Aware Mitigation** β€” reduces false positives by detecting nearby sanitization code
### Languages
Python Β· JavaScript Β· TypeScript Β· PHP Β· Java Β· Go Β· Ruby Β· C# Β· Shell Β·
YAML Β· XML Β· HTML Β· JSP Β· ASP Β· Vue Β· Svelte
---
*Built by Rootsystem2101 β€” Helmi S H ALAGHA*
*GitHub: https://github.com/rOOtsystem2010/ZeroCyber-SLM*
""")
demo.queue().launch(server_name="0.0.0.0", server_port=7860, show_error=True)