diff --git a/src/protect/engine/engine.js b/src/protect/engine/engine.js index d616b80..d61ae85 100644 --- a/src/protect/engine/engine.js +++ b/src/protect/engine/engine.js @@ -536,6 +536,13 @@ function collectLeafValues(root, nodeCap = 20000, maxDepth = 1000) { return out; } +// The inspection limits one evaluation reached, as `{ skips: [reason, …] }`, or nothing when it +// reached none. Callers count each reason in their coverage. +function skipsOf(resolver) { + const skips = resolver?.skips; + return skips && skips.length > 0 ? { skips } : {}; +} + // Emit a warning at most once per distinct key (keeps a persistent misconfiguration from spamming). const warnedKeys = new Set(); function warnOnce(key, message) { @@ -1185,8 +1192,11 @@ export class RuleEngine { let normalizedReq; let resolver; try { - normalizedReq = { ...req, ...normalizeRequest(req) }; + // Past the normalizer's depth bound a value is matched un-normalized; that is reported. + let limited = false; + normalizedReq = { ...req, ...normalizeRequest(req, { onLimit: () => { limited = true; } }) }; resolver = new RequestResolver(normalizedReq); + if (limited) resolver.noteSkip('container-cap'); } catch (err) { this.#reportError(err); return { blocked: false, rule: null, message: null }; // fail open @@ -1219,7 +1229,8 @@ export class RuleEngine { // that re-read the request instead would be reading it a second time: a getter, a stream or // anything else that answers once can give a different value, and evidence that disagrees // with the match it belongs to is worse than none. - resolver + resolver, + ...skipsOf(resolver) }; } } catch (err) { @@ -1229,7 +1240,7 @@ export class RuleEngine { } } - return { blocked: false, rule: null, message: null }; + return { blocked: false, rule: null, message: null, ...skipsOf(resolver) }; } #evaluateRule(conditions, resolver) { diff --git a/src/protect/engine/index.d.ts b/src/protect/engine/index.d.ts index cef6924..8287bfd 100644 --- a/src/protect/engine/index.d.ts +++ b/src/protect/engine/index.d.ts @@ -44,6 +44,8 @@ export interface EvaluateResult { blocked: boolean; rule: FirewallRule | null; message: string | null; + /** Inspection limits this evaluation reached (e.g. `container-cap`), when it reached any. */ + skips?: string[]; } export declare class RuleEngine { @@ -56,6 +58,8 @@ export declare class RequestResolver { constructor(req: any); resolve(parameter: string): any[]; applyMutations(mutations: string[], value: any): any; + noteSkip(reason: string): void; + readonly skips: string[]; } // Middleware @@ -111,6 +115,8 @@ export interface NormalizeOptions { sqlComments?: boolean; nullBytes?: boolean; whitespace?: boolean; + /** Called when a nested value is past the depth bound and is kept un-normalized. */ + onLimit?: () => void; } export declare function normalize(value: string, options?: NormalizeOptions): string; diff --git a/src/protect/engine/normalizer.js b/src/protect/engine/normalizer.js index 7b47c4f..186a15c 100644 --- a/src/protect/engine/normalizer.js +++ b/src/protect/engine/normalizer.js @@ -80,28 +80,45 @@ export function urlDecode(value) { while (result !== previous && iterations < MAX_DECODE_ITERATIONS) { previous = result; iterations++; - - try { - result = decodeURIComponent(result); - } catch { - result = safeUrlDecode(result); - break; - } + result = safeUrlDecode(result); } return result; } -function safeUrlDecode(value) { - return value.replace(/%([0-9A-Fa-f]{2})/g, (match, hex) => { +const utf8 = new TextDecoder(); + +/** + * Percent-decode every well-formed escape, whatever else the value contains. + * + * A `%` that does not start an escape is kept as it is, and does not stop the escapes around it from being + * decoded. Each run of escapes is decoded as UTF-8; bytes that are not valid UTF-8 become U+FFFD. + */ +export function safeUrlDecode(value) { + return value.replace(/(?:%[0-9A-Fa-f]{2})+/g, (run) => { try { - return String.fromCharCode(parseInt(hex, 16)); + return decodeURIComponent(run); } catch { - return match; + const bytes = new Uint8Array(run.length / 3); + for (let i = 0; i < bytes.length; i++) bytes[i] = parseInt(run.slice(i * 3 + 1, i * 3 + 3), 16); + return utf8.decode(bytes); } }); } +/** + * A request target with `+` in its query read as a space, the way query-string parsers read it. + * + * Only the query is affected: a `+` in the path is a literal `+`. Apply this before percent-decoding, so + * an encoded `%2B` still becomes a literal `+`. + */ +export function decodeQueryPlus(target) { + if (typeof target !== 'string') return target; + const query = target.indexOf('?'); + if (query === -1) return target; + return target.slice(0, query + 1) + target.slice(query + 1).replace(/\+/g, ' '); +} + export function htmlEntityDecode(value) { if (typeof value !== 'string') { return value; @@ -113,17 +130,40 @@ export function htmlEntityDecode(value) { result = result.split(entity).join(char); } - result = result.replace(/&#(\d+);/g, (match, code) => { - const num = parseInt(code, 10); - return num > 0 && num < 65536 ? String.fromCharCode(num) : match; - }); + return decodeHtmlEntities(result); +} - result = result.replace(/&#x([0-9A-Fa-f]+);/g, (match, hex) => { - const num = parseInt(hex, 16); - return num > 0 && num < 65536 ? String.fromCharCode(num) : match; - }); +/** + * Named entities decoded by `decodeHtmlEntities`: the ones that matter in injection contexts plus the + * handful every encoder emits. Numeric references are decoded generally, decimal and hex. + */ +const NAMED_ENTITIES = { + lt: '<', gt: '>', amp: '&', quot: '"', apos: "'", + nbsp: '\u00a0', sol: '/', bsol: '\\', colon: ':', lpar: '(', rpar: ')', equals: '=', grave: '`', + Tab: '\t', NewLine: '\n', semi: ';', excl: '!', num: '#', dollar: '$', percnt: '%', ast: '*', +}; - return result; +/** + * Decode HTML character references in one pass. + * + * The terminating `;` is optional, as it is for a browser reading a numeric reference: `:` and + * `:` decode like `:`. One pass means `&lt;` becomes `<`, not `<`. + */ +export function decodeHtmlEntities(input) { + return input.replace(/&(#[xX][0-9a-fA-F]+|#\d+|[A-Za-z][A-Za-z0-9]*);?/g, (whole, body) => { + if (body[0] === '#') { + const hex = body[1] === 'x' || body[1] === 'X'; + const code = Number.parseInt(hex ? body.slice(2) : body.slice(1), hex ? 16 : 10); + if (!Number.isFinite(code) || code < 0 || code > 0x10ffff) return whole; + try { + return String.fromCodePoint(code); + } catch { + return whole; + } + } + + return Object.prototype.hasOwnProperty.call(NAMED_ENTITIES, body) ? NAMED_ENTITIES[body] : whole; + }); } export function removeSqlComments(value) { @@ -352,41 +392,69 @@ export function normalizeRequest(req, options = {}) { query: normalizeObject(requestField(req, 'query') || {}, options), body: normalizeObject(body || {}, options), headers: normalizeObject(requestField(req, 'headers') || {}, options), - url: normalize(url || '', options), - originalUrl: normalize(requestField(req, 'originalUrl') || url || '', options), + url: normalize(decodeQueryPlus(url || ''), options), + originalUrl: normalize(decodeQueryPlus(requestField(req, 'originalUrl') || url || ''), options), _rawBody: rawBody }; } -// Depth bound for the recursive walk: a pathologically deep object would otherwise overflow the -// stack, and the engine's per-rule catch would swallow that into a fail-open. Beyond the bound the -// sub-value is left un-normalized (still matched, just in its raw form) rather than crashing. -const MAX_NORMALIZE_DEPTH = 200; +// The same depth bound as the engine's leaf walk, so every value that walk reaches is normalized. Past +// it a sub-value is kept as it is (still matched, in its raw form) and `options.onLimit` is called. +const MAX_NORMALIZE_DEPTH = 1000; +/** + * A copy of `value` with every string inside it normalized. + * + * Iterative, so depth cannot overflow the stack; the work is linear in the size of the value. A shared + * or cyclic node is copied once, and array holes stay holes. + */ export function normalizeObject(value, options = {}, depth = 0) { if (typeof value === 'string') { return normalize(value, options); } - if (depth >= MAX_NORMALIZE_DEPTH) { + if (value === null || typeof value !== 'object') { return value; } - if (Array.isArray(value)) { - return value.map(item => normalizeObject(item, options, depth + 1)); + if (depth >= MAX_NORMALIZE_DEPTH) { + options.onLimit?.(); + return value; } - if (typeof value === 'object' && value !== null) { - const result = {}; - - for (const [key, val] of Object.entries(value)) { - setOwn(result, key, normalizeObject(val, options, depth + 1)); + const copies = new Map(); + const copyOf = (node) => { + const copy = Array.isArray(node) ? new Array(node.length) : {}; + copies.set(node, copy); + return copy; + }; + const top = copyOf(value); + const stack = [[value, top, depth]]; + + while (stack.length > 0) { + const [node, copy, level] = stack.pop(); + + for (const key of Object.keys(node)) { + const child = node[key]; + + if (typeof child === 'string') { + setOwn(copy, key, normalize(child, options)); + } else if (child === null || typeof child !== 'object') { + setOwn(copy, key, child); + } else if (copies.has(child)) { + setOwn(copy, key, copies.get(child)); + } else if (level + 1 >= MAX_NORMALIZE_DEPTH) { + setOwn(copy, key, child); + options.onLimit?.(); + } else { + const childCopy = copyOf(child); + setOwn(copy, key, childCopy); + stack.push([child, childCopy, level + 1]); + } } - - return result; } - return value; + return top; } export function createMatchVariants(value) { diff --git a/src/protect/engine/request.js b/src/protect/engine/request.js index 8f9c947..dba6a01 100644 --- a/src/protect/engine/request.js +++ b/src/protect/engine/request.js @@ -1,3 +1,4 @@ +import { decodeHtmlEntities, safeUrlDecode } from './normalizer.js'; import { setOwn } from './own.js'; // Resolvable DATA attributes of an uploaded file part (files..). The engine only exposes @@ -22,55 +23,70 @@ export function base64DecodeUtf8(value) { return new TextDecoder().decode(bytes); } +// Mutations that decode one string into another. +const TEXT_MUTATIONS = new Set(['base64_decode', 'urldecode', 'htmlentitydecode']); + +// Same bounds as the engine's leaf walk: beyond them a node is kept as it is, still matched undecoded. +const MAX_MAPPED_NODES = 20000; +const MAX_MAPPED_DEPTH = 1000; + /** - * Decode HTML entities, so a payload written as `<script>` is screened as `