From ef80a68de54cc4992b4861866f2baab2eed903cc Mon Sep 17 00:00:00 2001 From: Dave Jong Date: Mon, 28 Sep 2026 16:10:46 +0200 Subject: [PATCH 1/2] Decode request values the way apps read them Percent-decoding keeps decoding the escapes around a % that starts no escape, and decodes each run as UTF-8. The urldecode mutation reads + as a space, as form decoding does, and REQUEST_URI and all read + in the query as a space while keeping a + in the path literal. HTML character references decode the same way in normalization and in the htmlentitydecode mutation: named references such as : and numeric references without a closing semicolon. Text mutations (urldecode, htmlentitydecode, base64_decode) applied to a structured value decode each string inside it and keep the structure, within the same bounds as the leaf walk. Co-Authored-By: Claude Opus 5.5 --- src/protect/engine/normalizer.js | 84 ++++++++---- src/protect/engine/request.js | 86 +++++++------ tests/protect/request-decoding.test.ts | 169 +++++++++++++++++++++++++ 3 files changed, 278 insertions(+), 61 deletions(-) create mode 100644 tests/protect/request-decoding.test.ts diff --git a/src/protect/engine/normalizer.js b/src/protect/engine/normalizer.js index 7b47c4f..01eb85b 100644 --- a/src/protect/engine/normalizer.js +++ b/src/protect/engine/normalizer.js @@ -80,28 +80,45 @@ export function urlDecode(value) { while (result !== previous && iterations < MAX_DECODE_ITERATIONS) { previous = result; iterations++; - - try { - result = decodeURIComponent(result); - } catch { - result = safeUrlDecode(result); - break; - } + result = safeUrlDecode(result); } return result; } -function safeUrlDecode(value) { - return value.replace(/%([0-9A-Fa-f]{2})/g, (match, hex) => { +const utf8 = new TextDecoder(); + +/** + * Percent-decode every well-formed escape, whatever else the value contains. + * + * A `%` that does not start an escape is kept as it is, and does not stop the escapes around it from being + * decoded. Each run of escapes is decoded as UTF-8; bytes that are not valid UTF-8 become U+FFFD. + */ +export function safeUrlDecode(value) { + return value.replace(/(?:%[0-9A-Fa-f]{2})+/g, (run) => { try { - return String.fromCharCode(parseInt(hex, 16)); + return decodeURIComponent(run); } catch { - return match; + const bytes = new Uint8Array(run.length / 3); + for (let i = 0; i < bytes.length; i++) bytes[i] = parseInt(run.slice(i * 3 + 1, i * 3 + 3), 16); + return utf8.decode(bytes); } }); } +/** + * A request target with `+` in its query read as a space, the way query-string parsers read it. + * + * Only the query is affected: a `+` in the path is a literal `+`. Apply this before percent-decoding, so + * an encoded `%2B` still becomes a literal `+`. + */ +export function decodeQueryPlus(target) { + if (typeof target !== 'string') return target; + const query = target.indexOf('?'); + if (query === -1) return target; + return target.slice(0, query + 1) + target.slice(query + 1).replace(/\+/g, ' '); +} + export function htmlEntityDecode(value) { if (typeof value !== 'string') { return value; @@ -113,17 +130,40 @@ export function htmlEntityDecode(value) { result = result.split(entity).join(char); } - result = result.replace(/&#(\d+);/g, (match, code) => { - const num = parseInt(code, 10); - return num > 0 && num < 65536 ? String.fromCharCode(num) : match; - }); + return decodeHtmlEntities(result); +} - result = result.replace(/&#x([0-9A-Fa-f]+);/g, (match, hex) => { - const num = parseInt(hex, 16); - return num > 0 && num < 65536 ? String.fromCharCode(num) : match; - }); +/** + * Named entities decoded by `decodeHtmlEntities`: the ones that matter in injection contexts plus the + * handful every encoder emits. Numeric references are decoded generally, decimal and hex. + */ +const NAMED_ENTITIES = { + lt: '<', gt: '>', amp: '&', quot: '"', apos: "'", + nbsp: '\u00a0', sol: '/', bsol: '\\', colon: ':', lpar: '(', rpar: ')', equals: '=', grave: '`', + Tab: '\t', NewLine: '\n', semi: ';', excl: '!', num: '#', dollar: '$', percnt: '%', ast: '*', +}; - return result; +/** + * Decode HTML character references in one pass. + * + * The terminating `;` is optional, as it is for a browser reading a numeric reference: `:` and + * `:` decode like `:`. One pass means `&lt;` becomes `<`, not `<`. + */ +export function decodeHtmlEntities(input) { + return input.replace(/&(#[xX][0-9a-fA-F]+|#\d+|[A-Za-z][A-Za-z0-9]*);?/g, (whole, body) => { + if (body[0] === '#') { + const hex = body[1] === 'x' || body[1] === 'X'; + const code = Number.parseInt(hex ? body.slice(2) : body.slice(1), hex ? 16 : 10); + if (!Number.isFinite(code) || code < 0 || code > 0x10ffff) return whole; + try { + return String.fromCodePoint(code); + } catch { + return whole; + } + } + + return Object.prototype.hasOwnProperty.call(NAMED_ENTITIES, body) ? NAMED_ENTITIES[body] : whole; + }); } export function removeSqlComments(value) { @@ -352,8 +392,8 @@ export function normalizeRequest(req, options = {}) { query: normalizeObject(requestField(req, 'query') || {}, options), body: normalizeObject(body || {}, options), headers: normalizeObject(requestField(req, 'headers') || {}, options), - url: normalize(url || '', options), - originalUrl: normalize(requestField(req, 'originalUrl') || url || '', options), + url: normalize(decodeQueryPlus(url || ''), options), + originalUrl: normalize(decodeQueryPlus(requestField(req, 'originalUrl') || url || ''), options), _rawBody: rawBody }; } diff --git a/src/protect/engine/request.js b/src/protect/engine/request.js index 8f9c947..294e8f4 100644 --- a/src/protect/engine/request.js +++ b/src/protect/engine/request.js @@ -1,3 +1,4 @@ +import { decodeHtmlEntities, safeUrlDecode } from './normalizer.js'; import { setOwn } from './own.js'; // Resolvable DATA attributes of an uploaded file part (files..). The engine only exposes @@ -22,44 +23,48 @@ export function base64DecodeUtf8(value) { return new TextDecoder().decode(bytes); } +// Mutations that decode one string into another. +const TEXT_MUTATIONS = new Set(['base64_decode', 'urldecode', 'htmlentitydecode']); + +// Same bounds as the engine's leaf walk: beyond them a node is kept as it is, still matched undecoded. +const MAX_MAPPED_NODES = 20000; +const MAX_MAPPED_DEPTH = 1000; + /** - * Decode HTML entities, so a payload written as `<script>` is screened as `