diff --git a/src/protect/engine/cookies.js b/src/protect/engine/cookies.js new file mode 100644 index 00000000..bed009df --- /dev/null +++ b/src/protect/engine/cookies.js @@ -0,0 +1,36 @@ +import { setOwn } from './own.js'; + +/** + * A `Cookie` header as name → value. + * + * A value wrapped in double quotes loses them, as cookie parsers strip them before an application reads + * the value. A name sent more than once keeps every value, as an array in the order sent: parsers differ + * on whether the first or the last one wins, so a rule on that cookie inspects all of them. + */ +export function parseCookieHeader(header) { + const cookies = {}; + if (typeof header !== 'string' || header === '') return cookies; + + const repeated = new Map(); + for (const pair of header.split(';')) { + const eq = pair.indexOf('='); + if (eq === -1) continue; + const name = pair.slice(0, eq).trim(); + if (name === '') continue; + let value = pair.slice(eq + 1).trim(); + if (value.length >= 2 && value[0] === '"' && value[value.length - 1] === '"') value = value.slice(1, -1); + + if (!Object.hasOwn(cookies, name)) { + setOwn(cookies, name, value); + } else { + let values = repeated.get(name); + if (values === undefined) { + values = [cookies[name]]; + repeated.set(name, values); + setOwn(cookies, name, values); + } + values.push(value); + } + } + return cookies; +} diff --git a/src/protect/engine/engine.js b/src/protect/engine/engine.js index d61ae856..fb49bf8d 100644 --- a/src/protect/engine/engine.js +++ b/src/protect/engine/engine.js @@ -514,18 +514,23 @@ const WHOLE_VALUE_MATCH_TYPES = new Set(['isset', 'array_in_array', 'array_key_v // Iteratively collect every scalar (non-object) leaf of a structured value. Iterative + bounded // (depth and node caps) so a pathologically deep/large attacker payload STOPS at the bound rather // than throwing a RangeError that the per-rule catch would swallow into a fail-open bypass. +// `truncated` says the bound was reached: containers past it contributed no leaves. function collectLeafValues(root, nodeCap = 20000, maxDepth = 1000) { - const out = []; + const leaves = []; const stack = [[root, 0]]; let visited = 0; + let truncated = false; while (stack.length) { const [node, depth] = stack.pop(); if (node === null || node === undefined) continue; if (typeof node !== 'object') { - out.push(node); + leaves.push(node); + continue; + } + if (depth >= maxDepth || visited >= nodeCap) { + truncated = true; continue; } - if (depth >= maxDepth || visited >= nodeCap) continue; visited++; if (Array.isArray(node)) { for (let i = node.length - 1; i >= 0; i--) stack.push([node[i], depth + 1]); @@ -533,7 +538,7 @@ function collectLeafValues(root, nodeCap = 20000, maxDepth = 1000) { for (const k of Object.keys(node)) stack.push([node[k], depth + 1]); } } - return out; + return { leaves, truncated }; } // The inspection limits one evaluation reached, as `{ skips: [reason, …] }`, or nothing when it @@ -543,6 +548,16 @@ function skipsOf(resolver) { return skips && skips.length > 0 ? { skips } : {}; } +// The whole value as text, for a container too large to walk leaf by leaf. `undefined` when it has +// no text form (a cycle, or nesting deeper than the serialiser allows). +function serialisedValue(value) { + try { + return JSON.stringify(value); + } catch { + return undefined; + } +} + // Emit a warning at most once per distinct key (keeps a persistent misconfiguration from spamming). const warnedKeys = new Set(); function warnOnce(key, message) { @@ -585,16 +600,34 @@ function warnUnsupportedMatchType(type) { * value that is already a bare host passes through untouched, which is what keeps the egress path and * the built-in default rule behaving exactly as before. */ +// Schemes a URL parser treats as hierarchical whatever follows the colon: `http:x`, `http:/x` and +// `http:\\x` all name host `x`. +const SPECIAL_SCHEMES = new Set(['http', 'https', 'ws', 'wss', 'ftp', 'file']); + +// Strip C0 controls and spaces (U+0000–U+0020) from both ends, in linear time. +function trimControls(text) { + let start = 0; + let end = text.length; + while (start < end && text.charCodeAt(start) <= 0x20) start++; + while (end > start && text.charCodeAt(end - 1) <= 0x20) end--; + return text.slice(start, end); +} + function hostFromValue(value) { - const raw = String(value ?? '').trim(); + // A URL parser drops tabs and newlines anywhere, and C0 controls and spaces at either end, before it + // reads anything else — so they are dropped here too, or they would hide the scheme. + const raw = trimControls(String(value ?? '').replace(/[\t\n\r]/g, '')).trim(); if (raw === '') return ''; - // A scheme (`http://`, and deliberately any other) or a protocol-relative URL. Parsing rather than - // string-slicing is what makes the userinfo evasion (`http://trusted@169.254.169.254/`) resolve to the - // host actually contacted, and keeps `http://evil.com#@127.0.0.1` resolving to evil.com. - if (/^[a-z][a-z0-9+.-]*:\/\//i.test(raw) || raw.startsWith('//')) { + // A scheme (`http://`, and deliberately any other), a special scheme in any of the shorter spellings a + // URL parser still resolves to a host, or a protocol-relative URL (either slash direction). Parsing + // rather than string-slicing is what makes the userinfo evasion (`http://trusted@169.254.169.254/`) + // resolve to the host actually contacted, and keeps `http://evil.com#@127.0.0.1` resolving to evil.com. + const scheme = /^([a-z][a-z0-9+.-]*):/i.exec(raw)?.[1]?.toLowerCase(); + const relative = /^[\\/]{2}/.test(raw); + if (relative || /^[a-z][a-z0-9+.-]*:\/\//i.test(raw) || (scheme !== undefined && SPECIAL_SCHEMES.has(scheme))) { try { - return new URL(raw.startsWith('//') ? `http:${raw}` : raw).hostname; + return new URL(relative ? `http:${raw}` : raw).hostname; } catch { // Unparseable: hand the raw value on, where the host check rejects it rather than guessing. return raw; @@ -1328,9 +1361,17 @@ export class RuleEngine { if (WHOLE_VALUE_MATCH_TYPES.has(match.type)) { if (matchValue(match.type, value, match.value, match)) return true; } else { - for (const leaf of collectLeafValues(value)) { + const { leaves, truncated } = collectLeafValues(value); + for (const leaf of leaves) { if (matchValue(match.type, leaf, match.value, match)) return true; } + // Past the walk's bound, the rest of the value is matched as its serialised text, and the + // bound is reported: the leaves beyond it were not inspected individually. + if (truncated) { + resolver.noteSkip('container-cap'); + const text = serialisedValue(value); + if (text !== undefined && matchValue(match.type, text, match.value, match)) return true; + } } continue; } diff --git a/src/protect/engine/fetch.js b/src/protect/engine/fetch.js index 1ccb2192..2e558490 100644 --- a/src/protect/engine/fetch.js +++ b/src/protect/engine/fetch.js @@ -9,6 +9,7 @@ import { resolveClientIp } from '../client-ip.js'; import { RuleEngine } from './engine.js'; import { notify } from '../notify.js'; +import { parseCookieHeader } from './cookies.js'; import { appendOwn, setOwn } from './own.js'; // Cap how much request body we buffer for inspection. A larger body is left UNSCANNED @@ -67,7 +68,7 @@ export async function fromFetchRequest(request, options = {}) { // that guessed differently would attribute one request to two addresses. ip: client.ip ?? '', _clientIp: client, - cookies: parseCookies(headers.cookie), + cookies: parseCookieHeader(headers.cookie), // Verbatim body text: preserves literal keys (e.g. `__proto__`) that JSON.stringify // drops, so prototype-pollution rules on `raw` are robust. _rawBody: rawBody, @@ -272,20 +273,6 @@ function decodeExtendedValue(value) { } } -function parseCookies(header) { - const cookies = {}; - if (!header) { - return cookies; - } - for (const pair of header.split(';')) { - const idx = pair.indexOf('='); - if (idx === -1) { - continue; - } - setOwn(cookies, pair.slice(0, idx).trim(), pair.slice(idx + 1).trim()); - } - return cookies; -} function defaultBlockResponse(result) { return new Response( @@ -319,6 +306,9 @@ export function createFetchMiddleware(rulesData, options = {}) { notify(options.onSkip, { phase: 'request', reason: req._bodyInspectionSkip }, 'onSkip'); } result = engine.evaluate(req); + for (const reason of result.skips ?? []) { + notify(options.onSkip, { phase: 'request', reason }, 'onSkip'); + } } catch (err) { notify(options.onError, err, 'onError'); return null; // fail open diff --git a/src/protect/engine/index.d.ts b/src/protect/engine/index.d.ts index 8287bfdb..303f0ef6 100644 --- a/src/protect/engine/index.d.ts +++ b/src/protect/engine/index.d.ts @@ -124,6 +124,7 @@ export declare function normalizeRequest(req: any, options?: NormalizeOptions): query: Record; body: Record; headers: Record; + cookies: Record; url: string; originalUrl: string; _rawBody: string; diff --git a/src/protect/engine/node.js b/src/protect/engine/node.js index 4f742dfc..1db8586e 100644 --- a/src/protect/engine/node.js +++ b/src/protect/engine/node.js @@ -10,6 +10,7 @@ import { resolveClientIp } from '../client-ip.js'; import { RuleEngine } from './engine.js'; import { parseBody } from './fetch.js'; import { notify } from '../notify.js'; +import { parseCookieHeader } from './cookies.js'; import { appendOwn, setOwn } from './own.js'; // Build the engine's request shape from a Node IncomingMessage + its raw body text. @@ -70,26 +71,12 @@ export function fromNodeRequest(req, rawBody = '', options = {}) { headers, ip: client.ip ?? '', _clientIp: client, - cookies: parseCookies(headers.cookie), + cookies: parseCookieHeader(headers.cookie), // Verbatim body text: preserves literal keys (e.g. `__proto__`) that JSON.stringify drops. _rawBody: rawBody }; } -function parseCookies(header) { - const cookies = {}; - if (!header) { - return cookies; - } - for (const pair of header.split(';')) { - const idx = pair.indexOf('='); - if (idx === -1) { - continue; - } - setOwn(cookies, pair.slice(0, idx).trim(), pair.slice(idx + 1).trim()); - } - return cookies; -} function defaultBlock(res, result) { res.statusCode = 403; diff --git a/src/protect/engine/normalizer.js b/src/protect/engine/normalizer.js index 186a15c8..dab3208a 100644 --- a/src/protect/engine/normalizer.js +++ b/src/protect/engine/normalizer.js @@ -6,6 +6,7 @@ * obfuscation techniques that attackers use to evade pattern matching. */ import { setOwn } from './own.js'; +import { parseCookieHeader } from './cookies.js'; const HTML_ENTITIES = { @@ -388,16 +389,29 @@ export function normalizeRequest(req, options = {}) { ? req._rawBody : serializeForRawDetection(body ?? null); + const headers = requestField(req, 'headers') || {}; + return { query: normalizeObject(requestField(req, 'query') || {}, options), body: normalizeObject(body || {}, options), - headers: normalizeObject(requestField(req, 'headers') || {}, options), + headers: normalizeObject(headers, options), + // Cookies are split into name/value pairs BEFORE their values are normalised, and normalised the + // same way whether a framework parsed them or they come from the header — so a rule on a cookie + // sees one value, independent of which cookie parser (if any) ran first. + cookies: normalizeObject(cookieValues(requestField(req, 'cookies'), headers), options), url: normalize(decodeQueryPlus(url || ''), options), originalUrl: normalize(decodeQueryPlus(requestField(req, 'originalUrl') || url || ''), options), _rawBody: rawBody }; } +// The request's cookies as name → value: the ones a cookie parser already produced when there are any, +// otherwise the `Cookie` header's pairs. +function cookieValues(parsed, headers) { + if (parsed !== null && typeof parsed === 'object' && !Array.isArray(parsed)) return parsed; + return parseCookieHeader(headers?.cookie); +} + // The same depth bound as the engine's leaf walk, so every value that walk reaches is normalized. Past // it a sub-value is kept as it is (still matched, in its raw form) and `options.onLimit` is called. const MAX_NORMALIZE_DEPTH = 1000; diff --git a/src/protect/engine/request.js b/src/protect/engine/request.js index dba6a01e..9e7c0323 100644 --- a/src/protect/engine/request.js +++ b/src/protect/engine/request.js @@ -1,3 +1,4 @@ +import { parseCookieHeader } from './cookies.js'; import { decodeHtmlEntities, safeUrlDecode } from './normalizer.js'; import { setOwn } from './own.js'; @@ -6,6 +7,30 @@ import { setOwn } from './own.js'; // in rules (see the triage-vpatch-npm skill), not hardcoded here. const FILE_ATTRS = new Set(['content', 'filename', 'type']); +const ownValue = (obj, key) => + obj !== null && typeof obj === 'object' && Object.prototype.hasOwnProperty.call(obj, key) ? obj[key] : undefined; + +/** + * The spellings one field path can arrive under: as written (resolved as a nested path), in bracket + * form as a flat key (`a.b` → `a[b]`, `a` → `a[]`), and — for a path written in bracket form — as the + * nested path it expands to (`a[b]` → `a.b`, `a[]` → `a`). + */ +function fieldSpellings(key) { + const spellings = [{ key, nested: true }]; + if (key.includes('[')) { + const dotted = key.replace(/\[\]$/, '').replace(/\[([^\][]*)\]/g, '.$1'); + if (dotted !== key && dotted !== '' && !dotted.includes('[') && !dotted.includes(']')) { + spellings.push({ key: dotted, nested: true }); + } + return spellings; + } + const [head, ...rest] = key.split('.'); + const bracketed = head + rest.map((part) => `[${part}]`).join(''); + if (bracketed !== key) spellings.push({ key: bracketed, nested: false }); + spellings.push({ key: `${bracketed}[]`, nested: false }); + return spellings; +} + // A captured file part is { filename, type, content }; tolerate the legacy bare-filename string. const fileFilename = (f) => (f && typeof f === 'object' ? f.filename : f); const fileAttribute = (f, attr) => (f && typeof f === 'object' ? f[attr] : attr === 'filename' ? f : undefined); @@ -32,9 +57,10 @@ const MAX_MAPPED_DEPTH = 1000; /** * A copy of `root` with `fn` applied to every string leaf. Iterative and bounded, so an oversized or - * deeply nested value cannot overflow the stack; a shared or cyclic node is copied once. + * deeply nested value cannot overflow the stack; a shared or cyclic node is copied once. `onLimit` is + * called when a node past the bounds is kept as it is, with `fn` not applied inside it. */ -function mapStringLeaves(root, fn) { +function mapStringLeaves(root, fn, onLimit) { const copies = new Map(); const copyOf = (node) => { const copy = Array.isArray(node) ? [] : {}; @@ -57,6 +83,7 @@ function mapStringLeaves(root, fn) { setOwn(copy, key, copies.get(child)); } else if (depth + 1 >= MAX_MAPPED_DEPTH || visited + stack.length >= MAX_MAPPED_NODES) { setOwn(copy, key, child); + onLimit(); } else { const childCopy = copyOf(child); setOwn(copy, key, childCopy); @@ -194,7 +221,8 @@ export class RequestResolver { // A text decoder applied to a structured value decodes each string inside it and keeps the structure, // so the matcher still sees every leaf. if (typeof value === 'object' && TEXT_MUTATIONS.has(mutation)) { - return mapStringLeaves(value, (leaf) => this.#applyMutation(mutation, leaf)); + // Past the walk's bounds the value is still matched, undecoded; that is reported like the leaf walk's. + return mapStringLeaves(value, (leaf) => this.#applyMutation(mutation, leaf), () => this.noteSkip('container-cap')); } switch (mutation) { @@ -247,8 +275,7 @@ export class RequestResolver { return this.#resolveWildcard(query, key); } - const value = this.#getNestedValue(query, key); - return value !== undefined ? [value] : []; + return this.#lookup(query, key); } #resolvePost(key) { @@ -258,8 +285,24 @@ export class RequestResolver { return this.#resolveWildcard(body, key); } - const value = this.#getNestedValue(body, key); - return value !== undefined ? [value] : []; + return this.#lookup(body, key); + } + + /** + * Every value a field path names in `obj`, in either spelling a form or query field can take. + * + * A parser that expands brackets (`qs`, as Express uses) turns `user[name]=x` into `{ user: { name } }` + * and `id[]=x` into `{ id: [x] }`; a flat parser (`URLSearchParams`, as the Fetch and Node adapters use) + * keeps `user[name]` and `id[]` as literal keys. The rule names the field once, so both shapes answer + * to both spellings: `user.name` and `user[name]`, `id` and `id[]`. + */ + #lookup(obj, key) { + const values = []; + for (const candidate of fieldSpellings(key)) { + const value = candidate.nested ? this.#getNestedValue(obj, candidate.key) : ownValue(obj, candidate.key); + if (value !== undefined) values.push(value); + } + return values; } #resolveRequest(key) { @@ -275,11 +318,12 @@ export class RequestResolver { ]; } - const value = this.#getNestedValue(query, key) - ?? this.#getNestedValue(body, key) - ?? (Object.hasOwn(cookies, key) ? cookies[key] : undefined); + const fromQuery = this.#lookup(query, key); + if (fromQuery.length > 0) return fromQuery; + const fromBody = this.#lookup(body, key); + if (fromBody.length > 0) return fromBody; - return value !== undefined ? [value] : []; + return Object.hasOwn(cookies, key) ? [cookies[key]] : []; } #resolveCookie(key) { @@ -494,25 +538,7 @@ export class RequestResolver { return this.#cookies; } - const header = this.#req.headers?.cookie; - if (!header) { - this.#cookies = {}; - return this.#cookies; - } - - const cookies = {}; - - for (const pair of header.split(';')) { - const eqIndex = pair.indexOf('='); - if (eqIndex === -1) { - continue; - } - const name = pair.substring(0, eqIndex).trim(); - const value = pair.substring(eqIndex + 1).trim(); - setOwn(cookies, name, value); - } - - this.#cookies = cookies; + this.#cookies = parseCookieHeader(this.#req.headers?.cookie); return this.#cookies; } } diff --git a/tests/protect/internal-host-spellings.test.ts b/tests/protect/internal-host-spellings.test.ts new file mode 100644 index 00000000..4279be65 --- /dev/null +++ b/tests/protect/internal-host-spellings.test.ts @@ -0,0 +1,62 @@ +import { describe, expect, it } from 'vitest'; +import { matchValue, RuleEngine } from '../../src/protect/engine/engine.js'; +import { fromFetchRequest } from '../../src/protect/engine/fetch.js'; + +// `internal_host` classifies the host a value names. A URL parser resolves a host from more spellings +// than `scheme://host`, so each of those has to be read the same way. + +describe('internal_host on URL spellings a parser resolves to a host', () => { + it.each([ + 'http:/169.254.169.254/latest', + 'http:\\\\169.254.169.254/latest', + 'HTTP:169.254.169.254', + 'http:127.0.0.1', + 'https:/\\127.0.0.1', + 'ws:127.0.0.1', + 'ftp:127.0.0.1', + '\\\\127.0.0.1/path', + '/\\127.0.0.1/path', + '\\/127.0.0.1/path', + 'ht\ttp://127.0.0.1/', + 'http://127.0.0.1\n/', + '\u0000http://127.0.0.1/', + ])('classifies %j as internal', (value) => { + expect(new URL(value, 'https://app.test/').hostname).toMatch(/^(127\.0\.0\.1|169\.254\.169\.254)$/); + expect(matchValue('internal_host', value, null)).toBe(true); + }); + + it.each([ + 'http:/example.test/', + 'HTTPS:example.test', + '\\\\example.test/path', + 'mailto:someone@127.0.0.1', + 'localhost-docs.example.test', + ])('does not classify %j as internal', (value) => { + expect(matchValue('internal_host', value, null)).toBe(false); + }); + + it('keeps reading a host:port pair as a host', () => { + expect(matchValue('internal_host', 'localhost:8080', null)).toBe(true); + expect(matchValue('internal_host', 'example.test:8080', null)).toBe(false); + }); + + it('trims a long value in linear time', () => { + const value = 'a' + ' '.repeat(200_000) + 'b'; + const started = performance.now(); + expect(matchValue('internal_host', value, null)).toBe(false); + expect(performance.now() - started).toBeLessThan(1_000); + }); + + it('blocks a request parameter written in a short spelling', async () => { + const engine = new RuleEngine({ + firewall: [{ id: 1, title: 'internal target', rule_v2: [{ parameter: 'get.target', match: { type: 'internal_host' } }] }], + whitelists: [], + whitelist_keys: {}, + } as any); + const blocked = async (target: string) => + engine.evaluate(await fromFetchRequest(new Request('https://app.test/?target=' + encodeURIComponent(target)))).blocked; + + expect(await blocked('http:/169.254.169.254/latest')).toBe(true); + expect(await blocked('http:/example.test/')).toBe(false); + }); +}); diff --git a/tests/protect/request-field-shapes.test.ts b/tests/protect/request-field-shapes.test.ts new file mode 100644 index 00000000..9c9c1096 --- /dev/null +++ b/tests/protect/request-field-shapes.test.ts @@ -0,0 +1,290 @@ +import { describe, expect, it } from 'vitest'; +import { RuleEngine } from '../../src/protect/engine/engine.js'; +import { createFetchMiddleware, fromFetchRequest } from '../../src/protect/engine/fetch.js'; +import { fromNodeRequest } from '../../src/protect/engine/node.js'; +import { createProtection } from '../../src/protect/runtime.js'; + +// A rule names a request field once. These cover the shapes the same field can arrive in — the parser +// that produced it, its spelling, its size — and that each shape reaches the rule. + +const MARKER = 'sample-marker'; + +const engineFor = (parameter: string, match: object = { type: 'contains', value: MARKER }) => + new RuleEngine({ + firewall: [{ id: 1, title: 'field shape', rule_v2: [{ parameter, match }] }], + whitelists: [], + whitelist_keys: {}, + } as any); + +const blocks = async (engine: RuleEngine, request: Request | Record) => + engine.evaluate(request instanceof Request ? await fromFetchRequest(request) : request).blocked; + +const shaped = (overrides: Record) => ({ + method: 'GET', + url: '/', + originalUrl: '/', + headers: {}, + query: {}, + body: {}, + ...overrides, +}); + +describe('cookie values', () => { + const rule = engineFor('cookie.pref'); + + it('reads a percent-encoded value from the Cookie header as the application does', async () => { + const request = new Request('https://app.test/', { headers: { cookie: 'pref=sample%2Dmarker' } }); + expect(await blocks(rule, request)).toBe(true); + }); + + it('normalises cookies a framework already parsed the same way', async () => { + expect(await blocks(rule, shaped({ cookies: { pref: 'sample%252Dmarker' } }))).toBe(true); + }); + + it('strips the double quotes around a quoted value', async () => { + const exact = engineFor('cookie.pref', { type: 'equals', value: MARKER }); + const request = new Request('https://app.test/', { headers: { cookie: `pref="${MARKER}"` } }); + expect(await blocks(exact, request)).toBe(true); + }); + + it.each([ + ['first', `pref=${MARKER}; pref=plain`], + ['last', `pref=plain; pref=${MARKER}`], + ])('inspects every value of a repeated name (%s)', async (_label, cookie) => { + expect(await blocks(rule, new Request('https://app.test/', { headers: { cookie } }))).toBe(true); + }); + + it('splits pairs before decoding their values', async () => { + // An encoded separator is part of the value; it does not start another cookie. + const present = engineFor('cookie.other', { type: 'isset' }); + const request = new Request('https://app.test/', { headers: { cookie: 'pref=one%3B%20other%3Dtwo' } }); + expect(await blocks(present, request)).toBe(false); + expect(await blocks(engineFor('cookie.pref', { type: 'contains', value: 'other=two' }), request)).toBe(true); + }); + + it('reads the Cookie header the same way on the Node adapter', () => { + const request = fromNodeRequest({ + method: 'GET', + url: '/', + headers: { host: 'app.test', cookie: `pref=plain; pref="${MARKER}"` }, + socket: { remoteAddress: '198.51.100.7' }, + }); + expect(request.cookies).toEqual({ pref: ['plain', MARKER] }); + }); + + it('leaves an ordinary cookie alone', async () => { + const request = new Request('https://app.test/', { headers: { cookie: 'pref=plain' } }); + expect(await blocks(rule, request)).toBe(false); + }); +}); + +describe('bracketed field names', () => { + it.each([ + ['a nested query field', 'get.user.name', 'https://app.test/?user[name]=sample-marker'], + ['an array query field', 'get.id', 'https://app.test/?id[]=sample-marker'], + ['a nested array query field', 'get.user.tags', 'https://app.test/?user[tags][]=sample-marker'], + ])('resolves %s sent in bracket form', async (_label, parameter, url) => { + expect(await blocks(engineFor(parameter), new Request(url))).toBe(true); + }); + + it('resolves a form field sent in bracket form', async () => { + const request = new Request('https://app.test/', { + method: 'POST', + headers: { 'content-type': 'application/x-www-form-urlencoded' }, + body: 'user%5Bname%5D=sample-marker', + }); + expect(await blocks(engineFor('post.user.name'), request)).toBe(true); + }); + + it('resolves the request source in bracket form, from the query or the body', async () => { + const rule = engineFor('request.user.name'); + expect(await blocks(rule, new Request('https://app.test/?user[name]=sample-marker'))).toBe(true); + const form = new Request('https://app.test/', { + method: 'POST', + headers: { 'content-type': 'application/x-www-form-urlencoded' }, + body: 'user%5Bname%5D=sample-marker', + }); + expect(await blocks(rule, form)).toBe(true); + }); + + it('resolves a bracket-form rule against a parser that expanded the brackets', async () => { + expect(await blocks(engineFor('get.user[name]'), shaped({ query: { user: { name: MARKER } } }))).toBe(true); + expect(await blocks(engineFor('get.id[]'), shaped({ query: { id: [MARKER] } }))).toBe(true); + }); + + it('does not treat a different field as the named one', async () => { + const rule = engineFor('get.user.name'); + expect(await blocks(rule, new Request('https://app.test/?user[nickname]=sample-marker'))).toBe(false); + expect(await blocks(rule, new Request('https://app.test/?username=sample-marker'))).toBe(false); + }); +}); + +describe('a structured value larger than the leaf walk', () => { + // More containers than the walk visits, with the marker inside one past the bound. + const oversized = () => { + const items: unknown[] = Array.from({ length: 20_500 }, () => ({})); + items.push({ note: MARKER }); + return JSON.stringify({ items }); + }; + const post = (body: string) => + new Request('https://app.test/', { method: 'POST', headers: { 'content-type': 'application/json' }, body }); + + it('still matches a value past the bound', async () => { + expect(await blocks(engineFor('post.items'), post(oversized()))).toBe(true); + }); + + it('reports that the bound was reached', async () => { + const skips: any[] = []; + const protection: any = await createProtection({ + mode: 'block', + rules: { + firewall: [{ id: 1, title: 'field shape', rule_v2: [{ parameter: 'post.items', match: { type: 'contains', value: 'absent-marker' } }] }], + whitelists: [], + whitelist_keys: {}, + }, + onSkip: (skip: any) => skips.push(skip), + }); + + await protection.fetchGuard()(post(oversized())); + + expect(protection.coverage().skipped['request:container-cap']).toBe(1); + expect(skips).toEqual([expect.objectContaining({ phase: 'request', reason: 'container-cap' })]); + }); + + it('reports the bound on a request it blocks', async () => { + const protection: any = await createProtection({ + mode: 'block', + rules: { + firewall: [{ id: 1, title: 'field shape', rule_v2: [{ parameter: 'post.items', match: { type: 'contains', value: MARKER } }] }], + whitelists: [], + whitelist_keys: {}, + }, + }); + + const blocked = await protection.fetchGuard()(post(oversized())); + + expect(blocked?.status).toBe(403); + expect(protection.coverage().skipped['request:container-cap']).toBe(1); + }); + + it('reports the bound through the standalone fetch middleware', async () => { + const skips: any[] = []; + const guard = createFetchMiddleware( + { + firewall: [{ id: 1, title: 'field shape', rule_v2: [{ parameter: 'post.items', match: { type: 'contains', value: 'absent-marker' } }] }], + whitelists: [], + whitelist_keys: {}, + } as any, + { onSkip: (skip: any) => skips.push(skip) }, + ); + + expect(await guard(post(oversized()))).toBeNull(); + expect(skips).toEqual([{ phase: 'request', reason: 'container-cap' }]); + }); + + it.each(['node', 'express'])('reports the bound through the %s guard', async (guard) => { + const protection: any = await createProtection({ + mode: 'block', + rules: { + firewall: [{ id: 1, title: 'field shape', rule_v2: [{ parameter: 'post.items', match: { type: 'contains', value: 'absent-marker' } }] }], + whitelists: [], + whitelist_keys: {}, + }, + }); + // A body another parser already produced, so neither guard reads the stream. + const req = { + method: 'POST', + url: '/', + originalUrl: '/', + headers: { 'content-type': 'application/json' }, + query: {}, + body: JSON.parse(oversized()), + socket: { remoteAddress: '198.51.100.7' }, + readableEnded: true, + }; + const res: any = { statusCode: 200, setHeader() {}, getHeader() {}, end() {}, status() { return this; }, json() { return this; } }; + let passed = false; + protection[guard]()(req, res, () => { passed = true; }); + + expect(passed).toBe(true); + expect(protection.coverage().skipped['request:container-cap']).toBe(1); + protection.stop(); + }); + + it('reports the bound once per screened response', async () => { + const protection: any = await createProtection({ + mode: 'block', + rules: { firewall: [], whitelists: [], whitelist_keys: {} }, + responseRules: ['first', 'second'].map((id) => ({ + id, + phase: 'response', + action: 'block', + rule_v2: [{ parameter: 'response.body', mutations: ['json_decode'], match: { type: 'contains', value: 'absent-marker' } }], + })), + }); + const response = await protection.screenResponse( + new Response(oversized(), { headers: { 'content-type': 'application/json' } }), + ); + + expect(response.status).toBe(200); + expect(protection.coverage().skipped['response:container-cap']).toBe(1); + }); + + describe('when a decoding mutation reaches it', () => { + // A whole-value matcher reads the decoded structure directly, without the leaf walk, so the decoding + // walk's own bound is the only one this value meets. + const decodedRule = (list: unknown[]) => ({ + rules: { + firewall: [{ + id: 1, + title: 'decoded field', + rule_v2: [{ + parameter: 'post.data', + mutations: ['base64_decode'], + match: { type: 'array_key_value', key: 'list.v', match: { type: 'contains', value: MARKER } }, + }], + }], + whitelists: [], + whitelist_keys: {}, + }, + body: JSON.stringify({ data: { list } }), + }); + const listOf = (padding: number) => { + const list: unknown[] = Array.from({ length: padding }, () => ({})); + list.push({ v: Buffer.from(MARKER).toString('base64') }); + return list; + }; + + it('reports that the decoding bound was reached', async () => { + const { rules, body } = decodedRule(listOf(20_500)); + const protection: any = await createProtection({ mode: 'block', rules }); + await protection.fetchGuard()(post(body)); + expect(protection.coverage().skipped['request:container-cap']).toBe(1); + }); + + it('decodes and matches a value within it, reporting nothing', async () => { + const { rules, body } = decodedRule(listOf(100)); + const protection: any = await createProtection({ mode: 'block', rules }); + expect((await protection.fetchGuard()(post(body)))?.status).toBe(403); + expect(protection.coverage().skipped['request:container-cap']).toBeUndefined(); + }); + }); + + it('reports nothing for a value within the bound', async () => { + const skips: any[] = []; + const protection: any = await createProtection({ + mode: 'block', + rules: { + firewall: [{ id: 1, title: 'field shape', rule_v2: [{ parameter: 'post.items', match: { type: 'contains', value: 'absent-marker' } }] }], + whitelists: [], + whitelist_keys: {}, + }, + onSkip: (skip: any) => skips.push(skip), + }); + + await protection.fetchGuard()(post(JSON.stringify({ items: [{ note: 'plain' }] }))); + + expect(protection.coverage().skipped['request:container-cap']).toBeUndefined(); + expect(skips).toEqual([]); + }); +});