diff --git a/CMakeLists.txt b/CMakeLists.txt index a468e9c..5c4d176 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -222,4 +222,7 @@ add_subdirectory(src/laige-core) # --------------------------------------------------------------------------- if(LAIGE_BUILD_TESTS) add_subdirectory(tests) + # Dev tools that need the built engine library (gated with tests: a + # library-only build does not need them). + add_subdirectory(tools/fuzz) endif() diff --git a/docs/api/json.md b/docs/api/json.md new file mode 100644 index 0000000..45de2a8 --- /dev/null +++ b/docs/api/json.md @@ -0,0 +1,207 @@ +# Bounded JSON parser + serializer (`laige::JsonValue`, `parseJson`) + +The engine's declarative-config format and the home of all engine JSON +(M0-CORE-07; ADR 0003, FR-1.5). Public header: +`src/laige-core/include/laige/json.h`; implementation: +`src/laige-core/json.cpp`. Unit suite: `ctest -R config_json` +(`tests/laige-core/config_json_tests.cpp`). Fuzz target: `json_parse` +(`tools/fuzz/laige-fuzz.cpp`; CTest entry `fuzz_json_parse`). + +ADR 0003's decision: a hand-rolled, *bounded* parser + serializer in +`laige-core`, **no new dependency** (PRD §11 unchanged). The needed +subset — objects, arrays, strings, numbers, booleans, null — is exactly +what this API covers. + +## Quick start + +```cpp +#include + +// Parse a config document (whole view must be consumed). +auto doc = laige::parseJson(text); // Result +if (doc.isError()) { + // Log the NFR-13.3 line (LOG-002) — the error is always + // ErrorCode::MalformedInput (3); no other code is reachable. + engineLog(laige::ErrorCode::MalformedInput, doc.errorText()); + return; +} + +const laige::JsonValue& root = doc.value(); + +// Read a member with a default (findMember is null-safe on any value). +const laige::JsonValue* tick = root.findMember("tick_rate"); +double tickRate = (tick != nullptr && tick->isNumber()) ? tick->asNumber() + : 60.0; + +// Build a document (API manifest tooling, budget files) and serialize. +laige::JsonValue budgets = laige::JsonValue::makeObject(); +budgets.setMember("update", laige::JsonValue::fromNumber(1.5)); +const std::string text2 = laige::serializeJson(budgets); +// parse(serialize(v)) == v always (see Round-trip below). +``` + +## Accepted grammar (RFC 8259, strict) + +Whitespace = `SPACE` / `TAB` / `LF` / `CR`, between tokens only. Values = +object / array / string / number / `true` / `false` / `null`. Strings +accept the six two-character escapes, `\/`, and `\uXXXX`. Numbers follow +RFC 8259 §6 exactly: `[ - ] int [ frac ] [ exp ]` with no leading zeros, +no leading `+`, and at least one digit after `.` and after `e`/`E`. + +Strictness choices above the RFC floor — each returned as +`ErrorCode::MalformedInput` (never a crash, never silent, CORE-008): + +| Rule | Rationale | +|---|---| +| Object member names must be unique (duplicates rejected) | Config hygiene: silent last-wins is a footgun for dev-authored documents. | +| Raw control characters `U+0000..U+001F` in strings rejected | The JSON grammar requires them escaped; `\u0000` escapes are accepted and stored as-is. | +| UTF-8 validated strictly | No overlong encodings, no raw surrogate codepoints, nothing above `U+10FFFF`. | +| Surrogate halves only as `\uD800..\uDBFF` + `\uDC00..\uDFFF` pairs | A lone half has no UTF-8 encoding (ADR 0003: "UTF-8 validated"). | + +Anything else — trailing data after the top-level value, unterminated +values, bad escapes, bad numbers, invalid bytes — is `MalformedInput`. +A failed parse produces **no** `JsonValue` (all-or-nothing). + +## Bounds (ADR 0003) + +`parseJson(input, options)` is bounded by `laige::JsonOptions`: + +| Option | Default | Meaning | +|---|---|---| +| `maxDocumentBytes` | `1 MiB` (`1 << 20`) | the raw input must not exceed this (inclusive) | +| `maxDepth` | `32` | maximum nested container depth (`maxDepth <= 0` rejects all containers) | + +The parser is recursive descent, but recursion depth is bounded by +`maxDepth`, so **no input causes recursion blowup**. Every failure path +is a bounded return — no crash, no partial document. The bounds are +engine-configurable per call (API-006); the defaults are the documented +ones. + +## Number semantics + +- **JSON number → `double`** (ADR 0003). Config values that need exact + integers must lie in `±2^53`, where doubles are exact; beyond that, + integer literals lose low-order bits (e.g. `9007199254740993` stores + `2^53`). That is a documented precision policy, not an error — the + config validation layer (M1) owns range checks. +- **Overflow stores `±inf`** (IEEE 754): a well-formed token that does + not fit a double (e.g. `1e999`) parses to `±inf`. That is a *valid* + parse result; callers consuming or serializing the value must reject + non-finite numbers (see Performance and failure below). NaN is + unreachable from a well-formed document (JSON has no NaN literal). +- **Serializer numbers** are the *shortest correctly rounded decimal*: + the least precision `p` in `1..17` whose `%g` rendering round-trips to + the bit-identical double (`1.5` → `1.5`, `2.0` → `2`, `1e21` → + `1e+21`, `123456.75` → `123456.75`). `-0.0` serializes as `0` (negative + zero is not preserved). The output is deterministic on every platform + (correctly-rounded decimal conversion; round-trip equality is + bit-exact IEEE 754). + +## `JsonValue` + +A parsed or hand-built document. Value semantics: **copy is deep** +(`O(size)`, allocates), **move is `O(1)`**; a moved-from value is `Null`. +The value always owns exactly the payload its kind names — every kind +change releases the old payload (no stale state). + +| Operation | Behavior | Cost | +|---|---|---| +| `kind()` / `isX()` | kind queries | `O(1)` | +| `asBool` / `asNumber` / `asString` / `asArray` / `asObject` | kind accessors; **precondition: matching kind** (debug assert; documented UB in release) | `O(1)` | +| `findMember(name)` | null-safe lookup: `nullptr` when not an object or absent (total on any value) | `O(members)` | +| `hasMember(name)` | `findMember(name) != nullptr` | `O(members)` | +| `setNull` / `setBool` / `setNumber` / `setString` | make this the given value, releasing the old payload | `O(1)` / `O(len)` | +| `append(element)` | array element; **precondition: `isArray()`** | `O(1)` amortized | +| `setMember(name, value)` | replace in place (position preserved) or append; **precondition: `isObject()`** | `O(members)` | +| `operator==` | deep structural equality | `O(size)` | + +**Equality** is deep and structural: objects compare **independent of +member order** (`{"a":1,"b":2} == {"b":2,"a":1}` — configs in different +key order are equal), arrays are **order-sensitive**, and a `Number` +holding NaN compares unequal to itself (IEEE 754 `==`). + +**Ownership and lifetime (CPP-002, CPP-009, CONC-001).** A `JsonValue` +owns its payload and has exactly one owner thread while mutable — it is +**not thread-safe**. A fully constructed value is safe to read from any +thread (no internal synchronization; the same publish contract as +`Result`/`Status`). + +## Serializer + +`serializeJson(value)` emits the canonical compact form: no insignificant +whitespace; strings ASCII-safe (`\b \f \n \r \t` for the printable +controls, `\uXXXX` for the rest, and `\uXXXX` — surrogate pairs above +`U+FFFF` — for every codepoint above `0x7F`); numbers as documented +above; objects and arrays in stored order (for parsed documents: document +order). The output is pure ASCII, so it is byte-identical on every +platform and re-parses to the bit-identical value. + +## Round-trip + +`parse(serialize(v)) == v` for every value whose numbers are finite, and +`serialize(parse(serialize(parse(s)))) == serialize(parse(s))` — the +serializer is idempotent on its own output. The `ConfigJsonRoundTrip` +suite pins both properties on a corpus that includes deep nesting (32), +escaped strings, surrogate pairs, and control characters. + +## Performance + +**Cold path, O(n).** Parsing and serialization are config-load, +manifest-generation, and budget-file work — **never frame or tick loops** +(PERF-003, PERF-007). Time and space are `O(n)` in document bytes; +object member insertion is `O(members)` per insert (`O(m^2)` per object — +negligible at config scale). One allocation per string value and per +array/object container; the serialized document is one `std::string`. + +**Traps:** + +- `serializeJson` **recurses** to the value's nesting depth: bounded by + `JsonOptions::maxDepth` for parsed documents, but a hand-built value of + pathological depth risks stack overflow (build documents at normal + depths; the parser cannot produce deeper values than `maxDepth`). +- Serializing a value containing a `Number` holding NaN or `±inf` is a + documented precondition violation (debug assert; UB in release) — + non-finite values are not representable in JSON. +- `fromString`/`setString` with invalid UTF-8 is a documented + precondition violation (debug assert; UB in release). The parser never + produces invalid UTF-8, so parser-built values are always safe to + serialize. + +## Determinism + +Parsing is a pure function of the input bytes: same input → same value, +same serialization, on every platform (no floating point beyond the +documented `double` number semantics, no platform intrinsics, no +ordering that depends on anything but the input). Object and array +iteration order is document order for parsed values and insertion order +for built values — deterministic in both cases. + +## Errors + +`parseJson` returns `laige::Result`; **every failure is +`ErrorCode::MalformedInput` (3)** — the registry entry for code 3 +already names the ADR 0003 size/depth bounds, so no new codes exist +(CORE-004). `Status::errorText()` / `errorText(code)` renders the +NFR-13.3 five-field line for logging (LOG-002). There is no other +failure channel: discarding the `Result` is a likely logic bug +(CORE-008). + +| Condition | Code | +|---|---| +| Input over `maxDocumentBytes` | `ErrorCode::MalformedInput` (3) | +| Nesting over `maxDepth` | `ErrorCode::MalformedInput` (3) | +| Grammar violation (bad token, escape, number, key) | `ErrorCode::MalformedInput` (3) | +| Duplicate object key | `ErrorCode::MalformedInput` (3) | +| Invalid UTF-8 / lone surrogate / raw control character | `ErrorCode::MalformedInput` (3) | +| Trailing data after the top-level value | `ErrorCode::MalformedInput` (3) | + +## Fuzzing (NFR-8.7, TEST-005, SCALE-005) + +The parser is a malformed-input surface and is fuzzed in CI: the +`json_parse` target feeds `laige::parseJson` a deterministic, +Prng-seeded stream of mutated, truncated, and random byte inputs +(`tools/fuzz/laige-fuzz.cpp`). The bounded run — `laige-fuzz +json_parse --runs=1000` — is registered as the `fuzz_json_parse` CTest +entry and runs in **every build tree**, instrumented in the ASan tree +(the step's Verify gate; PRD §14: fuzz "every commit (bounded), nightly +(long)" — the nightly long-run lane lands with M0-TEST-01). diff --git a/docs/getting-started/building.md b/docs/getting-started/building.md index e7c6aea..9d83809 100644 --- a/docs/getting-started/building.md +++ b/docs/getting-started/building.md @@ -42,7 +42,9 @@ Notes: - `Debug` is the canonical `CMAKE_BUILD_TYPE`; `Release` is supported. - The four rows above the lint row name tools that land in later M0 steps — - `laige-fuzz` (M0-TEST-01), `laige-bench` (M0-CORE-08), `laige-detcheck` + `laige-fuzz` (minimal form from M0-CORE-07: the `json_parse` target and + deterministic bounded runs; M0-TEST-01 extends it with CI lane semantics + and nightly long runs), `laige-bench` (M0-CORE-08), `laige-detcheck` (M0-TOOL-02), target `laige-api` (M0-TOOL-01). Their command forms are fixed here now so later steps cannot drift. - Include-graph lint (M0-CI-03): platform-independent (Python 3 stdlib @@ -156,10 +158,7 @@ pinned set and the NaN/Inf policy): and the fpx16_16 rounding/saturation policy in [docs/api/sim_math.md](../api/sim_math.md), pinned flags via `laige_apply_simmath_policy()`), and the memory pools from M0-CORE-05 - (`include/laige/pools.h`: `laige::ArenaPool` and `laige::Pool` - with `laige::PoolStats` accounting; API contract in - [docs/api/pools.md](../api/pools.md)). -- `tests/laige-core/laige-core_tests` is a CTest link smoke test (a + (`include/laige/pools.h`: `laige::ArenaPool` and `laige::Pool` with `laige::PoolStats` accounting; API contract in [docs/api/pools.md](../api/pools.md)), and the bounded JSON parser + serializer from M0-CORE-07 (`include/laige/json.h`, `json.cpp`: `laige::JsonValue`, `parseJson`, `serializeJson`, `JsonOptions`; API contract in [docs/api/json.md](../api/json.md)).- `tests/laige-core/laige-core_tests` is a CTest link smoke test (a GoogleTest suite since M0-DEP-01) that runs in every build tree above: it verifies the static/shared link and checks the NFR-8.10 policy flags with `static_assert` (a policy violation fails the build). @@ -188,6 +187,16 @@ pinned set and the NaN/Inf policy): `ctest -R pools` (budget exhaustion, reset semantics, and generation-checked stale handles; the stale-handle assert runs in a forked child on the POSIX jobs). +- `config_json` is the M0-CORE-07 CTest entry: a filtered view of the + same `laige-core_tests` executable covering the `ConfigJsonValid`, + `ConfigJsonInvalid`, `ConfigJsonRoundTrip`, `ConfigJsonValue`, and + `ConfigJsonOptions` suites — the step's Verify command is + `ctest -R config_json`. +- `fuzz_json_parse` is the M0-CORE-07 bounded-fuzz CTest entry + (`laige-fuzz json_parse --runs=1000`, registered in `tools/fuzz`): it + runs in every build tree — in the ASan tree it is instrumented and is + the step's sanitizer gate (NFR-8.7; PRD §14: fuzz "every commit + (bounded), nightly (long)"). - Every configure verifies the vendored dependency lock (`cmake/laige-deps-lock.cmake` against `deps.lock`); a tampered or unlisted file under `deps/` fails the configure loudly. GoogleTest is the diff --git a/roadmap/M0-foundations.md b/roadmap/M0-foundations.md index b6180ba..759fc29 100644 --- a/roadmap/M0-foundations.md +++ b/roadmap/M0-foundations.md @@ -573,7 +573,7 @@ No rendering, no physics, no networking yet — `laige-core` only. proof — the step's "period sanity" clause — is a full algebraic proof rather than a sample check, which the GF(2) machinery costs) -- [ ] **M0-CORE-07 · Config JSON: value type + parser** +- [x] **M0-CORE-07 · Config JSON: value type + parser** - **Refs:** FR-1.5 (M1 consumes it), M0-DEC-03; PRD §8.2 (NFR-8.7 fuzz), AGENTS TEST-005 - **Depends:** M0-CORE-01, M0-DEC-03 - **Scope:** @@ -581,8 +581,71 @@ No rendering, no physics, no networking yet — `laige-core` only. - Serializer for round-trip of simple structures. - Fuzz target `json_parse` registered with the fuzz runner. - Unit tests: valid/invalid/malformed corpus (nested depth limit, huge number, truncated input, encoding edge cases). - - **Verify:** `ctest -R config_json` green; `laige-fuzz json_parse --runs=1000` clean under ASan. - - **Size:** ~400 lines + tests (split here if the parser exceeds budget: `M0-CORE-07a` parser, `M0-CORE-07b` serializer+fuzz) + - **Decision (2026-09-11):** hand-rolled bounded parser in + `laige-core` — no dependency (ADR 0003). Recursive descent with an + RAII depth guard; bounds from `JsonOptions` (defaults: 1 MiB document, + depth 32, both inclusive; `maxDepth <= 0` rejects every container). + **Every** parse failure — grammar, bad escapes, duplicate object + keys, raw control characters, invalid UTF-8 (overlong / raw + surrogate / > U+10FFFF), lone surrogate halves in `\uXXXX`, over + size/depth — is `ErrorCode::MalformedInput` (3): no new codes, + because the registry entry for code 3 already names the ADR 0003 + bounds (CORE-004). Numbers: JSON number → `double` via + `std::strtod`; a well-formed overflow token (e.g. `1e999`) stores + ±inf — a *valid parse result*, callers must reject non-finite + (config validation, M1); integer literals beyond ±2^53 lose + precision (documented policy). `JsonValue`: plain members; deep + copy (O(size)), O(1) move with moved-from == Null (documented); + the "owns exactly the payload its kind names" invariant is kept by + `clearPayload()` on every kind change and every assignment. + Equality: deep; objects order-insensitive, arrays order-sensitive, + NaN != NaN. Serializer: canonical compact ASCII (control chars and + every codepoint > 0x7F as `\uXXXX`, surrogate pairs above U+FFFF; + numbers the shortest correctly rounded decimal via a `%.*g` + search over p = 1..17 — deliberately *not* `std::to_chars`, which + AppleClang 15 (macos-14 lane) lacks for floats; `-0.0` → `0`); + `parse(serialize(v)) == v` for all finite values; serializing a + non-finite number is a documented precondition violation. + `laige-fuzz`: the *minimal* deterministic fuzz runner lands in this + step because the Verify gate requires it (Prng-seeded, default + seed `0x1F055EED`, `--runs`/`--seed`, three input modes — mutate / + truncate / random bytes — over a 19-document ASCII corpus; any + Status is acceptable, only a crash fails); target `json_parse` + registered. M0-TEST-01 extends it (CI lane semantics, nightly long + runs, seed documentation). API contract: `docs/api/json.md`. + - **Verify:** `ctest -R config_json` green — 25 GTest cases across + `ConfigJsonValid` (all six kinds; DBL_MAX / denorm_min / ±inf + overflow / 2^53+1 rounding; escapes, surrogate pairs, strict + UTF-8, DEL; containers, document order, depth-32 and 1 MiB + boundary documents), `ConfigJsonInvalid` (empty/trailing data, + truncation, bad numbers, bad escapes, lone surrogates, raw + controls, invalid UTF-8, duplicate keys, depth 33, 1 MiB+2 — + every case asserts `MalformedInput`, not merely an error), + `ConfigJsonRoundTrip` (parse→serialize→parse value stability and + serializer idempotence over a corpus incl. 32-deep nesting and + escaped strings; canonical forms pinned), `ConfigJsonValue` + (factories, deep copy, moved-from Null, in-place replacement, kind + transitions, deep equality incl. NaN != NaN and object + order-insensitivity, total `findMember`, churn), and + `ConfigJsonOptions` (maxDepth 1/2; maxDocumentBytes inclusive bound + and 0). `laige-fuzz json_parse --runs=1000` clean under ASan: the + instrumented `fuzz_json_parse` CTest entry passed in the ASan tree + and a direct `ASAN_OPTIONS=detect_leaks=1:halt_on_error=1` run is + clean (1000 deterministic Prng-driven runs, no crash, no sanitizer + report). Verified locally 2026-09-11: full 14/14 ctest on + GCC 16.2.1 (`build` static, `build-shared` shared, `build-asan` + ASan+UBSan fatal, `build-tsan` TSan `halt_on_error=1`) and a + Clang 22.1.8 tree — zero warnings under the NFR-8.10 policy; + CI will additionally prove MSVC (windows lane) and AppleClang + (macos lane) compilation of the new sources. + - **Size:** 270 lines header (`json.h` — full AGENTS §9 contracts + next to the code) + 826 lines implementation + 620 lines tests + + 208 lines fuzz runner + ~22 lines fuzz CMake + ~12 lines CMake + wiring (over the ~400-line estimate; kept cohesive rather than + split into `M0-CORE-07a/07b`, same pattern as + M0-CORE-01…06: the header carries the API contract and the tests + prove the step's Verify clauses — depth/size bounds, the malformed + corpus, round-trip, value semantics — in one suite) - [ ] **M0-CORE-08 · Budget harness (histogram + budget checks)** - **Refs:** PRD §8.1, CORE-001; AGENTS §12 (benchmark report requirements) diff --git a/src/laige-core/CMakeLists.txt b/src/laige-core/CMakeLists.txt index 5f24c9f..4b747ca 100644 --- a/src/laige-core/CMakeLists.txt +++ b/src/laige-core/CMakeLists.txt @@ -15,10 +15,10 @@ if(LAIGE_BUILD_SHARED) add_library(laige-core SHARED version.cpp errors.cpp logging.cpp - sim_math.cpp sim_math_fixed.cpp) + sim_math.cpp sim_math_fixed.cpp json.cpp) else() add_library(laige-core STATIC version.cpp errors.cpp logging.cpp - sim_math.cpp sim_math_fixed.cpp) + sim_math.cpp sim_math_fixed.cpp json.cpp) endif() laige_apply_engine_policy(laige-core) diff --git a/src/laige-core/include/laige/json.h b/src/laige-core/include/laige/json.h new file mode 100644 index 0000000..3cf73e4 --- /dev/null +++ b/src/laige-core/include/laige/json.h @@ -0,0 +1,270 @@ +// laige-core bounded JSON parser + serializer (M0-CORE-07). +// +// FR-1.5: declarative game config in JSON. ADR 0003 (config JSON +// strategy): a hand-rolled *bounded* parser + serializer in laige-core, +// no new dependency (PRD §11), malformed input -> Status error +// (CORE-008), UTF-8 validated, a serializer for round-trip (used by the +// API manifest tooling and budget files), and a fuzz target `json_parse` +// (NFR-8.7, TEST-005). Full API contract: docs/api/json.md. +// +// --------------------------------------------------------------------------- +// Accepted grammar (RFC 8259, strict) +// --------------------------------------------------------------------------- +// +// whitespace SPACE / TAB / LF / CR, only *between* tokens +// value object / array / string / number / true / false / null +// object '{' [member (',' member)*] '}' member = string ':' value +// array '[' [value (',' value)*] ']' +// string '"' (unescaped-char | escape)* '"' +// escape '"' '\' '/' 'b' 'f' 'n' 'r' 't' | '\u' HEXDIGIT{4} +// number [ '-' ] int [ frac ] [ exp ] +// int = '0' | [1-9] DIGIT* (no leading zeros, +// no '+') +// frac = '.' DIGIT+ +// exp = ('e'|'E') ['+'|'-'] DIGIT+ +// +// Strictness choices above the RFC 8259 floor (each is a documented, +// actionable MalformedInput — never a crash, never silent, CORE-008): +// +// - Object member names must be unique; a duplicate key is rejected. +// (Config hygiene: silent last-wins is a footgun for the +// dev-authored documents this parser exists for.) +// - Raw control characters U+0000..U+001F in strings are rejected +// (the JSON grammar requires them escaped); the "\u0000" escape is +// accepted and stored as-is. +// - UTF-8 is validated strictly: no overlong encodings, no raw +// surrogate codepoints, nothing above U+10FFFF. +// - A "\uD800" high-surrogate escape is valid only when immediately +// followed by a "\uDC00" low-surrogate escape, together forming a +// codepoint above U+FFFF; lone surrogate halves are rejected (they +// have no UTF-8 encoding — ADR 0003 "UTF-8 validated"). +// +// Anything else — trailing data after the top-level value, unterminated +// values, bad escapes, bad numbers, invalid bytes — is MalformedInput. +// A failed parse produces no JsonValue (all-or-nothing, CORE-008). +// +// --------------------------------------------------------------------------- +// Bounds (ADR 0003) +// --------------------------------------------------------------------------- +// +// parseJson is bounded on raw input size and structural depth +// (JsonOptions; the documented defaults are engine-configurable): +// +// maxDocumentBytes 1 MiB (1 << 20) — the raw input must not exceed +// maxDepth 32 — nested container depth +// +// The parser is recursive descent, but recursion depth is bounded by +// maxDepth, so no input causes recursion blowup. Every failure path is +// a bounded return — no crash, no partial document. +// +// --------------------------------------------------------------------------- +// Number semantics (ADR 0003: "JSON number -> double") +// --------------------------------------------------------------------------- +// +// - A JSON number is stored as a `double`. Config values that need +// exact integers must lie in +/-2^53, where doubles are exact; +// beyond that, integer literals silently lose low-order bits. That +// is a *documented precision policy* (this header), not an error — +// the config validation layer (M1) owns range checks. +// - A number token that overflows double converts to +/-inf (IEEE 754; +// e.g. 1e999 -> +inf). That is a *valid parse result* — the token is +// well-formed JSON — and callers serializing or consuming the value +// must reject non-finite numbers (see Misuse warnings). NaN is +// unreachable from a well-formed document (JSON has no NaN literal). +// - serializeJson emits the *shortest correctly rounded decimal* for +// each finite number (the least precision p in 1..17 whose %g +// rendering round-trips to the bit-identical double), so +// parse(serialize(v)) == v always: serialized text re-parses to the +// same value. -0.0 serializes as "0" (negative zero is not +// preserved). +// +// --------------------------------------------------------------------------- +// JsonValue contract +// --------------------------------------------------------------------------- +// +// Ownership (CPP-002, CPP-009): a JsonValue owns its payload (string +// bytes, element vector, or member vector). Value semantics: copy is +// *deep* (O(size), allocates), move is O(1); a moved-from value is Null. +// The value always owns exactly the payload its kind names — every kind +// change releases the old payload (CPP-004: no stale active member). +// +// Threading (CONC-001): a JsonValue has exactly one owner thread while +// mutable and is NOT thread-safe. A fully constructed value is safe to +// read from any thread (no internal synchronization — the same publish +// contract as Result/Status, M0-CORE-01). +// +// Equality (operator==): deep and structural. Objects compare +// independent of member order ({"a":1,"b":2} == {"b":2,"a":1}); arrays +// are order-sensitive. A Number holding NaN compares unequal to itself +// (IEEE 754 ==). +// +// Performance (PERF-003/005, PERF-007): parsing and serialization are +// *cold paths* — config load, manifest generation, budget files — never +// frame or tick loops. Time and space are O(n) in document bytes (object +// member insertion is O(members) per insert, O(m^2) per object — +// negligible at config scale). serializeJson recurses to the value's +// nesting depth: bounded by JsonOptions::maxDepth for parsed documents; +// a hand-built value of pathological depth risks stack overflow (see +// Misuse warnings). +// +// Errors: parseJson returns Result; every failure is +// ErrorCode::MalformedInput — the registry entry for code 3 already +// names the ADR 0003 size/depth bounds, so no new codes (CORE-004). +// Status::errorText() renders the NFR-13.3 line for logging (LOG-002). +// +// --------------------------------------------------------------------------- +// Misuse warnings +// --------------------------------------------------------------------------- +// - fromString(s) / setString(s) require s to be valid UTF-8 (debug +// assert; documented undefined behavior in release). The parser +// never produces invalid UTF-8, so parser-built values are always +// safe to serialize. +// - serializeJson(v) requires v to contain no Number holding NaN or +// +/-inf (debug assert; documented undefined behavior in release). +// Parsed documents can carry +/-inf (number overflow); a config +// loader must reject non-finite values before serializing or +// consuming them. +// - The asX() accessors require kind() == X (debug assert; documented +// undefined behavior in release). findMember/hasMember are the +// null-safe forms for object lookups: nullptr / false when the value +// is not an object or the member is absent (total on any value). +// - append()/setMember() require the matching container kind (debug +// assert; documented undefined behavior in release). +// - Discarding the Result from parseJson is a likely logic bug +// (CORE-008); the API carries no other failure channel. + +#pragma once + +#include +#include +#include +#include +#include +#include + +#include "laige/result.h" + +namespace laige { + +// The six JSON value kinds (RFC 8259). +enum class JsonKind : std::uint8_t { + Null, + Bool, + Number, + String, + Array, + Object, +}; + +// Parse bounds (ADR 0003). The defaults are the documented ones; engine +// code may tighten (or, in principle, loosen) them per parse call. +// maxDepth <= 0 rejects every container; maxDocumentBytes is a bound on +// the raw input bytes. +struct JsonOptions { + std::size_t maxDocumentBytes = static_cast(1u) << 20; // 1 MiB + int maxDepth = 32; +}; + +// A parsed (or hand-built) JSON document. See the preamble for the full +// contract: grammar, bounds, number semantics, ownership, threading, +// equality, performance, and error behavior. +class JsonValue { + public: + // The Null value. + JsonValue() noexcept = default; + + // Value semantics (preamble): copy is deep (O(size), allocates); move is + // O(1) and leaves the moved-from value Null. Copy assignment and move + // assignment release this value's old payload first, so the "owns + // exactly the payload its kind names" invariant holds across + // assignment, not just across kind-changing mutation. + JsonValue(const JsonValue& other) noexcept; + JsonValue& operator=(const JsonValue& other) noexcept; + JsonValue(JsonValue&& other) noexcept; + JsonValue& operator=(JsonValue&& other) noexcept; + + // Factories (the engine builds with -fno-exceptions: a failed + // allocation terminates the process, it never throws). + [[nodiscard]] static JsonValue fromBool(bool value) noexcept; + [[nodiscard]] static JsonValue fromNumber(double value) noexcept; + [[nodiscard]] static JsonValue fromString(std::string_view value) noexcept; + [[nodiscard]] static JsonValue makeArray() noexcept; + [[nodiscard]] static JsonValue makeObject() noexcept; + + // Kind queries. + [[nodiscard]] JsonKind kind() const noexcept { return kind_; } + [[nodiscard]] bool isNull() const noexcept { return kind_ == JsonKind::Null; } + [[nodiscard]] bool isBool() const noexcept { return kind_ == JsonKind::Bool; } + [[nodiscard]] bool isNumber() const noexcept { return kind_ == JsonKind::Number; } + [[nodiscard]] bool isString() const noexcept { return kind_ == JsonKind::String; } + [[nodiscard]] bool isArray() const noexcept { return kind_ == JsonKind::Array; } + [[nodiscard]] bool isObject() const noexcept { return kind_ == JsonKind::Object; } + + // Kind accessors. Precondition: the matching kind (debug assert; + // documented undefined behavior in release). + [[nodiscard]] bool asBool() const noexcept; + [[nodiscard]] double asNumber() const noexcept; + [[nodiscard]] std::string_view asString() const noexcept; + [[nodiscard]] const std::vector& asArray() const noexcept; + [[nodiscard]] const std::vector>& + asObject() const noexcept; + + // Null-safe object lookups (total on any value): nullptr / false when + // this is not an object or the member is absent. + [[nodiscard]] const JsonValue* findMember(std::string_view name) const noexcept; + [[nodiscard]] bool hasMember(std::string_view name) const noexcept; + + // Mutations. The setX() forms make this the given kind/value, releasing + // the old payload first. append()/setMember() require the matching + // container kind (debug assert; documented undefined behavior in + // release). setMember() replaces an existing member in place (position + // preserved) or appends it. + void setNull() noexcept; + void setBool(bool value) noexcept; + void setNumber(double value) noexcept; + void setString(std::string_view value) noexcept; + void append(JsonValue element); + void setMember(std::string_view name, JsonValue value); + + // Deep structural equality (see the preamble: objects order-insensitive, + // arrays order-sensitive, NaN != NaN). + [[nodiscard]] bool operator==(const JsonValue& other) const noexcept; + [[nodiscard]] bool operator!=(const JsonValue& other) const noexcept { + return !(*this == other); + } + + private: + // Releases the payload of the current kind (destroy + release), so the + // value always owns exactly the payload its kind_ names (CPP-004). + void clearPayload() noexcept; + + JsonKind kind_ = JsonKind::Null; + bool boolean_ = false; // Bool + double number_ = 0.0; // Number + std::string str_; // String: valid + // UTF-8, may hold NUL + std::vector arr_; // Array: document order + std::vector> members_; // Object: document + // order, unique keys +}; + +// Parses exactly one JSON document from `input` (the whole view must be +// consumed; trailing non-whitespace is MalformedInput). Bounded by +// `options` (defaults: 1 MiB, depth 32 — see the preamble). Every failure +// is ErrorCode::MalformedInput; a failed parse produces no value. +[[nodiscard]] +Result parseJson(std::string_view input, + JsonOptions options = {}); + +// Serializes `value` to canonical compact JSON (no insignificant +// whitespace): strings ASCII-safe (\uXXXX for control characters and +// every codepoint above 0x7F; the two-character escapes for the six +// printable ones), numbers shortest-round-trip decimal (see the +// preamble), objects and arrays in stored order. Precondition: no Number +// holding NaN or +/-inf anywhere in the value (debug assert; documented +// undefined behavior in release). Cold path: allocates one output string +// plus recursive calls per nesting level. +[[nodiscard]] +std::string serializeJson(const JsonValue& value); + +} // namespace laige diff --git a/src/laige-core/json.cpp b/src/laige-core/json.cpp new file mode 100644 index 0000000..0b230aa --- /dev/null +++ b/src/laige-core/json.cpp @@ -0,0 +1,826 @@ +// laige-core bounded JSON parser + serializer (M0-CORE-07). +// +// Implementation of include/laige/json.h. The parser is a strict, +// depth-bounded recursive descent over RFC 8259 (grammar, bounds, and +// strictness choices are documented in the header preamble); every +// failure is a bounded return of ErrorCode::MalformedInput (CORE-008: +// never a crash, never silent). The serializer emits the canonical +// compact form documented there. Fuzz target: `json_parse` +// (tools/fuzz/laige-fuzz.cpp, NFR-8.7). + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "laige/json.h" + +namespace laige { + +namespace { + +// --------------------------------------------------------------------------- +// Strict UTF-8 decoding (shared by the parser, the serializer, and the +// string factories). Validates: no overlong encodings, no raw surrogate +// codepoints, nothing above U+10FFFF. On success advances `pos` past the +// sequence and sets `cp`. +// --------------------------------------------------------------------------- +bool decodeUtf8(std::string_view in, std::size_t& pos, std::uint32_t& cp) { + const std::size_t n = in.size(); + const auto byteAt = [&](std::size_t i) -> std::uint32_t { + return static_cast(in[i]); + }; + const std::size_t p0 = pos; + if (p0 >= n) return false; + const std::uint32_t b0 = byteAt(p0); + if (b0 < 0x80u) { + cp = b0; + pos = p0 + 1; + return true; + } + if (b0 >= 0xC2u && b0 <= 0xDFu) { + if (p0 + 1 >= n) return false; + const std::uint32_t b1 = byteAt(p0 + 1); + if ((b1 & 0xC0u) != 0x80u) return false; + cp = ((b0 & 0x1Fu) << 6) | (b1 & 0x3Fu); + pos = p0 + 2; + return true; + } + if (b0 >= 0xE0u && b0 <= 0xEFu) { + if (p0 + 2 >= n) return false; + const std::uint32_t b1 = byteAt(p0 + 1); + const std::uint32_t b2 = byteAt(p0 + 2); + if ((b1 & 0xC0u) != 0x80u || (b2 & 0xC0u) != 0x80u) return false; + cp = ((b0 & 0x0Fu) << 12) | ((b1 & 0x3Fu) << 6) | (b2 & 0x3Fu); + if (cp < 0x800u) return false; // overlong + if (cp >= 0xD800u && cp <= 0xDFFFu) return false; // raw surrogate + pos = p0 + 3; + return true; + } + if (b0 >= 0xF0u && b0 <= 0xF4u) { + if (p0 + 3 >= n) return false; + const std::uint32_t b1 = byteAt(p0 + 1); + const std::uint32_t b2 = byteAt(p0 + 2); + const std::uint32_t b3 = byteAt(p0 + 3); + if ((b1 & 0xC0u) != 0x80u || (b2 & 0xC0u) != 0x80u || + (b3 & 0xC0u) != 0x80u) { + return false; + } + cp = ((b0 & 0x07u) << 18) | ((b1 & 0x3Fu) << 12) | + ((b2 & 0x3Fu) << 6) | (b3 & 0x3Fu); + if (cp < 0x10000u) return false; // overlong (b0 <= 0xF4 => cp <= 0x10FFFF) + pos = p0 + 4; + return true; + } + return false; // 0x80-0xC1 and 0xF5-0xFF: never valid +} + +bool isValidUtf8(std::string_view s) { + std::size_t pos = 0; + while (pos < s.size()) { + std::uint32_t cp = 0; + if (!decodeUtf8(s, pos, cp)) return false; + } + return true; +} + +// UTF-8 encode a codepoint (precondition: cp <= 0x10FFFF, never a +// surrogate). +void appendUtf8(std::string& out, std::uint32_t cp) { + if (cp < 0x80u) { + out.push_back(static_cast(cp)); + return; + } + if (cp < 0x800u) { + out.push_back(static_cast(0xC0u | (cp >> 6))); + out.push_back(static_cast(0x80u | (cp & 0x3Fu))); + return; + } + if (cp < 0x10000u) { + out.push_back(static_cast(0xE0u | (cp >> 12))); + out.push_back(static_cast(0x80u | ((cp >> 6) & 0x3Fu))); + out.push_back(static_cast(0x80u | (cp & 0x3Fu))); + return; + } + out.push_back(static_cast(0xF0u | (cp >> 18))); + out.push_back(static_cast(0x80u | ((cp >> 12) & 0x3Fu))); + out.push_back(static_cast(0x80u | ((cp >> 6) & 0x3Fu))); + out.push_back(static_cast(0x80u | (cp & 0x3Fu))); +} + +// --------------------------------------------------------------------------- +// Parser state. All helpers below take this by reference; every failure +// sets `failed` (idempotent) and returns false. +// --------------------------------------------------------------------------- +struct ParserState { + std::string_view in; + std::size_t pos = 0; + int maxDepth = 0; + int depth = 0; // currently open containers (0 at top level) + bool failed = false; +}; + +void fail(ParserState& p) { p.failed = true; } + +bool isDigit(char c) { return c >= '0' && c <= '9'; } + +void skipWhitespace(ParserState& p) { + while (p.pos < p.in.size()) { + const char c = p.in[p.pos]; + if (c != ' ' && c != '\t' && c != '\n' && c != '\r') break; + ++p.pos; + } +} + +bool consume(ParserState& p, char c) { + if (p.pos < p.in.size() && p.in[p.pos] == c) { + ++p.pos; + return true; + } + return false; +} + +// Bounded container-depth increment (ADR 0003: no recursion blowup). +struct DepthGuard { + int& depth; + bool ok; + explicit DepthGuard(int& d, int max) : depth(d) { + ok = (d + 1 <= max); + if (ok) ++d; + } + ~DepthGuard() { + if (ok) --depth; + } + explicit operator bool() const { return ok; } +}; + +bool parseLiteral(ParserState& p, const char* lit, JsonValue& out) { + // out is always a fresh (Null) local at the call site. + const std::size_t n = std::char_traits::length(lit); + if (p.in.compare(p.pos, n, lit) != 0) { + fail(p); + return false; + } + p.pos += n; + if (lit[0] == 't') { + out = JsonValue::fromBool(true); + } else if (lit[0] == 'f') { + out = JsonValue::fromBool(false); + } else { + out = JsonValue(); // "null" + } + return true; +} + +// RFC 8259 §6 number grammar: [ '-' ] int [ frac ] [ exp ], with the +// strict sub-rules (no leading zeros, at least one digit after '.' and +// after 'e'/'E'). The token is converted with std::strtod; an overflow +// token (e.g. 1e999) stores +/-inf (documented in the header). +bool parseNumber(ParserState& p, JsonValue& out) { + const std::size_t start = p.pos; + if (p.in[p.pos] == '-') ++p.pos; + if (p.pos >= p.in.size() || !isDigit(p.in[p.pos])) { + fail(p); + return false; + } + if (p.in[p.pos] == '0') { + ++p.pos; + if (p.pos < p.in.size() && isDigit(p.in[p.pos])) { + fail(p); // leading zero ("01") + return false; + } + } else { + while (p.pos < p.in.size() && isDigit(p.in[p.pos])) ++p.pos; + } + if (p.pos < p.in.size() && p.in[p.pos] == '.') { + ++p.pos; + if (p.pos >= p.in.size() || !isDigit(p.in[p.pos])) { + fail(p); // "." with no fractional digits + return false; + } + while (p.pos < p.in.size() && isDigit(p.in[p.pos])) ++p.pos; + } + if (p.pos < p.in.size() && (p.in[p.pos] == 'e' || p.in[p.pos] == 'E')) { + ++p.pos; + if (p.pos < p.in.size() && (p.in[p.pos] == '+' || p.in[p.pos] == '-')) { + ++p.pos; + } + if (p.pos >= p.in.size() || !isDigit(p.in[p.pos])) { + fail(p); // 'e' with no exponent digits + return false; + } + while (p.pos < p.in.size() && isDigit(p.in[p.pos])) ++p.pos; + } + const std::string token(p.in.data() + start, p.in.data() + p.pos); + out = JsonValue::fromNumber(std::strtod(token.c_str(), nullptr)); + return true; +} + +bool parseHex4(ParserState& p, std::uint32_t& cp) { + std::uint32_t v = 0; + for (int i = 0; i < 4; ++i) { + if (p.pos >= p.in.size()) { + fail(p); + return false; + } + const char c = p.in[p.pos++]; + std::uint32_t d; + if (c >= '0' && c <= '9') { + d = static_cast(c - '0'); + } else if (c >= 'a' && c <= 'f') { + d = static_cast(c - 'a') + 10u; + } else if (c >= 'A' && c <= 'F') { + d = static_cast(c - 'A') + 10u; + } else { + fail(p); + return false; + } + v = (v << 4) | d; + } + cp = v; + return true; +} + +// A JSON string: strict UTF-8, control characters rejected raw, +// surrogate pairs only. Builds a local decoded buffer; on success the +// result goes to `out` via setString (out is a fresh local at the call +// site). +bool parseString(ParserState& p, JsonValue& out) { + consume(p, '"'); + std::string decoded; + for (;;) { + if (p.pos >= p.in.size()) { + fail(p); // unterminated + return false; + } + const char c = p.in[p.pos]; + if (c == '"') { + ++p.pos; + out.setString(std::string_view(decoded)); + return true; + } + if (c == '\\') { + ++p.pos; + if (p.pos >= p.in.size()) { + fail(p); + return false; + } + const char e = p.in[p.pos]; + switch (e) { + case '"': + decoded.push_back('"'); + ++p.pos; + continue; + case '\\': + decoded.push_back('\\'); + ++p.pos; + continue; + case '/': + decoded.push_back('/'); + ++p.pos; + continue; + case 'b': + decoded.push_back('\b'); + ++p.pos; + continue; + case 'f': + decoded.push_back('\f'); + ++p.pos; + continue; + case 'n': + decoded.push_back('\n'); + ++p.pos; + continue; + case 'r': + decoded.push_back('\r'); + ++p.pos; + continue; + case 't': + decoded.push_back('\t'); + ++p.pos; + continue; + case 'u': { + ++p.pos; + std::uint32_t hi = 0; + if (!parseHex4(p, hi)) return false; + if (hi >= 0xD800u && hi <= 0xDBFFu) { + // High surrogate: a low surrogate half must immediately follow. + if (p.pos + 1 >= p.in.size() || p.in[p.pos] != '\\' || + p.in[p.pos + 1] != 'u') { + fail(p); + return false; // lone high surrogate + } + p.pos += 2; + std::uint32_t lo = 0; + if (!parseHex4(p, lo)) return false; + if (lo < 0xDC00u || lo > 0xDFFFu) { + fail(p); + return false; // not a low surrogate + } + appendUtf8(decoded, + 0x10000u + ((hi - 0xD800u) << 10) + (lo - 0xDC00u)); + } else if (hi >= 0xDC00u && hi <= 0xDFFFu) { + fail(p); + return false; // lone low surrogate + } else { + appendUtf8(decoded, hi); + } + continue; + } + default: + fail(p); + return false; + } + } + // Raw UTF-8 sequence (strict decode); control characters rejected. + const std::size_t start = p.pos; + std::uint32_t cp = 0; + if (!decodeUtf8(p.in, p.pos, cp)) { + fail(p); + return false; + } + if (cp < 0x20u) { + fail(p); // raw control character (JSON grammar requires \uXXXX) + return false; + } + decoded.append(p.in.data() + start, p.pos - start); + } +} + +bool parseObjectMembers(ParserState& p, JsonValue& obj); +bool parseValue(ParserState& p, JsonValue& out); + +bool parseObject(ParserState& p, JsonValue& out) { + consume(p, '{'); // guaranteed by the caller + out = JsonValue::makeObject(); + DepthGuard guard(p.depth, p.maxDepth); + if (!guard) { + fail(p); // depth limit + return false; + } + skipWhitespace(p); + if (consume(p, '}')) return true; + return parseObjectMembers(p, out); +} + +bool parseObjectMembers(ParserState& p, JsonValue& obj) { + for (;;) { + JsonValue key; + if (!parseString(p, key)) return false; // keys must be quoted strings + skipWhitespace(p); + if (!consume(p, ':')) { + fail(p); + return false; + } + JsonValue value; + if (!parseValue(p, value)) return false; + if (obj.findMember(key.asString()) != nullptr) { + fail(p); // duplicate key (strictness choice, see the header) + return false; + } + obj.setMember(key.asString(), std::move(value)); + skipWhitespace(p); + if (consume(p, ',')) continue; + if (consume(p, '}')) return true; + fail(p); + return false; + } +} + +bool parseArray(ParserState& p, JsonValue& out) { + consume(p, '['); // guaranteed by the caller + out = JsonValue::makeArray(); + DepthGuard guard(p.depth, p.maxDepth); + if (!guard) { + fail(p); // depth limit + return false; + } + skipWhitespace(p); + if (consume(p, ']')) return true; + for (;;) { + JsonValue element; + if (!parseValue(p, element)) return false; + out.append(std::move(element)); + skipWhitespace(p); + if (consume(p, ',')) continue; + if (consume(p, ']')) return true; + fail(p); + return false; + } +} + +bool parseValue(ParserState& p, JsonValue& out) { + skipWhitespace(p); + if (p.pos >= p.in.size()) { + fail(p); // empty / truncated + return false; + } + const char c = p.in[p.pos]; + switch (c) { + case '{': + return parseObject(p, out); + case '[': + return parseArray(p, out); + case '"': + return parseString(p, out); + case 't': + return parseLiteral(p, "true", out); + case 'f': + return parseLiteral(p, "false", out); + case 'n': + return parseLiteral(p, "null", out); + default: + if (c == '-' || isDigit(c)) return parseNumber(p, out); + fail(p); + return false; + } +} + +// --------------------------------------------------------------------------- +// Serializer (canonical compact form). ASCII-safe strings; shortest +// correctly rounded decimal numbers; stored order for arrays/objects. +// --------------------------------------------------------------------------- +constexpr char kHexDigits[] = "0123456789abcdef"; + +void appendHex4(std::string& out, std::uint32_t v) { + out.push_back(kHexDigits[(v >> 12) & 0xFu]); + out.push_back(kHexDigits[(v >> 8) & 0xFu]); + out.push_back(kHexDigits[(v >> 4) & 0xFu]); + out.push_back(kHexDigits[v & 0xFu]); +} + +void appendEscapedString(std::string& out, std::string_view s) { + out.push_back('"'); + std::size_t i = 0; + while (i < s.size()) { + const std::uint8_t b = static_cast(s[i]); + if (b < 0x20u) { + switch (b) { + case 0x08u: + out += "\\b"; + break; + case 0x0Cu: + out += "\\f"; + break; + case 0x0Au: + out += "\\n"; + break; + case 0x0Du: + out += "\\r"; + break; + case 0x09u: + out += "\\t"; + break; + default: + out += "\\u00"; + out.push_back(kHexDigits[(b >> 4) & 0xFu]); + out.push_back(kHexDigits[b & 0xFu]); + break; + } + ++i; + continue; + } + if (b >= 0x80u) { + // Canonical form escapes all codepoints above 0x7F (\uXXXX, + // surrogate pairs above U+FFFF) so the output is pure ASCII and + // re-parses bit-identically on every platform. + std::uint32_t cp = 0; + if (!decodeUtf8(s, i, cp)) { + // Unreachable: JsonValue strings are valid UTF-8 (the parser + // guarantees it; fromString/setString assert it). + assert(false && "serializeJson: string is not valid UTF-8"); + return; + } + if (cp > 0xFFFFu) { + const std::uint32_t v = cp - 0x10000u; + out += "\\u"; + appendHex4(out, 0xD800u + (v >> 10)); + out += "\\u"; + appendHex4(out, 0xDC00u + (v & 0x3FFu)); + } else { + out += "\\u"; + appendHex4(out, cp); + } + continue; + } + if (b == '"' || b == '\\') { + out.push_back('\\'); + out.push_back(static_cast(b)); + } else { + out.push_back(static_cast(b)); + } + ++i; + } + out.push_back('"'); +} + +// Shortest correctly rounded decimal: the least precision p in 1..17 +// whose %g rendering round-trips to the bit-identical double. 17 +// significant digits always suffice for a finite IEEE 754 double, so the +// loop cannot fall through; the tail is a defensive fallback only. +void appendNumber(std::string& out, double v) { + assert(std::isfinite(v) && + "serializeJson: non-finite number is not serializable JSON"); + if (v == 0.0) { + out += "0"; // -0.0 serializes as "0" (documented) + return; + } + char buf[32]; + for (int precision = 1; precision <= 17; ++precision) { + const int n = std::snprintf(buf, sizeof(buf), "%.*g", precision, v); + if (n <= 0 || static_cast(n) >= sizeof(buf)) continue; + if (std::strtod(buf, nullptr) == v) { + out.append(buf, static_cast(n)); + return; + } + } + const int n = std::snprintf(buf, sizeof(buf), "%.17g", v); + out.append(buf, n < 0 ? 0 : static_cast(n)); +} + +void appendValue(std::string& out, const JsonValue& v) { + switch (v.kind()) { + case JsonKind::Null: + out += "null"; + break; + case JsonKind::Bool: + out += v.asBool() ? "true" : "false"; + break; + case JsonKind::Number: + appendNumber(out, v.asNumber()); + break; + case JsonKind::String: + appendEscapedString(out, v.asString()); + break; + case JsonKind::Array: { + out.push_back('['); + const std::vector& elements = v.asArray(); + for (std::size_t i = 0; i < elements.size(); ++i) { + if (i != 0) out.push_back(','); + appendValue(out, elements[i]); + } + out.push_back(']'); + break; + } + case JsonKind::Object: { + out.push_back('{'); + const std::vector>& members = + v.asObject(); + for (std::size_t i = 0; i < members.size(); ++i) { + if (i != 0) out.push_back(','); + appendEscapedString(out, members[i].first); + out.push_back(':'); + appendValue(out, members[i].second); + } + out.push_back('}'); + break; + } + } +} + +} // namespace + +// --------------------------------------------------------------------------- +// JsonValue member implementations +// --------------------------------------------------------------------------- +void JsonValue::clearPayload() noexcept { + str_ = std::string(); + arr_ = std::vector(); + members_ = std::vector>(); +} + +JsonValue::JsonValue(const JsonValue& other) noexcept + : kind_(other.kind_), + boolean_(other.boolean_), + number_(other.number_), + str_(other.str_), + arr_(other.arr_), + members_(other.members_) {} + +JsonValue& JsonValue::operator=(const JsonValue& other) noexcept { + if (this != &other) { + clearPayload(); // release the old kind's payload before the kind change + kind_ = other.kind_; + boolean_ = other.boolean_; + number_ = other.number_; + str_ = other.str_; + arr_ = other.arr_; + members_ = other.members_; + } + return *this; +} + +JsonValue::JsonValue(JsonValue&& other) noexcept + : kind_(other.kind_), + boolean_(other.boolean_), + number_(other.number_), + str_(std::move(other.str_)), + arr_(std::move(other.arr_)), + members_(std::move(other.members_)) { + // Moved-from value is Null (preamble); the moved payloads are already + // empty (std::string/vector move leaves the source empty). + other.kind_ = JsonKind::Null; + other.boolean_ = false; + other.number_ = 0.0; +} + +JsonValue& JsonValue::operator=(JsonValue&& other) noexcept { + if (this != &other) { + clearPayload(); + kind_ = other.kind_; + boolean_ = other.boolean_; + number_ = other.number_; + str_ = std::move(other.str_); + arr_ = std::move(other.arr_); + members_ = std::move(other.members_); + other.kind_ = JsonKind::Null; + other.boolean_ = false; + other.number_ = 0.0; + } + return *this; +} + +JsonValue JsonValue::fromBool(bool value) noexcept { + JsonValue v; + v.kind_ = JsonKind::Bool; + v.boolean_ = value; + return v; +} + +JsonValue JsonValue::fromNumber(double value) noexcept { + JsonValue v; + v.kind_ = JsonKind::Number; + v.number_ = value; + return v; +} + +JsonValue JsonValue::fromString(std::string_view value) noexcept { + JsonValue v; + v.kind_ = JsonKind::String; + v.str_ = std::string(value); + // Precondition (header): valid UTF-8; debug assert, documented UB in + // release. + assert(isValidUtf8(v.str_) && + "JsonValue::fromString: string is not valid UTF-8"); + return v; +} + +JsonValue JsonValue::makeArray() noexcept { + JsonValue v; + v.kind_ = JsonKind::Array; + return v; +} + +JsonValue JsonValue::makeObject() noexcept { + JsonValue v; + v.kind_ = JsonKind::Object; + return v; +} + +bool JsonValue::asBool() const noexcept { + assert(kind_ == JsonKind::Bool && "asBool() called on a non-bool value"); + return boolean_; +} + +double JsonValue::asNumber() const noexcept { + assert(kind_ == JsonKind::Number && + "asNumber() called on a non-number value"); + return number_; +} + +std::string_view JsonValue::asString() const noexcept { + assert(kind_ == JsonKind::String && + "asString() called on a non-string value"); + return std::string_view(str_); +} + +const std::vector& JsonValue::asArray() const noexcept { + assert(kind_ == JsonKind::Array && "asArray() called on a non-array value"); + return arr_; +} + +const std::vector>& JsonValue::asObject() + const noexcept { + assert(kind_ == JsonKind::Object && + "asObject() called on a non-object value"); + return members_; +} + +const JsonValue* JsonValue::findMember(std::string_view name) const noexcept { + if (kind_ != JsonKind::Object) return nullptr; + for (const auto& [key, value] : members_) { + if (key == name) return &value; + } + return nullptr; +} + +bool JsonValue::hasMember(std::string_view name) const noexcept { + return findMember(name) != nullptr; +} + +void JsonValue::setNull() noexcept { + clearPayload(); + kind_ = JsonKind::Null; + boolean_ = false; + number_ = 0.0; +} + +void JsonValue::setBool(bool value) noexcept { + clearPayload(); + kind_ = JsonKind::Bool; + boolean_ = value; +} + +void JsonValue::setNumber(double value) noexcept { + clearPayload(); + kind_ = JsonKind::Number; + number_ = value; +} + +void JsonValue::setString(std::string_view value) noexcept { + clearPayload(); + kind_ = JsonKind::String; + str_ = std::string(value); + // Precondition (header): valid UTF-8; debug assert, documented UB in + // release. + assert(isValidUtf8(str_) && "JsonValue::setString: not valid UTF-8"); +} + +void JsonValue::append(JsonValue element) { + assert(kind_ == JsonKind::Array && + "append() called on a non-array value"); + arr_.push_back(std::move(element)); +} + +void JsonValue::setMember(std::string_view name, JsonValue value) { + assert(kind_ == JsonKind::Object && + "setMember() called on a non-object value"); + for (auto& [key, memberValue] : members_) { + if (key == name) { + memberValue = std::move(value); // replace in place + return; + } + } + members_.emplace_back(std::string(name), std::move(value)); +} + +bool JsonValue::operator==(const JsonValue& other) const noexcept { + if (kind_ != other.kind_) return false; + switch (kind_) { + case JsonKind::Null: + return true; + case JsonKind::Bool: + return boolean_ == other.boolean_; + case JsonKind::Number: + return number_ == other.number_; // NaN != NaN (IEEE 754) + case JsonKind::String: + return str_ == other.str_; + case JsonKind::Array: + return arr_ == other.arr_; + case JsonKind::Object: { + if (members_.size() != other.members_.size()) return false; + for (const auto& [key, value] : members_) { + const JsonValue* otherValue = other.findMember(key); + if (otherValue == nullptr || *otherValue != value) return false; + } + return true; + } + } + return false; // unreachable: every JsonKind is handled above +} + +// --------------------------------------------------------------------------- +// parseJson / serializeJson +// --------------------------------------------------------------------------- +Result parseJson(std::string_view input, JsonOptions options) { + if (input.size() > options.maxDocumentBytes) { + return Result::failure(ErrorCode::MalformedInput); + } + ParserState p; + p.in = input; + p.maxDepth = options.maxDepth; + JsonValue out; + if (!parseValue(p, out)) { + return Result::failure(ErrorCode::MalformedInput); + } + skipWhitespace(p); + if (p.pos != p.in.size()) { + return Result::failure(ErrorCode::MalformedInput); // trailing + } + return Result::success(std::move(out)); +} + +std::string serializeJson(const JsonValue& value) { + std::string out; + out.reserve(128); + appendValue(out, value); + return out; +} + +} // namespace laige diff --git a/tests/laige-core/CMakeLists.txt b/tests/laige-core/CMakeLists.txt index cc1c622..957f4eb 100644 --- a/tests/laige-core/CMakeLists.txt +++ b/tests/laige-core/CMakeLists.txt @@ -13,7 +13,8 @@ set(LAIGE_CORE_TEST_SOURCES laige-core_tests.cpp result_status_tests.cpp logging_tests.cpp math_float_tests.cpp math_fixed_tests.cpp pools_tests.cpp - prng_tests.cpp) + prng_tests.cpp + config_json_tests.cpp) # M0-CORE-02: the test-only allocation counter overrides the global # operator new/new[]; the sanitizer runtimes define their own new/delete # (strong symbols in the Clang/GCC TSan runtime archives, interposed by @@ -109,10 +110,20 @@ add_test(NAME prng COMMAND laige-core_tests --gtest_filter=PrngGolden.*:PrngRange.*:PrngFloat01.*:PrngSubstreams.*:PrngPeriod.*) +# M0-CORE-07: bounded JSON parser + serializer (ADR 0003). The step's +# Verify command is `ctest -R config_json`; this entry selects exactly +# the ConfigJson* suites from the shared laige-core_tests executable. +# The fuzz gate of the step is the `fuzz_json_parse` entry registered in +# tools/fuzz (runs instrumented in the ASan tree). +add_test(NAME config_json + COMMAND laige-core_tests + --gtest_filter=ConfigJson*) + if(LAIGE_TSAN) # Make the first data race report fatal to the test process (NFR-8.2), # so ctest fails loudly on any TSan report. set_tests_properties( laige-core_tests result_status logging math_float math_fixed pools prng + config_json PROPERTIES ENVIRONMENT "TSAN_OPTIONS=halt_on_error=1") endif() diff --git a/tests/laige-core/config_json_tests.cpp b/tests/laige-core/config_json_tests.cpp new file mode 100644 index 0000000..75cf38e --- /dev/null +++ b/tests/laige-core/config_json_tests.cpp @@ -0,0 +1,620 @@ +// laige-core bounded JSON parser + serializer suite (M0-CORE-07). +// +// Step Verify scope (roadmap/M0-foundations.md): +// - `ctest -R config_json` green (suites: ConfigJsonValid, +// ConfigJsonInvalid, ConfigJsonRoundTrip, ConfigJsonValue, +// ConfigJsonOptions) +// - `laige-fuzz json_parse --runs=1000` clean under ASan — the +// `fuzz_json_parse` CTest entry (tools/fuzz), which the ASan tree +// runs instrumented. +// +// Coverage per the step: the valid / invalid / malformed corpus — +// nested depth limits, huge numbers, truncated input, encoding edge +// cases (strict UTF-8, surrogate pairs, control characters, escapes) — +// plus serializer round-trip, canonical forms, and JsonValue value +// semantics. Non-ASCII and control bytes are built explicitly with the +// Raw() helper: narrow-literal \x escapes greedily consume hex digits +// and are range-checked against `char` (both rejected under -Wall), and +// explicit bytes keep the test platform-portable (the corpus in +// tools/fuzz follows the same convention). + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "gtest/gtest.h" +#include "laige/errors.h" +#include "laige/json.h" + +// --------------------------------------------------------------------------- +// NFR-8.10 policy self-checks (compile-time; a violation fails the build) +// --------------------------------------------------------------------------- + +#if defined(__cpp_exceptions) +static_assert(false, + "config_json_tests must be built with exceptions disabled " + "(NFR-8.10); see laige_apply_engine_policy()."); +#elif defined(__EXCEPTIONS) && __EXCEPTIONS +static_assert(false, + "config_json_tests must be built with exceptions disabled " + "(NFR-8.10); see laige_apply_engine_policy()."); +#endif + +#if defined(__cpp_rtti) && __cpp_rtti +static_assert(false, + "config_json_tests must be built with RTTI disabled " + "(NFR-8.10); see laige_apply_engine_policy()."); +#endif + +namespace { + +using laige::ErrorCode; +using laige::JsonOptions; +using laige::JsonValue; +using laige::Result; + +Result Parse(std::string_view text, + const JsonOptions& options = {}) { + return laige::parseJson(text, options); +} + +// Every parse failure must be MalformedInput: the registry entry for +// code 3 names the ADR 0003 size/depth bounds, so no other code is +// reachable from parseJson (CORE-004). +void ExpectMalformed(std::string_view text, const JsonOptions& options = {}) { + const Result r = Parse(text, options); + EXPECT_TRUE(r.isError()) << "expected a parse failure for: " << text; + if (r.isError()) { + EXPECT_EQ(ErrorCode::MalformedInput, r.error()) << "for: " << text; + } +} + +// `depth` nested arrays around a null leaf ("[[[null]]]" at depth 3). +std::string NestedArray(int depth) { + std::string s(static_cast(depth), '['); + s += "null"; + s.append(static_cast(depth), ']'); + return s; +} + +// Explicit raw bytes (see the file header for why \x escapes are not +// used for these). +std::string Raw(std::initializer_list bytes) { + std::string s; + for (const std::uint16_t b : bytes) s.push_back(static_cast(b)); + return s; +} + +constexpr std::size_t kMiB = static_cast(1u) << 20; + +} // namespace + +// --------------------------------------------------------------------------- +// ConfigJsonValid — the parser accepts exactly the documented grammar +// --------------------------------------------------------------------------- + +TEST(ConfigJsonValid, Scalars) { + { + const auto r = Parse("null"); + ASSERT_TRUE(r.ok()); + EXPECT_TRUE(r.value().isNull()); + } + { + const auto r = Parse("true"); + ASSERT_TRUE(r.ok()); + EXPECT_TRUE(r.value().isBool()); + EXPECT_TRUE(r.value().asBool()); + } + { + const auto r = Parse("false"); + ASSERT_TRUE(r.ok()); + EXPECT_TRUE(r.value().asBool() == false); + } + { + const auto r = Parse("42"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(42.0, r.value().asNumber()); + } + { + const auto r = Parse("-0"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(-0.0, r.value().asNumber()); + } + { + const auto r = Parse("1.5"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(1.5, r.value().asNumber()); + } + { + const auto r = Parse("-2.5e-7"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(-2.5e-7, r.value().asNumber()); + } + { + const auto r = Parse("1E+3"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(1000.0, r.value().asNumber()); + } +} + +TEST(ConfigJsonValid, HugeNumbers) { + // DBL_MAX / denormal minimum: exact boundary values (KAT). + { + const auto r = Parse("1.7976931348623157e308"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(std::numeric_limits::max(), r.value().asNumber()); + } + { + const auto r = Parse("5e-324"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(std::numeric_limits::denorm_min(), r.value().asNumber()); + } + // Overflow: a well-formed token that does not fit a double stores + // +/-inf (documented IEEE 754 behavior, json.h preamble). + { + const auto r = Parse("1e999"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(std::numeric_limits::infinity(), r.value().asNumber()); + } + { + const auto r = Parse("-1e999999999999999999999999"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(-std::numeric_limits::infinity(), + r.value().asNumber()); + } + // Integer exactness policy (json.h): beyond +/-2^53 integer literals + // lose low-order bits; 2^53 + 1 stores 2^53 (round to even). + { + const auto r = Parse("9007199254740993"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(static_cast(1ull << 53), r.value().asNumber()); + } +} + +TEST(ConfigJsonValid, Strings) { + { + const auto r = Parse("\"\""); + ASSERT_TRUE(r.ok()); + EXPECT_TRUE(r.value().asString().empty()); + } + { + const auto r = Parse("\"hello world\""); + ASSERT_TRUE(r.ok()); + EXPECT_EQ("hello world", r.value().asString()); + } + // The two-character escapes plus \uXXXX for quote, backslash, and + // slash. + { + const auto r = Parse(R"("a\"b\\c\/d")"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ("a\"b\\c/d", r.value().asString()); + } + { + const auto r = Parse(R"("Ae\u00e9")"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(Raw({0x41, 0x65, 0xC3, 0xA9}), r.value().asString()); + } + // Surrogate pair -> codepoint above U+FFFF. + { + const auto r = Parse(R"("\ud83d\ude00")"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(Raw({0xF0, 0x9F, 0x98, 0x80}), r.value().asString()); + } + // Escaped control characters are stored as-is. + { + const auto r = Parse(R"("a\u0000b\u001fc")"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(Raw({0x61, 0x00, 0x62, 0x1F, 0x63}), r.value().asString()); + } + // Raw UTF-8 bytes in strings (validated strictly). + { + const std::string hello = + Raw({0x68, 0xC3, 0xA9, 0x6C, 0x6C, 0x6F, 0x20, 0xF0, 0x9F, 0x8E, 0xAE}); + const auto r = Parse("\"" + hello + "\""); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(std::string_view(hello), r.value().asString()); + } + // DEL (U+007F) is not a control character: allowed raw. + { + const auto r = Parse("\"" + Raw({0x61, 0x7F, 0x62}) + "\""); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(std::string_view(Raw({0x61, 0x7F, 0x62})), r.value().asString()); + } + // Whitespace inside a string key. + { + const auto r = Parse("{\" \": 1}"); + ASSERT_TRUE(r.ok()); + EXPECT_TRUE(r.value().hasMember(" ")); + } +} + +TEST(ConfigJsonValid, Containers) { + { + const auto r = Parse("[]"); + ASSERT_TRUE(r.ok()); + EXPECT_TRUE(r.value().isArray()); + EXPECT_TRUE(r.value().asArray().empty()); + } + { + const auto r = Parse("{}"); + ASSERT_TRUE(r.ok()); + EXPECT_TRUE(r.value().isObject()); + EXPECT_TRUE(r.value().asObject().empty()); + } + { + const auto r = Parse("[1,2,3]"); + ASSERT_TRUE(r.ok()); + const auto& a = r.value().asArray(); + ASSERT_EQ(3u, a.size()); + EXPECT_EQ(1.0, a[0].asNumber()); + EXPECT_EQ(3.0, a[2].asNumber()); + } + { + const auto r = Parse(R"({"a":1,"b":{"c":[true,null]}})"); + ASSERT_TRUE(r.ok()); + const JsonValue* b = r.value().findMember("b"); + ASSERT_NE(nullptr, b); + const JsonValue* c = b->findMember("c"); + ASSERT_NE(nullptr, c); + ASSERT_EQ(2u, c->asArray().size()); + EXPECT_TRUE(c->asArray()[0].asBool()); + EXPECT_TRUE(c->asArray()[1].isNull()); + } + // Whitespace between every token; document order is preserved. + { + const auto r = Parse(" \t\r\n [ 1 , 2 ] \r\n "); + ASSERT_TRUE(r.ok()); + ASSERT_EQ(2u, r.value().asArray().size()); + } + { + const auto r = Parse(R"({"a":1,"b":2,"c":3})"); + ASSERT_TRUE(r.ok()); + const auto& members = r.value().asObject(); + ASSERT_EQ(3u, members.size()); + EXPECT_EQ("a", members[0].first); + EXPECT_EQ("b", members[1].first); + EXPECT_EQ("c", members[2].first); + } +} + +TEST(ConfigJsonValid, DepthAtLimit) { + // Default maxDepth is 32: exactly 32 nested containers parse, and the + // walk reaches the null leaf at level 32. + const auto r = Parse(NestedArray(32)); + ASSERT_TRUE(r.ok()); + const JsonValue* node = &r.value(); + for (int i = 0; i < 32; ++i) { + ASSERT_TRUE(node->isArray()) << "level " << i; + ASSERT_EQ(1u, node->asArray().size()); + node = &node->asArray()[0]; + } + EXPECT_TRUE(node->isNull()); +} + +TEST(ConfigJsonValid, SizeAtLimit) { + // A document of exactly kMiB bytes parses (the bound is inclusive). + std::string doc; + doc.reserve(kMiB); + doc.push_back('"'); + doc.append(kMiB - 2, 'a'); + doc.push_back('"'); + ASSERT_EQ(kMiB, doc.size()); + const auto r = Parse(doc); + ASSERT_TRUE(r.ok()); + EXPECT_EQ(kMiB - 2, r.value().asString().size()); +} + +// --------------------------------------------------------------------------- +// ConfigJsonInvalid — the malformed corpus (all MalformedInput) +// --------------------------------------------------------------------------- + +TEST(ConfigJsonInvalid, EmptyAndTrailing) { + ExpectMalformed(""); + ExpectMalformed(" \t\r\n"); + ExpectMalformed("1 2"); + ExpectMalformed("[] x"); + ExpectMalformed("1]"); + ExpectMalformed("\"a\"b"); + ExpectMalformed("tru"); + ExpectMalformed("fals"); + ExpectMalformed("nul"); + ExpectMalformed("truex"); + ExpectMalformed("1e"); + ExpectMalformed("-1."); + ExpectMalformed(Raw({0xFF})); // bare invalid byte outside a string +} + +TEST(ConfigJsonInvalid, Truncated) { + ExpectMalformed("["); + ExpectMalformed("[1"); + ExpectMalformed("[1,"); + ExpectMalformed("{"); + ExpectMalformed("{\"a\""); + ExpectMalformed("{\"a\":"); + ExpectMalformed("\"abc"); + ExpectMalformed("[["); + ExpectMalformed("1.5e"); + ExpectMalformed("-1e+"); + ExpectMalformed("01"); + ExpectMalformed("-"); +} + +TEST(ConfigJsonInvalid, BadNumbers) { + ExpectMalformed("+1"); + ExpectMalformed(".5"); + ExpectMalformed("1."); + ExpectMalformed("1e"); + ExpectMalformed("1e+"); + ExpectMalformed("1e-"); + ExpectMalformed("1.2.3"); + ExpectMalformed("0x1"); + ExpectMalformed("--1"); + ExpectMalformed("1 2"); + ExpectMalformed("1e5.5"); + ExpectMalformed("1.5e5e5"); +} + +TEST(ConfigJsonInvalid, BadEscapes) { + ExpectMalformed(R"("a\z")"); + ExpectMalformed(R"("a\")"); + ExpectMalformed(R"("a\u12")"); // fewer than 4 hex digits + ExpectMalformed(R"("a\uD800")"); // lone high surrogate + ExpectMalformed(R"("a\uDC00")"); // lone low surrogate + ExpectMalformed(R"("\ud83d")"); // high surrogate at end of string + ExpectMalformed(R"("\ud83d\u0041")"); // high not followed by a low +} + +TEST(ConfigJsonInvalid, ControlAndEncoding) { + // Raw control characters are rejected inside strings. + ExpectMalformed("\"" + Raw({0x61, 0x01, 0x62}) + "\""); + ExpectMalformed("\"" + Raw({0x61, 0x1F, 0x62}) + "\""); + ExpectMalformed("\"a\nb\""); + // Strict UTF-8: overlong, surrogate, out-of-range, and truncated. + ExpectMalformed("\"" + Raw({0xFF}) + "\""); + ExpectMalformed("\"" + Raw({0xC0, 0x80}) + "\""); // overlong NUL + ExpectMalformed("\"" + Raw({0xC1, 0xBF}) + "\""); // overlong + ExpectMalformed("\"" + Raw({0xE0, 0x9F, 0xBF}) + "\""); // overlong U+007F + ExpectMalformed("\"" + Raw({0xED, 0xA0, 0x80}) + "\""); // raw surrogate + ExpectMalformed("\"" + Raw({0xF5, 0x80, 0x80, 0x80}) + "\""); // above U+10FFFF + ExpectMalformed("\"" + Raw({0xF0, 0x80, 0x80, 0x80}) + "\""); // overlong NUL + ExpectMalformed("\"" + Raw({0x61, 0x62, 0x63, 0xE2, 0x82}) + + "\""); // truncated sequence at end + ExpectMalformed("\"" + Raw({0x80}) + "\""); // lone continuation byte +} + +TEST(ConfigJsonInvalid, DuplicateKeys) { + ExpectMalformed(R"({"a":1,"a":2})"); + ExpectMalformed(R"({"a":{"b":1},"a":{"b":2}})"); // top-level duplicate +} + +TEST(ConfigJsonInvalid, DepthOverLimit) { + ExpectMalformed(NestedArray(33)); // 33 nested containers > default 32 +} + +TEST(ConfigJsonInvalid, SizeOverLimit) { + // kMiB + 2 bytes: over the inclusive document-size bound. + std::string doc; + doc.reserve(kMiB + 2); + doc.push_back('"'); + doc.append(kMiB, 'a'); + doc.push_back('"'); + ASSERT_EQ(kMiB + 2, doc.size()); + ExpectMalformed(doc); +} + +// --------------------------------------------------------------------------- +// ConfigJsonRoundTrip — parse -> serialize -> parse is stable +// --------------------------------------------------------------------------- + +TEST(ConfigJsonRoundTrip, ValueRoundTrip) { + // Note: documents whose numbers overflow to +/-inf (e.g. "1e999") + // parse but are deliberately excluded here — serializing a non-finite + // number is a documented precondition violation, not a supported + // form. Every other parsed document round-trips. + const std::string deepNested = NestedArray(32); + const std::string utf8Doc = "\"" + + Raw({0x68, 0xC3, 0xA9, 0x6C, 0x6C, 0x6F, 0x20, 0xF0, 0x9F, 0x8E, 0xAE}) + + "\""; + const std::vector roundTripCorpus = { + "null", + "true", + "false", + "0", + "-0", + "1.5", + "-2.5e-7", + utf8Doc, + R"("a\u0001b")", + "[1,[2,3],{\"k\":\"v\"}]", + R"({"a":1,"b":[true,null],"c":"x","d":[]})", + deepNested, + }; + for (const std::string& text : roundTripCorpus) { + const auto r1 = Parse(text); + ASSERT_TRUE(r1.ok()) << text; + const std::string s1 = laige::serializeJson(r1.value()); + const auto r2 = Parse(s1); + ASSERT_TRUE(r2.ok()) << "re-parse of: " << s1; + EXPECT_TRUE(r1.value() == r2.value()) << "round trip of: " << text; + const std::string s2 = laige::serializeJson(r2.value()); + EXPECT_EQ(s1, s2) << "serializer idempotence for: " << text; + } +} + +TEST(ConfigJsonRoundTrip, CanonicalForms) { + const auto r = Parse(R"({"a":1,"b":[true,null],"c":"x"})"); + ASSERT_TRUE(r.ok()); + EXPECT_EQ("{\"a\":1,\"b\":[true,null],\"c\":\"x\"}", + laige::serializeJson(r.value())); + + // Numbers: shortest round-trip decimal; -0.0 serializes as "0". + EXPECT_EQ("1.5", laige::serializeJson(JsonValue::fromNumber(1.5))); + EXPECT_EQ("2", laige::serializeJson(JsonValue::fromNumber(2.0))); + EXPECT_EQ("0", laige::serializeJson(JsonValue::fromNumber(-0.0))); + EXPECT_EQ("1e+21", laige::serializeJson(JsonValue::fromNumber(1e21))); + EXPECT_EQ("1e-07", laige::serializeJson(JsonValue::fromNumber(1e-7))); + EXPECT_EQ("1.1", laige::serializeJson(JsonValue::fromNumber(1.1))); + EXPECT_EQ("123456.75", + laige::serializeJson(JsonValue::fromNumber(123456.75))); + EXPECT_EQ("0.1", laige::serializeJson(JsonValue::fromNumber(0.1))); + + // Strings: ASCII-safe canonical escaping. + EXPECT_EQ("\"h\\u00e9llo \\ud83c\\udfae\"", laige::serializeJson( + JsonValue::fromString( + Raw({0x68, 0xC3, 0xA9, 0x6C, 0x6C, 0x6F, 0x20, + 0xF0, 0x9F, 0x8E, 0xAE})))); + EXPECT_EQ("\"a\\u0001b\"", + laige::serializeJson(JsonValue::fromString(Raw({0x61, 0x01, 0x62})))); + EXPECT_EQ("\"q\\\"u\\\\ote\"", + laige::serializeJson(JsonValue::fromString("q\"u\\ote"))); + EXPECT_EQ("\"/\"", laige::serializeJson(JsonValue::fromString("/"))); +} + +// --------------------------------------------------------------------------- +// ConfigJsonValue — the value type's mechanics +// --------------------------------------------------------------------------- + +TEST(ConfigJsonValue, FactoriesAndKinds) { + EXPECT_TRUE(JsonValue().isNull()); + EXPECT_TRUE(JsonValue::fromBool(true).isBool()); + EXPECT_TRUE(JsonValue::fromNumber(1.0).isNumber()); + EXPECT_TRUE(JsonValue::fromString("x").isString()); + EXPECT_TRUE(JsonValue::makeArray().isArray()); + EXPECT_TRUE(JsonValue::makeObject().isObject()); +} + +TEST(ConfigJsonValue, CopyIsDeep) { + const auto r = Parse(R"({"k":[1,2]})"); + ASSERT_TRUE(r.ok()); + JsonValue doc = r.value(); + JsonValue copy = doc; + doc.setMember("k", JsonValue::fromNumber(99.0)); + EXPECT_EQ(99.0, doc.findMember("k")->asNumber()); + const JsonValue* copiedKey = copy.findMember("k"); + ASSERT_NE(nullptr, copiedKey); + EXPECT_TRUE(copiedKey->isArray()); + EXPECT_EQ(2u, copiedKey->asArray().size()); +} + +TEST(ConfigJsonValue, MoveSemantics) { + const auto r = Parse("[1,2,3]"); + ASSERT_TRUE(r.ok()); + JsonValue source = r.value(); + JsonValue moved(std::move(source)); + EXPECT_TRUE(source.isNull()); // moved-from value is Null + ASSERT_EQ(3u, moved.asArray().size()); + + JsonValue target; + target = std::move(moved); + EXPECT_TRUE(moved.isNull()); + ASSERT_EQ(3u, target.asArray().size()); + EXPECT_EQ(3.0, target.asArray()[2].asNumber()); +} + +TEST(ConfigJsonValue, Mutation) { + JsonValue obj = JsonValue::makeObject(); + obj.setMember("a", JsonValue::fromNumber(1.0)); + obj.setMember("b", JsonValue::fromBool(true)); + obj.setMember("a", JsonValue::fromNumber(2.0)); // replace in place + ASSERT_EQ(2u, obj.asObject().size()); + EXPECT_EQ(2.0, obj.findMember("a")->asNumber()); + EXPECT_EQ("a", obj.asObject().front().first); // position preserved + + JsonValue arr = JsonValue::makeArray(); + arr.append(JsonValue::fromNumber(1.0)); + arr.append(JsonValue::fromString("x")); + ASSERT_EQ(2u, arr.asArray().size()); + EXPECT_EQ("x", arr.asArray()[1].asString()); + + // Kind transitions release the old payload (the churn test below plus + // the ASan tree verify this). + JsonValue v = JsonValue::makeArray(); + v.append(JsonValue::fromNumber(1.0)); + v.setString("s"); + EXPECT_TRUE(v.isString()); + v.setNull(); + EXPECT_TRUE(v.isNull()); +} + +TEST(ConfigJsonValue, Equality) { + EXPECT_TRUE(JsonValue() == JsonValue()); + EXPECT_TRUE(JsonValue::fromNumber(1.0) == JsonValue::fromNumber(1.0)); + EXPECT_TRUE(JsonValue::fromNumber(1.0) != JsonValue::fromNumber(2.0)); + // NaN is unequal to itself (IEEE 754 ==). + const double nanValue = std::nan(""); + EXPECT_TRUE(JsonValue::fromNumber(nanValue) != + JsonValue::fromNumber(nanValue)); + + // Objects compare independent of member order. + JsonValue a = JsonValue::makeObject(); + a.setMember("x", JsonValue::fromNumber(1.0)); + a.setMember("y", JsonValue::fromNumber(2.0)); + JsonValue b = JsonValue::makeObject(); + b.setMember("y", JsonValue::fromNumber(2.0)); + b.setMember("x", JsonValue::fromNumber(1.0)); + EXPECT_TRUE(a == b); + + // Arrays are order-sensitive; kinds must match. + EXPECT_TRUE(Parse("[1,2]").value() != Parse("[2,1]").value()); + EXPECT_TRUE(Parse("{\"a\":1}").value() == Parse("{\"a\":1}").value()); + EXPECT_TRUE(JsonValue::fromBool(true) != JsonValue::fromNumber(1.0)); +} + +TEST(ConfigJsonValue, FindMemberTotal) { + // Total on any value: non-objects report absence, not an error. + JsonValue arr = JsonValue::makeArray(); + EXPECT_TRUE(arr.findMember("a") == nullptr); + EXPECT_FALSE(arr.hasMember("a")); + + JsonValue obj = JsonValue::makeObject(); + obj.setMember("a", JsonValue::fromNumber(1.0)); + EXPECT_TRUE(obj.findMember("a") != nullptr); + EXPECT_TRUE(obj.findMember("b") == nullptr); + EXPECT_FALSE(obj.hasMember("b")); +} + +TEST(ConfigJsonValue, ChurnNoLeak) { + // Construct/destroy churn: the ASan/LSan trees prove the leak-freeness; + // on the plain tree this exercises the copy/move/dtor paths. + for (int i = 0; i < 10000; ++i) { + JsonValue v = JsonValue::makeObject(); + v.setMember("k", JsonValue::fromString("value")); + v.setMember("n", JsonValue::fromNumber(static_cast(i))); + JsonValue copy = v; + (void)copy; + } + SUCCEED(); +} + +// --------------------------------------------------------------------------- +// ConfigJsonOptions — the documented bounds are engine-configurable +// --------------------------------------------------------------------------- + +TEST(ConfigJsonOptions, DepthLimit) { + JsonOptions opts; + opts.maxDepth = 1; + EXPECT_TRUE(Parse("[1]", opts).ok()); + ExpectMalformed("{{}}", opts); // inner object would be depth 2 + + opts.maxDepth = 2; + EXPECT_TRUE(Parse("[[1]]", opts).ok()); + ExpectMalformed("[[[1]]]", opts); // depth 3 +} + +TEST(ConfigJsonOptions, SizeLimit) { + JsonOptions opts; + opts.maxDocumentBytes = 10; + EXPECT_TRUE(Parse("[1,2,3,45]", opts).ok()); // exactly 10 bytes + ExpectMalformed("[1,2,3,4,5]", opts); // 11 bytes + EXPECT_TRUE(Parse("1", opts).ok()); + ExpectMalformed("1 2", opts); // trailing data + + JsonOptions zero; + zero.maxDocumentBytes = 0; + ExpectMalformed("1", zero); // any non-empty document +} diff --git a/tools/README.md b/tools/README.md index b7420b1..f98b14a 100644 --- a/tools/README.md +++ b/tools/README.md @@ -2,7 +2,11 @@ Engine tools and CI scripts, each landing with its roadmap step: -- `laige-fuzz` — fuzz runner (M0-TEST-01) +- `laige-fuzz` — deterministic bounded fuzz runner (minimal form from + M0-CORE-07, in `tools/fuzz`: the `json_parse` target, `--runs`/`--seed`, + built with `LAIGE_BUILD_TESTS=ON`, registered as the `fuzz_json_parse` + CTest entry — bounded fuzz in every commit, PRD §14; M0-TEST-01 + extends it: CI lane semantics, nightly long runs, seed documentation) - `laige-include-lint` — include-graph lint + vendored-dependency-count metric over `src/**` (M0-CI-03). Pure Python 3 stdlib; run it as `python3 tools/laige-include-lint [--root REPO_ROOT]`. Enforces the PRD diff --git a/tools/fuzz/CMakeLists.txt b/tools/fuzz/CMakeLists.txt new file mode 100644 index 0000000..4f92a4b --- /dev/null +++ b/tools/fuzz/CMakeLists.txt @@ -0,0 +1,22 @@ +# laige-fuzz — deterministic bounded fuzz runner (M0-CORE-07, minimal +# form; M0-TEST-01 extends it: CI lane semantics, nightly long runs, +# seed-handling documentation). +# +# Gated with LAIGE_BUILD_TESTS (root CMakeLists): a dev tool that links +# the engine library — a library-only build does not need it. + +add_executable(laige-fuzz laige-fuzz.cpp) +laige_apply_engine_policy(laige-fuzz) +target_link_libraries(laige-fuzz PRIVATE laige-core) + +# Bounded fuzz in every commit (PRD §14: "every commit (bounded), nightly +# (long)"): 1000 deterministic inputs to the json_parse target (NFR-8.7, +# ADR 0003). In an ASan tree this entry is the M0-CORE-07 Verify gate — +# the run is instrumented, so any crash/UB fails ctest loudly. +add_test(NAME fuzz_json_parse COMMAND laige-fuzz json_parse --runs=1000) +set_tests_properties(fuzz_json_parse PROPERTIES TIMEOUT 120) +if(LAIGE_TSAN) + # Same fatal-race policy as the module test entries (NFR-8.2). + set_tests_properties(fuzz_json_parse PROPERTIES + ENVIRONMENT "TSAN_OPTIONS=halt_on_error=1") +endif() diff --git a/tools/fuzz/laige-fuzz.cpp b/tools/fuzz/laige-fuzz.cpp new file mode 100644 index 0000000..384fa72 --- /dev/null +++ b/tools/fuzz/laige-fuzz.cpp @@ -0,0 +1,208 @@ +// laige-fuzz — deterministic bounded fuzz runner (M0-CORE-07, minimal +// form; M0-TEST-01 extends it: CI lane semantics, nightly long runs, +// seed-handling documentation). +// +// Roadmap step M0-CORE-07 lands the *minimal* runner this Verify gate +// requires: `laige-fuzz json_parse --runs=1000` clean under ASan +// (NFR-8.7: parsers are fuzzed in CI, bounded every commit — PRD §14). +// The `json_parse` target is registered below; later steps add their +// targets to kTargets. +// +// Design (deterministic by construction): +// - Every input is generated from laige::Prng (M0-CORE-06): a fixed +// default seed (kDefaultSeed, overridable with --seed) drives the +// whole run, so the same (target, runs, seed) reproduces the exact +// same input sequence on every platform (Prng's cross-platform +// bit-exact determinism, ARCH-010). +// - Per input, one of three modes (Prng-selected): +// mutate pick a corpus document, apply 1..8 random byte edits +// (flip / insert / delete) +// truncate pick a corpus document, keep a random prefix +// random a pure random byte string, length 1..kMaxInputBytes +// - The target is called with (bytes, size); any Status the target +// returns is acceptable. The harness fails only on process death +// (a crash or sanitizer report), which ctest turns into a test +// failure — no silent failure (CORE-008). +// +// Usage: +// laige-fuzz [--runs=N] [--seed=HEX|DEC] +// +// Exit codes: 0 = all runs clean · 2 = usage error or unknown target. +// (A crash exits non-zero on its own before any summary is printed.) + +#include +#include +#include +#include +#include +#include + +#include "laige/json.h" +#include "laige/prng.h" + +namespace { + +constexpr std::uint64_t kDefaultSeed = 0x1F055EEDull; // "one-fuzz-seed" +constexpr int kDefaultRuns = 1000; +constexpr std::size_t kMaxInputBytes = 64; + +// Valid-document corpus: the base for mutate/truncate inputs. Kept ASCII +// (explicit bytes for the one non-ASCII document) so the corpus bytes are +// identical on every platform (narrow \x escapes are not portable). +const std::string kCorpusUtf8 = + "\"\\ud83d\\ude00 raw " + + std::string({static_cast(0xE2), static_cast(0x82), + static_cast(0xAC)}) + + "\""; +const std::vector kCorpus = { + "null", + "true", + "false", + "0", + "-1", + "1.5", + "1e999", + "-2.5e-7", + "\"\"", + "\"hello\"", + "\"\\u0041\\u00e9\"", + kCorpusUtf8, + "[]", + "[1,2,3]", + "[[1],[2,3]]", + "{}", + "{\"a\":1,\"b\":null}", + "{\"a\":{\"b\":[1,{\"c\":\"x\"}]}}", + " \t\r\n [ 1 , 2 ] \r\n ", +}; + +// --------------------------------------------------------------------------- +// Fuzz targets: (name, entry point). The entry point receives the raw +// input; any Status outcome is acceptable — only a crash fails the run. +// --------------------------------------------------------------------------- +struct FuzzTarget { + const char* name; + void (*run)(const std::uint8_t* data, std::size_t size); +}; + +void fuzzJsonParse(const std::uint8_t* data, std::size_t size) { + const std::string_view input(reinterpret_cast(data), size); + const laige::Result result = laige::parseJson(input); + (void)result; // any Status is acceptable; only a crash fails the run +} + +const FuzzTarget kTargets[] = { + {"json_parse", &fuzzJsonParse}, +}; + +void printUsage() { + std::printf("usage: laige-fuzz [--runs=N] [--seed=HEX|DEC]\n"); + std::printf("targets:\n"); + for (const FuzzTarget& target : kTargets) { + std::printf(" %s\n", target.name); + } +} + +} // namespace + +int main(int argc, char** argv) { + const char* targetName = nullptr; + int runs = kDefaultRuns; + std::uint64_t seed = kDefaultSeed; + + for (int i = 1; i < argc; ++i) { + const std::string_view arg = argv[i]; + if (arg.starts_with("--runs=")) { + const int v = std::atoi(arg.data() + 7); + if (v <= 0) { + printUsage(); + return 2; + } + runs = v; + } else if (arg.starts_with("--seed=")) { + const char* value = arg.data() + 7; + if (*value == '\0') { + printUsage(); + return 2; + } + seed = std::strtoull(value, nullptr, 0); // 0x-prefix or decimal + } else if (arg.starts_with("-")) { + printUsage(); + return 2; + } else if (targetName != nullptr) { + printUsage(); + return 2; + } else { + targetName = arg.data(); + } + } + + if (targetName == nullptr) { + printUsage(); + return 2; + } + const FuzzTarget* target = nullptr; + for (const FuzzTarget& candidate : kTargets) { + if (std::string_view(candidate.name) == targetName) { + target = &candidate; + break; + } + } + if (target == nullptr) { + printUsage(); + return 2; + } + + laige::Prng rng(seed); + for (int i = 0; i < runs; ++i) { + std::vector input; + const int mode = rng.next_range(0, 10); // 0..9 + if (mode < 5) { + // ~50%: mutate a corpus document with 1..8 random byte edits. + const std::string& doc = + kCorpus[rng.next_range(0, static_cast(kCorpus.size()))]; + input.assign(doc.begin(), doc.end()); + const int edits = 1 + static_cast(rng.next_range(0, 8)); + for (int e = 0; e < edits; ++e) { + // Position in [0, input.size()]: insert-at-end is a valid edit. + const std::size_t where = + rng.next_range(0, static_cast(input.size() + 1)); + const int op = static_cast(rng.next_range(0, 3)); + if (op == 0) { // flip a byte + if (!input.empty()) { + input[where % input.size()] ^= + static_cast(1 + rng.next_range(0, 255)); + } + } else if (op == 1) { // insert a byte + input.insert(input.begin() + where, + static_cast(rng.next_u64())); + } else { // delete a byte + if (!input.empty()) { + input.erase(input.begin() + (where % input.size())); + } + } + } + } else if (mode < 8) { + // ~30%: truncate a corpus document to a random prefix. + const std::string& doc = + kCorpus[rng.next_range(0, static_cast(kCorpus.size()))]; + const std::size_t len = doc.size(); + const std::size_t keep = + len == 0 ? 0 : rng.next_range(0, static_cast(len + 1)); + input.assign(doc.begin(), doc.begin() + static_cast(keep)); + } else { + // ~20%: pure random bytes. + const std::size_t len = + rng.next_range(1, static_cast(kMaxInputBytes + 1)); + input.resize(len); + for (std::uint8_t& byte : input) { + byte = static_cast(rng.next_u64()); + } + } + target->run(input.data(), input.size()); + } + + std::printf("laige-fuzz: target '%s' runs=%d seed=0x%016llx — ok\n", + target->name, runs, static_cast(seed)); + return 0; +}