From 6c80354ab78822d73f5a9386b6da7e6dcde1d2a2 Mon Sep 17 00:00:00 2001 From: Affan Khan Date: Thu, 2 Jul 2026 10:47:34 -0400 Subject: [PATCH] perf: SIMD-accelerate JSON string paths (from_json/to_json) Add 128-bit SIMD (ARM NEON + x86 SSE2, scalar fallback) to the two byte-at-a-time string loops in the msgpack<->JSON conversion, the one place in the extension where raw byte throughput dominates: - mpJpParseString (from_json): SIMD scan for '"'/'\' then bulk-copy the safe run in a single append. - mpJsonEscapeStr (to_json/pretty): SIMD scan for '"'/'\'/control bytes then jump straight to the next byte needing escaping. New helpers mpScanStr/mpScanEsc use full 16-byte loads only (no out-of-bounds over-read), a scalar tail, and only the always-available baseline ISA (no runtime CPU dispatch). A cheap first-byte guard avoids any regression on escape-dense input. Define MSGPACK_DISABLE_SIMD to force the portable scalar path. MessagePack navigation (extract/set/remove/type) is a serial length-prefixed pointer-chase that cannot be vectorized, and key/buffer ops already use libc SIMD memcmp/memcpy, so those are left untouched. Measured on arm64/NEON, ~4 KB strings (A/B vs MSGPACK_DISABLE_SIMD): - from_json, no escapes: ~2670 -> ~579 ns/op (~4.6x) - to_json, no escapes: ~2590 -> ~510 ns/op (~5.0x) - to_json, all escapes (worst case): parity, no regression Small strings are unchanged (bound by SQLite call overhead). Correctness: 16/16 ctest pass; a 6000-case deterministic differential fuzz and 15 boundary edge cases (escapes at 16-byte chunk offsets, control bytes, tails) produce byte-identical output to the scalar path. SSE2 branch cross-compiles clean and is MSVC-safe (_BitScanForward). Also adds large-string rows to bench_msgpack_vs_json.c. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- src/msgpack.c | 163 +++++++++++++++++++++++++++++++++- tests/bench_msgpack_vs_json.c | 26 ++++++ 2 files changed, 186 insertions(+), 3 deletions(-) diff --git a/src/msgpack.c b/src/msgpack.c index 305f597..54e1a43 100644 --- a/src/msgpack.c +++ b/src/msgpack.c @@ -164,6 +164,150 @@ static void mpWrite64(u8 *p, u64 v){ mpWrite32(p,(u32)(v>>32)); mpWrite32(p+4,(u32)v); } +/* +** ============================================================ +** SIMD byte scanners (JSON string fast paths) +** ============================================================ +** +** MessagePack navigation (mpSkipOne/mpLookup) is a serial length-prefixed +** pointer-chase and cannot be vectorized. The one place byte throughput +** dominates is scanning JSON *string* bodies during msgpack<->JSON +** conversion: +** +** mpJpParseString (from_json) — scan for '"' or '\\' +** mpJsonEscapeStr (to_json) — scan for '"', '\\' or a control byte (<0x20) +** +** Both loops previously advanced one byte at a time. The helpers below find +** the first "interesting" byte in a run using 128-bit SIMD (ARM NEON or x86 +** SSE2), letting the callers bulk-copy the safe span in between. Only full +** 16-byte loads are used (i+16<=len) so we never read past the buffer; the +** short tail falls back to a scalar loop. Only the always-available 128-bit +** baseline ISA is used, so no runtime CPU dispatch is required. +** +** Define MSGPACK_DISABLE_SIMD to force the portable scalar path (used to +** A/B benchmark the speedup, and for exotic targets). +*/ +#if !defined(MSGPACK_DISABLE_SIMD) +# if defined(__ARM_NEON) || defined(__ARM_NEON__) +# include +# define MSGPACK_HAVE_NEON 1 +# elif defined(__SSE2__) || (defined(_MSC_VER) && (defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP>=2))) +# include +# define MSGPACK_HAVE_SSE2 1 +# endif +#endif +#ifndef MSGPACK_HAVE_NEON +# define MSGPACK_HAVE_NEON 0 +#endif +#ifndef MSGPACK_HAVE_SSE2 +# define MSGPACK_HAVE_SSE2 0 +#endif + +#if MSGPACK_HAVE_NEON +/* Index (0..15) of the first matching lane in a NEON compare result whose +** lanes are 0x00 (no match) or 0xff (match), or 16 if none matched. +** Uses the shift-narrow trick: each source byte becomes a nibble in a 64-bit +** word, so (ctz/4) is the byte index of the first match. */ +static inline u32 mpSimdFirstNeon(uint8x16_t cmp){ + uint8x8_t narrowed = vshrn_n_u16(vreinterpretq_u16_u8(cmp), 4); + u64 m = vget_lane_u64(vreinterpret_u64_u8(narrowed), 0); + if( m==0 ) return 16; + return (u32)(__builtin_ctzll(m) >> 2); +} +#endif +#if MSGPACK_HAVE_SSE2 +# if defined(_MSC_VER) +# include +# endif +/* Index (0..15) of the first matching byte in an SSE2 compare result, or 16. */ +static inline u32 mpSimdFirstSse(__m128i cmp){ + unsigned m = (unsigned)_mm_movemask_epi8(cmp); + if( m==0 ) return 16; +# if defined(_MSC_VER) + { unsigned long idx; _BitScanForward(&idx, m); return (u32)idx; } +# else + return (u32)__builtin_ctz(m); +# endif +} +#endif + +/* Return the offset of the first byte in s[0..len) equal to '"' or '\\', +** or len if there is none. */ +static u32 mpScanStr(const u8 *s, u32 len){ + u32 i = 0; +#if MSGPACK_HAVE_NEON || MSGPACK_HAVE_SSE2 + /* Cheap first-byte check: on escape-dense input (matches back-to-back) this + ** returns immediately, avoiding SIMD load/compare setup cost per byte. */ + if( len && (s[0]=='"' || s[0]=='\\') ) return 0; +#endif +#if MSGPACK_HAVE_NEON + const uint8x16_t vq = vdupq_n_u8('"'); + const uint8x16_t vbs = vdupq_n_u8('\\'); + for( ; i+16<=len; i+=16 ){ + uint8x16_t v = vld1q_u8(s+i); + uint8x16_t m = vorrq_u8(vceqq_u8(v,vq), vceqq_u8(v,vbs)); + u32 k = mpSimdFirstNeon(m); + if( k<16 ) return i+k; + } +#elif MSGPACK_HAVE_SSE2 + const __m128i vq = _mm_set1_epi8('"'); + const __m128i vbs = _mm_set1_epi8('\\'); + for( ; i+16<=len; i+=16 ){ + __m128i v = _mm_loadu_si128((const __m128i*)(s+i)); + __m128i m = _mm_or_si128(_mm_cmpeq_epi8(v,vq), _mm_cmpeq_epi8(v,vbs)); + u32 k = mpSimdFirstSse(m); + if( k<16 ) return i+k; + } +#endif + for( ; i min(v,0x1f)==v */ + __m128i lt = _mm_cmpeq_epi8(_mm_min_epu8(v, vlo), v); + __m128i m = _mm_or_si128(lt, _mm_cmpeq_epi8(v,vq)); + m = _mm_or_si128(m, _mm_cmpeq_epi8(v,vbs)); + u32 k = mpSimdFirstSse(m); + if( k<16 ) return i+k; + } +#endif + for( ; i=0x20 && c!='"' && c!='\\' ) continue; + j = 0; + while( j=len ) break; + c = s[j]; /* Flush buffered safe characters */ if( j>start ) mpBufAppend(out, s+start, j-start); if( c=='"' ){ @@ -2076,6 +2224,7 @@ static void mpJsonEscapeStr(MpBuf *out, const u8 *s, u32 len){ mpBufAppend(out,(const u8*)esc,6); } start = j+1; + j++; } /* Flush remaining safe characters */ if( len>start ) mpBufAppend(out, s+start, len-start); @@ -2292,6 +2441,14 @@ static int mpJpParseString(MpJsonParser *p, MpBuf *out){ mpBufInit(&sb, out->pCtx); p->i++; /* skip '"' */ while(p->in){ + /* SIMD fast path: copy the run of ordinary bytes up to the next + ** '"' (end) or '\\' (escape) in a single bulk append. */ + u32 run = mpScanStr((const u8*)p->z + p->i, (u32)(p->n - p->i)); + if( run ){ + mpBufAppend(&sb, (const u8*)p->z + p->i, run); + p->i += (int)run; + if( p->i>=p->n ) break; + } unsigned char c=(unsigned char)p->z[p->i]; if(c=='"'){ p->i++; break; } if(c=='\\'){ diff --git a/tests/bench_msgpack_vs_json.c b/tests/bench_msgpack_vs_json.c index cb73081..ce81989 100644 --- a/tests/bench_msgpack_vs_json.c +++ b/tests/bench_msgpack_vs_json.c @@ -165,6 +165,17 @@ int main(void){ " jsonb_object('id',1,'name','Alice','score',9.5,'active',1) AS jb;", NULL, NULL, NULL); + /* ── Setup: large string payloads for the SIMD string-path benchmarks ── + ** hex(zeroblob(2048)) is a 4096-char ASCII string with no bytes that need + ** JSON escaping (best case for SIMD bulk-copy); replacing every '0' with a + ** '"' yields an all-escapes worst case where safe runs are empty. */ + sqlite3_exec(g_db, + "CREATE TEMP TABLE bigstr AS SELECT" + " '\"' || hex(zeroblob(2048)) || '\"' AS from_plain_js," + " msgpack_quote(hex(zeroblob(2048))) AS to_plain_mp," + " msgpack_quote(replace(hex(zeroblob(2048)),'0','\"')) AS to_esc_mp;", + NULL, NULL, NULL); + /* Iteration counts tuned so each scenario takes ~0.5–2 s on a modern machine */ const int N_BUILD = 200000; const int N_EXTRACT = 400000; @@ -270,6 +281,21 @@ int main(void){ bench_expr("SELECT msgpack_to_json(mp) FROM src", N_EXTRACT), -1.0 /* baseline */, -1.0); + /* ── Large-string SIMD string paths (msgpack-only baseline) ─────────────── */ + const int N_BIGSTR = 50000; + + print_row("from_json large str (~4 KB, no escapes)", + bench_expr("SELECT msgpack_from_json(from_plain_js) FROM bigstr", N_BIGSTR), + -1.0, -1.0); + + print_row("to_json large str (~4 KB, no escapes)", + bench_expr("SELECT msgpack_to_json(to_plain_mp) FROM bigstr", N_BIGSTR), + -1.0, -1.0); + + print_row("to_json large str (~4 KB, all escapes)", + bench_expr("SELECT msgpack_to_json(to_esc_mp) FROM bigstr", N_BIGSTR), + -1.0, -1.0); + printf("\n\n"); sqlite3_close(g_db);