From 8c68bed2b4de0f4f2f1b65985c7f70f02747ffb4 Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Fri, 11 Sep 2026 00:57:25 -0700 Subject: [PATCH] fix(tangle): truncate oversized tool output instead of discarding a paid turn MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 1.1.9 widened the bound on tool output from the 16 KiB metadata limit to the 1 MiB content limit. That removed the instance and left the class: the producer still serializes each tool value up to 4 MiB, and the consumer bound is 1 MiB across the WHOLE record, so a single 2 MiB fetch still died, and so did two 0.6 MiB fetches together. A 256x mismatch became a 4x one. The mismatch is not the real defect. This validator runs inside the terminal result read, after the live stream has drained and after the usage receipt has been credited, so a throw here cannot prevent the work or the charge. It can only destroy a finished, fully paid turn, which a supervisor then reports as a child that did nothing at all. Any bound that throws on this path is one large page away from doing that again. So oversized tool output is now replaced rather than refused. Each pass cuts the widest `toolInvocations[].result` by the record's measured overflow and leaves a marker naming both byte counts, so a reader can tell a truncated result from a tool that genuinely returned little. The caller still receives the turn, its response text, its usage, and every tool call it made. Only tool output is truncatable. Every other field is identity, accounting, or control material where a silently shortened value would be worse than a refusal, so a record whose overflow is elsewhere still refuses — as does one that breaks the node, depth, or array limits, which trimming a string cannot satisfy. Tests: one added, failing on 1.1.9. It pins a 2 MiB single result, the producer's own 4 MiB per-value maximum, several results that each fit but together do not, the marker's byte counts, that every tool call survives, and that an oversized `response` still refuses. Provider suite 264 pass, typecheck clean. Co-Authored-By: Claude Opus 5 (1M context) --- packages/agent-provider-tangle/package.json | 2 +- .../src/tangle-events.test.ts | 49 +++++++++- .../src/tangle-prompt.ts | 98 ++++++++++++++++++- 3 files changed, 144 insertions(+), 5 deletions(-) diff --git a/packages/agent-provider-tangle/package.json b/packages/agent-provider-tangle/package.json index 4ff8b2d..ca04462 100644 --- a/packages/agent-provider-tangle/package.json +++ b/packages/agent-provider-tangle/package.json @@ -1,6 +1,6 @@ { "name": "@tangle-network/agent-provider-tangle", - "version": "1.1.9", + "version": "1.1.10", "description": "AgentEnvironmentProvider adapter for Tangle sandboxes", "type": "module", "license": "MIT", diff --git a/packages/agent-provider-tangle/src/tangle-events.test.ts b/packages/agent-provider-tangle/src/tangle-events.test.ts index 773ff82..d2e9218 100644 --- a/packages/agent-provider-tangle/src/tangle-events.test.ts +++ b/packages/agent-provider-tangle/src/tangle-events.test.ts @@ -1,6 +1,7 @@ import { readFileSync } from "node:fs"; import type { SandboxEvent } from "@tangle-network/sandbox"; import { AgentTurnResultSchema, type AgentEnvironmentEvent } from "@tangle-network/agent-interface/environment-provider"; +import { isBoundedEventContentJson } from "@tangle-network/agent-interface"; import { describe, expect, it } from "vitest"; import { createTangleProvider } from "./index.js"; import { assertBoundedJson, MAX_STRING_LENGTH as CONTRACT_MAX_STRING_LENGTH } from "./tangle-contract-safety.js"; @@ -147,9 +148,53 @@ describe("Sandbox stream event content", () => { // The exact wall the failures piled against: one character over the metadata bound. expect(() => validatedSandboxPromptResult(toolResult(CONTRACT_MAX_STRING_LENGTH + 1))).not.toThrow(); expect(() => validatedSandboxPromptResult(toolResult(200_000))).not.toThrow(); - // The content bound still governs it, so an unbounded tool result is still refused. - expect(() => validatedSandboxPromptResult(toolResult(2 * 1024 * 1024))).toThrow(/JSON bound/); // Every other field keeps the metadata bound; only what the agent produced is content. expect(() => validatedSandboxPromptResult({ ...result, traceId: "x".repeat(CONTRACT_MAX_STRING_LENGTH + 1) })).toThrow(/JSON bound/); }); + + it("truncates oversized tool output instead of discarding a paid turn", () => { + // Widening the bound did not remove the failure mode. This validator runs in the terminal + // result read, AFTER the stream drained and the usage receipt was credited, so a throw here + // cannot prevent the work or the charge — it can only destroy a finished, paid turn. The + // producer serializes each tool value up to 4 MiB while this bound is 1 MiB across the WHOLE + // record, so a single 2 MiB fetch, or two 0.6 MiB fetches together, still died. + const base = { success: true, status: "success" as const, durationMs: 1, response: "ok" }; + const withResults = (length: number, count = 1) => ({ + ...base, + toolInvocations: Array.from({ length: count }, (_unused, index) => ({ + toolName: "webfetch", + args: { url: `https://example.test/${index}` }, + result: "x".repeat(length), + })), + }); + const marked = (out: unknown) => + ((out as { toolInvocations: { result: string }[] }).toolInvocations ?? []).filter((entry) => + entry.result.includes("[truncated by the Tangle provider"), + ).length; + + // A single result larger than the whole-record bound: the turn survives, the tail is cut. + const single = validatedSandboxPromptResult(withResults(2 * 1024 * 1024)); + expect(marked(single)).toBe(1); + // The producer's own per-value maximum must not be able to kill a turn either. + expect(marked(validatedSandboxPromptResult(withResults(4 * 1024 * 1024)))).toBe(1); + // Several results that each fit but together do not: the record converges rather than + // destroying the first one it meets. + const many = validatedSandboxPromptResult(withResults(600_000, 2)); + expect(marked(many)).toBeGreaterThanOrEqual(1); + expect(isBoundedEventContentJson(many as unknown as Record)).toBe(true); + const wide = validatedSandboxPromptResult(withResults(900_000, 5)); + expect(isBoundedEventContentJson(wide as unknown as Record)).toBe(true); + + // The marker names both byte counts, so a reader can tell a truncated result from a tool that + // genuinely returned little, and every tool call the turn made is still present. + const first = (single as unknown as { toolInvocations: { result: string; toolName: string }[] }).toolInvocations[0]; + expect(first.toolName).toBe("webfetch"); + expect(first.result).toMatch(/kept \d+ of 2097152 bytes/u); + + // A record whose overflow is NOT tool output still refuses: only tool output is truncatable, + // because every other field is identity, accounting, or control material. + expect(() => + validatedSandboxPromptResult({ ...base, response: "x".repeat(2 * 1024 * 1024) }), + ).toThrow(/JSON bound/); + }); }); diff --git a/packages/agent-provider-tangle/src/tangle-prompt.ts b/packages/agent-provider-tangle/src/tangle-prompt.ts index 3519e70..ada7010 100644 --- a/packages/agent-provider-tangle/src/tangle-prompt.ts +++ b/packages/agent-provider-tangle/src/tangle-prompt.ts @@ -13,7 +13,9 @@ import type { } from "@tangle-network/agent-interface"; import { agentProfileSchema, + CONTRACT_MAX_JSON_BYTES, boundedEventContentRecordSchema, + isBoundedEventContentJson, AgentExactRunControlRefSchema, AgentTurnInputSchema, ContextTransferReceiptSchema, @@ -554,6 +556,77 @@ const SANDBOX_OPTIONAL_RESULT_FIELDS = new Set([ "costUsd", ]); +/** + * The marker left in place of a tool result's discarded tail. + * + * It names the byte counts so a reader can tell a truncated result from a tool that genuinely + * returned little, and so a caller can decide to re-fetch rather than reason from a partial page. + */ +export function toolOutputTruncationMarker(keptBytes: number, originalBytes: number): string { + return `\n\n[truncated by the Tangle provider: kept ${keptBytes} of ${originalBytes} bytes to stay inside the ${CONTRACT_MAX_JSON_BYTES}-byte record content bound]`; +} + +/** + * Shrink oversized tool output until the whole record fits the content bound. + * + * Only `toolInvocations[].result` is touched, because it is the one field that carries arbitrary + * fetched material rather than identity, accounting, or control state. The largest result is cut + * first and the loop repeats, so a record with several big results converges instead of destroying + * the first one it meets. + * + * Returns the record unchanged when it already fits, so the common path allocates nothing. + */ +function withTruncatedToolOutput(record: Record): { + record: Record; + truncated: number; +} { + if (isBoundedEventContentJson(record)) return { record, truncated: 0 }; + const invocations = record.toolInvocations; + if (!Array.isArray(invocations)) return { record, truncated: 0 }; + const next = invocations.map((entry) => + entry && typeof entry === "object" && !Array.isArray(entry) + ? { ...(entry as Record) } + : entry, + ); + const originalLengths = new Map(); + let truncated = 0; + // Each pass cuts the current widest result by the record's measured overflow, so a single huge + // value converges immediately and several large ones shrink in turn rather than the first being + // destroyed. The cap is a backstop: `isBoundedEventContentJson` also enforces node, depth, and + // array limits that trimming a string cannot satisfy, and those must fall through to the refusal + // rather than spin here. + for (let pass = 0; pass < next.length * 4 + 8; pass += 1) { + const candidate = { ...record, toolInvocations: next }; + if (isBoundedEventContentJson(candidate)) return { record: candidate, truncated }; + let widest = -1; + let widestLength = 0; + for (const [index, entry] of next.entries()) { + const value = (entry as Record | undefined)?.result; + if (typeof value === "string" && value.length > widestLength) { + widest = index; + widestLength = value.length; + } + } + // Nothing left to shrink: the overflow is elsewhere, and the caller refuses rather than + // silently altering a field that is not tool output. + if (widest < 0 || widestLength === 0) return { record, truncated }; + const entry = next[widest] as Record; + const current = entry.result as string; + const original = originalLengths.get(widest) ?? current.length; + originalLengths.set(widest, original); + // Cut by what the record is actually over, plus room for the marker and for JSON escaping, + // which can widen a byte count well past the character count. + const serialized = Buffer.byteLength(JSON.stringify(candidate) ?? "", "utf8"); + const overflow = Math.max(0, serialized - CONTRACT_MAX_JSON_BYTES); + const marker = toolOutputTruncationMarker(0, original).length; + const cut = Math.max(1, overflow + marker + 1024); + const kept = Math.max(0, current.length - cut); + entry.result = current.slice(0, kept) + toolOutputTruncationMarker(kept, original); + truncated += 1; + } + return { record, truncated }; +} + export function validatedSandboxPromptResult( result: PromptResult, ): ValidatedSandboxPromptResult { @@ -572,16 +645,37 @@ export function validatedSandboxPromptResult( // The Sandbox SDK materializes absent optional response fields as // `undefined`. They were absent on the JSON wire and must stay absent in the // provider-neutral result before the strict JSON check runs. - const record = Object.fromEntries( + let record = Object.fromEntries( Object.entries(source).filter( ([field, value]) => value !== undefined || !SANDBOX_OPTIONAL_RESULT_FIELDS.has(field), ), ); - const content = boundedEventContentRecordSchema.safeParse(record); + // Oversized tool output is TRUNCATED, never thrown. + // + // This validator runs inside the terminal result read — after the live stream has drained and + // after the usage receipt has been credited — so a throw here cannot prevent the work or the + // charge. It can only discard a finished, fully paid turn, which a supervisor then reports as a + // child that did nothing at all. That failure mode cost one Discovery Lab 143 of 199 children + // across 16 pursuits, and its six-stage sourcing graph blocked in 24 of 24 invocations. + // + // Widening the bound alone does not remove it. The Sandbox SDK serializes each tool value up to + // MAX_SERIALIZED_TOOL_VALUE_BYTES (4 MiB) while this bound is CONTRACT_MAX_JSON_BYTES (1 MiB) + // across the WHOLE record, so a single 2 MiB fetch still dies, and so do two 0.6 MiB fetches + // together. Any bound that throws on this path is one large page away from discarding a paid + // turn again. + // + // So the bound is enforced by replacing what does not fit, and saying so in the record. The + // caller still receives the turn, its response text, its usage, and every tool call it made; + // what it loses is the tail of an oversized tool result, marked where it was cut. Only tool + // output is truncatable: every other field is identity, accounting, or control material where a + // silently shortened value would be worse than a refusal, so those still refuse below. + const bounded = withTruncatedToolOutput(record); + const content = boundedEventContentRecordSchema.safeParse(bounded.record); if (!content.success) { throw new Error("Tangle prompt result exceeded its JSON bound", { cause: content.error }); } + record = bounded.record; // What the agent produced is content and is bounded as content by the check above; the fields // that describe the turn keep their metadata limits. //