diff --git a/README.md b/README.md index 5ec81a0..81bc8af 100644 --- a/README.md +++ b/README.md @@ -92,6 +92,8 @@ The simulator can call a deployed AWS Lambda function that invokes Amazon Bedroc The [recorded September 4 deployment verification](docs/verification/aws-protection-2026-09-04.md) documents `flo-bedrock-narrator` reaching `UPDATE_COMPLETE` in `us-west-2`, successful signed narration, rejected unsigned/invalid requests, seven-day log retention and a retained DynamoDB model-attempt allowance. Caller authentication is IAM/SigV4; a build marker is not authentication. The allowance and throttling are not an account-wide dollar cap. The shop simulator needs an authorized server-side AWS identity for optional narration and otherwise falls back locally. The vehicle-owner preview never calls Bedrock. No AWS credentials belong in browser code. +The [October 3 browser-path verification](docs/verification/browser-narration-2026-10-03.md) exercised one real shop comparison through the simulator's signer to AWS and back to the visible response, within the unchanged 2.5-second deadline. The route worked, but the lead incorrectly applied the recommended option's quality tier to every option; narration accuracy remains an open verification gap. The record also retains an earlier authorization failure and a prompt-concision deviation. The narration badge now distinguishes configuration awaiting a result, the last accepted narration, and local fallback after a failed attempt; configuration alone does not show success. + ### Intended full AWS deployment ```mermaid @@ -278,8 +280,9 @@ and scheduling, resumed context and customer-only estimates. Run it only against a disposable local demo. The database suite uses DynamoDB Local in an isolated network with synthetic identities; it does not verify live AWS IAM or Amazon sign-in. Platform-specific filesystem tests run on Linux; Windows refusal tests -run on Windows. CI checks both platforms. Hosted AWS observations above remain -dated September evidence. +run on Windows. CI checks both platforms. Hosted AWS observations are dated in +their linked records; the October 3 narrator check covers one browser request, +not general Alexa+ deployment or model quality. ## Adding an integration diff --git a/apps/alexa-simulator/public/app.js b/apps/alexa-simulator/public/app.js index 028415c..8d7b1b6 100644 --- a/apps/alexa-simulator/public/app.js +++ b/apps/alexa-simulator/public/app.js @@ -12,8 +12,10 @@ const traceItems = $("#traceItems"); const contextList = $("#contextList"); const send = form.querySelector(".send"); const talk = $("#talk"); +const awsStatus = $("#awsStatus"); const state = { workOrder: null, asset: null, approval: "Not requested", pending: "None" }; let lastResult = null; +let lastNarrationOk = null; const escapeHtml = (value) => String(value ?? "").replace(/[&<>'"]/g, (character) => ({ "&": "&", "<": "<", ">": ">", "'": "'", '"': """ })[character]); const dollars = (cents) => new Intl.NumberFormat("en-US", { style: "currency", currency: "USD" }).format(Number(cents ?? 0) / 100); @@ -41,6 +43,14 @@ function renderTrace(items = []) {
${index + 1}. ${escapeHtml((item.kind ?? "mcp").toUpperCase())} · ${escapeHtml(item.tool)}
${item.ok ? "SUCCESS" : "FAILED"}${escapeHtml(item.durationMs)} ms
`).join("") : `

No tools invoked in this turn.

`; } +function renderNarrationStatus(configured = false) { + if (!awsStatus) return; + const label = lastNarrationOk === true ? "Bedrock · last narration succeeded" + : lastNarrationOk === false ? "Local fallback · last Bedrock attempt failed" + : configured ? "Bedrock configured · awaiting first result" : "AWS optional"; + awsStatus.innerHTML = `${label}`; +} + function offersFrom(data) { const ranked = data?.ranked ?? data?.recommendations ?? data?.offers ?? []; if (!Array.isArray(ranked)) return []; @@ -111,6 +121,11 @@ async function runCommand(command) { const response = await browserRequest("/api/command", { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify({ command }) }); const result = await response.json(); lastResult = result; + const narration = result.invocations?.findLast((item) => item.tool === "amazon_bedrock_narration" && item.kind === "aws" && typeof item.ok === "boolean"); + if (narration) { + lastNarrationOk = narration.ok; + renderNarrationStatus(); + } addMessage("assistant", result.voice); renderTrace(result.invocations); render(result.view, result.data); return { ok: result.ok, voice: result.voice, view: result.view, tools: result.invocations?.map((item) => item.tool) ?? [] }; } catch (error) { @@ -169,8 +184,7 @@ fetch("/api/health").then(async (response) => { const health = await response.json(); connection.className = `connection ${response.ok ? "ready" : "offline"}`; connection.innerHTML = `${response.ok ? `MCP ${escapeHtml(health.protocol)} · ${health.toolCount} tools` : "MCP unavailable"}`; - const awsStatus = $("#awsStatus"); - if (awsStatus) awsStatus.innerHTML = `${health.bedrockNarration ? "Amazon Bedrock narration" : "AWS optional"}`; + renderNarrationStatus(response.ok && health.bedrockNarration === true); }).catch(() => { connection.className = "connection offline"; connection.innerHTML = "MCP unavailable"; }); const modelContext = document.modelContext; diff --git a/apps/alexa-simulator/public/index.html b/apps/alexa-simulator/public/index.html index a4aff54..adf140e 100644 --- a/apps/alexa-simulator/public/index.html +++ b/apps/alexa-simulator/public/index.html @@ -35,7 +35,7 @@

Keep your hands on the work.

Live MCP Deterministic pricing Confirmation gated - AWS optional + Narration status pending diff --git a/docs/verification/browser-narration-2026-10-03.md b/docs/verification/browser-narration-2026-10-03.md new file mode 100644 index 0000000..a25b14f --- /dev/null +++ b/docs/verification/browser-narration-2026-10-03.md @@ -0,0 +1,105 @@ +# Browser narration verification — October 3, 2026 + +One shop-simulator comparison completed the real browser → local simulator → +IAM/SigV4 request → AWS narrator → visible browser response path. This is bounded +integration evidence for the optional narration sentence. It does not establish +Alexa+ deployment, account linking, certification, general model quality, or +reliability across repeated requests. + +## Real request + +The test used Chrome 154.0.8037.98 on Windows and service/runtime source +`4eb1cc7bc6e206d4b577b12895ee8161370d7d2a`. It opened work order 1842, recorded +“The alternator failed,” then submitted “Find compatible replacements under $300 +that can arrive tomorrow” through the visible command form. The simulator's +signer, model configuration, response validation and 2,500 ms deadline were +unchanged. + +At 02:31 UTC, the single comparison returned HTTP 200 from the AWS endpoint in +1,807 ms. The simulator recorded `amazon_bedrock_narration`, `kind: aws`, +`ok: true` in 1,819.5 ms and displayed this lead before its deterministic +comparison: + +> Evaluating four premium service-part choices aids a technician in making a confident selection. + +The route and display checks passed, but the lead has a factual scope error: +the four eligible offers have budget, premium, OEM and premium tiers. The caller +sends `comparison.ranked.length` together with +`comparison.recommendation.part.qualityTier`; the Lambda prompt incorrectly +applies the recommendation's tier to every option. Saying “four premium” therefore +mischaracterizes the comparison. The tier composition was checked against the +same source and command, without another paid request. Narration accuracy remains +an open verification gap. + +The source correction labels the outgoing model context as `optionCount` and +`recommendedQualityTier`, explicitly restricts the tier to the recommendation, +and requests a tier-neutral lead. A regression inspects the actual Converse +input for each allowed tier through the existing local Lambda test harness. +This correction was deployed as recorded below. No further paid model request +was made for this deployment-only follow-up, so the corrected production lead +remains unverified; the historical output above remains unchanged. + +The lead also has 13 words. It satisfies the runtime's limit of 16 words and its +no-digits/no-prices display contract, but exceeds the prompt's request for at +most 12 words. The narrator did not independently inspect or rank the offers; +accepted narration is not proof of factual accuracy or full prompt compliance. + +A consistent allowance read changed from 97 remaining / 3 used to 96 remaining +/ 4 used. No purchase, scheduling, customer approval or monetary calculation was +delegated to the model. The browser reported no page errors or external browser +requests. All six local services and the isolated browser were stopped afterward. + +## Retained failure and authorization boundary + +An earlier attempt at 02:02 UTC returned HTTP 403 in 188 ms. Its narration trace +reported `ok: false`, the simulator displayed deterministic fallback, and the +allowance stayed at 97 remaining / 3 used. The caller lacked permission to invoke +that route; IAM simulation reported an implicit denial. + +The successful attempt used a separately approved, temporary permission for the +exact narrator route. That permission was removed immediately afterward. The +caller again had no inline policies, its existing sign-in policy was unchanged, +and simulation again denied route invocation. This verification does not leave +the caller authorized or make AWS narration available to every local installation. + +## Status badge regression checks + +The previous badge turned green when narration was configured, even after the +403 caused local fallback. The updated badge uses actual narration outcomes: + +- Configuration alone: “Bedrock configured · awaiting first result,” with an amber dot. +- Latest successful narration: “Bedrock · last narration succeeded,” with a green dot. +- Latest failed attempt: “Local fallback · last Bedrock attempt failed,” with an amber dot. + +Only a Boolean `ok` value from an `aws` invocation of +`amazon_bedrock_narration` updates this history. Unrelated commands preserve it; +a delayed health response cannot replace a recorded result with configuration. +Here, success means the service returned a lead accepted by the display validator; +it does not certify the model's factual accuracy. + +Seven separate model-free browser cases exercised initial/configured state, +failure and unrelated commands, success followed by failure, both delayed-health +races, optional/unrecognized invocations, and responsive accessibility. They +passed 17 state assertions through eight actual form submissions. Six screenshots +at 1440 and 390 pixels were visually checked, with no clipping or horizontal +overflow, page/console errors, or unexpected requests. Browser and listener +cleanup passed. These cases used local fixture responses, including a synthetic +success trace; they are display tests, separate from the single real AWS request +above. The focused permanent regression is +`tests/integration/narration-status-ui.test.ts`. + +## Deployment-only follow-up — October 4, 2026 + +At 06:04:49 UTC, `flo-bedrock-narrator` was observed in `UPDATE_COMPLETE` and +change set `flo-narration-context-20261003T0304Z` in `EXECUTE_COMPLETE`. The +deployed template exactly matched the reviewed candidate +`2c297e9ea1802c135730126e9fe4f19b7cf1a8ef`; comparison with the previous template +found only `NarratorFunction.Properties.Code.ZipFile` changed. CloudFormation +also recorded a dependent integration URI update; neither resource was replaced. + +The function reported `Active` and `Successful`. Configured model, runtime name, +IAM role, memory and timeout, plus stack parameters and outputs, were unchanged. +The allowance remained 96 remaining / 4 used. Existing local and CI results are +retained; no additional Invoke grant or paid model request was made. The corrected +production lead was not resampled, so this deployment does not establish improved +model accuracy or prompt compliance. diff --git a/infra/aws/bedrock-narrator/template.yaml b/infra/aws/bedrock-narrator/template.yaml index bc9bdf1..cbba7cf 100644 --- a/infra/aws/bedrock-narrator/template.yaml +++ b/infra/aws/bedrock-narrator/template.yaml @@ -236,8 +236,8 @@ Resources: } throw new Error("Narration allowance unavailable"); } - const qualityTier = body.qualityTier; - const prompt = `Write one calm sentence of at most twelve words explaining that comparing ${body.optionCount} ${qualityTier} service-part options helps a technician choose confidently. Do not use digits, prices, dates, supplier names, part numbers, claims of completion, or markdown.`; + const comparisonContext = { optionCount: body.optionCount, recommendedQualityTier: body.qualityTier }; + const prompt = `Write one calm sentence of at most twelve words explaining that comparing service-part options helps a technician choose confidently. Mention the total option count in words; do not describe the options' quality tiers. Context: ${JSON.stringify(comparisonContext)}. optionCount counts all eligible options. recommendedQualityTier describes only the recommended option, not every option. Do not use digits, prices, dates, supplier names, part numbers, claims of completion, or markdown.`; const command = new ConverseCommand({ modelId: process.env.MODEL_ID, messages: [{ role: "user", content: [{ text: prompt }] }], diff --git a/tests/integration/narration-status-ui.test.ts b/tests/integration/narration-status-ui.test.ts new file mode 100644 index 0000000..d06cec0 --- /dev/null +++ b/tests/integration/narration-status-ui.test.ts @@ -0,0 +1,67 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { setImmediate } from "node:timers/promises"; +import { describe, it } from "node:test"; +import { runInNewContext } from "node:vm"; + +const source = readFileSync(new URL("../../../apps/alexa-simulator/public/app.js", import.meta.url), "utf8") + .replace(/^import .* from '.+';\r?\n/gm, ""); +const html = readFileSync(new URL("../../../apps/alexa-simulator/public/index.html", import.meta.url), "utf8"); +type Listener = (event: { preventDefault(): void }) => void; +class Element { + innerHTML = ""; textContent = ""; className = ""; value = ""; + disabled = false; hidden = false; scrollTop = 0; scrollHeight = 0; + listeners = new Map(); + addEventListener(name: string, listener: Listener) { this.listeners.set(name, listener); } + querySelector() { return new Element(); } + append() { /* Transcript rendering is outside these status assertions. */ } + focus() { /* No real focus in the unit DOM. */ } + async emit(name: string) { this.listeners.get(name)?.({ preventDefault() {} }); await setImmediate(); } +} +function harness() { + const elements = new Map([...html.matchAll(/\bid="([^"]+)"/g)].map(match => [match[1]!, new Element()])); + const get = (id: string) => { const element = elements.get(id); assert.ok(element, `Missing HTML element: ${id}`); return element; }; + const pending: unknown[][] = []; + let finishHealth!: (response: Response) => void; + const health = new Promise(resolve => { finishHealth = resolve; }); + runInNewContext(source, { + document: { querySelector: (selector: string) => get(selector.slice(1)), createElement: () => new Element() }, + window: {}, HTMLButtonElement: Element, + fetch: (path: string) => { assert.equal(path, "/api/health"); return health; }, + browserRequest: (path: string, options: RequestInit) => { + assert.equal(path, "/api/command"); assert.equal(options.method, "POST"); + assert.deepEqual(JSON.parse(options.body as string), { command: "Check narration" }); + const invocations = pending.shift(); assert.ok(invocations, "Unexpected command"); + return Promise.resolve(Response.json({ ok: true, voice: "Test response", view: "help", data: { examples: [] }, invocations })); + } + }); + return { + badge: get("awsStatus"), + async health(configured: boolean) { finishHealth(Response.json({ bedrockNarration: configured, protocol: "test", toolCount: 0 })); await setImmediate(); }, + async command(invocations: unknown[]) { pending.push(invocations); get("command").value = "Check narration"; await get("commandForm").emit("submit"); } + }; +} +const narration = (ok: unknown) => ({ tool: "amazon_bedrock_narration", kind: "aws", ok, durationMs: 1 }); +const live = (badge: Element) => /class="dot live"/.test(badge.innerHTML); + +describe("shop narration status reflects results rather than configuration", () => { + it("requires an actual boolean AWS result and retains it across unrelated commands", async () => { + const ui = harness(); await ui.health(true); + assert.match(ui.badge.innerHTML, /awaiting first result/); assert.equal(live(ui.badge), false); + await ui.command([{ ...narration(true), tool: "other_tool" }, { ...narration(true), kind: "mcp" }, narration("true")]); + assert.match(ui.badge.innerHTML, /awaiting first result/); assert.equal(live(ui.badge), false); + await ui.command([narration(true)]); + assert.match(ui.badge.innerHTML, /last narration succeeded/); assert.equal(live(ui.badge), true); + await ui.command([]); assert.match(ui.badge.innerHTML, /last narration succeeded/); + await ui.command([narration(false)]); + assert.match(ui.badge.innerHTML, /Local fallback/); assert.equal(live(ui.badge), false); + }); + it("does not let late configured health replace a failed narration", async () => { + const ui = harness(); await ui.command([narration(false)]); await ui.health(true); + assert.match(ui.badge.innerHTML, /last Bedrock attempt failed/); assert.equal(live(ui.badge), false); + }); + it("does not let late unconfigured health erase the last successful narration", async () => { + const ui = harness(); await ui.command([narration(true)]); await ui.health(false); + assert.match(ui.badge.innerHTML, /last narration succeeded/); assert.equal(live(ui.badge), true); + }); +}); diff --git a/tests/integration/narrator-boundary.test.ts b/tests/integration/narrator-boundary.test.ts index 2b55ceb..36ec3c8 100644 --- a/tests/integration/narrator-boundary.test.ts +++ b/tests/integration/narrator-boundary.test.ts @@ -13,12 +13,14 @@ const valid = { requestContext: { authorizer: { iam: { userArn: "arn:aws:iam::12 const harness = (state: Ledger, modelFails = false, configured = true) => { let calls = 0; + const modelInputs: Record[] = []; const result = { handler: undefined as unknown as (event: unknown) => Promise }; class Command { constructor(readonly input: Record) {} } class Bedrock { constructor(options: { maxAttempts: number }) { assert.equal(options.maxAttempts, 1); } - async send() { + async send(command: Command) { calls++; + modelInputs.push(command.input); if (modelFails) throw new Error("timeout after possible charge"); return { output: { message: { content: [{ text: "Comparing quality options helps technicians choose confidently." }] } } }; } @@ -46,10 +48,24 @@ const harness = (state: Ledger, modelFails = false, configured = true) => { ? { DynamoDBClient: Dynamo, UpdateItemCommand: Command } : { BedrockRuntimeClient: Bedrock, ConverseCommand: Command } }); - return { invoke: result.handler, calls: () => calls }; + return { invoke: result.handler, calls: () => calls, modelInputs }; }; describe("deployed narrator authentication and lifetime allowance", () => { + it("keeps the total option count separate from the recommendation's quality tier", async () => { + for (const qualityTier of ["budget", "standard", "premium", "oem"]) { + const h = harness({ remaining: 1, used: 0 }); + const response = await h.invoke({ ...valid, body: JSON.stringify({ task: "part-comparison-lead", optionCount: 4, qualityTier }) }); + assert.equal(response.statusCode, 200); assert.equal(h.calls(), 1); + const messages = h.modelInputs[0]!.messages as { role: string; content: { text: string }[] }[]; + const prompt = messages[0]!.content[0]!.text; + const context = /Context: (\{[^\n]+?\})\./.exec(prompt); + assert.ok(context, "Model input must label comparison context"); + assert.deepEqual(JSON.parse(context[1]!), { optionCount: 4, recommendedQualityTier: qualityTier }); + assert.match(prompt, /recommendedQualityTier describes only the recommended option, not every option/); + assert.match(prompt, /do not describe the options' quality tiers/); + } + }); it("bounds a stalled credential signer and never accepts credential-leaking destinations", async () => { const source = readFileSync(new URL("../../../apps/alexa-simulator/src/aws-narrator-auth.ts", import.meta.url), "utf8"); const compiled = ts.transpileModule(source, { compilerOptions: { module: ts.ModuleKind.CommonJS, target: ts.ScriptTarget.ES2022 } }).outputText;