Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 5 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -92,6 +92,8 @@ The simulator can call a deployed AWS Lambda function that invokes Amazon Bedroc

The [recorded September 4 deployment verification](docs/verification/aws-protection-2026-09-04.md) documents `flo-bedrock-narrator` reaching `UPDATE_COMPLETE` in `us-west-2`, successful signed narration, rejected unsigned/invalid requests, seven-day log retention and a retained DynamoDB model-attempt allowance. Caller authentication is IAM/SigV4; a build marker is not authentication. The allowance and throttling are not an account-wide dollar cap. The shop simulator needs an authorized server-side AWS identity for optional narration and otherwise falls back locally. The vehicle-owner preview never calls Bedrock. No AWS credentials belong in browser code.

The [October 3 browser-path verification](docs/verification/browser-narration-2026-10-03.md) exercised one real shop comparison through the simulator's signer to AWS and back to the visible response, within the unchanged 2.5-second deadline. The route worked, but the lead incorrectly applied the recommended option's quality tier to every option; narration accuracy remains an open verification gap. The record also retains an earlier authorization failure and a prompt-concision deviation. The narration badge now distinguishes configuration awaiting a result, the last accepted narration, and local fallback after a failed attempt; configuration alone does not show success.

### Intended full AWS deployment

```mermaid
Expand Down Expand Up @@ -278,8 +280,9 @@ and scheduling, resumed context and customer-only estimates. Run it only against
a disposable local demo. The database suite uses DynamoDB Local in an isolated
network with synthetic identities; it does not verify live AWS IAM or Amazon
sign-in. Platform-specific filesystem tests run on Linux; Windows refusal tests
run on Windows. CI checks both platforms. Hosted AWS observations above remain
dated September evidence.
run on Windows. CI checks both platforms. Hosted AWS observations are dated in
their linked records; the October 3 narrator check covers one browser request,
not general Alexa+ deployment or model quality.

## Adding an integration

Expand Down
18 changes: 16 additions & 2 deletions apps/alexa-simulator/public/app.js
Original file line number Diff line number Diff line change
Expand Up @@ -12,8 +12,10 @@ const traceItems = $("#traceItems");
const contextList = $("#contextList");
const send = form.querySelector(".send");
const talk = $("#talk");
const awsStatus = $("#awsStatus");
const state = { workOrder: null, asset: null, approval: "Not requested", pending: "None" };
let lastResult = null;
let lastNarrationOk = null;

const escapeHtml = (value) => String(value ?? "").replace(/[&<>'"]/g, (character) => ({ "&": "&amp;", "<": "&lt;", ">": "&gt;", "'": "&#39;", '"': "&quot;" })[character]);
const dollars = (cents) => new Intl.NumberFormat("en-US", { style: "currency", currency: "USD" }).format(Number(cents ?? 0) / 100);
Expand Down Expand Up @@ -41,6 +43,14 @@ function renderTrace(items = []) {
<div class="trace-item"><code>${index + 1}. ${escapeHtml((item.kind ?? "mcp").toUpperCase())} · ${escapeHtml(item.tool)}</code><div class="trace-meta"><span>${item.ok ? "SUCCESS" : "FAILED"}</span><span>${escapeHtml(item.durationMs)} ms</span></div></div>`).join("") : `<p class="muted">No tools invoked in this turn.</p>`;
}

function renderNarrationStatus(configured = false) {
if (!awsStatus) return;
const label = lastNarrationOk === true ? "Bedrock · last narration succeeded"
: lastNarrationOk === false ? "Local fallback · last Bedrock attempt failed"
: configured ? "Bedrock configured · awaiting first result" : "AWS optional";
awsStatus.innerHTML = `<i class="dot ${lastNarrationOk === true ? "live" : ""}"></i>${label}`;
}

function offersFrom(data) {
const ranked = data?.ranked ?? data?.recommendations ?? data?.offers ?? [];
if (!Array.isArray(ranked)) return [];
Expand Down Expand Up @@ -111,6 +121,11 @@ async function runCommand(command) {
const response = await browserRequest("/api/command", { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify({ command }) });
const result = await response.json();
lastResult = result;
const narration = result.invocations?.findLast((item) => item.tool === "amazon_bedrock_narration" && item.kind === "aws" && typeof item.ok === "boolean");
if (narration) {
lastNarrationOk = narration.ok;
renderNarrationStatus();
}
addMessage("assistant", result.voice); renderTrace(result.invocations); render(result.view, result.data);
return { ok: result.ok, voice: result.voice, view: result.view, tools: result.invocations?.map((item) => item.tool) ?? [] };
} catch (error) {
Expand Down Expand Up @@ -169,8 +184,7 @@ fetch("/api/health").then(async (response) => {
const health = await response.json();
connection.className = `connection ${response.ok ? "ready" : "offline"}`;
connection.innerHTML = `<i></i>${response.ok ? `MCP ${escapeHtml(health.protocol)} · ${health.toolCount} tools` : "MCP unavailable"}`;
const awsStatus = $("#awsStatus");
if (awsStatus) awsStatus.innerHTML = `<i class="dot ${health.bedrockNarration ? "live" : ""}"></i>${health.bedrockNarration ? "Amazon Bedrock narration" : "AWS optional"}`;
renderNarrationStatus(response.ok && health.bedrockNarration === true);
}).catch(() => { connection.className = "connection offline"; connection.innerHTML = "<i></i>MCP unavailable"; });

const modelContext = document.modelContext;
Expand Down
2 changes: 1 addition & 1 deletion apps/alexa-simulator/public/index.html
Original file line number Diff line number Diff line change
Expand Up @@ -35,7 +35,7 @@ <h1 id="page-title">Keep your hands on the work.</h1>
<span><i class="dot live"></i>Live MCP</span>
<span><i class="dot"></i>Deterministic pricing</span>
<span><i class="dot"></i>Confirmation gated</span>
<span id="awsStatus"><i class="dot"></i>AWS optional</span>
<span id="awsStatus" role="status"><i class="dot"></i>Narration status pending</span>
</div>
</section>

Expand Down
105 changes: 105 additions & 0 deletions docs/verification/browser-narration-2026-10-03.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,105 @@
# Browser narration verification — October 3, 2026

One shop-simulator comparison completed the real browser → local simulator →
IAM/SigV4 request → AWS narrator → visible browser response path. This is bounded
integration evidence for the optional narration sentence. It does not establish
Alexa+ deployment, account linking, certification, general model quality, or
reliability across repeated requests.

## Real request

The test used Chrome 154.0.8037.98 on Windows and service/runtime source
`4eb1cc7bc6e206d4b577b12895ee8161370d7d2a`. It opened work order 1842, recorded
“The alternator failed,” then submitted “Find compatible replacements under $300
that can arrive tomorrow” through the visible command form. The simulator's
signer, model configuration, response validation and 2,500 ms deadline were
unchanged.

At 02:31 UTC, the single comparison returned HTTP 200 from the AWS endpoint in
1,807 ms. The simulator recorded `amazon_bedrock_narration`, `kind: aws`,
`ok: true` in 1,819.5 ms and displayed this lead before its deterministic
comparison:

> Evaluating four premium service-part choices aids a technician in making a confident selection.

The route and display checks passed, but the lead has a factual scope error:
the four eligible offers have budget, premium, OEM and premium tiers. The caller
sends `comparison.ranked.length` together with
`comparison.recommendation.part.qualityTier`; the Lambda prompt incorrectly
applies the recommendation's tier to every option. Saying “four premium” therefore
mischaracterizes the comparison. The tier composition was checked against the
same source and command, without another paid request. Narration accuracy remains
an open verification gap.

The source correction labels the outgoing model context as `optionCount` and
`recommendedQualityTier`, explicitly restricts the tier to the recommendation,
and requests a tier-neutral lead. A regression inspects the actual Converse
input for each allowed tier through the existing local Lambda test harness.
This correction was deployed as recorded below. No further paid model request
was made for this deployment-only follow-up, so the corrected production lead
remains unverified; the historical output above remains unchanged.

The lead also has 13 words. It satisfies the runtime's limit of 16 words and its
no-digits/no-prices display contract, but exceeds the prompt's request for at
most 12 words. The narrator did not independently inspect or rank the offers;
accepted narration is not proof of factual accuracy or full prompt compliance.

A consistent allowance read changed from 97 remaining / 3 used to 96 remaining
/ 4 used. No purchase, scheduling, customer approval or monetary calculation was
delegated to the model. The browser reported no page errors or external browser
requests. All six local services and the isolated browser were stopped afterward.

## Retained failure and authorization boundary

An earlier attempt at 02:02 UTC returned HTTP 403 in 188 ms. Its narration trace
reported `ok: false`, the simulator displayed deterministic fallback, and the
allowance stayed at 97 remaining / 3 used. The caller lacked permission to invoke
that route; IAM simulation reported an implicit denial.

The successful attempt used a separately approved, temporary permission for the
exact narrator route. That permission was removed immediately afterward. The
caller again had no inline policies, its existing sign-in policy was unchanged,
and simulation again denied route invocation. This verification does not leave
the caller authorized or make AWS narration available to every local installation.

## Status badge regression checks

The previous badge turned green when narration was configured, even after the
403 caused local fallback. The updated badge uses actual narration outcomes:

- Configuration alone: “Bedrock configured · awaiting first result,” with an amber dot.
- Latest successful narration: “Bedrock · last narration succeeded,” with a green dot.
- Latest failed attempt: “Local fallback · last Bedrock attempt failed,” with an amber dot.

Only a Boolean `ok` value from an `aws` invocation of
`amazon_bedrock_narration` updates this history. Unrelated commands preserve it;
a delayed health response cannot replace a recorded result with configuration.
Here, success means the service returned a lead accepted by the display validator;
it does not certify the model's factual accuracy.

Seven separate model-free browser cases exercised initial/configured state,
failure and unrelated commands, success followed by failure, both delayed-health
races, optional/unrecognized invocations, and responsive accessibility. They
passed 17 state assertions through eight actual form submissions. Six screenshots
at 1440 and 390 pixels were visually checked, with no clipping or horizontal
overflow, page/console errors, or unexpected requests. Browser and listener
cleanup passed. These cases used local fixture responses, including a synthetic
success trace; they are display tests, separate from the single real AWS request
above. The focused permanent regression is
`tests/integration/narration-status-ui.test.ts`.

## Deployment-only follow-up — October 4, 2026

At 06:04:49 UTC, `flo-bedrock-narrator` was observed in `UPDATE_COMPLETE` and
change set `flo-narration-context-20261003T0304Z` in `EXECUTE_COMPLETE`. The
deployed template exactly matched the reviewed candidate
`2c297e9ea1802c135730126e9fe4f19b7cf1a8ef`; comparison with the previous template
found only `NarratorFunction.Properties.Code.ZipFile` changed. CloudFormation
also recorded a dependent integration URI update; neither resource was replaced.

The function reported `Active` and `Successful`. Configured model, runtime name,
IAM role, memory and timeout, plus stack parameters and outputs, were unchanged.
The allowance remained 96 remaining / 4 used. Existing local and CI results are
retained; no additional Invoke grant or paid model request was made. The corrected
production lead was not resampled, so this deployment does not establish improved
model accuracy or prompt compliance.
4 changes: 2 additions & 2 deletions infra/aws/bedrock-narrator/template.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -236,8 +236,8 @@ Resources:
}
throw new Error("Narration allowance unavailable");
}
const qualityTier = body.qualityTier;
const prompt = `Write one calm sentence of at most twelve words explaining that comparing ${body.optionCount} ${qualityTier} service-part options helps a technician choose confidently. Do not use digits, prices, dates, supplier names, part numbers, claims of completion, or markdown.`;
const comparisonContext = { optionCount: body.optionCount, recommendedQualityTier: body.qualityTier };
const prompt = `Write one calm sentence of at most twelve words explaining that comparing service-part options helps a technician choose confidently. Mention the total option count in words; do not describe the options' quality tiers. Context: ${JSON.stringify(comparisonContext)}. optionCount counts all eligible options. recommendedQualityTier describes only the recommended option, not every option. Do not use digits, prices, dates, supplier names, part numbers, claims of completion, or markdown.`;
const command = new ConverseCommand({
modelId: process.env.MODEL_ID,
messages: [{ role: "user", content: [{ text: prompt }] }],
Expand Down
67 changes: 67 additions & 0 deletions tests/integration/narration-status-ui.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,67 @@
import assert from "node:assert/strict";
import { readFileSync } from "node:fs";
import { setImmediate } from "node:timers/promises";
import { describe, it } from "node:test";
import { runInNewContext } from "node:vm";

const source = readFileSync(new URL("../../../apps/alexa-simulator/public/app.js", import.meta.url), "utf8")
.replace(/^import .* from '.+';\r?\n/gm, "");
const html = readFileSync(new URL("../../../apps/alexa-simulator/public/index.html", import.meta.url), "utf8");
type Listener = (event: { preventDefault(): void }) => void;
class Element {
innerHTML = ""; textContent = ""; className = ""; value = "";
disabled = false; hidden = false; scrollTop = 0; scrollHeight = 0;
listeners = new Map<string, Listener>();
addEventListener(name: string, listener: Listener) { this.listeners.set(name, listener); }
querySelector() { return new Element(); }
append() { /* Transcript rendering is outside these status assertions. */ }
focus() { /* No real focus in the unit DOM. */ }
async emit(name: string) { this.listeners.get(name)?.({ preventDefault() {} }); await setImmediate(); }
}
function harness() {
const elements = new Map([...html.matchAll(/\bid="([^"]+)"/g)].map(match => [match[1]!, new Element()]));
const get = (id: string) => { const element = elements.get(id); assert.ok(element, `Missing HTML element: ${id}`); return element; };
const pending: unknown[][] = [];
let finishHealth!: (response: Response) => void;
const health = new Promise<Response>(resolve => { finishHealth = resolve; });
runInNewContext(source, {
document: { querySelector: (selector: string) => get(selector.slice(1)), createElement: () => new Element() },
window: {}, HTMLButtonElement: Element,
fetch: (path: string) => { assert.equal(path, "/api/health"); return health; },
browserRequest: (path: string, options: RequestInit) => {
assert.equal(path, "/api/command"); assert.equal(options.method, "POST");
assert.deepEqual(JSON.parse(options.body as string), { command: "Check narration" });
const invocations = pending.shift(); assert.ok(invocations, "Unexpected command");
return Promise.resolve(Response.json({ ok: true, voice: "Test response", view: "help", data: { examples: [] }, invocations }));
}
});
return {
badge: get("awsStatus"),
async health(configured: boolean) { finishHealth(Response.json({ bedrockNarration: configured, protocol: "test", toolCount: 0 })); await setImmediate(); },
async command(invocations: unknown[]) { pending.push(invocations); get("command").value = "Check narration"; await get("commandForm").emit("submit"); }
};
}
const narration = (ok: unknown) => ({ tool: "amazon_bedrock_narration", kind: "aws", ok, durationMs: 1 });
const live = (badge: Element) => /class="dot live"/.test(badge.innerHTML);

describe("shop narration status reflects results rather than configuration", () => {
it("requires an actual boolean AWS result and retains it across unrelated commands", async () => {
const ui = harness(); await ui.health(true);
assert.match(ui.badge.innerHTML, /awaiting first result/); assert.equal(live(ui.badge), false);
await ui.command([{ ...narration(true), tool: "other_tool" }, { ...narration(true), kind: "mcp" }, narration("true")]);
assert.match(ui.badge.innerHTML, /awaiting first result/); assert.equal(live(ui.badge), false);
await ui.command([narration(true)]);
assert.match(ui.badge.innerHTML, /last narration succeeded/); assert.equal(live(ui.badge), true);
await ui.command([]); assert.match(ui.badge.innerHTML, /last narration succeeded/);
await ui.command([narration(false)]);
assert.match(ui.badge.innerHTML, /Local fallback/); assert.equal(live(ui.badge), false);
});
it("does not let late configured health replace a failed narration", async () => {
const ui = harness(); await ui.command([narration(false)]); await ui.health(true);
assert.match(ui.badge.innerHTML, /last Bedrock attempt failed/); assert.equal(live(ui.badge), false);
});
it("does not let late unconfigured health erase the last successful narration", async () => {
const ui = harness(); await ui.command([narration(true)]); await ui.health(false);
assert.match(ui.badge.innerHTML, /last narration succeeded/); assert.equal(live(ui.badge), true);
});
});
Loading
Loading