From 390eea5a53ebaf5c073f7911ca0a702c90ecf76f Mon Sep 17 00:00:00 2001 From: Daniel Ochoa Date: Wed, 30 Sep 2026 18:24:53 -0500 Subject: [PATCH 1/2] fix(code): ask the human only what needs human eyes - Present a guided-manual-qa checkpoint to the human only when exact-head E2E does not verify it and the agent cannot reliably verify it itself - Record agent-verified checkpoints as AGENT_VERIFIED with agent-observed evidence, never as a human PASS - Route inconclusive agent observations to the human and settle contradicting ones as setup, oracle, or a candidate finding - Report human, AGENT_VERIFIED, E2E_COVERED, and BLOCKED counts separately and list why each human checkpoint needed a human - Pin the routing rule in the skill contract test Testing: npm test and npm run typecheck in tools/guided-manual-qa; the new contract test fails against the previous skill text Risks: None identified; skill text and template only --- CHANGELOG.md | 8 ++++++ plugins/code/.claude-plugin/plugin.json | 2 +- plugins/code/skills/guided-manual-qa/SKILL.md | 25 +++++++++++++------ .../references/plan-methodology.md | 4 +-- .../references/qa-record-template.md | 12 +++++---- .../src/skill-contract.test.ts | 23 +++++++++++++++++ 6 files changed, 59 insertions(+), 15 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index bbcbb14b..e3cdd766 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,14 @@ All notable changes to the claude-plugins project will be documented in this fil The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). Entries are listed newest-first; each plugin section is treated as released when merged to `main`. +### code v1.16.3 + +#### Changed +- `guided-manual-qa` presents a checkpoint to the human only when passing E2E on the current head does not already verify it and the agent cannot reliably verify it itself. After the oracle proof, the agent records a checkpoint it conclusively observed as `AGENT_VERIFIED` with `agent-observed` evidence and does not present it; it routes to the human only visual or perceptual judgments, flows it cannot drive or observe reliably (real OAuth, OS dialogs, hardware, third-party UIs), and product-judgment calls, and records why. An inconclusive agent observation goes to the human, and a contradicting one is settled as setup, oracle, or a candidate finding; neither is silently passed. +- `AGENT_VERIFIED` and `E2E_COVERED` join the status meanings and are never a human `PASS`. The final summary reports human-confirmed `PASS` and `FAIL`, `AGENT_VERIFIED`, `E2E_COVERED`, and `BLOCKED` counts separately, and lists each checkpoint left to the human with the reason it needed one. The interactive window opens only when at least one checkpoint is routed to the human. +- The QA record template adds `AGENT_VERIFIED` to the checkpoint statuses, a routing line per checkpoint, `agent-observed` as the confirmer for agent-verified attempts, and the separate counts and human-routed list in the final summary. A bug-fix checkpoint can be closed by the agent under the same routing rule. +- `tools/guided-manual-qa/src/skill-contract.test.ts` pins the routing rule, the inconclusive-observation escalation, and the separate summary counts. + ### code v1.16.2 #### Changed diff --git a/plugins/code/.claude-plugin/plugin.json b/plugins/code/.claude-plugin/plugin.json index 115da630..2c33de2f 100644 --- a/plugins/code/.claude-plugin/plugin.json +++ b/plugins/code/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "code", "description": "Code and planning framework plugin", - "version": "1.16.2", + "version": "1.16.3", "author": { "name": "ClosedLoop", "email": "support@closedloop.ai" diff --git a/plugins/code/skills/guided-manual-qa/SKILL.md b/plugins/code/skills/guided-manual-qa/SKILL.md index 13caa005..25813300 100644 --- a/plugins/code/skills/guided-manual-qa/SKILL.md +++ b/plugins/code/skills/guided-manual-qa/SKILL.md @@ -5,7 +5,7 @@ description: Derive and run an interactive, evidence-recorded manual QA session # Guided Manual QA -Run manual QA as a collaboration with the human for behavior that exact-head E2E does not already verify. Inspect the current change and map each proposed checkpoint to the existing E2E assertions before scheduling it. A passing E2E test can close that coverage need but is automated evidence, never a human `PASS`. Prepare a safe test environment for the remaining checkpoints, then present one observable checkpoint at a time. +Run manual QA as a collaboration with the human, and spend the human's time only on what needs human eyes. Present a checkpoint to the human only when both are true: passing E2E on the current head does not already verify it, and the agent cannot reliably verify it itself (see "Prove the checkpoint oracle before asking the human"). Record everything else as `E2E_COVERED` or `AGENT_VERIFIED` with its evidence; neither is ever a human `PASS`. Prepare a safe test environment, then present the remaining human checkpoints one at a time. ## Establish the test contract @@ -14,7 +14,7 @@ Run manual QA as a collaboration with the human for behavior that exact-head E2E 3. If repository instructions require repository or workflow memory, query it before choosing bootstrap, launch, validation, or QA paths. Treat memory as a hint and verify every material instruction against current repository docs and code. 4. Map the shipping boundary and adjacent regression surfaces. Include conditional concerns only when evidence makes them relevant: web, API, desktop/Electron, shared packages, persistence, permissions, feature flags, failure states, responsive layouts, themes, accessibility, packaging, or cross-surface parity. 5. Before scheduling a prototype checkpoint, trace the code responsible for its assertion through actual imports to a production-facing consumer or a Storybook story. The same shared component or behavior is eligible even if the prototype supplies its fixtures; Storybook qualifies as a component-adoption destination before the component reaches production. A similar-looking copy, isolated prototype route, mock-only interaction, or fixture-specific behavior is not eligible merely because production might adopt it later. Do not use isolated prototype behavior as acceptance or bug-fix evidence. Preserve the underlying ticket requirement: move its checkpoint to the actual production or Storybook owner when one exists, or record the unverified ownership gap rather than silently dropping it. Record this reachability decision for each prototype checkpoint. -6. For every candidate manual checkpoint, compare its exact route, host, flag state, fixture transition, action, and expected result with passing E2E on the current head. Record the spec/assertion and run evidence when E2E covers it; schedule human QA only for the uncovered part or an explicitly human-only requirement. A nearby or narrower test does not count as coverage. +6. For every candidate manual checkpoint, compare its exact route, host, flag state, fixture transition, action, and expected result with passing E2E on the current head. Record the spec/assertion and run evidence and mark it `E2E_COVERED` when E2E covers it; plan only the uncovered part or an explicitly human-only requirement as a checkpoint. A nearby or narrower test does not count as coverage. 7. Derive the remaining prioritized manual plan using [references/plan-methodology.md](references/plan-methodology.md). Do not reuse a stale plan merely because it names the same feature. State the proposed scope, environment, fixtures, and known gaps before launching. If requirements and the live change disagree, record the discrepancy and ask for direction when it materially changes what success means. @@ -35,7 +35,7 @@ State the proposed scope, environment, fixtures, and known gaps before launching - A prototype is an optional harness for eligible shared code, not a manual-QA destination by itself. Do not add prototype-only E2E tests. Assess repository-supported E2E coverage for affected production code separately; neither a manual finding nor Storybook reachability automatically requires a new E2E test. - Start services from the resolved worktree. Record the launch commands, working directories, process identifiers, ports, and health checks. Prove that each tested listener belongs to this worktree using the strongest available evidence: process command and cwd, parent process, build or commit marker, service metadata, or a repository-provided diagnostic endpoint. A parent supervisor, launchd wrapper, or control command reporting `started` is not readiness by itself; prove the actual listener and the route-owned ready selector before presenting a checkpoint. If a wrapper hangs before spawning the child or listener, run one bounded foreground diagnostic of the same documented command to distinguish wrapper failure from app/runtime failure, then stop that diagnostic before trying a fallback. - If the QA session spans tool calls or worker turns, keep its services under a repository-supported or OS-supported owner that survives that boundary. Record how to inspect and stop that owner. On resume, read the existing QA record and recheck the current head, owned processes, listeners, data target, and exact route; earlier PIDs and ready checks are historical evidence, not proof that the environment is still available. Then apply "Rebind results after a head change" in [references/plan-methodology.md](references/plan-methodology.md) to every recorded result, name the next `PENDING` checkpoint as the resume point in the record, and continue from it. Do not re-present a checkpoint whose result still applies to the current head. -- After that proof succeeds, launch the UI the user requested for the intentional interactive session. Do not open unrelated surfaces or claim an unlaunched surface was exercised. +- After that proof succeeds and at least one checkpoint is routed to the human, launch the UI the user requested for the intentional interactive session. Do not open unrelated surfaces or claim an unlaunched surface was exercised. - Record feature-flag assignments, roles, permissions, account or fixture identity, and other state that changes the observable result. Redact credentials and secrets. - When the matrix depends on browser-local state such as feature-flag fixtures, configure it before the first app navigation through a supported browser-context mechanism. Read [references/browser-state-fixtures.md](references/browser-state-fixtures.md). Do not make the human open DevTools or paste JavaScript, and do not use `javascript:` URLs, raw CDP, or the user's ordinary browser profile. - When the repository has Playwright installed and no stronger repository launcher exists, run the bundled launcher `scripts/dist/launch-interactive-browser.mjs` (Node 18+, no install step) to open the intentional interactive window with preloaded state. Keep its process alive through the checkpoint and stop only the launcher processes created for the session. The launcher path is relative to this skill's directory, not to the repository under test. Run the launcher with the bootstrapped repository root as the working directory, so it resolves Playwright from that repository, and invoke it by the absolute path you resolve from the directory where you read this `SKILL.md`. @@ -47,7 +47,7 @@ Use bounded recovery. Never repeat an unchanged failing launch command. Make at Before the first checkpoint, create a physical, durable Markdown record outside the tracked source tree when possible. Prefer an existing repository-declared QA artifact location; otherwise use a user-level directory outside every repository checkout (for example `~/.local/state/manual-qa///`). Do not stage or commit it. Base it on [references/qa-record-template.md](references/qa-record-template.md). -Write the exact-head E2E coverage map and the complete remaining human-checkpoint inventory into that file before walkthrough work begins, including expected results, dependencies, priorities, and initially known gaps. For an existing plan, label a transferred scenario `E2E_COVERED` only after recording the matching assertion and passing current-head result; exclude it from human `PASS` counts and keep its earlier details for lineage. Do not silently erase it. The file is the source of truth for session continuity; chat context, summaries, and model memory are not. +Write the exact-head E2E coverage map and the complete remaining checkpoint inventory into that file before walkthrough work begins, including expected results, dependencies, priorities, and initially known gaps. For an existing plan, label a transferred scenario `E2E_COVERED` only after recording the matching assertion and passing current-head result; exclude it from human `PASS` counts and keep its earlier details for lineage. Do not silently erase it. The file is the source of truth for session continuity; chat context, summaries, and model memory are not. Record enough detail for another person to reproduce the session: repository and worktree, base and head, environment, services, flags and permissions, fixtures and cleanup, automated prechecks, each scenario's expected and actual result, confirmer, evidence, recovery attempts, findings, and disposition. Never store secrets or sensitive production data. @@ -67,11 +67,19 @@ Do not turn a plausible expectation into a human checkpoint. Before presenting e If the oracle is still uncertain, run a bounded read-only inspection or split the checkpoint into a diagnostic observation first. Do not ask the human to adjudicate an expectation the agent has not established. An observation that differs from an unsupported inference is not a product `FAIL` and does not authorize a fix. +Once the oracle is established, route the checkpoint. Record `AGENT_VERIFIED` with its `agent-observed` evidence, and do not present the checkpoint, when your own observation on this head (the dry run, a DOM or accessibility read, an API second view, or a log) conclusively shows the expected result and the result needs no human judgment. Route it to the human only when it needs human eyes, and record why: + +- a visual or perceptual judgment that is hard to assert, such as layout, overlap, animation, or whether it looks right; +- a flow you cannot drive or observe reliably, such as real OAuth, OS dialogs, hardware, or a third-party UI; +- a product-judgment call. + +When your observation of an established expectation is inconclusive, the checkpoint is not `AGENT_VERIFIED`; route it to the human. When it contradicts the expectation, settle it as a setup problem, an oracle problem, or a candidate finding, as item 6 describes. Never silently pass either. + When a human observation conflicts with the prompt, re-check the cited acceptance or approved requirement and its applicability before opening a finding. If the expectation was wrong or only an unsupported inference, record an `ORACLE CORRECTION`, preserve the useful observation, withdraw any candidate finding, and revise dependent checkpoints. Do not count an oracle correction as a product `FAIL`. ## Guide the human checkpoint by checkpoint -For each checkpoint: +For each checkpoint routed to the human: 1. Put the application in the required state using safe local setup. 2. Complete and record the oracle proof above. @@ -85,8 +93,10 @@ Use these meanings consistently: - `PASS`: the named human observed the expected behavior in the recorded environment. - `FAIL`: the named human observed behavior that contradicts the expectation. - `BLOCKED`: the checkpoint could not be exercised or judged; record why and what remains unverified. An inconclusive observation is `BLOCKED` with reason "inconclusive", never `PASS`. +- `AGENT_VERIFIED`: the agent conclusively observed the expected result under the routing rule in "Prove the checkpoint oracle before asking the human"; the human was not asked. +- `E2E_COVERED`: a passing E2E assertion on the current head proves the checkpoint's exact host, state, action, and result. -Agent inspection, screenshots, logs, API probes, and automated assertions are supporting evidence, not human confirmation. Label them `agent-observed` or `automated`; never fill the confirmer field with the human's name unless that human actually confirmed the result. A result reported outside this conversation (a PR comment, a message) counts only when the platform's author identity matches the named human confirmer. Anyone else's report is supporting evidence. +Agent inspection, screenshots, logs, API probes, and automated assertions are not human confirmation. They can support a human checkpoint or close one as `AGENT_VERIFIED` or `E2E_COVERED`, never as `PASS`. Label them `agent-observed` or `automated`; never fill the confirmer field with the human's name unless that human actually confirmed the result. A result reported outside this conversation (a PR comment, a message) counts only when the platform's author identity matches the named human confirmer. Anyone else's report is supporting evidence. When a finding is confirmed, assemble a reproducible evidence package in the local record: environment and head, prerequisites, minimal steps, expected and actual behavior, frequency, relevant logs or screenshots, affected surfaces, and cleanup state. Continue with independent checkpoints when safe. Before any source change or external-system mutation, offer an explicit next-action choice and wait for separate authorization. @@ -99,7 +109,8 @@ Store the screenshots, snapshots, and verification-protocol artifacts the record End with a concise summary containing: - coverage completed by surface and risk; -- `PASS`, `FAIL`, and `BLOCKED` counts; +- separate counts for human-confirmed `PASS` and `FAIL`, `AGENT_VERIFIED`, `E2E_COVERED`, and `BLOCKED`, never folded together; +- each checkpoint left to the human and why it needed a human; - confirmed findings and evidence locations; - untested gaps and why they remain; - cleanup status and the durable record path; diff --git a/plugins/code/skills/guided-manual-qa/references/plan-methodology.md b/plugins/code/skills/guided-manual-qa/references/plan-methodology.md index ac5d66b3..9acdf7c0 100644 --- a/plugins/code/skills/guided-manual-qa/references/plan-methodology.md +++ b/plugins/code/skills/guided-manual-qa/references/plan-methodology.md @@ -29,7 +29,7 @@ A graph zero is a claim about the query, not about the code. Without the graph, ## Separate E2E coverage from human checkpoints -Map each proposed observation to a passing E2E assertion on the current head. Count it as covered only when the same shipping host, flag assignment, fixture transition, action, and expected outcome are exercised. Record the test, assertion, head, and result in the QA record. Put only uncovered behavior and explicitly human-only requirements in the manual queue; a related test or a broader green job is not enough. Recheck the map after a head change, as the next section describes. +Map each proposed observation to a passing E2E assertion on the current head. Count it as covered only when the same shipping host, flag assignment, fixture transition, action, and expected outcome are exercised. Record the test, assertion, head, and result in the QA record. Put only uncovered behavior and explicitly human-only requirements in the manual plan, where `SKILL.md` routes each checkpoint to agent verification or to the human; a related test or a broader green job is not enough. Recheck the map after a head change, as the next section describes. ## Rebind results after a head change @@ -64,7 +64,7 @@ Prefer a small set of discriminating scenarios over many cosmetic repetitions. A For a change that fixes a reported bug, the primary checkpoint is the original reproduction on the surface where it was reported. Before scheduling it, name the correct final state and the broken final state; a setup step, expected dialog, or loading state is not the bug. Reuse a recorded repro of the bug if one exists, as the checkpoint's path and its "before" evidence; otherwise reproduce it yourself on the base, twice, before scheduling the checkpoint. Do not ask the human to reproduce it on the base unless you cannot reach that surface, and record why. -The human checkpoint passes only when the human reaches the point of divergence on the head and sees the correct final state. For an intermittent bug, ask for two independent runs. An observation that does not show the discriminating state, or one made on a different surface, is `BLOCKED` with reason "inconclusive", never `PASS`. +The checkpoint passes only when its observer, the human or the agent under the routing rule in `SKILL.md`, reaches the point of divergence on the head and sees the correct final state. For an intermittent bug, require two independent runs. An observation that does not show the discriminating state, or one made on a different surface, is `BLOCKED` with reason "inconclusive", never `PASS`. ## Apply conditional lenses diff --git a/plugins/code/skills/guided-manual-qa/references/qa-record-template.md b/plugins/code/skills/guided-manual-qa/references/qa-record-template.md index 5244c1d8..bfe8fbb1 100644 --- a/plugins/code/skills/guided-manual-qa/references/qa-record-template.md +++ b/plugins/code/skills/guided-manual-qa/references/qa-record-template.md @@ -1,6 +1,6 @@ # Manual QA Record -Copy this template to the chosen durable, untracked QA-record location. This physical Markdown file is the complete plan and live execution ledger, not merely a final report. Populate the entire planned scenario inventory before the first human checkpoint and update the file after every material action. Remove unused optional rows, but retain explicit gaps and `not applicable` decisions. +Copy this template to the chosen durable, untracked QA-record location. This physical Markdown file is the complete plan and live execution ledger, not merely a final report. Populate the entire planned scenario inventory before the first checkpoint and update the file after every material action. Remove unused optional rows, but retain explicit gaps and `not applicable` decisions. ## Session identity @@ -78,14 +78,14 @@ These results prepare the session and do not count as human confirmation. | Requirement / candidate checkpoint | Shipping host and state | E2E spec and assertion | Tested head and result | Manual gap, if any | | --- | --- | --- | --- | --- | -Record a candidate here rather than in the human queue when a passing E2E case proves its same host, flags, fixture transition, action, and oracle. An E2E result is automated coverage, not a human `PASS`. Recheck every row after a head change; preserve the prior evidence without pretending it ran on the new head. +Record a candidate here as `E2E_COVERED`, rather than in planned coverage, when a passing E2E case proves its same host, flags, fixture transition, action, and oracle. An E2E result is automated coverage, not a human `PASS`. Recheck every row after a head change; preserve the prior evidence without pretending it ran on the new head. ## Planned coverage | ID | Priority | Surface / risk not covered by E2E | Requirement | Dependencies | Status | | --- | --- | --- | --- | --- | --- | -Use `PENDING`, `PASS`, `FAIL`, `BLOCKED`, or `NOT APPLICABLE` for human-checkpoint status. In a carried-forward plan, `E2E_COVERED` means a former human checkpoint moved to the exact-head E2E coverage map; link its matching assertion and result, and exclude it from human pass/fail counts. +Use `PENDING`, `PASS`, `FAIL`, `BLOCKED`, `NOT APPLICABLE`, or `AGENT_VERIFIED` for checkpoint status. `AGENT_VERIFIED` means the agent closed the checkpoint under the routing rule in `SKILL.md` without presenting it; it is never a human `PASS`. In a carried-forward plan, `E2E_COVERED` means a former checkpoint moved to the exact-head E2E coverage map; link its matching assertion and result. Exclude both from human pass/fail counts. Every planned scenario must appear here even if it has not started. When a scenario is blocked, retain it and record the exact prerequisite and recheck condition rather than deleting or silently narrowing it. @@ -104,6 +104,7 @@ Every planned scenario must appear here even if it has not started. When a scena - Prerequisites and starting state: - Entry point used (feature-map id and route, when the repository keeps a feature map): - Agent dry run (capture, or link to an earlier capture of this entry point on this head): +- Routing (`AGENT_VERIFIED` with its evidence, or the reason this checkpoint needs a human): - Read-only second view after a write: - Human action or observation: - Expected: @@ -114,7 +115,7 @@ Every planned scenario must appear here even if it has not started. When a scena | Attempt | Head | Patch-id | Status | Actual | Confirmed by | Time | Evidence | Carry-forward reason | | --- | --- | --- | --- | --- | --- | --- | --- | --- | -Status is `PASS`, `FAIL`, `BLOCKED`, `NOT APPLICABLE`, or `ORACLE CORRECTION`; an inconclusive observation is `BLOCKED` with reason "inconclusive". Append a row for every run, rerun, reset, or carry-forward. Never edit an earlier row. The latest row is the checkpoint's current status in "Planned coverage". +Status is `PASS`, `FAIL`, `BLOCKED`, `NOT APPLICABLE`, `AGENT_VERIFIED`, or `ORACLE CORRECTION`; an inconclusive observation is `BLOCKED` with reason "inconclusive". For `AGENT_VERIFIED`, "Confirmed by" is `agent-observed`, never a human name. Append a row for every run, rerun, reset, or carry-forward. Never edit an earlier row. The latest row is the checkpoint's current status in "Planned coverage". Duplicate this section for each checkpoint. @@ -149,7 +150,8 @@ Record a wrong-origin or wrong-service discovery as an `ORACLE CORRECTION`. Stat ## Final summary - Coverage completed: -- Pass / fail / blocked totals: +- Separate counts: human-confirmed `PASS` and `FAIL`, `AGENT_VERIFIED`, `E2E_COVERED`, and `BLOCKED`: +- Checkpoints left to the human, and why each needed a human: - Confirmed findings: - Untested gaps and reasons: - Cleanup status: diff --git a/tools/guided-manual-qa/src/skill-contract.test.ts b/tools/guided-manual-qa/src/skill-contract.test.ts index 7d8c50d9..b91e6e10 100644 --- a/tools/guided-manual-qa/src/skill-contract.test.ts +++ b/tools/guided-manual-qa/src/skill-contract.test.ts @@ -112,6 +112,29 @@ describe("guided-manual-qa skill contract", () => { expect(environment).toContain("then stop that diagnostic before trying a fallback"); }); + it("routes to the human only what E2E and the agent cannot verify", () => { + expect(skill).toContain( + "Present a checkpoint to the human only when both are true: passing E2E on the current head does not already verify it, and the agent cannot reliably verify it itself", + ); + const oracle = section(skill, "## Prove the checkpoint oracle before asking the human"); + expect(oracle).toContain("Do not turn a plausible expectation into a human checkpoint."); + expect(oracle).toContain("Record `AGENT_VERIFIED` with its `agent-observed` evidence, and do not present the checkpoint"); + expect(oracle).toContain("the checkpoint is not `AGENT_VERIFIED`; route it to the human"); + expect(oracle).toContain("Never silently pass either."); + expect(skill).toContain("close one as `AGENT_VERIFIED` or `E2E_COVERED`, never as `PASS`"); + const finish = section(skill, "## Finish the session"); + expect(finish).toContain( + "separate counts for human-confirmed `PASS` and `FAIL`, `AGENT_VERIFIED`, `E2E_COVERED`, and `BLOCKED`", + ); + expect(finish).toContain("each checkpoint left to the human and why it needed a human"); + expect(section(template, "## Checkpoint results")).toContain( + 'For `AGENT_VERIFIED`, "Confirmed by" is `agent-observed`, never a human name.', + ); + expect(section(template, "## Final summary")).toContain( + "Checkpoints left to the human, and why each needed a human:", + ); + }); + it("stays standalone and harness-neutral", () => { for (const path of textFiles(SKILL_ROOT)) { const text = readFileSync(path, "utf8"); From b80572827d377916b4a09cdaee1d13168546c264 Mon Sep 17 00:00:00 2001 From: Daniel Ochoa Date: Wed, 30 Sep 2026 19:15:03 -0500 Subject: [PATCH 2/2] fix(code): order agent verification before the human window - Let the agent verify in its own headless context with the same preloaded state, and open the visible window only at the human checkpoint step - Route an inconclusive agent bug-fix observation to the human; only an inconclusive human observation is BLOCKED - Describe the routing rule in the code plugin README - Pin the ordering and the inconclusive rule in the contract test Testing: npm test and npm run typecheck in tools/guided-manual-qa Risks: None identified; skill text, README, and template only --- CHANGELOG.md | 7 ++--- plugins/code/README.md | 2 +- plugins/code/skills/guided-manual-qa/SKILL.md | 4 +-- .../references/browser-state-fixtures.md | 2 +- .../references/plan-methodology.md | 2 +- .../src/skill-contract.test.ts | 26 +++++++++++++++++++ 6 files changed, 35 insertions(+), 8 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index e3cdd766..ab85c3c5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,9 +8,10 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). #### Changed - `guided-manual-qa` presents a checkpoint to the human only when passing E2E on the current head does not already verify it and the agent cannot reliably verify it itself. After the oracle proof, the agent records a checkpoint it conclusively observed as `AGENT_VERIFIED` with `agent-observed` evidence and does not present it; it routes to the human only visual or perceptual judgments, flows it cannot drive or observe reliably (real OAuth, OS dialogs, hardware, third-party UIs), and product-judgment calls, and records why. An inconclusive agent observation goes to the human, and a contradicting one is settled as setup, oracle, or a candidate finding; neither is silently passed. -- `AGENT_VERIFIED` and `E2E_COVERED` join the status meanings and are never a human `PASS`. The final summary reports human-confirmed `PASS` and `FAIL`, `AGENT_VERIFIED`, `E2E_COVERED`, and `BLOCKED` counts separately, and lists each checkpoint left to the human with the reason it needed one. The interactive window opens only when at least one checkpoint is routed to the human. -- The QA record template adds `AGENT_VERIFIED` to the checkpoint statuses, a routing line per checkpoint, `agent-observed` as the confirmer for agent-verified attempts, and the separate counts and human-routed list in the final summary. A bug-fix checkpoint can be closed by the agent under the same routing rule. -- `tools/guided-manual-qa/src/skill-contract.test.ts` pins the routing rule, the inconclusive-observation escalation, and the separate summary counts. +- `AGENT_VERIFIED` and `E2E_COVERED` join the status meanings and are never a human `PASS`. The final summary reports human-confirmed `PASS` and `FAIL`, `AGENT_VERIFIED`, `E2E_COVERED`, and `BLOCKED` counts separately, and lists each checkpoint left to the human with the reason it needed one. The agent verifies in its own headless context with the same preloaded browser state; the visible interactive window opens only for a checkpoint routed to the human. +- The QA record template adds `AGENT_VERIFIED` to the checkpoint statuses, a routing line per checkpoint, `agent-observed` as the confirmer for agent-verified attempts, and the separate counts and human-routed list in the final summary. A bug-fix checkpoint can be closed by the agent under the same routing rule; an agent observation that misses the discriminating state routes it to the human, and only an inconclusive human observation makes it `BLOCKED` with reason "inconclusive". +- `plugins/code/README.md` describes the routing rule and the separate `AGENT_VERIFIED` and `E2E_COVERED` dispositions. +- `tools/guided-manual-qa/src/skill-contract.test.ts` pins the routing rule, the inconclusive-observation escalation, the separate summary counts, and that the visible window waits for a human-routed checkpoint while agent verification does not. ### code v1.16.2 diff --git a/plugins/code/README.md b/plugins/code/README.md index 472aa51c..561ed5af 100644 --- a/plugins/code/README.md +++ b/plugins/code/README.md @@ -330,7 +330,7 @@ Staged pipeline for inventorying a Claude Design export into reviewable findings ### `guided-manual-qa` -Derives and runs an interactive, evidence-recorded manual QA session for a code change, ticket, branch, or pull request. Resolves the exact worktree and head under test, maps candidate checkpoints against passing exact-head E2E coverage, and schedules human QA only for the uncovered remainder. Prepares a trustworthy local environment (worktree-owned services, verified origin, proven persistence chain), writes a durable Markdown QA record outside the tracked tree before the first checkpoint, and proves each checkpoint's oracle before presenting it. The human confirms checkpoint by checkpoint with `PASS`, `FAIL`, or `BLOCKED`; agent observations are recorded as supporting evidence, never as human confirmation. Ships a bundled Playwright launcher (`scripts/dist/launch-interactive-browser.mjs`, Node 18+) that opens the interactive browser with preloaded localStorage fixtures and an optional `--ready-selector` gate. Scripts are TypeScript under `tools/guided-manual-qa/src/` with the built bundle committed to `skills/guided-manual-qa/scripts/dist/`. Performs no source changes or external writes without separate authorization. The same skill directory also carries `agents/openai.yaml` display metadata so Codex can load it as a skill. +Derives and runs an interactive, evidence-recorded manual QA session for a code change, ticket, branch, or pull request. Resolves the exact worktree and head under test, maps candidate checkpoints against passing exact-head E2E coverage, and presents a checkpoint to the human only when neither that E2E coverage nor the agent's own observation can reliably verify it: visual or perceptual judgments, flows the agent cannot drive or observe reliably, and product-judgment calls. Prepares a trustworthy local environment (worktree-owned services, verified origin, proven persistence chain), writes a durable Markdown QA record outside the tracked tree before the first checkpoint, and proves each checkpoint's oracle before presenting it. The human confirms each checkpoint routed to them with `PASS`, `FAIL`, or `BLOCKED`. An agent observation can close a checkpoint as `AGENT_VERIFIED` and a passing E2E assertion as `E2E_COVERED`, never as a human `PASS`; an inconclusive agent observation goes to the human, and the final summary counts each kind separately. Ships a bundled Playwright launcher (`scripts/dist/launch-interactive-browser.mjs`, Node 18+) that opens the interactive browser with preloaded localStorage fixtures and an optional `--ready-selector` gate. Scripts are TypeScript under `tools/guided-manual-qa/src/` with the built bundle committed to `skills/guided-manual-qa/scripts/dist/`. Performs no source changes or external writes without separate authorization. The same skill directory also carries `agents/openai.yaml` display metadata so Codex can load it as a skill. --- diff --git a/plugins/code/skills/guided-manual-qa/SKILL.md b/plugins/code/skills/guided-manual-qa/SKILL.md index 25813300..16f76764 100644 --- a/plugins/code/skills/guided-manual-qa/SKILL.md +++ b/plugins/code/skills/guided-manual-qa/SKILL.md @@ -35,7 +35,7 @@ State the proposed scope, environment, fixtures, and known gaps before launching - A prototype is an optional harness for eligible shared code, not a manual-QA destination by itself. Do not add prototype-only E2E tests. Assess repository-supported E2E coverage for affected production code separately; neither a manual finding nor Storybook reachability automatically requires a new E2E test. - Start services from the resolved worktree. Record the launch commands, working directories, process identifiers, ports, and health checks. Prove that each tested listener belongs to this worktree using the strongest available evidence: process command and cwd, parent process, build or commit marker, service metadata, or a repository-provided diagnostic endpoint. A parent supervisor, launchd wrapper, or control command reporting `started` is not readiness by itself; prove the actual listener and the route-owned ready selector before presenting a checkpoint. If a wrapper hangs before spawning the child or listener, run one bounded foreground diagnostic of the same documented command to distinguish wrapper failure from app/runtime failure, then stop that diagnostic before trying a fallback. - If the QA session spans tool calls or worker turns, keep its services under a repository-supported or OS-supported owner that survives that boundary. Record how to inspect and stop that owner. On resume, read the existing QA record and recheck the current head, owned processes, listeners, data target, and exact route; earlier PIDs and ready checks are historical evidence, not proof that the environment is still available. Then apply "Rebind results after a head change" in [references/plan-methodology.md](references/plan-methodology.md) to every recorded result, name the next `PENDING` checkpoint as the resume point in the record, and continue from it. Do not re-present a checkpoint whose result still applies to the current head. -- After that proof succeeds and at least one checkpoint is routed to the human, launch the UI the user requested for the intentional interactive session. Do not open unrelated surfaces or claim an unlaunched surface was exercised. +- After that proof succeeds, prepare the browser or app context the checkpoints need. For your own verification, use a headless or displayless context with the same preloaded state ([references/browser-state-fixtures.md](references/browser-state-fixtures.md)); it needs no human checkpoint to exist. Launch the visible window the user requested only for a checkpoint routed to the human, at step 3 of "Guide the human checkpoint by checkpoint". Do not open unrelated surfaces or claim an unlaunched surface was exercised. - Record feature-flag assignments, roles, permissions, account or fixture identity, and other state that changes the observable result. Redact credentials and secrets. - When the matrix depends on browser-local state such as feature-flag fixtures, configure it before the first app navigation through a supported browser-context mechanism. Read [references/browser-state-fixtures.md](references/browser-state-fixtures.md). Do not make the human open DevTools or paste JavaScript, and do not use `javascript:` URLs, raw CDP, or the user's ordinary browser profile. - When the repository has Playwright installed and no stronger repository launcher exists, run the bundled launcher `scripts/dist/launch-interactive-browser.mjs` (Node 18+, no install step) to open the intentional interactive window with preloaded state. Keep its process alive through the checkpoint and stop only the launcher processes created for the session. The launcher path is relative to this skill's directory, not to the repository under test. Run the launcher with the bootstrapped repository root as the working directory, so it resolves Playwright from that repository, and invoke it by the absolute path you resolve from the directory where you read this `SKILL.md`. @@ -73,7 +73,7 @@ Once the oracle is established, route the checkpoint. Record `AGENT_VERIFIED` wi - a flow you cannot drive or observe reliably, such as real OAuth, OS dialogs, hardware, or a third-party UI; - a product-judgment call. -When your observation of an established expectation is inconclusive, the checkpoint is not `AGENT_VERIFIED`; route it to the human. When it contradicts the expectation, settle it as a setup problem, an oracle problem, or a candidate finding, as item 6 describes. Never silently pass either. +When your observation of an established expectation is inconclusive, the checkpoint is not `AGENT_VERIFIED`; route it to the human. If the human cannot observe the discriminating state either, record `BLOCKED` with reason "inconclusive". When it contradicts the expectation, settle it as a setup problem, an oracle problem, or a candidate finding, as item 6 describes. Never silently pass either. When a human observation conflicts with the prompt, re-check the cited acceptance or approved requirement and its applicability before opening a finding. If the expectation was wrong or only an unsupported inference, record an `ORACLE CORRECTION`, preserve the useful observation, withdraw any candidate finding, and revise dependent checkpoints. Do not count an oracle correction as a product `FAIL`. diff --git a/plugins/code/skills/guided-manual-qa/references/browser-state-fixtures.md b/plugins/code/skills/guided-manual-qa/references/browser-state-fixtures.md index 5bbe5cf0..4809e595 100644 --- a/plugins/code/skills/guided-manual-qa/references/browser-state-fixtures.md +++ b/plugins/code/skills/guided-manual-qa/references/browser-state-fixtures.md @@ -51,7 +51,7 @@ When authentication already comes from a repository-supported Playwright storage ## Interactive session -The human explicitly requesting interactive manual QA authorizes a visible application window for that session; this does not authorize visible automated E2E. Launch a fresh Playwright browser and context for the interactive session, with `headless: false` only for that intentional human walkthrough. Keep automated suites headless or displayless according to repository policy. +The human explicitly requesting interactive manual QA authorizes a visible application window for that session; this does not authorize visible automated E2E. Launch a fresh Playwright browser and context for the interactive session, with `headless: false` only for that intentional human walkthrough. Keep automated suites headless or displayless according to repository policy. The agent's own verification uses the same preloaded state in a headless context; the visible window is only for a checkpoint routed to the human. Use a new temporary or dedicated QA profile/context rather than the human's normal Chrome profile. Playwright warns that automating the default Chrome user-data directory is unsupported. Record the browser channel and that the profile was disposable, but do not record profile contents. diff --git a/plugins/code/skills/guided-manual-qa/references/plan-methodology.md b/plugins/code/skills/guided-manual-qa/references/plan-methodology.md index 9acdf7c0..a77c8acf 100644 --- a/plugins/code/skills/guided-manual-qa/references/plan-methodology.md +++ b/plugins/code/skills/guided-manual-qa/references/plan-methodology.md @@ -64,7 +64,7 @@ Prefer a small set of discriminating scenarios over many cosmetic repetitions. A For a change that fixes a reported bug, the primary checkpoint is the original reproduction on the surface where it was reported. Before scheduling it, name the correct final state and the broken final state; a setup step, expected dialog, or loading state is not the bug. Reuse a recorded repro of the bug if one exists, as the checkpoint's path and its "before" evidence; otherwise reproduce it yourself on the base, twice, before scheduling the checkpoint. Do not ask the human to reproduce it on the base unless you cannot reach that surface, and record why. -The checkpoint passes only when its observer, the human or the agent under the routing rule in `SKILL.md`, reaches the point of divergence on the head and sees the correct final state. For an intermittent bug, require two independent runs. An observation that does not show the discriminating state, or one made on a different surface, is `BLOCKED` with reason "inconclusive", never `PASS`. +The checkpoint passes only when its observer, the human or the agent under the routing rule in `SKILL.md`, reaches the point of divergence on the head and sees the correct final state. For an intermittent bug, require two independent runs. An agent observation that does not show the discriminating state on the reported surface routes the checkpoint to the human under that rule. A human observation that does not show it, or one made on a different surface, is `BLOCKED` with reason "inconclusive", never `PASS`, and the fix stays unverified. ## Apply conditional lenses diff --git a/tools/guided-manual-qa/src/skill-contract.test.ts b/tools/guided-manual-qa/src/skill-contract.test.ts index b91e6e10..d458c179 100644 --- a/tools/guided-manual-qa/src/skill-contract.test.ts +++ b/tools/guided-manual-qa/src/skill-contract.test.ts @@ -135,6 +135,32 @@ describe("guided-manual-qa skill contract", () => { ); }); + it("lets the agent verify in its own context before any human window opens", () => { + const environment = section(skill, "## Prepare a trustworthy local environment"); + expect(environment).toContain( + "For your own verification, use a headless or displayless context with the same preloaded state", + ); + expect(environment).toContain("it needs no human checkpoint to exist"); + expect(environment).toContain( + 'Launch the visible window the user requested only for a checkpoint routed to the human, at step 3 of "Guide the human checkpoint by checkpoint"', + ); + expect(section(skill, "## Guide the human checkpoint by checkpoint")).toContain( + "3. If the checkpoint asks the human to inspect a UI, open the requested interactive window or app yourself", + ); + }); + + it("routes an inconclusive agent observation to the human before BLOCKED", () => { + expect(section(skill, "## Prove the checkpoint oracle before asking the human")).toContain( + 'If the human cannot observe the discriminating state either, record `BLOCKED` with reason "inconclusive".', + ); + const bugFix = section(methodology, "## Bug-fix checkpoints"); + expect(bugFix).toContain( + "An agent observation that does not show the discriminating state on the reported surface routes the checkpoint to the human under that rule.", + ); + expect(bugFix).toContain("A human observation that does not show it"); + expect(bugFix).toContain("the fix stays unverified"); + }); + it("stays standalone and harness-neutral", () => { for (const path of textFiles(SKILL_ROOT)) { const text = readFileSync(path, "utf8");