From 70c6e443ab1908f004c990672d985de109210273 Mon Sep 17 00:00:00 2001 From: "operator-stack-publisher[bot]" Date: Sat, 1 Aug 2026 16:20:56 +0000 Subject: [PATCH] Sync Yield from Intelligence Flow @ 381aea9c6d15 --- UPSTREAM.json | 13 +++++-- evals/cases/README.md | 12 +++---- evals/cases/actions-auditor/README.md | 34 +++++++++++++++++++ evals/cases/doc-coauthoring/README.md | 34 +++++++++++++++++++ evals/cases/gstack-review/README.md | 34 +++++++++++++++++++ evals/cases/mcp-builder/README.md | 34 +++++++++++++++++++ evals/cases/systematic-debugging/README.md | 34 +++++++++++++++++++ evals/cases/vercel-deploy/README.md | 34 +++++++++++++++++++ evals/scripts/validate.mjs | 4 +++ .../2026-08-01-evaluation-case-guides.md | 9 +++++ 10 files changed, 233 insertions(+), 9 deletions(-) create mode 100644 evals/cases/actions-auditor/README.md create mode 100644 evals/cases/doc-coauthoring/README.md create mode 100644 evals/cases/gstack-review/README.md create mode 100644 evals/cases/mcp-builder/README.md create mode 100644 evals/cases/systematic-debugging/README.md create mode 100644 evals/cases/vercel-deploy/README.md create mode 100644 release-notes/2026-08-01-evaluation-case-guides.md diff --git a/UPSTREAM.json b/UPSTREAM.json index 457e0f2..f4f3009 100644 --- a/UPSTREAM.json +++ b/UPSTREAM.json @@ -41,18 +41,24 @@ "docs/tutorials/environment-repair.md": "289c9b261e3768c2c58d60d9ef7a09e85f89038438a08febb4b9c12ebada88d1", "evals/.gitignore": "0d5020173666118bafe31c857b96aa325809f41d159ca51324ceaf239e043347", "evals/README.md": "c6fb08381877d3cddce22afe59746902ba73deebd78e52a414f85bcbbf441b22", - "evals/cases/README.md": "6f478571ce1850942c3797f31eab13059b97dbdfb67bbbaf62b95cbb50eaa115", + "evals/cases/README.md": "03bd5a5bca02c9052a8413b5d30f463d37e30d130a99093d9642a1e0bbede29d", + "evals/cases/actions-auditor/README.md": "e9efdfa7657840ed1829b2831abc1ef988c7f3faac5de98e9444d5751df9b913", "evals/cases/actions-auditor/SKILL.md": "bdc81816f25144bcc44badb14ff3b46a234365466a8eeb6b3a6f6128ed57fd9b", "evals/cases/actions-auditor/workflow.ts": "622f6dd037548a5ae0eb097a77392dbcc92eaa294691a4c4e30c4abfaa9544ec", + "evals/cases/doc-coauthoring/README.md": "d72fc22a0d7134c0764efb190fad7f7a6a7d93b0f55ce09447f469cb1711ecda", "evals/cases/doc-coauthoring/SKILL.md": "8ab3e342ef202f22fe2763d1d5e0d5d6918ccdf5a30426abfc02f71b8a0b00d8", "evals/cases/doc-coauthoring/workflow.ts": "df78a605f5c0d7e52ba63ef2b91757dda1f4a5229fbe66fcb7c9adcb9aac9efb", + "evals/cases/gstack-review/README.md": "fbfcba7aca73c0fc13813e3e85edf86b4988ac3e0e538209d8d3a89df7e35cd2", "evals/cases/gstack-review/SKILL.md": "9b4914d4e569cf70fdc794d88c1c1714d4e88b4646ab9762d29da2cc2b892fba", "evals/cases/gstack-review/workflow.ts": "0743492bc14aeacd941cfbe90ef909ffe47f474941fd1cc28c33a38147ae6910", "evals/cases/index.json": "b363426b292939f977360daaaad3ea4875a447f59657fed74275f6cc4084de43", + "evals/cases/mcp-builder/README.md": "db708ccf258d6888173e573c02455a40c689ab51e09f7c57311fe1cefea32f14", "evals/cases/mcp-builder/SKILL.md": "0a5c45ae7ad2e902f2a5c2c782e78db995b8e42869f36fad42839993e346dd5e", "evals/cases/mcp-builder/workflow.ts": "fdf1ed76dc0cbe0eafd95ce1c8da7f8e65419e1f655c2df9ce88eaf285eee8a5", + "evals/cases/systematic-debugging/README.md": "5654154dd172c6405fcd8833d73c2ec2deb273416208b29890eb35b84fa0b438", "evals/cases/systematic-debugging/SKILL.md": "2d130b11d680d43f02ef126a77b462c585101e4df1632b89873597acdd7b189b", "evals/cases/systematic-debugging/workflow.ts": "86aaacc7d1fab4e320e5ce0fab7bd9972c137d6b17e8ebddc6d118299d2a1a80", + "evals/cases/vercel-deploy/README.md": "e585f09b81b3a3a2f23624f1d8e7c816af5b50d88f4204513072128bdef39b0c", "evals/cases/vercel-deploy/SKILL.md": "4ac80ad71f42c327f63685530b6b98ce3e287459527d2c0bafb5e1c716c313da", "evals/cases/vercel-deploy/workflow.ts": "78667cea9b4b4d3dfc54ef6944ca9aa75d837b0170a4d3811b8f55172f63d6f2", "evals/package-lock.json": "22c099aa5f2d9959084d6703e4a94ff7fa34216cac21b9d81137e2dd155c09d5", @@ -60,7 +66,7 @@ "evals/results/README.md": "8eda7b901660466e60e5e32609cb2a3dcca7d9d512a93f2f379090b16da15525", "evals/results/latest.json": "339a15841921fa270c668aa5c55cd33a271493735b4b85beaeaba1b25807216a", "evals/scripts/measure-source.mjs": "064a62073422c067e12b0f2b77cc65a38f7d3440efb1a91020595a941470db7a", - "evals/scripts/validate.mjs": "eff5f24a95ebb40f86d561930b8ec7f47ec1437c0d358feb040f81b4fcc7d160", + "evals/scripts/validate.mjs": "2616fdf2ebbfd940dff7e06bcb33f4a362e063484a4beb62f54ef2856747abd7", "examples/convert-skill/SKILL.md": "e6376f34365d4ac030d316db55e91f0c606a501668099f4ecf1e27d43ec806a2", "examples/convert-skill/fixtures/responses.json": "5b5f27b0ae5962360ac7b0e779992f430c655f352a38da37debb3505c32cd60a", "examples/convert-skill/main.go": "eb987ed1ce9139e6d0d40a2fae20c532c2fba3b27357e31555db2f0f52db4e39", @@ -277,6 +283,7 @@ "ir/yield.v1/request-envelope.schema.json": "2d5f34b04638450f1bd87ec1305ba7ecf0de2ec0ef6948365e6f46c2f2ee1c35", "ir/yield.v1/response-envelope.schema.json": "698fc20510bf1362cac17f332b8ec4b4dd336d2bee4294b1949ed3257b535639", "release-notes/2026-08-01-docs-and-typescript-package.md": "93382375cb47187092aff9447b280853151c738b140879828c74576a109195b8", + "release-notes/2026-08-01-evaluation-case-guides.md": "570bc996eb4d2892456a02d938fb6299d106c431bb842ce726db1df550f779d7", "release-notes/2026-08-01-evaluation-surface.md": "64fb5e2fdad3ccd41967e028cf4675c45939174000281c443f4b28d96dc07549", "release-notes/2026-08-01-example-library.md": "14d6ca40529a6aeeb72872295e57bb0d7dcc7824df31e029e497602060ce4c97", "release-notes/2026-08-01-initial-projection.md": "d38f0832b5552fb97237b30d19bb63369442ddde69e678b752ac07eedeb7ba3d", @@ -295,7 +302,7 @@ "generator": "operatorstack/yield:project", "schema_version": 1, "source": { - "commit": "ba7ac8e086564039b94ac3dd4cf5b82a247ba82f", + "commit": "381aea9c6d153503e68ad96e82ac236a354db432", "path": "labs/22-yield", "repository": "operatorstack/intelligence-flow" } diff --git a/evals/cases/README.md b/evals/cases/README.md index 0f606b1..4d194a5 100644 --- a/evals/cases/README.md +++ b/evals/cases/README.md @@ -7,9 +7,9 @@ repository and is identified by commit plus SHA-256 in `index.json`. | Case | Thin skill | Yield program | Pinned original | |---|---|---|---| -| GStack review | [SKILL.md](gstack-review/SKILL.md) | [workflow.ts](gstack-review/workflow.ts) | [source](https://github.com/garrytan/gstack/blob/a3259400a366593e0c909dd9ac3e59752efd2488/review/SKILL.md) | -| Anthropic doc co-authoring | [SKILL.md](doc-coauthoring/SKILL.md) | [workflow.ts](doc-coauthoring/workflow.ts) | [source](https://github.com/anthropics/skills/blob/b29e7cf65e5cb78a5ac33d582270551bc74a14eb/skills/doc-coauthoring/SKILL.md) | -| Superpowers systematic debugging | [SKILL.md](systematic-debugging/SKILL.md) | [workflow.ts](systematic-debugging/workflow.ts) | [source](https://github.com/obra/superpowers/blob/44c9b2d6e889982ac18c27d05a19fefe335194e1/skills/systematic-debugging/SKILL.md) | -| Vercel deploy | [SKILL.md](vercel-deploy/SKILL.md) | [workflow.ts](vercel-deploy/workflow.ts) | [source](https://github.com/vercel-labs/agent-skills/blob/7c180d9044c9ae2b442b567aad4e42a28dd5ed62/skills/deploy-to-vercel/SKILL.md) | -| Microsoft MCP builder | [SKILL.md](mcp-builder/SKILL.md) | [workflow.ts](mcp-builder/workflow.ts) | [source](https://github.com/microsoft/skills/blob/4a2873faffc1b101a33a0b59c24713d4ed78142f/.github/skills/mcp-builder/SKILL.md) | -| Trail of Bits actions auditor | [SKILL.md](actions-auditor/SKILL.md) | [workflow.ts](actions-auditor/workflow.ts) | [source](https://github.com/trailofbits/skills/blob/1256982d4d925a0acfe11e26c2253c32052c6247/plugins/agentic-actions-auditor/skills/agentic-actions-auditor/SKILL.md) | +| [GStack review](gstack-review/) | [SKILL.md](gstack-review/SKILL.md) | [workflow.ts](gstack-review/workflow.ts) | [source](https://github.com/garrytan/gstack/blob/a3259400a366593e0c909dd9ac3e59752efd2488/review/SKILL.md) | +| [Anthropic doc co-authoring](doc-coauthoring/) | [SKILL.md](doc-coauthoring/SKILL.md) | [workflow.ts](doc-coauthoring/workflow.ts) | [source](https://github.com/anthropics/skills/blob/b29e7cf65e5cb78a5ac33d582270551bc74a14eb/skills/doc-coauthoring/SKILL.md) | +| [Superpowers systematic debugging](systematic-debugging/) | [SKILL.md](systematic-debugging/SKILL.md) | [workflow.ts](systematic-debugging/workflow.ts) | [source](https://github.com/obra/superpowers/blob/44c9b2d6e889982ac18c27d05a19fefe335194e1/skills/systematic-debugging/SKILL.md) | +| [Vercel deploy](vercel-deploy/) | [SKILL.md](vercel-deploy/SKILL.md) | [workflow.ts](vercel-deploy/workflow.ts) | [source](https://github.com/vercel-labs/agent-skills/blob/7c180d9044c9ae2b442b567aad4e42a28dd5ed62/skills/deploy-to-vercel/SKILL.md) | +| [Microsoft MCP builder](mcp-builder/) | [SKILL.md](mcp-builder/SKILL.md) | [workflow.ts](mcp-builder/workflow.ts) | [source](https://github.com/microsoft/skills/blob/4a2873faffc1b101a33a0b59c24713d4ed78142f/.github/skills/mcp-builder/SKILL.md) | +| [Trail of Bits actions auditor](actions-auditor/) | [SKILL.md](actions-auditor/SKILL.md) | [workflow.ts](actions-auditor/workflow.ts) | [source](https://github.com/trailofbits/skills/blob/1256982d4d925a0acfe11e26c2253c32052c6247/plugins/agentic-actions-auditor/skills/agentic-actions-auditor/SKILL.md) | diff --git a/evals/cases/actions-auditor/README.md b/evals/cases/actions-auditor/README.md new file mode 100644 index 0000000..e5ff368 --- /dev/null +++ b/evals/cases/actions-auditor/README.md @@ -0,0 +1,34 @@ +# Trail of Bits actions auditor — Yield conversion + +This is an independent, measured conversion of +[Trail of Bits' agentic actions auditor](https://github.com/trailofbits/skills/blob/1256982d4d925a0acfe11e26c2253c32052c6247/plugins/agentic-actions-auditor/skills/agentic-actions-auditor/SKILL.md). +It is not a Trail of Bits artifact. + +We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to +split security judgment from repeatable control flow, then reviewed the +checked-in result and measured it against the pinned original. + +## The converted version + +- [`SKILL.md`](SKILL.md) keeps attack-path analysis and finding quality. +- [`workflow.ts`](workflow.ts) owns discovery, coverage, evidence requirements, + report generation, and completion. + +## Result + +| Original | Thin skill | Yield program | Maintained change | +|---:|---:|---:|---:| +| 4,846 tokens | 132 tokens | 222 tokens | **−92.7%** | + +## Benefits in this conversion + +- Workflow files are discovered by a real command instead of inferred. +- Completion requires every discovered workflow to be reviewed. +- Every finding must carry concrete evidence. +- Report generation is an observed command with a checked exit code. + +## Claim boundary + +This result measures source size, not audit coverage or behavioral equivalence. +See the [evaluation methodology](../../README.md) for the evidence required +before making a behavior claim. diff --git a/evals/cases/doc-coauthoring/README.md b/evals/cases/doc-coauthoring/README.md new file mode 100644 index 0000000..ea82e05 --- /dev/null +++ b/evals/cases/doc-coauthoring/README.md @@ -0,0 +1,34 @@ +# Anthropic doc co-authoring — Yield conversion + +This is an independent, measured conversion of +[Anthropic's doc co-authoring skill](https://github.com/anthropics/skills/blob/b29e7cf65e5cb78a5ac33d582270551bc74a14eb/skills/doc-coauthoring/SKILL.md). +It is not an Anthropic artifact. + +We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to +split model judgment from repeatable control flow, then reviewed the checked-in +result and measured it against the pinned original. + +## The converted version + +- [`SKILL.md`](SKILL.md) keeps writing judgment, reader perspective, and voice. +- [`workflow.ts`](workflow.ts) owns the questions, outline approval, draft stages, + reader test, saved answers, and completion gate. + +## Result + +| Original | Thin skill | Yield program | Maintained change | +|---:|---:|---:|---:| +| 3,289 tokens | 137 tokens | 209 tokens | **−89.5%** | + +## Benefits in this conversion + +- A returning run can continue from saved audience, outcome, and source context. +- Outline approval is an explicit gate instead of a prose suggestion. +- Reader testing happens before the final revision. +- The model spends its context on the document, not on remembering stage order. + +## Claim boundary + +This result measures source size, not writing quality or behavioral equivalence. +See the [evaluation methodology](../../README.md) for the evidence required +before making a behavior claim. diff --git a/evals/cases/gstack-review/README.md b/evals/cases/gstack-review/README.md new file mode 100644 index 0000000..41eca41 --- /dev/null +++ b/evals/cases/gstack-review/README.md @@ -0,0 +1,34 @@ +# GStack review — Yield conversion + +This is an independent, measured conversion of +[GStack's review skill](https://github.com/garrytan/gstack/blob/a3259400a366593e0c909dd9ac3e59752efd2488/review/SKILL.md). +It is not an upstream GStack artifact. + +We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to +split model judgment from repeatable control flow, then reviewed the checked-in +result and measured it against the pinned original. + +## The converted version + +- [`SKILL.md`](SKILL.md) keeps review judgment, severity rules, and finding shape. +- [`workflow.ts`](workflow.ts) owns the diff, tests, typecheck, critical-finding + gate, saved result, and completion. + +## Result + +| Original | Thin skill | Yield program | Maintained change | +|---:|---:|---:|---:| +| 27,162 tokens | 146 tokens | 160 tokens | **−98.9%** | + +## Benefits in this conversion + +- Repository checks run as commands instead of instructions the model must remember. +- A review cannot complete while a critical finding remains. +- The diff and check output become explicit evidence passed into the review. +- The model-facing prompt is small enough to focus on finding real defects. + +## Claim boundary + +The token result measures source size. A separate early GStack behavior study is +summarized in [`results/latest.json`](../../results/latest.json), but its raw +artifact is not yet published and it does not prove general equivalence. diff --git a/evals/cases/mcp-builder/README.md b/evals/cases/mcp-builder/README.md new file mode 100644 index 0000000..5c52d32 --- /dev/null +++ b/evals/cases/mcp-builder/README.md @@ -0,0 +1,34 @@ +# Microsoft MCP builder — Yield conversion + +This is an independent, measured conversion of +[Microsoft's MCP builder skill](https://github.com/microsoft/skills/blob/4a2873faffc1b101a33a0b59c24713d4ed78142f/.github/skills/mcp-builder/SKILL.md). +It is not a Microsoft artifact. + +We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to +split tool-design judgment from repeatable control flow, then reviewed the +checked-in result and measured it against the pinned original. + +## The converted version + +- [`SKILL.md`](SKILL.md) keeps tool-design and review judgment. +- [`workflow.ts`](workflow.ts) owns user inputs, the tool-count gate, design + approval, scaffold, tests, evaluations, and completion. + +## Result + +| Original | Thin skill | Yield program | Maintained change | +|---:|---:|---:|---:| +| 2,648 tokens | 131 tokens | 291 tokens | **−84.1%** | + +## Benefits in this conversion + +- The tool surface stays bounded before code generation begins. +- Building requires explicit design approval. +- Scaffold, tests, review, and evaluations happen in a fixed order. +- Completion requires both review and evaluation gates to pass. + +## Claim boundary + +This result measures source size, not MCP server quality or behavioral +equivalence. See the [evaluation methodology](../../README.md) for the evidence +required before making a behavior claim. diff --git a/evals/cases/systematic-debugging/README.md b/evals/cases/systematic-debugging/README.md new file mode 100644 index 0000000..66e8838 --- /dev/null +++ b/evals/cases/systematic-debugging/README.md @@ -0,0 +1,34 @@ +# Systematic debugging — Yield conversion + +This is an independent, measured conversion of +[Superpowers' systematic debugging skill](https://github.com/obra/superpowers/blob/44c9b2d6e889982ac18c27d05a19fefe335194e1/skills/systematic-debugging/SKILL.md). +It is not an upstream Superpowers artifact. + +We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to +split diagnostic judgment from repeatable control flow, then reviewed the +checked-in result and measured it against the pinned original. + +## The converted version + +- [`SKILL.md`](SKILL.md) keeps the standard for a falsifiable root-cause hypothesis. +- [`workflow.ts`](workflow.ts) owns reproduction, experiment bounds, approval, + verification, and completion. + +## Result + +| Original | Thin skill | Yield program | Maintained change | +|---:|---:|---:|---:| +| 2,226 tokens | 137 tokens | 226 tokens | **−83.7%** | + +## Benefits in this conversion + +- The failure must reproduce before diagnosis starts. +- A proposed cause must include an executable falsification experiment. +- Applying a fix requires user approval. +- Completion requires the full test suite to pass. + +## Claim boundary + +This result measures source size, not diagnosis quality or behavioral +equivalence. See the [evaluation methodology](../../README.md) for the evidence +required before making a behavior claim. diff --git a/evals/cases/vercel-deploy/README.md b/evals/cases/vercel-deploy/README.md new file mode 100644 index 0000000..0bbd19d --- /dev/null +++ b/evals/cases/vercel-deploy/README.md @@ -0,0 +1,34 @@ +# Vercel deploy — Yield conversion + +This is an independent, measured conversion of +[Vercel's deploy skill](https://github.com/vercel-labs/agent-skills/blob/7c180d9044c9ae2b442b567aad4e42a28dd5ed62/skills/deploy-to-vercel/SKILL.md). +It is not a Vercel artifact. + +We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to +split deployment explanation from repeatable control flow, then reviewed the +checked-in result and measured it against the pinned original. + +## The converted version + +- [`SKILL.md`](SKILL.md) keeps interpretation of project state and failures. +- [`workflow.ts`](workflow.ts) owns state detection, approval, authentication, + linking, deployment, HTTPS verification, and completion. + +## Result + +| Original | Thin skill | Yield program | Maintained change | +|---:|---:|---:|---:| +| 2,898 tokens | 108 tokens | 290 tokens | **−86.3%** | + +## Benefits in this conversion + +- The deployment command runs only after explicit approval. +- Missing authentication produces an honest blocked outcome. +- A successful command is not enough; the deployed URL must answer over HTTPS. +- Cancellation and failure are recorded separately from success. + +## Claim boundary + +This result measures source size, not deployment reliability or behavioral +equivalence. See the [evaluation methodology](../../README.md) for the evidence +required before making a behavior claim. diff --git a/evals/scripts/validate.mjs b/evals/scripts/validate.mjs index 651f007..20ba098 100644 --- a/evals/scripts/validate.mjs +++ b/evals/scripts/validate.mjs @@ -24,7 +24,11 @@ for (const item of index.cases) { if (!isSha256(item.source.sha256)) fail(`${item.id}: source sha256 is invalid`) const skill = await readFile(join(root, "cases", item.thin_skill), "utf8") const workflow = await readFile(join(root, "cases", item.workflow), "utf8") + const readme = await readFile(join(root, "cases", item.id, "README.md"), "utf8") if (!skill.trim() || !workflow.trim()) fail(`${item.id}: conversion source is empty`) + for (const required of [item.source.repo, "convert-skill", "](SKILL.md)", "](workflow.ts)"]) { + if (!readme.includes(required)) fail(`${item.id}: README is missing ${required}`) + } localMeasurements.set(item.id, { prompt_tokens: countTokens(skill), workflow_tokens: countTokens(workflow), diff --git a/release-notes/2026-08-01-evaluation-case-guides.md b/release-notes/2026-08-01-evaluation-case-guides.md new file mode 100644 index 0000000..b10c5d1 --- /dev/null +++ b/release-notes/2026-08-01-evaluation-case-guides.md @@ -0,0 +1,9 @@ +### Inspect every evaluation conversion + +Each evaluation case now has its own README linking the pinned upstream skill, +thin model-facing `SKILL.md`, Yield program, measured source-size result, and +case-specific benefits. The guides also state what the current evidence does +and does not establish. + +Evaluation validation now requires these guides and their provenance and code +links, so a projected case cannot silently lose its explanation.