Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 10 additions & 3 deletions UPSTREAM.json
Original file line number Diff line number Diff line change
Expand Up @@ -41,26 +41,32 @@
"docs/tutorials/environment-repair.md": "289c9b261e3768c2c58d60d9ef7a09e85f89038438a08febb4b9c12ebada88d1",
"evals/.gitignore": "0d5020173666118bafe31c857b96aa325809f41d159ca51324ceaf239e043347",
"evals/README.md": "c6fb08381877d3cddce22afe59746902ba73deebd78e52a414f85bcbbf441b22",
"evals/cases/README.md": "6f478571ce1850942c3797f31eab13059b97dbdfb67bbbaf62b95cbb50eaa115",
"evals/cases/README.md": "03bd5a5bca02c9052a8413b5d30f463d37e30d130a99093d9642a1e0bbede29d",
"evals/cases/actions-auditor/README.md": "e9efdfa7657840ed1829b2831abc1ef988c7f3faac5de98e9444d5751df9b913",
"evals/cases/actions-auditor/SKILL.md": "bdc81816f25144bcc44badb14ff3b46a234365466a8eeb6b3a6f6128ed57fd9b",
"evals/cases/actions-auditor/workflow.ts": "622f6dd037548a5ae0eb097a77392dbcc92eaa294691a4c4e30c4abfaa9544ec",
"evals/cases/doc-coauthoring/README.md": "d72fc22a0d7134c0764efb190fad7f7a6a7d93b0f55ce09447f469cb1711ecda",
"evals/cases/doc-coauthoring/SKILL.md": "8ab3e342ef202f22fe2763d1d5e0d5d6918ccdf5a30426abfc02f71b8a0b00d8",
"evals/cases/doc-coauthoring/workflow.ts": "df78a605f5c0d7e52ba63ef2b91757dda1f4a5229fbe66fcb7c9adcb9aac9efb",
"evals/cases/gstack-review/README.md": "fbfcba7aca73c0fc13813e3e85edf86b4988ac3e0e538209d8d3a89df7e35cd2",
"evals/cases/gstack-review/SKILL.md": "9b4914d4e569cf70fdc794d88c1c1714d4e88b4646ab9762d29da2cc2b892fba",
"evals/cases/gstack-review/workflow.ts": "0743492bc14aeacd941cfbe90ef909ffe47f474941fd1cc28c33a38147ae6910",
"evals/cases/index.json": "b363426b292939f977360daaaad3ea4875a447f59657fed74275f6cc4084de43",
"evals/cases/mcp-builder/README.md": "db708ccf258d6888173e573c02455a40c689ab51e09f7c57311fe1cefea32f14",
"evals/cases/mcp-builder/SKILL.md": "0a5c45ae7ad2e902f2a5c2c782e78db995b8e42869f36fad42839993e346dd5e",
"evals/cases/mcp-builder/workflow.ts": "fdf1ed76dc0cbe0eafd95ce1c8da7f8e65419e1f655c2df9ce88eaf285eee8a5",
"evals/cases/systematic-debugging/README.md": "5654154dd172c6405fcd8833d73c2ec2deb273416208b29890eb35b84fa0b438",
"evals/cases/systematic-debugging/SKILL.md": "2d130b11d680d43f02ef126a77b462c585101e4df1632b89873597acdd7b189b",
"evals/cases/systematic-debugging/workflow.ts": "86aaacc7d1fab4e320e5ce0fab7bd9972c137d6b17e8ebddc6d118299d2a1a80",
"evals/cases/vercel-deploy/README.md": "e585f09b81b3a3a2f23624f1d8e7c816af5b50d88f4204513072128bdef39b0c",
"evals/cases/vercel-deploy/SKILL.md": "4ac80ad71f42c327f63685530b6b98ce3e287459527d2c0bafb5e1c716c313da",
"evals/cases/vercel-deploy/workflow.ts": "78667cea9b4b4d3dfc54ef6944ca9aa75d837b0170a4d3811b8f55172f63d6f2",
"evals/package-lock.json": "22c099aa5f2d9959084d6703e4a94ff7fa34216cac21b9d81137e2dd155c09d5",
"evals/package.json": "3361b8d579664bf9d6642eaeca6a906a1bca43fd0b1b543b00b2d5e0aa021b17",
"evals/results/README.md": "8eda7b901660466e60e5e32609cb2a3dcca7d9d512a93f2f379090b16da15525",
"evals/results/latest.json": "339a15841921fa270c668aa5c55cd33a271493735b4b85beaeaba1b25807216a",
"evals/scripts/measure-source.mjs": "064a62073422c067e12b0f2b77cc65a38f7d3440efb1a91020595a941470db7a",
"evals/scripts/validate.mjs": "eff5f24a95ebb40f86d561930b8ec7f47ec1437c0d358feb040f81b4fcc7d160",
"evals/scripts/validate.mjs": "2616fdf2ebbfd940dff7e06bcb33f4a362e063484a4beb62f54ef2856747abd7",
"examples/convert-skill/SKILL.md": "e6376f34365d4ac030d316db55e91f0c606a501668099f4ecf1e27d43ec806a2",
"examples/convert-skill/fixtures/responses.json": "5b5f27b0ae5962360ac7b0e779992f430c655f352a38da37debb3505c32cd60a",
"examples/convert-skill/main.go": "eb987ed1ce9139e6d0d40a2fae20c532c2fba3b27357e31555db2f0f52db4e39",
Expand Down Expand Up @@ -277,6 +283,7 @@
"ir/yield.v1/request-envelope.schema.json": "2d5f34b04638450f1bd87ec1305ba7ecf0de2ec0ef6948365e6f46c2f2ee1c35",
"ir/yield.v1/response-envelope.schema.json": "698fc20510bf1362cac17f332b8ec4b4dd336d2bee4294b1949ed3257b535639",
"release-notes/2026-08-01-docs-and-typescript-package.md": "93382375cb47187092aff9447b280853151c738b140879828c74576a109195b8",
"release-notes/2026-08-01-evaluation-case-guides.md": "570bc996eb4d2892456a02d938fb6299d106c431bb842ce726db1df550f779d7",
"release-notes/2026-08-01-evaluation-surface.md": "64fb5e2fdad3ccd41967e028cf4675c45939174000281c443f4b28d96dc07549",
"release-notes/2026-08-01-example-library.md": "14d6ca40529a6aeeb72872295e57bb0d7dcc7824df31e029e497602060ce4c97",
"release-notes/2026-08-01-initial-projection.md": "d38f0832b5552fb97237b30d19bb63369442ddde69e678b752ac07eedeb7ba3d",
Expand All @@ -295,7 +302,7 @@
"generator": "operatorstack/yield:project",
"schema_version": 1,
"source": {
"commit": "ba7ac8e086564039b94ac3dd4cf5b82a247ba82f",
"commit": "381aea9c6d153503e68ad96e82ac236a354db432",
"path": "labs/22-yield",
"repository": "operatorstack/intelligence-flow"
}
Expand Down
12 changes: 6 additions & 6 deletions evals/cases/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,9 +7,9 @@ repository and is identified by commit plus SHA-256 in `index.json`.

| Case | Thin skill | Yield program | Pinned original |
|---|---|---|---|
| GStack review | [SKILL.md](gstack-review/SKILL.md) | [workflow.ts](gstack-review/workflow.ts) | [source](https://github.com/garrytan/gstack/blob/a3259400a366593e0c909dd9ac3e59752efd2488/review/SKILL.md) |
| Anthropic doc co-authoring | [SKILL.md](doc-coauthoring/SKILL.md) | [workflow.ts](doc-coauthoring/workflow.ts) | [source](https://github.com/anthropics/skills/blob/b29e7cf65e5cb78a5ac33d582270551bc74a14eb/skills/doc-coauthoring/SKILL.md) |
| Superpowers systematic debugging | [SKILL.md](systematic-debugging/SKILL.md) | [workflow.ts](systematic-debugging/workflow.ts) | [source](https://github.com/obra/superpowers/blob/44c9b2d6e889982ac18c27d05a19fefe335194e1/skills/systematic-debugging/SKILL.md) |
| Vercel deploy | [SKILL.md](vercel-deploy/SKILL.md) | [workflow.ts](vercel-deploy/workflow.ts) | [source](https://github.com/vercel-labs/agent-skills/blob/7c180d9044c9ae2b442b567aad4e42a28dd5ed62/skills/deploy-to-vercel/SKILL.md) |
| Microsoft MCP builder | [SKILL.md](mcp-builder/SKILL.md) | [workflow.ts](mcp-builder/workflow.ts) | [source](https://github.com/microsoft/skills/blob/4a2873faffc1b101a33a0b59c24713d4ed78142f/.github/skills/mcp-builder/SKILL.md) |
| Trail of Bits actions auditor | [SKILL.md](actions-auditor/SKILL.md) | [workflow.ts](actions-auditor/workflow.ts) | [source](https://github.com/trailofbits/skills/blob/1256982d4d925a0acfe11e26c2253c32052c6247/plugins/agentic-actions-auditor/skills/agentic-actions-auditor/SKILL.md) |
| [GStack review](gstack-review/) | [SKILL.md](gstack-review/SKILL.md) | [workflow.ts](gstack-review/workflow.ts) | [source](https://github.com/garrytan/gstack/blob/a3259400a366593e0c909dd9ac3e59752efd2488/review/SKILL.md) |
| [Anthropic doc co-authoring](doc-coauthoring/) | [SKILL.md](doc-coauthoring/SKILL.md) | [workflow.ts](doc-coauthoring/workflow.ts) | [source](https://github.com/anthropics/skills/blob/b29e7cf65e5cb78a5ac33d582270551bc74a14eb/skills/doc-coauthoring/SKILL.md) |
| [Superpowers systematic debugging](systematic-debugging/) | [SKILL.md](systematic-debugging/SKILL.md) | [workflow.ts](systematic-debugging/workflow.ts) | [source](https://github.com/obra/superpowers/blob/44c9b2d6e889982ac18c27d05a19fefe335194e1/skills/systematic-debugging/SKILL.md) |
| [Vercel deploy](vercel-deploy/) | [SKILL.md](vercel-deploy/SKILL.md) | [workflow.ts](vercel-deploy/workflow.ts) | [source](https://github.com/vercel-labs/agent-skills/blob/7c180d9044c9ae2b442b567aad4e42a28dd5ed62/skills/deploy-to-vercel/SKILL.md) |
| [Microsoft MCP builder](mcp-builder/) | [SKILL.md](mcp-builder/SKILL.md) | [workflow.ts](mcp-builder/workflow.ts) | [source](https://github.com/microsoft/skills/blob/4a2873faffc1b101a33a0b59c24713d4ed78142f/.github/skills/mcp-builder/SKILL.md) |
| [Trail of Bits actions auditor](actions-auditor/) | [SKILL.md](actions-auditor/SKILL.md) | [workflow.ts](actions-auditor/workflow.ts) | [source](https://github.com/trailofbits/skills/blob/1256982d4d925a0acfe11e26c2253c32052c6247/plugins/agentic-actions-auditor/skills/agentic-actions-auditor/SKILL.md) |
34 changes: 34 additions & 0 deletions evals/cases/actions-auditor/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
# Trail of Bits actions auditor — Yield conversion

This is an independent, measured conversion of
[Trail of Bits' agentic actions auditor](https://github.com/trailofbits/skills/blob/1256982d4d925a0acfe11e26c2253c32052c6247/plugins/agentic-actions-auditor/skills/agentic-actions-auditor/SKILL.md).
It is not a Trail of Bits artifact.

We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to
split security judgment from repeatable control flow, then reviewed the
checked-in result and measured it against the pinned original.

## The converted version

- [`SKILL.md`](SKILL.md) keeps attack-path analysis and finding quality.
- [`workflow.ts`](workflow.ts) owns discovery, coverage, evidence requirements,
report generation, and completion.

## Result

| Original | Thin skill | Yield program | Maintained change |
|---:|---:|---:|---:|
| 4,846 tokens | 132 tokens | 222 tokens | **−92.7%** |

## Benefits in this conversion

- Workflow files are discovered by a real command instead of inferred.
- Completion requires every discovered workflow to be reviewed.
- Every finding must carry concrete evidence.
- Report generation is an observed command with a checked exit code.

## Claim boundary

This result measures source size, not audit coverage or behavioral equivalence.
See the [evaluation methodology](../../README.md) for the evidence required
before making a behavior claim.
34 changes: 34 additions & 0 deletions evals/cases/doc-coauthoring/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
# Anthropic doc co-authoring — Yield conversion

This is an independent, measured conversion of
[Anthropic's doc co-authoring skill](https://github.com/anthropics/skills/blob/b29e7cf65e5cb78a5ac33d582270551bc74a14eb/skills/doc-coauthoring/SKILL.md).
It is not an Anthropic artifact.

We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to
split model judgment from repeatable control flow, then reviewed the checked-in
result and measured it against the pinned original.

## The converted version

- [`SKILL.md`](SKILL.md) keeps writing judgment, reader perspective, and voice.
- [`workflow.ts`](workflow.ts) owns the questions, outline approval, draft stages,
reader test, saved answers, and completion gate.

## Result

| Original | Thin skill | Yield program | Maintained change |
|---:|---:|---:|---:|
| 3,289 tokens | 137 tokens | 209 tokens | **−89.5%** |

## Benefits in this conversion

- A returning run can continue from saved audience, outcome, and source context.
- Outline approval is an explicit gate instead of a prose suggestion.
- Reader testing happens before the final revision.
- The model spends its context on the document, not on remembering stage order.

## Claim boundary

This result measures source size, not writing quality or behavioral equivalence.
See the [evaluation methodology](../../README.md) for the evidence required
before making a behavior claim.
34 changes: 34 additions & 0 deletions evals/cases/gstack-review/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
# GStack review — Yield conversion

This is an independent, measured conversion of
[GStack's review skill](https://github.com/garrytan/gstack/blob/a3259400a366593e0c909dd9ac3e59752efd2488/review/SKILL.md).
It is not an upstream GStack artifact.

We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to
split model judgment from repeatable control flow, then reviewed the checked-in
result and measured it against the pinned original.

## The converted version

- [`SKILL.md`](SKILL.md) keeps review judgment, severity rules, and finding shape.
- [`workflow.ts`](workflow.ts) owns the diff, tests, typecheck, critical-finding
gate, saved result, and completion.

## Result

| Original | Thin skill | Yield program | Maintained change |
|---:|---:|---:|---:|
| 27,162 tokens | 146 tokens | 160 tokens | **−98.9%** |

## Benefits in this conversion

- Repository checks run as commands instead of instructions the model must remember.
- A review cannot complete while a critical finding remains.
- The diff and check output become explicit evidence passed into the review.
- The model-facing prompt is small enough to focus on finding real defects.

## Claim boundary

The token result measures source size. A separate early GStack behavior study is
summarized in [`results/latest.json`](../../results/latest.json), but its raw
artifact is not yet published and it does not prove general equivalence.
34 changes: 34 additions & 0 deletions evals/cases/mcp-builder/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
# Microsoft MCP builder — Yield conversion

This is an independent, measured conversion of
[Microsoft's MCP builder skill](https://github.com/microsoft/skills/blob/4a2873faffc1b101a33a0b59c24713d4ed78142f/.github/skills/mcp-builder/SKILL.md).
It is not a Microsoft artifact.

We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to
split tool-design judgment from repeatable control flow, then reviewed the
checked-in result and measured it against the pinned original.

## The converted version

- [`SKILL.md`](SKILL.md) keeps tool-design and review judgment.
- [`workflow.ts`](workflow.ts) owns user inputs, the tool-count gate, design
approval, scaffold, tests, evaluations, and completion.

## Result

| Original | Thin skill | Yield program | Maintained change |
|---:|---:|---:|---:|
| 2,648 tokens | 131 tokens | 291 tokens | **−84.1%** |

## Benefits in this conversion

- The tool surface stays bounded before code generation begins.
- Building requires explicit design approval.
- Scaffold, tests, review, and evaluations happen in a fixed order.
- Completion requires both review and evaluation gates to pass.

## Claim boundary

This result measures source size, not MCP server quality or behavioral
equivalence. See the [evaluation methodology](../../README.md) for the evidence
required before making a behavior claim.
34 changes: 34 additions & 0 deletions evals/cases/systematic-debugging/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
# Systematic debugging — Yield conversion

This is an independent, measured conversion of
[Superpowers' systematic debugging skill](https://github.com/obra/superpowers/blob/44c9b2d6e889982ac18c27d05a19fefe335194e1/skills/systematic-debugging/SKILL.md).
It is not an upstream Superpowers artifact.

We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to
split diagnostic judgment from repeatable control flow, then reviewed the
checked-in result and measured it against the pinned original.

## The converted version

- [`SKILL.md`](SKILL.md) keeps the standard for a falsifiable root-cause hypothesis.
- [`workflow.ts`](workflow.ts) owns reproduction, experiment bounds, approval,
verification, and completion.

## Result

| Original | Thin skill | Yield program | Maintained change |
|---:|---:|---:|---:|
| 2,226 tokens | 137 tokens | 226 tokens | **−83.7%** |

## Benefits in this conversion

- The failure must reproduce before diagnosis starts.
- A proposed cause must include an executable falsification experiment.
- Applying a fix requires user approval.
- Completion requires the full test suite to pass.

## Claim boundary

This result measures source size, not diagnosis quality or behavioral
equivalence. See the [evaluation methodology](../../README.md) for the evidence
required before making a behavior claim.
34 changes: 34 additions & 0 deletions evals/cases/vercel-deploy/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
# Vercel deploy — Yield conversion

This is an independent, measured conversion of
[Vercel's deploy skill](https://github.com/vercel-labs/agent-skills/blob/7c180d9044c9ae2b442b567aad4e42a28dd5ed62/skills/deploy-to-vercel/SKILL.md).
It is not a Vercel artifact.

We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to
split deployment explanation from repeatable control flow, then reviewed the
checked-in result and measured it against the pinned original.

## The converted version

- [`SKILL.md`](SKILL.md) keeps interpretation of project state and failures.
- [`workflow.ts`](workflow.ts) owns state detection, approval, authentication,
linking, deployment, HTTPS verification, and completion.

## Result

| Original | Thin skill | Yield program | Maintained change |
|---:|---:|---:|---:|
| 2,898 tokens | 108 tokens | 290 tokens | **−86.3%** |

## Benefits in this conversion

- The deployment command runs only after explicit approval.
- Missing authentication produces an honest blocked outcome.
- A successful command is not enough; the deployed URL must answer over HTTPS.
- Cancellation and failure are recorded separately from success.

## Claim boundary

This result measures source size, not deployment reliability or behavioral
equivalence. See the [evaluation methodology](../../README.md) for the evidence
required before making a behavior claim.
4 changes: 4 additions & 0 deletions evals/scripts/validate.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,11 @@ for (const item of index.cases) {
if (!isSha256(item.source.sha256)) fail(`${item.id}: source sha256 is invalid`)
const skill = await readFile(join(root, "cases", item.thin_skill), "utf8")
const workflow = await readFile(join(root, "cases", item.workflow), "utf8")
const readme = await readFile(join(root, "cases", item.id, "README.md"), "utf8")
if (!skill.trim() || !workflow.trim()) fail(`${item.id}: conversion source is empty`)
for (const required of [item.source.repo, "convert-skill", "](SKILL.md)", "](workflow.ts)"]) {
if (!readme.includes(required)) fail(`${item.id}: README is missing ${required}`)
}
localMeasurements.set(item.id, {
prompt_tokens: countTokens(skill),
workflow_tokens: countTokens(workflow),
Expand Down
9 changes: 9 additions & 0 deletions release-notes/2026-08-01-evaluation-case-guides.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
### Inspect every evaluation conversion

Each evaluation case now has its own README linking the pinned upstream skill,
thin model-facing `SKILL.md`, Yield program, measured source-size result, and
case-specific benefits. The guides also state what the current evidence does
and does not establish.

Evaluation validation now requires these guides and their provenance and code
links, so a projected case cannot silently lose its explanation.
Loading