diff --git a/README.md b/README.md index 03565e0..78c78b5 100644 --- a/README.md +++ b/README.md @@ -81,8 +81,8 @@ use the documentation by job: - [tutorials](docs/tutorials/README.md) — review, approval, environment repair, bounded debugging, and migration; - [examples](docs/examples.md) — working programs in all four languages; -- [evaluations](evals/README.md) — pinned conversion cases, reproducible - measurements, and the raw-evidence publication boundary; +- [evaluations](evals/README.md) — first-party workflow conformance and runtime + invariant results, including the exact claim boundary; - [convert an existing skill](docs/convert-existing-skill.md) — move control flow into code without claiming that fixture execution proves every reading of the original prose; diff --git a/UPSTREAM.json b/UPSTREAM.json index f4f3009..4729ad9 100644 --- a/UPSTREAM.json +++ b/UPSTREAM.json @@ -2,11 +2,11 @@ "files": { ".gitignore": "803c5f79d6da7f2c5a1dc0ce27c53b8e5c059309b165782831c4c472af058a4c", "LICENSE": "fff261ce507eabd57666c283a621f33e183a3aedebda04c4ecbc6309a62f5edf", - "README.md": "d3a13ebcd016ab1ba6a4fac64ca4a555ff738e9ce4c5c9744ee4e046e9dc08ef", + "README.md": "acd9f37f0f7013a4bf99df382fd3b5c14d070914ec958e3198d9c70aa8317ce6", "cmd/yskill/main.go": "a88bdd133118aa7b86cdf021e67052bd8e64e63b3792c091f7612699af89648b", "docs/README.md": "cd8a4105f4f04143172661b39789e3c533d031ad46437aa2c078625a29a0a9db", "docs/convert-existing-skill.md": "0dd538e908a1f2d3a1958e4b2cc1efede2a8e7a73341e32c1a75fec69a9acb8a", - "docs/examples.md": "8c81f990b8d3a06b42cedcf34813a11dbe0606201782f72903a8c14d91d1ae23", + "docs/examples.md": "3c19bacae7ec7b31bf93417e61d01cd1872f26228fedf0ed6decfce7c41ed274", "docs/locus-conformance.md": "a71fd5a72678dcad344afc71cd7a788090ef008a8e5e82d925da5a3db41af6a0", "docs/locus-converter.md": "b8d1bdb63cb0283ace524f572fb13f0d1cb535b130c5ea3e112d2efae55a8681", "docs/locus-yield.md": "c29d0c5801f32591828fc348385fb25c1850b39afc22770098eae0614fc3a417", @@ -40,33 +40,13 @@ "docs/tutorials/data-migration.md": "6699fc47cda84c4f46a3703ca4a69bcced7cc4a9df339a308e807ea3a99578bd", "docs/tutorials/environment-repair.md": "289c9b261e3768c2c58d60d9ef7a09e85f89038438a08febb4b9c12ebada88d1", "evals/.gitignore": "0d5020173666118bafe31c857b96aa325809f41d159ca51324ceaf239e043347", - "evals/README.md": "c6fb08381877d3cddce22afe59746902ba73deebd78e52a414f85bcbbf441b22", - "evals/cases/README.md": "03bd5a5bca02c9052a8413b5d30f463d37e30d130a99093d9642a1e0bbede29d", - "evals/cases/actions-auditor/README.md": "e9efdfa7657840ed1829b2831abc1ef988c7f3faac5de98e9444d5751df9b913", - "evals/cases/actions-auditor/SKILL.md": "bdc81816f25144bcc44badb14ff3b46a234365466a8eeb6b3a6f6128ed57fd9b", - "evals/cases/actions-auditor/workflow.ts": "622f6dd037548a5ae0eb097a77392dbcc92eaa294691a4c4e30c4abfaa9544ec", - "evals/cases/doc-coauthoring/README.md": "d72fc22a0d7134c0764efb190fad7f7a6a7d93b0f55ce09447f469cb1711ecda", - "evals/cases/doc-coauthoring/SKILL.md": "8ab3e342ef202f22fe2763d1d5e0d5d6918ccdf5a30426abfc02f71b8a0b00d8", - "evals/cases/doc-coauthoring/workflow.ts": "df78a605f5c0d7e52ba63ef2b91757dda1f4a5229fbe66fcb7c9adcb9aac9efb", - "evals/cases/gstack-review/README.md": "fbfcba7aca73c0fc13813e3e85edf86b4988ac3e0e538209d8d3a89df7e35cd2", - "evals/cases/gstack-review/SKILL.md": "9b4914d4e569cf70fdc794d88c1c1714d4e88b4646ab9762d29da2cc2b892fba", - "evals/cases/gstack-review/workflow.ts": "0743492bc14aeacd941cfbe90ef909ffe47f474941fd1cc28c33a38147ae6910", - "evals/cases/index.json": "b363426b292939f977360daaaad3ea4875a447f59657fed74275f6cc4084de43", - "evals/cases/mcp-builder/README.md": "db708ccf258d6888173e573c02455a40c689ab51e09f7c57311fe1cefea32f14", - "evals/cases/mcp-builder/SKILL.md": "0a5c45ae7ad2e902f2a5c2c782e78db995b8e42869f36fad42839993e346dd5e", - "evals/cases/mcp-builder/workflow.ts": "fdf1ed76dc0cbe0eafd95ce1c8da7f8e65419e1f655c2df9ce88eaf285eee8a5", - "evals/cases/systematic-debugging/README.md": "5654154dd172c6405fcd8833d73c2ec2deb273416208b29890eb35b84fa0b438", - "evals/cases/systematic-debugging/SKILL.md": "2d130b11d680d43f02ef126a77b462c585101e4df1632b89873597acdd7b189b", - "evals/cases/systematic-debugging/workflow.ts": "86aaacc7d1fab4e320e5ce0fab7bd9972c137d6b17e8ebddc6d118299d2a1a80", - "evals/cases/vercel-deploy/README.md": "e585f09b81b3a3a2f23624f1d8e7c816af5b50d88f4204513072128bdef39b0c", - "evals/cases/vercel-deploy/SKILL.md": "4ac80ad71f42c327f63685530b6b98ce3e287459527d2c0bafb5e1c716c313da", - "evals/cases/vercel-deploy/workflow.ts": "78667cea9b4b4d3dfc54ef6944ca9aa75d837b0170a4d3811b8f55172f63d6f2", - "evals/package-lock.json": "22c099aa5f2d9959084d6703e4a94ff7fa34216cac21b9d81137e2dd155c09d5", - "evals/package.json": "3361b8d579664bf9d6642eaeca6a906a1bca43fd0b1b543b00b2d5e0aa021b17", - "evals/results/README.md": "8eda7b901660466e60e5e32609cb2a3dcca7d9d512a93f2f379090b16da15525", - "evals/results/latest.json": "339a15841921fa270c668aa5c55cd33a271493735b4b85beaeaba1b25807216a", - "evals/scripts/measure-source.mjs": "064a62073422c067e12b0f2b77cc65a38f7d3440efb1a91020595a941470db7a", - "evals/scripts/validate.mjs": "2616fdf2ebbfd940dff7e06bcb33f4a362e063484a4beb62f54ef2856747abd7", + "evals/README.md": "7b0a92908d4eff84f1716756a00536ed79b31ef542928a2fd0c17c234161bba6", + "evals/package-lock.json": "cfbd68d590e92b94233a807fd774e6667e9474096ab76e1c5cd037acfb4e3200", + "evals/package.json": "aa97f75cadbf6d1802dcf03f4b34dcbbc4aea8c2b5ef18130af26a9e63ecac51", + "evals/results/README.md": "65b74bfab83dfc51fc5b8a3a4824c37b722fb3a358a5dcfb217dc2af199f4801", + "evals/results/latest.json": "cc9346c0db8e8395464a7348ea440a891fad0080350ac5d677b1a2875047e4d4", + "evals/scripts/run.mjs": "119248cd4d383326d396f4459df28c51ce7c31088d97639851f2f5ab4c329892", + "evals/scripts/validate.mjs": "7ac2a3239cd1b52a604c8a8a52a1af4ac55c4007ac2751d4fbfe20d0e1f568dc", "examples/convert-skill/SKILL.md": "e6376f34365d4ac030d316db55e91f0c606a501668099f4ecf1e27d43ec806a2", "examples/convert-skill/fixtures/responses.json": "5b5f27b0ae5962360ac7b0e779992f430c655f352a38da37debb3505c32cd60a", "examples/convert-skill/main.go": "eb987ed1ce9139e6d0d40a2fae20c532c2fba3b27357e31555db2f0f52db4e39", @@ -286,6 +266,7 @@ "release-notes/2026-08-01-evaluation-case-guides.md": "570bc996eb4d2892456a02d938fb6299d106c431bb842ce726db1df550f779d7", "release-notes/2026-08-01-evaluation-surface.md": "64fb5e2fdad3ccd41967e028cf4675c45939174000281c443f4b28d96dc07549", "release-notes/2026-08-01-example-library.md": "14d6ca40529a6aeeb72872295e57bb0d7dcc7824df31e029e497602060ce4c97", + "release-notes/2026-08-01-first-party-evaluations.md": "9e74c48112343c409a88358138db80d731004f87900f1b23073f16c415163fcf", "release-notes/2026-08-01-initial-projection.md": "d38f0832b5552fb97237b30d19bb63369442ddde69e678b752ac07eedeb7ba3d", "release-notes/2026-08-01-multi-language-and-converter.md": "d0cf62d191e6a58e827f3b35d442f88b83be4dde98aa25ad54e4ae3ff787a453", "release-notes/2026-08-01-remove-stray-analysis-traces.md": "0567f78ee97ffd23b3f26b5c39606e9ff6659c50a3fdef04ed9b3aa86cfa99af", @@ -302,7 +283,7 @@ "generator": "operatorstack/yield:project", "schema_version": 1, "source": { - "commit": "381aea9c6d153503e68ad96e82ac236a354db432", + "commit": "4c338abc0d14f017eb3501885adca7594bd5b6f4", "path": "labs/22-yield", "repository": "operatorstack/intelligence-flow" } diff --git a/docs/examples.md b/docs/examples.md index e5638ad..31ce6c3 100644 --- a/docs/examples.md +++ b/docs/examples.md @@ -61,14 +61,12 @@ When adapting an example, change the repository-specific commands and model instructions. Keep stable operation IDs for existing steps so saved runs can replay them. -## Conversion cases used by evaluations +## Evaluated examples -The [evaluation cases](../evals/cases/) are different from the example library. -Each one pins a public third-party skill to an exact commit and digest, then -shows the smaller model-facing `SKILL.md` beside the TypeScript workflow that -owns its order, commands, gates, and completion. +The [first-party evaluation suite](../evals/) runs every example-library +pattern through TypeScript, Python, Go, and Rust. It also checks resume, replay, +changed behavior, blocking, and changed source. -The source-size harness and early summaries live in [`evals/`](../evals/). -Large transcripts and temporary repositories are external artifacts, never -committed source. A behavioral result is publishable only when its summary -binds the exact artifact URI and SHA-256. +These results show that the tested Yield version runs the workflow steps in +code. They do not compare Yield with prose or test whether an agent's judgment +is correct. diff --git a/evals/README.md b/evals/README.md index a8d144d..a7a813a 100644 --- a/evals/README.md +++ b/evals/README.md @@ -1,39 +1,55 @@ # Yield evaluations -This directory contains the public, reviewable part of Yield's evaluation -system: case definitions, pinned source identities, conversion programs, -measurement code, validation rules, and small result summaries. +These evaluations test Yield itself. They do not compare Yield with another +tool, company, prompt, or skill. -Raw agent transcripts, temporary repositories, command logs, and model -responses do not belong in Git. A full campaign uploads those files as one -immutable artifact bundle and records its URI and SHA-256 in the published -summary. Until that bundle exists, the summary must say `unpublished`. +The suite answers two questions: -## Layout +1. Can each checked-in example workflow reach its expected final result + through every supported SDK? +2. Does the runtime behave correctly when a run resumes, replays, blocks, or + encounters changed code? -- `cases/` — pinned public source identities plus the thin skill and Yield - program used for each conversion. -- `scripts/measure-source.mjs` — reproduces the source-size comparison from - pinned upstream files. -- `scripts/validate.mjs` — fail-closed validation for cases and summaries. -- `results/latest.json` — small website-safe summary. It is not raw evidence. -- `runs/` — local or CI output; ignored by Git and projected releases. +## Current coverage -## Run +- 10 workflow patterns written by this project. +- 4 SDKs: TypeScript, Python, Go, and Rust. +- 40 end-to-end workflow tests. +- 5 runtime checks: resume and complete, repeat the same saved step, stop when + behavior changes, block when a rule fails, and require approval for changed + source. + +Run the exact suite and refresh the checked-in result: + +```bash +cd evals +npm run eval +``` + +Check that the published result still matches the current source: ```bash -npm install -npm run validate -npm run measure +npm test ``` -`npm run measure` writes a fresh summary under `runs/`. Publishing that result -requires a separate promotion step that binds the raw artifact digest, the -exact Yield commit, model identity, harness version, and case-set digest. +## What a passing result proves + +A passing result proves that the tested Yield revision: + +- executes each owned workflow test to `completed`; +- runs command steps rather than asking the model to invent their outputs; +- presents requests in the program-defined order; +- resumes from recorded responses; +- returns to the same saved step during replay; +- stops on changed behavior or failed requirements. + +## What it does not prove -## Claim boundary +This suite does not prove that Yield is better than prose, that an agent's +judgment is correct, or that illustrative commands are production-safe. The +Fixed test data supplies agent and human responses so the suite can test only +the code-controlled workflow layer. -Source-size measurements show how much model-facing text and workflow source -the prototypes contain. They do not prove behavioral equivalence. Behavioral -claims require executable fixtures, held-out oracles, repeated model runs, and -the immutable raw artifact named by the result summary. +`results/latest.json` is a compact, website-safe result. Its source hash is +computed from the CLI, engine, protocol, SDKs, example workflows, fixtures, and +evaluation harness. CI reruns the suite instead of trusting that file alone. diff --git a/evals/cases/README.md b/evals/cases/README.md deleted file mode 100644 index 4d194a5..0000000 --- a/evals/cases/README.md +++ /dev/null @@ -1,15 +0,0 @@ -# Conversion cases - -These are measured rewrites, not automatic equivalence claims. Each case keeps -the model-facing judgment in a short `SKILL.md` and moves repeatable control -flow into a TypeScript Yield program. The pinned original remains in its source -repository and is identified by commit plus SHA-256 in `index.json`. - -| Case | Thin skill | Yield program | Pinned original | -|---|---|---|---| -| [GStack review](gstack-review/) | [SKILL.md](gstack-review/SKILL.md) | [workflow.ts](gstack-review/workflow.ts) | [source](https://github.com/garrytan/gstack/blob/a3259400a366593e0c909dd9ac3e59752efd2488/review/SKILL.md) | -| [Anthropic doc co-authoring](doc-coauthoring/) | [SKILL.md](doc-coauthoring/SKILL.md) | [workflow.ts](doc-coauthoring/workflow.ts) | [source](https://github.com/anthropics/skills/blob/b29e7cf65e5cb78a5ac33d582270551bc74a14eb/skills/doc-coauthoring/SKILL.md) | -| [Superpowers systematic debugging](systematic-debugging/) | [SKILL.md](systematic-debugging/SKILL.md) | [workflow.ts](systematic-debugging/workflow.ts) | [source](https://github.com/obra/superpowers/blob/44c9b2d6e889982ac18c27d05a19fefe335194e1/skills/systematic-debugging/SKILL.md) | -| [Vercel deploy](vercel-deploy/) | [SKILL.md](vercel-deploy/SKILL.md) | [workflow.ts](vercel-deploy/workflow.ts) | [source](https://github.com/vercel-labs/agent-skills/blob/7c180d9044c9ae2b442b567aad4e42a28dd5ed62/skills/deploy-to-vercel/SKILL.md) | -| [Microsoft MCP builder](mcp-builder/) | [SKILL.md](mcp-builder/SKILL.md) | [workflow.ts](mcp-builder/workflow.ts) | [source](https://github.com/microsoft/skills/blob/4a2873faffc1b101a33a0b59c24713d4ed78142f/.github/skills/mcp-builder/SKILL.md) | -| [Trail of Bits actions auditor](actions-auditor/) | [SKILL.md](actions-auditor/SKILL.md) | [workflow.ts](actions-auditor/workflow.ts) | [source](https://github.com/trailofbits/skills/blob/1256982d4d925a0acfe11e26c2253c32052c6247/plugins/agentic-actions-auditor/skills/agentic-actions-auditor/SKILL.md) | diff --git a/evals/cases/actions-auditor/README.md b/evals/cases/actions-auditor/README.md deleted file mode 100644 index e5ff368..0000000 --- a/evals/cases/actions-auditor/README.md +++ /dev/null @@ -1,34 +0,0 @@ -# Trail of Bits actions auditor — Yield conversion - -This is an independent, measured conversion of -[Trail of Bits' agentic actions auditor](https://github.com/trailofbits/skills/blob/1256982d4d925a0acfe11e26c2253c32052c6247/plugins/agentic-actions-auditor/skills/agentic-actions-auditor/SKILL.md). -It is not a Trail of Bits artifact. - -We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to -split security judgment from repeatable control flow, then reviewed the -checked-in result and measured it against the pinned original. - -## The converted version - -- [`SKILL.md`](SKILL.md) keeps attack-path analysis and finding quality. -- [`workflow.ts`](workflow.ts) owns discovery, coverage, evidence requirements, - report generation, and completion. - -## Result - -| Original | Thin skill | Yield program | Maintained change | -|---:|---:|---:|---:| -| 4,846 tokens | 132 tokens | 222 tokens | **−92.7%** | - -## Benefits in this conversion - -- Workflow files are discovered by a real command instead of inferred. -- Completion requires every discovered workflow to be reviewed. -- Every finding must carry concrete evidence. -- Report generation is an observed command with a checked exit code. - -## Claim boundary - -This result measures source size, not audit coverage or behavioral equivalence. -See the [evaluation methodology](../../README.md) for the evidence required -before making a behavior claim. diff --git a/evals/cases/actions-auditor/SKILL.md b/evals/cases/actions-auditor/SKILL.md deleted file mode 100644 index ffd83ce..0000000 --- a/evals/cases/actions-auditor/SKILL.md +++ /dev/null @@ -1,17 +0,0 @@ ---- -name: agentic-actions-auditor -description: Audit AI-enabled CI workflows for concrete security risks. ---- - -# Agentic actions auditor - -Inspect the supplied workflow files and repository context. Trace untrusted input -to agent prompts, tools, credentials, write permissions, network access, and -mutable dependencies. Report only findings with a concrete attack path. - -Each finding must include severity, workflow and line, source, capability reached, -impact, evidence, and remediation. Distinguish exploitable paths from hardening -advice. State coverage gaps explicitly. - -Yield owns file discovery, scope, evidence capture, required fields, report -generation, and completion. diff --git a/evals/cases/actions-auditor/workflow.ts b/evals/cases/actions-auditor/workflow.ts deleted file mode 100644 index 56aadc9..0000000 --- a/evals/cases/actions-auditor/workflow.ts +++ /dev/null @@ -1,13 +0,0 @@ -export default defineSkill((ctx) => { - const files = ctx.runCommand("discover", "find .github/workflows -type f -name '*.yml' -o -name '*.yaml'", 30) - ctx.require(files.exit_code === 0 && files.stdout.length > 0, "workflow files found", files) - - const context = ctx.runCommand("context", "git ls-files && git status --short", 30) - const audit = ctx.agentTask("audit", prompts.audit, { files: files.stdout, context }) - ctx.require(audit.coverage.reviewed === audit.coverage.discovered, "all workflows reviewed", audit.coverage) - ctx.require(audit.findings.every((finding) => finding.evidence), "every finding has evidence", audit) - - const report = ctx.runCommand("report", "node scripts/render-audit.mjs", 60, { stdin: audit }) - ctx.require(report.exit_code === 0, "report generated", report) - return ctx.complete({ findings: audit.findings, report: report.stdout }) -}) diff --git a/evals/cases/doc-coauthoring/README.md b/evals/cases/doc-coauthoring/README.md deleted file mode 100644 index ea82e05..0000000 --- a/evals/cases/doc-coauthoring/README.md +++ /dev/null @@ -1,34 +0,0 @@ -# Anthropic doc co-authoring — Yield conversion - -This is an independent, measured conversion of -[Anthropic's doc co-authoring skill](https://github.com/anthropics/skills/blob/b29e7cf65e5cb78a5ac33d582270551bc74a14eb/skills/doc-coauthoring/SKILL.md). -It is not an Anthropic artifact. - -We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to -split model judgment from repeatable control flow, then reviewed the checked-in -result and measured it against the pinned original. - -## The converted version - -- [`SKILL.md`](SKILL.md) keeps writing judgment, reader perspective, and voice. -- [`workflow.ts`](workflow.ts) owns the questions, outline approval, draft stages, - reader test, saved answers, and completion gate. - -## Result - -| Original | Thin skill | Yield program | Maintained change | -|---:|---:|---:|---:| -| 3,289 tokens | 137 tokens | 209 tokens | **−89.5%** | - -## Benefits in this conversion - -- A returning run can continue from saved audience, outcome, and source context. -- Outline approval is an explicit gate instead of a prose suggestion. -- Reader testing happens before the final revision. -- The model spends its context on the document, not on remembering stage order. - -## Claim boundary - -This result measures source size, not writing quality or behavioral equivalence. -See the [evaluation methodology](../../README.md) for the evidence required -before making a behavior claim. diff --git a/evals/cases/doc-coauthoring/SKILL.md b/evals/cases/doc-coauthoring/SKILL.md deleted file mode 100644 index 7e7d62b..0000000 --- a/evals/cases/doc-coauthoring/SKILL.md +++ /dev/null @@ -1,17 +0,0 @@ ---- -name: doc-coauthoring -description: Help a user turn context into a clear document for a named reader. ---- - -# Doc co-authoring - -At each model step, ask only the questions needed for the current stage. Keep -the author's voice. Make claims concrete, expose missing evidence, and organize -the document around what its intended reader must understand or decide. - -For reader testing, act as a fresh reader with no hidden context. List unclear -terms, unanswered questions, and assumptions the document makes. Suggest the -smallest edits that resolve them. - -Yield owns stage order, saved answers, iteration limits, user choices, and the -definition of done. diff --git a/evals/cases/doc-coauthoring/workflow.ts b/evals/cases/doc-coauthoring/workflow.ts deleted file mode 100644 index 7ba13a5..0000000 --- a/evals/cases/doc-coauthoring/workflow.ts +++ /dev/null @@ -1,15 +0,0 @@ -export default defineSkill((ctx) => { - const audience = ctx.askUser("audience", "Who will read this document?") - const outcome = ctx.askUser("outcome", "What should the reader know or decide?") - const context = ctx.askUser("context", "Paste the source context.") - - const outline = ctx.agentTask("outline", prompts.outline, { audience, outcome, context }) - const chosen = ctx.askUser("outline-approval", "Use this outline?", ["use", "revise"]) - ctx.require(chosen === "use", "outline approved", outline) - - const draft = ctx.agentTask("draft", prompts.draft, { outline, context }) - const test = ctx.agentTask("reader-test", prompts.readerTest, { audience, draft }) - const final = ctx.agentTask("revise", prompts.revise, { draft, test }) - ctx.require(test.blocking_questions.length === 0, "reader has no blocking questions", test) - return ctx.complete(final) -}) diff --git a/evals/cases/gstack-review/README.md b/evals/cases/gstack-review/README.md deleted file mode 100644 index 41eca41..0000000 --- a/evals/cases/gstack-review/README.md +++ /dev/null @@ -1,34 +0,0 @@ -# GStack review — Yield conversion - -This is an independent, measured conversion of -[GStack's review skill](https://github.com/garrytan/gstack/blob/a3259400a366593e0c909dd9ac3e59752efd2488/review/SKILL.md). -It is not an upstream GStack artifact. - -We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to -split model judgment from repeatable control flow, then reviewed the checked-in -result and measured it against the pinned original. - -## The converted version - -- [`SKILL.md`](SKILL.md) keeps review judgment, severity rules, and finding shape. -- [`workflow.ts`](workflow.ts) owns the diff, tests, typecheck, critical-finding - gate, saved result, and completion. - -## Result - -| Original | Thin skill | Yield program | Maintained change | -|---:|---:|---:|---:| -| 27,162 tokens | 146 tokens | 160 tokens | **−98.9%** | - -## Benefits in this conversion - -- Repository checks run as commands instead of instructions the model must remember. -- A review cannot complete while a critical finding remains. -- The diff and check output become explicit evidence passed into the review. -- The model-facing prompt is small enough to focus on finding real defects. - -## Claim boundary - -The token result measures source size. A separate early GStack behavior study is -summarized in [`results/latest.json`](../../results/latest.json), but its raw -artifact is not yet published and it does not prove general equivalence. diff --git a/evals/cases/gstack-review/SKILL.md b/evals/cases/gstack-review/SKILL.md deleted file mode 100644 index 0c64cbb..0000000 --- a/evals/cases/gstack-review/SKILL.md +++ /dev/null @@ -1,18 +0,0 @@ ---- -name: review -description: Review the current branch for important defects before shipping. ---- - -# Review - -Inspect the supplied diff and evidence. Look for failures tests may miss: -trust-boundary mistakes, unsafe side effects, incomplete error handling, -concurrency problems, and behavior that contradicts the surrounding code. - -Return structured findings with `severity`, `confidence`, `category`, `file`, -`line`, `problem`, and `fix`. Use only `critical` or `informational`. Count only -defects caused by this branch. Prefer a small number of specific findings over -general advice. - -Yield owns repository checks, ordering, policy validation, saved state, and -completion. Do not reproduce those rules in prose. diff --git a/evals/cases/gstack-review/workflow.ts b/evals/cases/gstack-review/workflow.ts deleted file mode 100644 index 14b7a61..0000000 --- a/evals/cases/gstack-review/workflow.ts +++ /dev/null @@ -1,12 +0,0 @@ -export default defineSkill((ctx) => { - const diff = ctx.runCommand("diff", "git diff --merge-base origin/main HEAD", 60) - ctx.require(diff.exit_code === 0 && diff.stdout.length > 0, "branch has a readable diff", diff) - - const checks = ctx.runCommand("checks", "npm test && npm run typecheck", 600) - ctx.require(checks.exit_code === 0, "tests and types pass", checks) - - const review = ctx.agentTask("review", prompts.review, { diff: diff.stdout, checks }) - const critical = review.findings.some((finding) => finding.severity === "critical") - ctx.require(!critical, "no unresolved critical findings", review) - return ctx.complete(review) -}) diff --git a/evals/cases/index.json b/evals/cases/index.json deleted file mode 100644 index c92575a..0000000 --- a/evals/cases/index.json +++ /dev/null @@ -1,84 +0,0 @@ -{ - "schema_version": 1, - "methodology_version": "0.1", - "cases": [ - { - "id": "gstack-review", - "label": "GStack review", - "source": { - "repo": "garrytan/gstack", - "commit": "a3259400a366593e0c909dd9ac3e59752efd2488", - "path": "review/SKILL.md", - "license": "MIT", - "sha256": "92ee16af71d5e0088326869b0a211c50f94b9261eeae75656bc21f9bcfae2031" - }, - "thin_skill": "gstack-review/SKILL.md", - "workflow": "gstack-review/workflow.ts" - }, - { - "id": "doc-coauthoring", - "label": "Anthropic doc co-authoring", - "source": { - "repo": "anthropics/skills", - "commit": "b29e7cf65e5cb78a5ac33d582270551bc74a14eb", - "path": "skills/doc-coauthoring/SKILL.md", - "license": "See source repository", - "sha256": "2e47d78846faeea4a56e9809c52700087a15a2155a3f293a3efbaded81398ef4" - }, - "thin_skill": "doc-coauthoring/SKILL.md", - "workflow": "doc-coauthoring/workflow.ts" - }, - { - "id": "systematic-debugging", - "label": "Superpowers systematic debugging", - "source": { - "repo": "obra/superpowers", - "commit": "44c9b2d6e889982ac18c27d05a19fefe335194e1", - "path": "skills/systematic-debugging/SKILL.md", - "license": "MIT", - "sha256": "808fc5717aa88ad65efff312b11c186294d3e6ee301afb584e2f86599b137787" - }, - "thin_skill": "systematic-debugging/SKILL.md", - "workflow": "systematic-debugging/workflow.ts" - }, - { - "id": "vercel-deploy", - "label": "Vercel deploy", - "source": { - "repo": "vercel-labs/agent-skills", - "commit": "7c180d9044c9ae2b442b567aad4e42a28dd5ed62", - "path": "skills/deploy-to-vercel/SKILL.md", - "license": "See source repository", - "sha256": "cfcc3dd479ab2e0ae721ddf39b8af84d977321487672f1487c8d6855f576927b" - }, - "thin_skill": "vercel-deploy/SKILL.md", - "workflow": "vercel-deploy/workflow.ts" - }, - { - "id": "mcp-builder", - "label": "Microsoft MCP builder", - "source": { - "repo": "microsoft/skills", - "commit": "4a2873faffc1b101a33a0b59c24713d4ed78142f", - "path": ".github/skills/mcp-builder/SKILL.md", - "license": "MIT", - "sha256": "621e771c22224140752ddf933923467b2d1148580194ad59d3f1f21f9f27bdc9" - }, - "thin_skill": "mcp-builder/SKILL.md", - "workflow": "mcp-builder/workflow.ts" - }, - { - "id": "actions-auditor", - "label": "Trail of Bits actions auditor", - "source": { - "repo": "trailofbits/skills", - "commit": "1256982d4d925a0acfe11e26c2253c32052c6247", - "path": "plugins/agentic-actions-auditor/skills/agentic-actions-auditor/SKILL.md", - "license": "CC-BY-SA-4.0", - "sha256": "80e36ab06e3ee667ac45bab036ed395fc425c4d54fa72c4b4981a5b982a389aa" - }, - "thin_skill": "actions-auditor/SKILL.md", - "workflow": "actions-auditor/workflow.ts" - } - ] -} diff --git a/evals/cases/mcp-builder/README.md b/evals/cases/mcp-builder/README.md deleted file mode 100644 index 5c52d32..0000000 --- a/evals/cases/mcp-builder/README.md +++ /dev/null @@ -1,34 +0,0 @@ -# Microsoft MCP builder — Yield conversion - -This is an independent, measured conversion of -[Microsoft's MCP builder skill](https://github.com/microsoft/skills/blob/4a2873faffc1b101a33a0b59c24713d4ed78142f/.github/skills/mcp-builder/SKILL.md). -It is not a Microsoft artifact. - -We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to -split tool-design judgment from repeatable control flow, then reviewed the -checked-in result and measured it against the pinned original. - -## The converted version - -- [`SKILL.md`](SKILL.md) keeps tool-design and review judgment. -- [`workflow.ts`](workflow.ts) owns user inputs, the tool-count gate, design - approval, scaffold, tests, evaluations, and completion. - -## Result - -| Original | Thin skill | Yield program | Maintained change | -|---:|---:|---:|---:| -| 2,648 tokens | 131 tokens | 291 tokens | **−84.1%** | - -## Benefits in this conversion - -- The tool surface stays bounded before code generation begins. -- Building requires explicit design approval. -- Scaffold, tests, review, and evaluations happen in a fixed order. -- Completion requires both review and evaluation gates to pass. - -## Claim boundary - -This result measures source size, not MCP server quality or behavioral -equivalence. See the [evaluation methodology](../../README.md) for the evidence -required before making a behavior claim. diff --git a/evals/cases/mcp-builder/SKILL.md b/evals/cases/mcp-builder/SKILL.md deleted file mode 100644 index a9b860e..0000000 --- a/evals/cases/mcp-builder/SKILL.md +++ /dev/null @@ -1,17 +0,0 @@ ---- -name: mcp-builder -description: Design and review a small MCP server around a defined use case. ---- - -# MCP builder - -Turn the supplied use case into a minimal tool surface. For each tool, define a -specific name, typed inputs, bounded output, error behavior, and one realistic -example. Prefer fewer composable tools. Keep secrets out of arguments and make -destructive effects explicit. - -During review, check schema clarity, transport errors, authentication boundaries, -pagination, idempotency, and whether evaluations cover success and failure paths. - -Yield owns research approval, scaffold and test commands, evaluation thresholds, -saved artifacts, and completion. diff --git a/evals/cases/mcp-builder/workflow.ts b/evals/cases/mcp-builder/workflow.ts deleted file mode 100644 index ac5cc77..0000000 --- a/evals/cases/mcp-builder/workflow.ts +++ /dev/null @@ -1,19 +0,0 @@ -export default defineSkill((ctx) => { - const useCase = ctx.askUser("use-case", "What must this MCP server enable?") - const constraints = ctx.askUser("constraints", "Which APIs, auth, and runtime apply?") - const design = ctx.agentTask("design", prompts.design, { useCase, constraints }) - ctx.require(design.tools.length <= 8, "tool surface stays small", design) - - const approval = ctx.askUser("design-approval", "Build this tool surface?", ["build", "revise"]) - if (approval !== "build") ctx.blocked("design needs revision") - const scaffold = ctx.runCommand("scaffold", "npm run scaffold:mcp", 120) - ctx.require(scaffold.exit_code === 0, "server scaffolds", scaffold) - const tests = ctx.runCommand("tests", "npm test", 600) - ctx.require(tests.exit_code === 0, "tests pass", tests) - - const review = ctx.agentTask("review", prompts.review, { design, tests }) - ctx.require(review.blockers.length === 0, "review has no blockers", review) - const evals = ctx.runCommand("evals", "npm run evals", 900) - ctx.require(evals.exit_code === 0, "evaluations pass", evals) - return ctx.complete({ design, review, evals }) -}) diff --git a/evals/cases/systematic-debugging/README.md b/evals/cases/systematic-debugging/README.md deleted file mode 100644 index 66e8838..0000000 --- a/evals/cases/systematic-debugging/README.md +++ /dev/null @@ -1,34 +0,0 @@ -# Systematic debugging — Yield conversion - -This is an independent, measured conversion of -[Superpowers' systematic debugging skill](https://github.com/obra/superpowers/blob/44c9b2d6e889982ac18c27d05a19fefe335194e1/skills/systematic-debugging/SKILL.md). -It is not an upstream Superpowers artifact. - -We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to -split diagnostic judgment from repeatable control flow, then reviewed the -checked-in result and measured it against the pinned original. - -## The converted version - -- [`SKILL.md`](SKILL.md) keeps the standard for a falsifiable root-cause hypothesis. -- [`workflow.ts`](workflow.ts) owns reproduction, experiment bounds, approval, - verification, and completion. - -## Result - -| Original | Thin skill | Yield program | Maintained change | -|---:|---:|---:|---:| -| 2,226 tokens | 137 tokens | 226 tokens | **−83.7%** | - -## Benefits in this conversion - -- The failure must reproduce before diagnosis starts. -- A proposed cause must include an executable falsification experiment. -- Applying a fix requires user approval. -- Completion requires the full test suite to pass. - -## Claim boundary - -This result measures source size, not diagnosis quality or behavioral -equivalence. See the [evaluation methodology](../../README.md) for the evidence -required before making a behavior claim. diff --git a/evals/cases/systematic-debugging/SKILL.md b/evals/cases/systematic-debugging/SKILL.md deleted file mode 100644 index 610fb8f..0000000 --- a/evals/cases/systematic-debugging/SKILL.md +++ /dev/null @@ -1,18 +0,0 @@ ---- -name: systematic-debugging -description: Diagnose a reproducible failure from evidence before proposing a fix. ---- - -# Systematic debugging - -Use the supplied failure output and code context to identify one falsifiable -root-cause hypothesis. Separate observations from inference. Name the mechanism, -the evidence that supports it, and the smallest experiment that could disprove -it. Do not recommend a fix until the hypothesis survives that experiment. - -When reviewing a candidate fix, check that it addresses the mechanism rather -than hiding the symptom and that the original failure now passes without a new -regression. - -Yield owns phase order, experiment bounds, commands, saved evidence, and exit -conditions. diff --git a/evals/cases/systematic-debugging/workflow.ts b/evals/cases/systematic-debugging/workflow.ts deleted file mode 100644 index 815e5b3..0000000 --- a/evals/cases/systematic-debugging/workflow.ts +++ /dev/null @@ -1,16 +0,0 @@ -export default defineSkill((ctx) => { - const failure = ctx.runCommand("reproduce", "npm test -- --runInBand", 300) - ctx.require(failure.exit_code !== 0, "failure reproduces", failure) - - const hypothesis = ctx.agentTask("hypothesis", prompts.hypothesis, { failure }) - ctx.require(Boolean(hypothesis.experiment), "hypothesis is falsifiable", hypothesis) - const experiment = ctx.runCommand("experiment", hypothesis.experiment, 300) - const verdict = ctx.agentTask("evaluate", prompts.evaluate, { hypothesis, experiment }) - ctx.require(verdict.supported, "root cause supported by evidence", verdict) - - const fix = ctx.askUser("fix", "Apply the proposed fix?", ["apply", "stop"]) - if (fix !== "apply") ctx.blocked("fix not approved") - const verification = ctx.runCommand("verify", "npm test", 600) - ctx.require(verification.exit_code === 0, "full test suite passes", verification) - return ctx.complete({ hypothesis, verification }) -}) diff --git a/evals/cases/vercel-deploy/README.md b/evals/cases/vercel-deploy/README.md deleted file mode 100644 index 0bbd19d..0000000 --- a/evals/cases/vercel-deploy/README.md +++ /dev/null @@ -1,34 +0,0 @@ -# Vercel deploy — Yield conversion - -This is an independent, measured conversion of -[Vercel's deploy skill](https://github.com/vercel-labs/agent-skills/blob/7c180d9044c9ae2b442b567aad4e42a28dd5ed62/skills/deploy-to-vercel/SKILL.md). -It is not a Vercel artifact. - -We used Yield's [`convert-skill`](../../../examples/convert-skill/) workflow to -split deployment explanation from repeatable control flow, then reviewed the -checked-in result and measured it against the pinned original. - -## The converted version - -- [`SKILL.md`](SKILL.md) keeps interpretation of project state and failures. -- [`workflow.ts`](workflow.ts) owns state detection, approval, authentication, - linking, deployment, HTTPS verification, and completion. - -## Result - -| Original | Thin skill | Yield program | Maintained change | -|---:|---:|---:|---:| -| 2,898 tokens | 108 tokens | 290 tokens | **−86.3%** | - -## Benefits in this conversion - -- The deployment command runs only after explicit approval. -- Missing authentication produces an honest blocked outcome. -- A successful command is not enough; the deployed URL must answer over HTTPS. -- Cancellation and failure are recorded separately from success. - -## Claim boundary - -This result measures source size, not deployment reliability or behavioral -equivalence. See the [evaluation methodology](../../README.md) for the evidence -required before making a behavior claim. diff --git a/evals/cases/vercel-deploy/SKILL.md b/evals/cases/vercel-deploy/SKILL.md deleted file mode 100644 index 159eebd..0000000 --- a/evals/cases/vercel-deploy/SKILL.md +++ /dev/null @@ -1,14 +0,0 @@ ---- -name: deploy-to-vercel -description: Interpret deployment evidence and explain a failed Vercel release. ---- - -# Deploy to Vercel - -Given detected project state and command output, explain the selected deployment -path in plain language. If deployment fails, identify the failing boundary and -suggest one next action grounded in the output. Never claim a deployment is -live without a successful command and HTTP verification. - -Yield owns state detection, method selection, team choice, command execution, -timeouts, verification, and the final success gate. diff --git a/evals/cases/vercel-deploy/workflow.ts b/evals/cases/vercel-deploy/workflow.ts deleted file mode 100644 index ef9425c..0000000 --- a/evals/cases/vercel-deploy/workflow.ts +++ /dev/null @@ -1,18 +0,0 @@ -export default defineSkill((ctx) => { - const git = ctx.runCommand("git-state", "git remote get-url origin", 20) - const linked = ctx.runCommand("vercel-state", "test -f .vercel/project.json", 20) - const auth = ctx.runCommand("auth", "vercel whoami", 30) - - const state = { git: git.exit_code === 0, linked: linked.exit_code === 0, auth: auth.exit_code === 0 } - const plan = ctx.agentTask("explain-plan", prompts.plan, state) - const approval = ctx.askUser("deploy", "Deploy using this plan?", ["deploy", "cancel"]) - if (approval !== "deploy") ctx.refused("deployment cancelled") - if (!state.auth) ctx.blocked("Vercel authentication required") - if (!state.linked) ctx.runCommand("link", "vercel link", 120) - - const deploy = ctx.runCommand("deploy", "vercel deploy --prod --yes", 900) - ctx.require(deploy.exit_code === 0, "deploy command succeeds", deploy) - const verify = ctx.runCommand("verify", `curl -fsS ${deploy.url}`, 120) - ctx.require(verify.exit_code === 0, "deployment responds over HTTPS", verify) - return ctx.complete({ url: deploy.url, plan }) -}) diff --git a/evals/package-lock.json b/evals/package-lock.json index 6b3084d..b6f6b21 100644 --- a/evals/package-lock.json +++ b/evals/package-lock.json @@ -1,21 +1,12 @@ { "name": "@operatorstack/yield-evals", - "version": "0.1.0", + "version": "1.0.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@operatorstack/yield-evals", - "version": "0.1.0", - "dependencies": { - "gpt-tokenizer": "3.4.0" - } - }, - "node_modules/gpt-tokenizer": { - "version": "3.4.0", - "resolved": "https://registry.npmjs.org/gpt-tokenizer/-/gpt-tokenizer-3.4.0.tgz", - "integrity": "sha512-wxFLnhIXTDjYebd9A9pGl3e31ZpSypbpIJSOswbgop5jLte/AsZVDvjlbEuVFlsqZixVKqbcoNmRlFDf6pz/UQ==", - "license": "MIT" + "version": "1.0.0" } } } diff --git a/evals/package.json b/evals/package.json index 7d3ba68..ad898e4 100644 --- a/evals/package.json +++ b/evals/package.json @@ -1,13 +1,11 @@ { "name": "@operatorstack/yield-evals", - "version": "0.1.0", + "version": "1.0.0", "private": true, "type": "module", "scripts": { - "validate": "node scripts/validate.mjs", - "measure": "node scripts/measure-source.mjs" - }, - "dependencies": { - "gpt-tokenizer": "3.4.0" + "eval": "node scripts/run.mjs --write", + "test": "node scripts/validate.mjs && node scripts/run.mjs --check", + "validate": "node scripts/validate.mjs" } } diff --git a/evals/results/README.md b/evals/results/README.md index cbe99db..e2389c8 100644 --- a/evals/results/README.md +++ b/evals/results/README.md @@ -1,17 +1,8 @@ -# Published summaries +# Published result -`latest.json` is intentionally small enough for documentation and websites. -It may point to a large raw artifact, but it never embeds transcripts or -temporary repositories. +`latest.json` records the latest first-party test run. It contains no +model transcripts and no third-party comparison data. -A promoted behavior result must contain: - -- the exact Yield commit and eval case-set digest; -- model and harness versions; -- fixture, repeat, and arm counts; -- aggregate metrics with uncertainty intervals; -- an immutable artifact URI and SHA-256; -- explicit exclusions and untested boundaries. - -The current result is marked early and its raw artifact is unpublished. It is -useful for building the reporting surface, not as a final launch claim. +The result is publishable only when `npm test` reruns all workflow tests and +runtime checks successfully and reproduces the same source hash, counts, +and case identities. diff --git a/evals/results/latest.json b/evals/results/latest.json index f7fac15..b521e6c 100644 --- a/evals/results/latest.json +++ b/evals/results/latest.json @@ -1,41 +1,306 @@ { - "schema_version": 1, - "status": "early", - "source_size": { - "methodology_version": "0.1", - "generated_at": "2026-08-01T15:57:22.533Z", - "tokenizer": "gpt-tokenizer 3.4.0 / cl100k_base", - "summary": { - "original_tokens": 43069, - "prompt_tokens": 791, - "workflow_tokens": 1398, - "maintained_tokens": 2189, - "maintained_reduction_pct": 94.9 - }, - "rows": [ - {"id":"gstack-review","label":"GStack review","original_tokens":27162,"prompt_tokens":146,"workflow_tokens":160,"maintained_tokens":306,"maintained_change_pct":-98.9}, - {"id":"doc-coauthoring","label":"Anthropic doc co-authoring","original_tokens":3289,"prompt_tokens":137,"workflow_tokens":209,"maintained_tokens":346,"maintained_change_pct":-89.5}, - {"id":"systematic-debugging","label":"Superpowers systematic debugging","original_tokens":2226,"prompt_tokens":137,"workflow_tokens":226,"maintained_tokens":363,"maintained_change_pct":-83.7}, - {"id":"vercel-deploy","label":"Vercel deploy","original_tokens":2898,"prompt_tokens":108,"workflow_tokens":290,"maintained_tokens":398,"maintained_change_pct":-86.3}, - {"id":"mcp-builder","label":"Microsoft MCP builder","original_tokens":2648,"prompt_tokens":131,"workflow_tokens":291,"maintained_tokens":422,"maintained_change_pct":-84.1}, - {"id":"actions-auditor","label":"Trail of Bits actions auditor","original_tokens":4846,"prompt_tokens":132,"workflow_tokens":222,"maintained_tokens":354,"maintained_change_pct":-92.7} + "schema_version": 2, + "methodology_version": "1.0", + "generated_at": "2026-08-01T17:53:20.762Z", + "source_digest": "8426ef59771921109fa7aa2d242e089d13f391aea60f2aebf509da93b7b534ae", + "status": "passed", + "workflow_conformance": { + "passed": 40, + "total": 40, + "patterns": 10, + "languages": [ + "typescript", + "python", + "go", + "rust" + ], + "cases": [ + { + "id": "typescript/review-branch", + "language": "typescript", + "pattern": "review-branch", + "status": "passed" + }, + { + "id": "typescript/investigate-failure", + "language": "typescript", + "pattern": "investigate-failure", + "status": "passed" + }, + { + "id": "typescript/qa-web-change", + "language": "typescript", + "pattern": "qa-web-change", + "status": "passed" + }, + { + "id": "typescript/release-package", + "language": "typescript", + "pattern": "release-package", + "status": "passed" + }, + { + "id": "typescript/triage-issue", + "language": "typescript", + "pattern": "triage-issue", + "status": "passed" + }, + { + "id": "typescript/repair-ci", + "language": "typescript", + "pattern": "repair-ci", + "status": "passed" + }, + { + "id": "typescript/upgrade-dependency", + "language": "typescript", + "pattern": "upgrade-dependency", + "status": "passed" + }, + { + "id": "typescript/migrate-database", + "language": "typescript", + "pattern": "migrate-database", + "status": "passed" + }, + { + "id": "typescript/audit-security", + "language": "typescript", + "pattern": "audit-security", + "status": "passed" + }, + { + "id": "typescript/publish-ios", + "language": "typescript", + "pattern": "publish-ios", + "status": "passed" + }, + { + "id": "python/review-branch", + "language": "python", + "pattern": "review-branch", + "status": "passed" + }, + { + "id": "python/investigate-failure", + "language": "python", + "pattern": "investigate-failure", + "status": "passed" + }, + { + "id": "python/qa-web-change", + "language": "python", + "pattern": "qa-web-change", + "status": "passed" + }, + { + "id": "python/release-package", + "language": "python", + "pattern": "release-package", + "status": "passed" + }, + { + "id": "python/triage-issue", + "language": "python", + "pattern": "triage-issue", + "status": "passed" + }, + { + "id": "python/repair-ci", + "language": "python", + "pattern": "repair-ci", + "status": "passed" + }, + { + "id": "python/upgrade-dependency", + "language": "python", + "pattern": "upgrade-dependency", + "status": "passed" + }, + { + "id": "python/migrate-database", + "language": "python", + "pattern": "migrate-database", + "status": "passed" + }, + { + "id": "python/audit-security", + "language": "python", + "pattern": "audit-security", + "status": "passed" + }, + { + "id": "python/publish-ios", + "language": "python", + "pattern": "publish-ios", + "status": "passed" + }, + { + "id": "go/review-branch", + "language": "go", + "pattern": "review-branch", + "status": "passed" + }, + { + "id": "go/investigate-failure", + "language": "go", + "pattern": "investigate-failure", + "status": "passed" + }, + { + "id": "go/qa-web-change", + "language": "go", + "pattern": "qa-web-change", + "status": "passed" + }, + { + "id": "go/release-package", + "language": "go", + "pattern": "release-package", + "status": "passed" + }, + { + "id": "go/triage-issue", + "language": "go", + "pattern": "triage-issue", + "status": "passed" + }, + { + "id": "go/repair-ci", + "language": "go", + "pattern": "repair-ci", + "status": "passed" + }, + { + "id": "go/upgrade-dependency", + "language": "go", + "pattern": "upgrade-dependency", + "status": "passed" + }, + { + "id": "go/migrate-database", + "language": "go", + "pattern": "migrate-database", + "status": "passed" + }, + { + "id": "go/audit-security", + "language": "go", + "pattern": "audit-security", + "status": "passed" + }, + { + "id": "go/publish-ios", + "language": "go", + "pattern": "publish-ios", + "status": "passed" + }, + { + "id": "rust/review-branch", + "language": "rust", + "pattern": "review-branch", + "status": "passed" + }, + { + "id": "rust/investigate-failure", + "language": "rust", + "pattern": "investigate-failure", + "status": "passed" + }, + { + "id": "rust/qa-web-change", + "language": "rust", + "pattern": "qa-web-change", + "status": "passed" + }, + { + "id": "rust/release-package", + "language": "rust", + "pattern": "release-package", + "status": "passed" + }, + { + "id": "rust/triage-issue", + "language": "rust", + "pattern": "triage-issue", + "status": "passed" + }, + { + "id": "rust/repair-ci", + "language": "rust", + "pattern": "repair-ci", + "status": "passed" + }, + { + "id": "rust/upgrade-dependency", + "language": "rust", + "pattern": "upgrade-dependency", + "status": "passed" + }, + { + "id": "rust/migrate-database", + "language": "rust", + "pattern": "migrate-database", + "status": "passed" + }, + { + "id": "rust/audit-security", + "language": "rust", + "pattern": "audit-security", + "status": "passed" + }, + { + "id": "rust/publish-ios", + "language": "rust", + "pattern": "publish-ios", + "status": "passed" + } ] }, - "behavior": { - "methodology_version": "0.3", - "generated_at": "2026-08-01T13:59:25.489Z", - "model": "gpt-5.6-terra", - "fixture_count": 4, - "repeats": 3, - "paid_model_runs": 36, - "arms": { - "original": {"oracle_passes":11,"samples":12,"detections":5,"positives":6,"clean_passes":6,"clean_controls":6,"median_input_tokens":235477.5}, - "compressed": {"oracle_passes":12,"samples":12,"detections":6,"positives":6,"clean_passes":6,"clean_controls":6,"median_input_tokens":71905.5}, - "yield": {"oracle_passes":12,"samples":12,"detections":6,"positives":6,"clean_passes":6,"clean_controls":6,"median_input_tokens":14027.5,"deterministic_replays":12} - }, - "artifact": { - "status": "unpublished", - "reason": "The precursor run completed locally; its raw bundle has not been promoted to immutable public storage." - } + "runtime_invariants": { + "passed": 5, + "total": 5, + "cases": [ + { + "id": "resume-complete", + "test": "TestEndToEndRunResumeComplete", + "assertion": "a recorded response advances the run to completion", + "status": "passed" + }, + { + "id": "deterministic-replay", + "test": "TestReplayIsDeterministic", + "assertion": "the saved log returns to the same next step", + "status": "passed" + }, + { + "id": "replay-divergence", + "test": "TestReplayDivergenceFailsLoudly", + "assertion": "changed behavior stops replay instead of reusing the wrong result", + "status": "passed" + }, + { + "id": "requirement-block", + "test": "TestFailedRequirementBlocksRun", + "assertion": "a failed rule ends the run as blocked", + "status": "passed" + }, + { + "id": "source-change", + "test": "TestDigestMismatchRefusedThenMigrates", + "assertion": "changed source is refused until the user accepts the change", + "status": "passed" + } + ] + }, + "claim_boundary": { + "summary": "Yield executes the tested workflows and passes the tested runtime checks.", + "compares_tools": false, + "tests_agent_judgment": false, + "exclusions": [ + "No claim that Yield is better than prompts or another tool.", + "Fixed test data replaces agent and human judgment.", + "Illustrative commands do not establish production safety for a real project." + ] } } diff --git a/evals/scripts/measure-source.mjs b/evals/scripts/measure-source.mjs deleted file mode 100644 index 273c0cf..0000000 --- a/evals/scripts/measure-source.mjs +++ /dev/null @@ -1,59 +0,0 @@ -import { createHash } from "node:crypto" -import { mkdir, readFile, writeFile } from "node:fs/promises" -import { dirname, join, resolve } from "node:path" -import { fileURLToPath } from "node:url" -import { countTokens } from "gpt-tokenizer/encoding/cl100k_base" - -const root = resolve(dirname(fileURLToPath(import.meta.url)), "..") -const index = JSON.parse(await readFile(join(root, "cases/index.json"), "utf8")) -const measure = (text) => ({ - bytes: Buffer.byteLength(text), - lines: text === "" ? 0 : text.split(/\r?\n/).length, - tokens: countTokens(text), -}) -const sha256 = (text) => createHash("sha256").update(text).digest("hex") -const round = (value) => Math.round(value * 10) / 10 -const rows = [] - -for (const item of index.cases) { - const rawUrl = `https://raw.githubusercontent.com/${item.source.repo}/${item.source.commit}/${item.source.path}` - const response = await fetch(rawUrl, { headers: { "user-agent": "yield-evals/0.1" } }) - if (!response.ok) throw new Error(`${item.id}: source fetch failed (${response.status})`) - const originalText = await response.text() - if (sha256(originalText) !== item.source.sha256) throw new Error(`${item.id}: pinned source digest changed`) - - const skillText = await readFile(join(root, "cases", item.thin_skill), "utf8") - const workflowText = await readFile(join(root, "cases", item.workflow), "utf8") - const original = measure(originalText) - const prompt = measure(skillText) - const workflow = measure(workflowText) - rows.push({ - id: item.id, - label: item.label, - source_url: `https://github.com/${item.source.repo}/blob/${item.source.commit}/${item.source.path}`, - original_tokens: original.tokens, - prompt_tokens: prompt.tokens, - workflow_tokens: workflow.tokens, - maintained_tokens: prompt.tokens + workflow.tokens, - maintained_change_pct: round(((prompt.tokens + workflow.tokens) / original.tokens - 1) * 100), - }) -} - -const sum = (field) => rows.reduce((total, row) => total + row[field], 0) -const output = { - schema_version: 1, - methodology_version: index.methodology_version, - generated_at: new Date().toISOString(), - tokenizer: "gpt-tokenizer 3.4.0 / cl100k_base", - summary: { - original_tokens: sum("original_tokens"), - prompt_tokens: sum("prompt_tokens"), - workflow_tokens: sum("workflow_tokens"), - maintained_tokens: sum("maintained_tokens"), - }, - rows, -} -const stamp = output.generated_at.replaceAll(":", "-") -await mkdir(join(root, "runs"), { recursive: true }) -await writeFile(join(root, "runs", `source-size-${stamp}.json`), JSON.stringify(output, null, 2) + "\n") -console.log(JSON.stringify(output.summary, null, 2)) diff --git a/evals/scripts/run.mjs b/evals/scripts/run.mjs new file mode 100644 index 0000000..9ad0f74 --- /dev/null +++ b/evals/scripts/run.mjs @@ -0,0 +1,140 @@ +import { createHash } from "node:crypto" +import { spawnSync } from "node:child_process" +import { mkdtemp, readdir, readFile, rm, stat, writeFile } from "node:fs/promises" +import { tmpdir } from "node:os" +import { dirname, join, relative, resolve } from "node:path" +import { fileURLToPath } from "node:url" + +const evalRoot = resolve(dirname(fileURLToPath(import.meta.url)), "..") +const yieldRoot = resolve(evalRoot, "..") +const libraryRoot = join(yieldRoot, "examples/library") +const languages = ["typescript", "python", "go", "rust"] +const runtimeCases = [ + ["resume-complete", "TestEndToEndRunResumeComplete", "a recorded response advances the run to completion"], + ["deterministic-replay", "TestReplayIsDeterministic", "the saved log returns to the same next step"], + ["replay-divergence", "TestReplayDivergenceFailsLoudly", "changed behavior stops replay instead of reusing the wrong result"], + ["requirement-block", "TestFailedRequirementBlocksRun", "a failed rule ends the run as blocked"], + ["source-change", "TestDigestMismatchRefusedThenMigrates", "changed source is refused until the user accepts the change"], +] +const excludedDirectories = new Set([ + ".git", ".yield", "node_modules", "runs", "raw", "artifacts", + "target", "build", "dist", "__pycache__", +]) +const digestRoots = ["cmd/yskill", "internal", "sdk", "examples/library", "evals/scripts", "evals/package.json"] + +function execute(command, args, cwd = yieldRoot) { + const result = spawnSync(command, args, { cwd, encoding: "utf8", env: process.env }) + if (result.status !== 0) { + const detail = [result.stdout, result.stderr].filter(Boolean).join("\n").trim() + throw new Error(`${command} ${args.join(" ")} failed${detail ? `:\n${detail}` : ""}`) + } + return result.stdout.trim() +} + +async function filesUnder(path) { + const info = await stat(path) + if (info.isFile()) return [path] + const files = [] + for (const entry of await readdir(path, { withFileTypes: true })) { + if (entry.isDirectory() && excludedDirectories.has(entry.name)) continue + files.push(...await filesUnder(join(path, entry.name))) + } + return files +} + +async function sourceDigest() { + const files = [] + for (const root of digestRoots) files.push(...await filesUnder(join(yieldRoot, root))) + files.sort() + const hash = createHash("sha256") + for (const path of files) { + hash.update(relative(yieldRoot, path)) + hash.update("\0") + hash.update(await readFile(path)) + hash.update("\0") + } + return hash.digest("hex") +} + +async function workflowCases(yskill) { + const catalog = JSON.parse(await readFile(join(libraryRoot, "catalog.json"), "utf8")) + const cases = [] + for (const language of languages) { + const languageRoot = join(libraryRoot, language) + for (const pattern of catalog) { + const skill = join(languageRoot, pattern.slug) + const output = execute(yskill, ["test", skill]) + if (!/reached completed$/.test(output)) throw new Error(`${language}/${pattern.slug}: missing completed result`) + cases.push({ id: `${language}/${pattern.slug}`, language, pattern: pattern.slug, status: "passed" }) + await rm(join(skill, ".yield"), { recursive: true, force: true }) + } + } + return { catalog, cases } +} + +function evaluateRuntime() { + return runtimeCases.map(([id, test, assertion]) => { + execute("go", ["test", "./internal/engine", "-run", `^${test}$`, "-count=1"]) + return { id, test, assertion, status: "passed" } + }) +} + +async function evaluate() { + const temporary = await mkdtemp(join(tmpdir(), "yield-evals-")) + try { + const yskill = join(temporary, "yskill") + execute("go", ["build", "-o", yskill, "./cmd/yskill"]) + const { catalog, cases } = await workflowCases(yskill) + const invariants = evaluateRuntime() + return { + schema_version: 2, + methodology_version: "1.0", + generated_at: new Date().toISOString(), + source_digest: await sourceDigest(), + status: "passed", + workflow_conformance: { + passed: cases.length, + total: cases.length, + patterns: catalog.length, + languages, + cases, + }, + runtime_invariants: { + passed: invariants.length, + total: invariants.length, + cases: invariants, + }, + claim_boundary: { + summary: "Yield executes the tested workflows and passes the tested runtime checks.", + compares_tools: false, + tests_agent_judgment: false, + exclusions: [ + "No claim that Yield is better than prompts or another tool.", + "Fixed test data replaces agent and human judgment.", + "Illustrative commands do not establish production safety for a real project.", + ], + }, + } + } finally { + await rm(temporary, { recursive: true, force: true }) + } +} + +const result = await evaluate() +const latestPath = join(evalRoot, "results/latest.json") +if (process.argv.includes("--write")) { + await writeFile(latestPath, JSON.stringify(result, null, 2) + "\n") + console.log(`wrote ${relative(process.cwd(), latestPath)}`) +} else if (process.argv.includes("--check")) { + const published = JSON.parse(await readFile(latestPath, "utf8")) + for (const field of ["schema_version", "methodology_version", "source_digest", "status"]) { + if (published[field] !== result[field]) throw new Error(`published ${field} is stale`) + } + for (const field of ["workflow_conformance", "runtime_invariants", "claim_boundary"]) { + if (JSON.stringify(published[field]) !== JSON.stringify(result[field])) throw new Error(`published ${field} is stale`) + } + console.log(`passed ${result.workflow_conformance.passed}/${result.workflow_conformance.total} workflow tests`) + console.log(`passed ${result.runtime_invariants.passed}/${result.runtime_invariants.total} runtime checks`) +} else { + console.log(JSON.stringify(result, null, 2)) +} diff --git a/evals/scripts/validate.mjs b/evals/scripts/validate.mjs index 20ba098..2ff0a75 100644 --- a/evals/scripts/validate.mjs +++ b/evals/scripts/validate.mjs @@ -1,86 +1,32 @@ -import { readdir, readFile, stat } from "node:fs/promises" +import { readFile } from "node:fs/promises" import { dirname, join, resolve } from "node:path" import { fileURLToPath } from "node:url" -import { countTokens } from "gpt-tokenizer/encoding/cl100k_base" const root = resolve(dirname(fileURLToPath(import.meta.url)), "..") -const readJson = async (path) => JSON.parse(await readFile(join(root, path), "utf8")) +const result = JSON.parse(await readFile(join(root, "results/latest.json"), "utf8")) const fail = (message) => { throw new Error(message) } -const isSha256 = (value) => /^[0-9a-f]{64}$/.test(value) -const isCommit = (value) => /^[0-9a-f]{40}$/.test(value) -const round = (value) => Math.round(value * 10) / 10 -const index = await readJson("cases/index.json") -const result = await readJson("results/latest.json") -if (index.schema_version !== 1) fail("unsupported case schema") -if (result.schema_version !== 1) fail("unsupported result schema") +if (result.schema_version !== 2) fail("unsupported result schema") +if (result.methodology_version !== "1.0") fail("unsupported methodology") +if (!/^[0-9a-f]{64}$/.test(result.source_digest)) fail("source hash is invalid") +if (result.status !== "passed") fail("published result is not passing") -const ids = new Set() -const localMeasurements = new Map() -for (const item of index.cases) { - if (!item.id || ids.has(item.id)) fail(`duplicate or empty case id: ${item.id}`) - ids.add(item.id) - if (!isCommit(item.source.commit)) fail(`${item.id}: source commit must be a full SHA`) - if (!isSha256(item.source.sha256)) fail(`${item.id}: source sha256 is invalid`) - const skill = await readFile(join(root, "cases", item.thin_skill), "utf8") - const workflow = await readFile(join(root, "cases", item.workflow), "utf8") - const readme = await readFile(join(root, "cases", item.id, "README.md"), "utf8") - if (!skill.trim() || !workflow.trim()) fail(`${item.id}: conversion source is empty`) - for (const required of [item.source.repo, "convert-skill", "](SKILL.md)", "](workflow.ts)"]) { - if (!readme.includes(required)) fail(`${item.id}: README is missing ${required}`) - } - localMeasurements.set(item.id, { - prompt_tokens: countTokens(skill), - workflow_tokens: countTokens(workflow), - }) +const workflows = result.workflow_conformance +if (workflows.patterns !== 10 || workflows.languages.length !== 4) { + fail("workflow matrix is incomplete") } - -const rows = result.source_size.rows -if (rows.length !== ids.size) fail("result row count does not match cases") -for (const row of rows) { - if (!ids.has(row.id)) fail(`result references unknown case: ${row.id}`) - const local = localMeasurements.get(row.id) - if (row.prompt_tokens !== local.prompt_tokens || row.workflow_tokens !== local.workflow_tokens) { - fail(`${row.id}: published source-size row does not match the committed conversion`) - } - if (row.maintained_tokens !== row.prompt_tokens + row.workflow_tokens) { - fail(`${row.id}: maintained token total is inconsistent`) - } - const change = round((row.maintained_tokens / row.original_tokens - 1) * 100) - if (change !== row.maintained_change_pct) fail(`${row.id}: maintained change is inconsistent`) +if (workflows.total !== 40 || workflows.passed !== workflows.total) { + fail("not every workflow test passed") } +if (workflows.cases.length !== workflows.total) fail("workflow case list is incomplete") -const sum = (field) => rows.reduce((total, row) => total + row[field], 0) -const totals = result.source_size.summary -for (const [field, rowField] of [ - ["original_tokens", "original_tokens"], - ["prompt_tokens", "prompt_tokens"], - ["workflow_tokens", "workflow_tokens"], - ["maintained_tokens", "maintained_tokens"], -]) { - if (totals[field] !== sum(rowField)) fail(`summary ${field} is inconsistent`) +const runtime = result.runtime_invariants +if (runtime.total !== 5 || runtime.passed !== runtime.total) { + fail("not every runtime check passed") } -const maintainedReduction = round((1 - totals.maintained_tokens / totals.original_tokens) * 100) -if (maintainedReduction !== totals.maintained_reduction_pct) fail("summary maintained reduction is inconsistent") +if (runtime.cases.length !== runtime.total) fail("runtime case list is incomplete") -const artifact = result.behavior.artifact -if (!["published", "unpublished"].includes(artifact.status)) fail("invalid artifact status") -if (artifact.status === "published" && (!artifact.uri || !isSha256(artifact.sha256))) { - fail("published behavior result requires an artifact URI and SHA-256") -} +if (result.claim_boundary.compares_tools !== false) fail("result must not compare tools") +if (result.claim_boundary.tests_agent_judgment !== false) fail("result must isolate agent judgment") -const forbidden = new Set(["runs", "raw", "artifacts", ".worktrees"]) -const walk = async (directory) => { - for (const entry of await readdir(directory, { withFileTypes: true })) { - if (entry.name === "node_modules") continue - const path = join(directory, entry.name) - if (entry.isDirectory()) { - if (directory !== root && forbidden.has(entry.name)) fail(`raw artifact directory is committed: ${path}`) - await walk(path) - } else if ((await stat(path)).size > 262144) { - fail(`evaluation source file exceeds 256 KiB: ${path}`) - } - } -} -await walk(root) -console.log(`validated ${ids.size} conversion cases and ${rows.length} result rows`) +console.log(`validated ${workflows.passed} workflow tests and ${runtime.passed} runtime checks`) diff --git a/release-notes/2026-08-01-first-party-evaluations.md b/release-notes/2026-08-01-first-party-evaluations.md new file mode 100644 index 0000000..665327a --- /dev/null +++ b/release-notes/2026-08-01-first-party-evaluations.md @@ -0,0 +1,11 @@ +# First-party conformance evaluations + +- Replace third-party conversion comparisons with 40 owned workflow fixture + runs across TypeScript, Python, Go, and Rust. +- Add five runtime-invariant checks for resume, replay, divergence, blocking, + and explicit source migration. +- Publish a source-digested result with an explicit claim boundary: the suite + verifies Yield's coding and supervision layer, not agent judgment or product + superiority. +- Rerun the same suite against the source tree and projected public surface in + CI.