Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 20 additions & 5 deletions UPSTREAM.json
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,8 @@
".gitignore": "803c5f79d6da7f2c5a1dc0ce27c53b8e5c059309b165782831c4c472af058a4c",
"LICENSE": "fff261ce507eabd57666c283a621f33e183a3aedebda04c4ecbc6309a62f5edf",
"README.md": "acd9f37f0f7013a4bf99df382fd3b5c14d070914ec958e3198d9c70aa8317ce6",
"cmd/yskill/main.go": "a88bdd133118aa7b86cdf021e67052bd8e64e63b3792c091f7612699af89648b",
"cmd/yskill/main.go": "37a1004022d3b4c6db4b0c0e31bebcf6ec15e0737975762e3cb5290d96c9c231",
"cmd/yskill/main_test.go": "9e0e5b157a8e34d56bada07f6af49e0cd24f8e310777f27666dd9f7e2c20a942",
"docs/README.md": "cd8a4105f4f04143172661b39789e3c533d031ad46437aa2c078625a29a0a9db",
"docs/convert-existing-skill.md": "0dd538e908a1f2d3a1958e4b2cc1efede2a8e7a73341e32c1a75fec69a9acb8a",
"docs/examples.md": "3c19bacae7ec7b31bf93417e61d01cd1872f26228fedf0ed6decfce7c41ed274",
Expand Down Expand Up @@ -40,11 +41,24 @@
"docs/tutorials/data-migration.md": "6699fc47cda84c4f46a3703ca4a69bcced7cc4a9df339a308e807ea3a99578bd",
"docs/tutorials/environment-repair.md": "289c9b261e3768c2c58d60d9ef7a09e85f89038438a08febb4b9c12ebada88d1",
"evals/.gitignore": "0d5020173666118bafe31c857b96aa325809f41d159ca51324ceaf239e043347",
"evals/README.md": "7b0a92908d4eff84f1716756a00536ed79b31ef542928a2fd0c17c234161bba6",
"evals/README.md": "db8461f9a2cc17cde6597149a16fd68b4450985b4f4b40b097b1e7059d2cbe34",
"evals/agent/README.md": "7936e06915bfc1ae755a9529d4a6c632a8d866b9c804e95af49a66cad6eb5038",
"evals/agent/cases.json": "98ae0ab3f099bd957739368de7bdb5b8b43997776ed6a11e406f81fd523c3130",
"evals/agent/fixtures/long/SKILL.md": "8826f89f40c17085c92d246043e3794565a230d578cc07de3b5f3c26ba661efe",
"evals/agent/fixtures/shared/.gitignore": "9a6da4f2c69a82d04438e358640bc5784d70b9e5dff9c6fdaaad62c38aa8b667",
"evals/agent/fixtures/shared/bin/record.mjs": "7def5e42bf55184c90cc8083473d283ba56bc4b1ce63ca7bc8a706862d8773a1",
"evals/agent/fixtures/shared/bin/step.mjs": "623824a2f06f9032af21a6f303e46904316e35e5630b662443d54f202fa16b30",
"evals/agent/fixtures/shared/bin/user-answer.mjs": "1e5e4d7c8690d8fffaa95081fe408cf4c9cf566573609de51f2b7048310cf7cc",
"evals/agent/fixtures/yield/SKILL.md": "71dc416eee6d23c01904694b765dc76eeaedb948ea41757044473bb9d652f78f",
"evals/agent/fixtures/yield/skills/release/main.ts": "197fa885b7e680f4d859fd091256d098799ee7a6ce2b8b323f6f24ee4e37b603",
"evals/agent/fixtures/yield/skills/release/skill.json": "5dd8e8c7532fc77aa07adc42ee212fbb48ed8ef179d1e47569eb07c9d6eb1773",
"evals/agent/scripts/run.mjs": "7df7429c897d0d1696326a0ae52e319b67926d9ccbbd40e346ba605046ac8d9a",
"evals/agent/scripts/validate.mjs": "4f31fb478919f1df400ad32f93ad4a0f9dd9a9f8e19139cbcf6aeb966133be67",
"evals/package-lock.json": "cfbd68d590e92b94233a807fd774e6667e9474096ab76e1c5cd037acfb4e3200",
"evals/package.json": "aa97f75cadbf6d1802dcf03f4b34dcbbc4aea8c2b5ef18130af26a9e63ecac51",
"evals/package.json": "84ac7bd7d1c0d1db2233dab54754296195b267e5e907d9f6c402d0d44d569adb",
"evals/results/README.md": "65b74bfab83dfc51fc5b8a3a4824c37b722fb3a358a5dcfb217dc2af199f4801",
"evals/results/latest.json": "cc9346c0db8e8395464a7348ea440a891fad0080350ac5d677b1a2875047e4d4",
"evals/results/latest-agent.json": "1484174819ae29bbcffeb615036fa165518669525e6d664123f6a38ca3614b69",
"evals/results/latest.json": "ba580e2f1d29ebfcb1f70700ed4148906686e9ded7350be0552b0fc30a91ff66",
"evals/scripts/run.mjs": "119248cd4d383326d396f4459df28c51ce7c31088d97639851f2f5ab4c329892",
"evals/scripts/validate.mjs": "7ac2a3239cd1b52a604c8a8a52a1af4ac55c4007ac2751d4fbfe20d0e1f568dc",
"examples/convert-skill/SKILL.md": "e6376f34365d4ac030d316db55e91f0c606a501668099f4ecf1e27d43ec806a2",
Expand Down Expand Up @@ -262,6 +276,7 @@
"ir/yield.v1/program-output.schema.json": "6b80f6a267cfeea3429cd95485195df037176726ff999e0d94bd20e07600051c",
"ir/yield.v1/request-envelope.schema.json": "2d5f34b04638450f1bd87ec1305ba7ecf0de2ec0ef6948365e6f46c2f2ee1c35",
"ir/yield.v1/response-envelope.schema.json": "698fc20510bf1362cac17f332b8ec4b4dd336d2bee4294b1949ed3257b535639",
"release-notes/2026-08-01-agent-workflow-evaluation.md": "61fe8b5f893d21dac9dba9fa759818637c81ce2cd844571274e50f763ac72d00",
"release-notes/2026-08-01-docs-and-typescript-package.md": "93382375cb47187092aff9447b280853151c738b140879828c74576a109195b8",
"release-notes/2026-08-01-evaluation-case-guides.md": "570bc996eb4d2892456a02d938fb6299d106c431bb842ce726db1df550f779d7",
"release-notes/2026-08-01-evaluation-surface.md": "64fb5e2fdad3ccd41967e028cf4675c45939174000281c443f4b28d96dc07549",
Expand All @@ -283,7 +298,7 @@
"generator": "operatorstack/yield:project",
"schema_version": 1,
"source": {
"commit": "4c338abc0d14f017eb3501885adca7594bd5b6f4",
"commit": "7b82c7095893931a2cef22e9a076dec16b4ba0fc",
"path": "labs/22-yield",
"repository": "operatorstack/intelligence-flow"
}
Expand Down
22 changes: 17 additions & 5 deletions cmd/yskill/main.go
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@ import (
"fmt"
"os"
"path/filepath"
"strings"

"github.com/operatorstack/yield/internal/engine"
"github.com/operatorstack/yield/internal/protocol"
Expand Down Expand Up @@ -60,7 +61,7 @@ func main() {
func cmdRun(args []string) error {
fs := flag.NewFlagSet("run", flag.ExitOnError)
input := fs.String("input", "", "path to a JSON input file")
if err := fs.Parse(args); err != nil {
if err := parseOnePositional(fs, args); err != nil {
return err
}
if fs.NArg() != 1 {
Expand Down Expand Up @@ -90,7 +91,7 @@ func cmdResume(args []string) error {
response := fs.String("response", "", "path to the response envelope JSON")
skillDir := fs.String("skill", ".", "skill directory the run belongs to")
migrate := fs.Bool("accept-new-digest", false, "explicitly rebind the run to the current skill source digest")
if err := fs.Parse(args); err != nil {
if err := parseOnePositional(fs, args); err != nil {
return err
}
if fs.NArg() != 1 || *response == "" {
Expand All @@ -114,7 +115,7 @@ func cmdResume(args []string) error {
func cmdInspect(args []string) error {
fs := flag.NewFlagSet("inspect", flag.ExitOnError)
skillDir := fs.String("skill", ".", "skill directory the run belongs to")
if err := fs.Parse(args); err != nil {
if err := parseOnePositional(fs, args); err != nil {
return err
}
e, err := engine.New(*skillDir)
Expand Down Expand Up @@ -144,7 +145,7 @@ func cmdInspect(args []string) error {
func cmdReplay(args []string) error {
fs := flag.NewFlagSet("replay", flag.ExitOnError)
skillDir := fs.String("skill", ".", "skill directory the run belongs to")
if err := fs.Parse(args); err != nil {
if err := parseOnePositional(fs, args); err != nil {
return err
}
if fs.NArg() != 1 {
Expand All @@ -167,7 +168,7 @@ func cmdReplay(args []string) error {
// real; everything else is answered from the script.
func cmdTest(args []string) error {
fs := flag.NewFlagSet("test", flag.ExitOnError)
if err := fs.Parse(args); err != nil {
if err := parseOnePositional(fs, args); err != nil {
return err
}
if fs.NArg() != 1 {
Expand Down Expand Up @@ -213,6 +214,17 @@ func cmdTest(args []string) error {
return nil
}

// parseOnePositional accepts the documented command shape where the target
// comes first and flags follow it. The standard flag package stops parsing at
// the first positional argument, so move that one target behind the flags.
// Flag-first calls keep working unchanged.
func parseOnePositional(fs *flag.FlagSet, args []string) error {
if len(args) > 0 && !strings.HasPrefix(args[0], "-") {
args = append(append([]string{}, args[1:]...), args[0])
}
return fs.Parse(args)
}

func printProgress(p *engine.Progress) error {
if p.Terminal != nil {
fmt.Printf("run %s: %s\n", p.RunID, p.Terminal.Status)
Expand Down
32 changes: 32 additions & 0 deletions cmd/yskill/main_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,32 @@
package main

import (
"flag"
"testing"
)

func TestParseOnePositionalAllowsDocumentedFlagOrder(t *testing.T) {
fs := flag.NewFlagSet("resume", flag.ContinueOnError)
response := fs.String("response", "", "response file")
skill := fs.String("skill", ".", "skill directory")
if err := parseOnePositional(fs, []string{"run_123", "--response", "response.json", "--skill", "skills/release"}); err != nil {
t.Fatal(err)
}
if fs.NArg() != 1 || fs.Arg(0) != "run_123" {
t.Fatalf("positional args = %q, want run_123", fs.Args())
}
if *response != "response.json" || *skill != "skills/release" {
t.Fatalf("flags = response %q skill %q", *response, *skill)
}
}

func TestParseOnePositionalKeepsFlagFirstOrder(t *testing.T) {
fs := flag.NewFlagSet("resume", flag.ContinueOnError)
response := fs.String("response", "", "response file")
if err := parseOnePositional(fs, []string{"--response", "response.json", "run_123"}); err != nil {
t.Fatal(err)
}
if fs.NArg() != 1 || fs.Arg(0) != "run_123" || *response != "response.json" {
t.Fatalf("args = %q response = %q", fs.Args(), *response)
}
}
18 changes: 15 additions & 3 deletions evals/README.md
Original file line number Diff line number Diff line change
@@ -1,9 +1,9 @@
# Yield evaluations

These evaluations test Yield itself. They do not compare Yield with another
tool, company, prompt, or skill.
tool or company.

The suite answers two questions:
The deterministic suite answers two questions:

1. Can each checked-in example workflow reach its expected final result
through every supported SDK?
Expand Down Expand Up @@ -47,9 +47,21 @@ A passing result proves that the tested Yield revision:

This suite does not prove that Yield is better than prose, that an agent's
judgment is correct, or that illustrative commands are production-safe. The
Fixed test data supplies agent and human responses so the suite can test only
fixed test data supplies agent and human responses so the suite can test only
the code-controlled workflow layer.

`results/latest.json` is a compact, website-safe result. Its source hash is
computed from the CLI, engine, protocol, SDKs, example workflows, fixtures, and
evaluation harness. CI reruns the suite instead of trusting that file alone.

## Coding-agent workflow check

The separate `agent/` suite runs the same owned workflow through a real coding
agent in two forms: a long skill, and a thin skill backed by Yield code. It
checks matching step order, gates, responses, and final status. It does not
score the agent's domain judgment or claim that one form is better.

```bash
npm run eval:agent
npm run test:agent
```
46 changes: 46 additions & 0 deletions evals/agent/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
# Coding-agent workflow equivalence

This suite checks one Yield product claim with a real coding agent:

> Moving workflow control from a long skill into Yield code preserves the
> tested step order and final result.

It does not score general coding ability. The release data, command outcomes,
and user answers are deliberately simple. The same Codex model runs two owned
representations:

1. a long `SKILL.md` containing every step and branch;
2. a thin `SKILL.md` that follows the same workflow in Yield code.

Six cases cover the important branches: failed tests, failed review rule, user
refusal, failed publish, failed verification, and successful completion.

## Evidence

The long-skill arm records actual command calls and structured step results.
The Yield arm is scored from its append-only run log. The scorer verifies:

- the same ordered step IDs;
- the same final status;
- real command evidence for every command step;
- accepted `agent_task` and `ask_user` responses in the Yield log;
- matching requirement results and terminal events;
- no rejected response envelopes.

Raw Codex transcripts and run logs stay under ignored `evals/runs/`. The compact
result in `results/latest-agent.json` contains counts, normalized traces, model
identity, CLI version, token usage, timings, and a source hash covering the
harness, fixtures, CLI, engine, protocol, and TypeScript SDK.

## Run

Use the current Codex login:

```bash
cd evals
npm run eval:agent
npm run test:agent
```

CI can instead provide `CODEX_API_KEY`. Set `EVAL_AGENT_MODEL` and
`EVAL_AGENT_REASONING` to make a different model configuration explicit.
68 changes: 68 additions & 0 deletions evals/agent/cases.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,68 @@
[
{
"id": "preflight-blocks",
"commands": {
"test-package": { "exit_code": 1, "stdout": "", "stderr": "tests failed\n" },
"publish-package": { "exit_code": 0, "stdout": "published\n", "stderr": "" },
"verify-package": { "exit_code": 0, "stdout": "verified\n", "stderr": "" }
},
"review": { "status": "pass", "critical": 0, "summary": "Release metadata is complete." },
"answers": { "approve-publish": "continue" },
"expected": { "steps": ["test-package"], "terminal": "blocked" }
},
{
"id": "review-blocks",
"commands": {
"test-package": { "exit_code": 0, "stdout": "tests passed\n", "stderr": "" },
"publish-package": { "exit_code": 0, "stdout": "published\n", "stderr": "" },
"verify-package": { "exit_code": 0, "stdout": "verified\n", "stderr": "" }
},
"review": { "status": "needs_work", "critical": 1, "summary": "The release notes omit a breaking API change." },
"answers": { "approve-publish": "continue" },
"expected": { "steps": ["test-package", "review-release"], "terminal": "blocked" }
},
{
"id": "approval-refuses",
"commands": {
"test-package": { "exit_code": 0, "stdout": "tests passed\n", "stderr": "" },
"publish-package": { "exit_code": 0, "stdout": "published\n", "stderr": "" },
"verify-package": { "exit_code": 0, "stdout": "verified\n", "stderr": "" }
},
"review": { "status": "pass", "critical": 0, "summary": "Release metadata is complete." },
"answers": { "approve-publish": "stop" },
"expected": { "steps": ["test-package", "review-release", "approve-publish"], "terminal": "refused" }
},
{
"id": "publish-blocks",
"commands": {
"test-package": { "exit_code": 0, "stdout": "tests passed\n", "stderr": "" },
"publish-package": { "exit_code": 1, "stdout": "", "stderr": "registry unavailable\n" },
"verify-package": { "exit_code": 0, "stdout": "verified\n", "stderr": "" }
},
"review": { "status": "pass", "critical": 0, "summary": "Release metadata is complete." },
"answers": { "approve-publish": "continue" },
"expected": { "steps": ["test-package", "review-release", "approve-publish", "publish-package"], "terminal": "blocked" }
},
{
"id": "verification-blocks",
"commands": {
"test-package": { "exit_code": 0, "stdout": "tests passed\n", "stderr": "" },
"publish-package": { "exit_code": 0, "stdout": "published\n", "stderr": "" },
"verify-package": { "exit_code": 1, "stdout": "", "stderr": "package not visible\n" }
},
"review": { "status": "pass", "critical": 0, "summary": "Release metadata is complete." },
"answers": { "approve-publish": "continue" },
"expected": { "steps": ["test-package", "review-release", "approve-publish", "publish-package", "verify-package"], "terminal": "blocked" }
},
{
"id": "success",
"commands": {
"test-package": { "exit_code": 0, "stdout": "tests passed\n", "stderr": "" },
"publish-package": { "exit_code": 0, "stdout": "published\n", "stderr": "" },
"verify-package": { "exit_code": 0, "stdout": "verified\n", "stderr": "" }
},
"review": { "status": "pass", "critical": 0, "summary": "Release metadata is complete." },
"answers": { "approve-publish": "continue" },
"expected": { "steps": ["test-package", "review-release", "approve-publish", "publish-package", "verify-package"], "terminal": "completed" }
}
]
34 changes: 34 additions & 0 deletions evals/agent/fixtures/long/SKILL.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
---
name: release-package-long
description: Test, review, approve, publish, and verify a package release.
---

Run this workflow in order. Do not skip a step. Do not continue after a failed
rule. The `node bin/record.mjs` calls are eval instrumentation and must run.

1. Run `node bin/step.mjs test-package`.
2. Record whether the command passed:
`node bin/record.mjs requirement package-tests '{"passed":true}'` or use
`false`. If it failed, run
`node bin/record.mjs terminal blocked '{"reason":"the package tests pass"}'`
and stop.
3. Read `.eval/release.json`. Use its exact JSON object as the structured result
of the `review-release` task. Record it with
`node bin/record.mjs agent_task review-release '<json>'`.
4. The review passes only when `status` is `pass` and `critical` is `0`. Record
that rule with `node bin/record.mjs requirement review-ready
'{"passed":true}'` or use `false`. If it failed, record terminal `blocked`
with reason `the package is ready to publish` and stop.
5. Get the user's answer by running
`node bin/user-answer.mjs approve-publish`. The helper records this user
step. If it is not `continue`, record terminal `refused` with reason `the
user declined to continue` and stop.
6. Run `node bin/step.mjs publish-package`. Record requirement `publish-passed`.
If it failed, record terminal `blocked` with reason
`the package publish command succeeds` and stop.
7. Run `node bin/step.mjs verify-package`. Record requirement `verify-passed`.
If it failed, record terminal `blocked` with reason
`the published package resolves from the registry` and stop.
8. Record terminal `completed` with the review summary.

Every recorder result must be valid JSON. End with a short status report.
3 changes: 3 additions & 0 deletions evals/agent/fixtures/shared/.gitignore
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
.eval/observed.jsonl
.eval/codex-last.json
skills/release/.yield/
10 changes: 10 additions & 0 deletions evals/agent/fixtures/shared/bin/record.mjs
Original file line number Diff line number Diff line change
@@ -0,0 +1,10 @@
#!/usr/bin/env node
import { appendFile } from "node:fs/promises"
import { dirname, join, resolve } from "node:path"
import { fileURLToPath } from "node:url"

const root = resolve(dirname(fileURLToPath(import.meta.url)), "..")
const [kind, id, raw = "null"] = process.argv.slice(2)
if (!["agent_task", "requirement", "terminal"].includes(kind)) throw new Error(`unsupported event kind: ${kind}`)
const event = { kind, id, result: JSON.parse(raw) }
await appendFile(join(root, ".eval/observed.jsonl"), JSON.stringify(event) + "\n")
14 changes: 14 additions & 0 deletions evals/agent/fixtures/shared/bin/step.mjs
Original file line number Diff line number Diff line change
@@ -0,0 +1,14 @@
#!/usr/bin/env node
import { appendFile, readFile } from "node:fs/promises"
import { dirname, join, resolve } from "node:path"
import { fileURLToPath } from "node:url"

const root = resolve(dirname(fileURLToPath(import.meta.url)), "..")
const id = process.argv[2]
const testCase = JSON.parse(await readFile(join(root, ".eval/case.json"), "utf8"))
const command = testCase.commands[id]
if (!command) throw new Error(`unknown command step: ${id}`)
await appendFile(join(root, ".eval/observed.jsonl"), JSON.stringify({ kind: "run_command", id, exit_code: command.exit_code }) + "\n")
process.stdout.write(command.stdout)
process.stderr.write(command.stderr)
process.exitCode = command.exit_code
12 changes: 12 additions & 0 deletions evals/agent/fixtures/shared/bin/user-answer.mjs
Original file line number Diff line number Diff line change
@@ -0,0 +1,12 @@
#!/usr/bin/env node
import { appendFile, readFile } from "node:fs/promises"
import { dirname, join, resolve } from "node:path"
import { fileURLToPath } from "node:url"

const root = resolve(dirname(fileURLToPath(import.meta.url)), "..")
const id = process.argv[2]
const testCase = JSON.parse(await readFile(join(root, ".eval/case.json"), "utf8"))
const value = testCase.answers[id]
if (typeof value !== "string") throw new Error(`no user answer for: ${id}`)
await appendFile(join(root, ".eval/observed.jsonl"), JSON.stringify({ kind: "ask_user", id, result: { value } }) + "\n")
process.stdout.write(value + "\n")
Loading
Loading