diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index afd3c0e6..210d7f38 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -3,12 +3,12 @@ "owner": { "name": "Martin Patino" }, - "description": "pstack for Claude Code and Codex on the models you choose: subscription CLIs or API-key gateway lanes.", + "description": "Shared pstack skills with coding-agent harness adapters and configurable model lanes.", "plugins": [ { "name": "pstack", "source": "./plugins/pstack", - "description": "if you want to go fast, go deep first. pstack helps you write less, but higher quality code. rigorous agent workflows you can parallelize with confidence.", + "description": "Shared pstack skills with coding-agent harness adapters and configurable model lanes.", "version": "1.5.0", "author": { "name": "Lauren Tan (original)" diff --git a/AGENTS.md b/AGENTS.md index 89410e5b..d6234ca5 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -2,7 +2,7 @@ Track all durable work in this repository's GitHub Issues. Do not create a parallel Linear queue. Read `UPSTREAM.md` before changing upstream-derived content. -Cursor's `cursor/plugins/pstack` tree is the content upstream. Keep one shared skill tree for Claude Code and Codex; adapt harness primitives at the existing mapping boundaries instead of forking skills or adding compatibility layers. The parent harness resolves provider routing once. Children do not detect or reroute themselves. +Cursor's `cursor/plugins/pstack` tree is the content upstream. Keep one shared skill tree for Claude Code, Codex, and OpenCode; adapt harness primitives at the existing mapping boundaries instead of forking skills or adding compatibility layers. The parent harness resolves provider routing once. Children do not detect or reroute themselves. Before opening a pull request, run the Bun tests, strict typecheck, static invariants, and plugin validation: `bash scripts/check.sh` runs all of them. diff --git a/CHANGES.md b/CHANGES.md index edbb69f5..ec3725c9 100644 --- a/CHANGES.md +++ b/CHANGES.md @@ -1,465 +1,18 @@ -# CHANGES — applied substitutions +# Changelog -## Unreleased: OpenRouter gateway, any model +## Unreleased -- New gateway provider `openrouter` ([#7](https://github.com/thisguymartin/pstack-flex/issues/7)). One `OPENROUTER_API_KEY` reaches any model in OpenRouter's catalog through the stock `claude` binary, the same path DeepSeek and MiniMax use. There is no allowlist: the flex matrix carries one open `openrouter` row, each distinct model ID is its own family, and setup's live probe on the named model is the gate. The OpenRouter probe reads its marker from a file, so it proves a tool call. -- The runner refuses an OpenRouter ID without a namespace and OpenRouter's own `openrouter/*` routers, which pick the model server-side. OpenRouter reports must match the requested ID exactly apart from case; the other gateways keep their prefix rule. Only OpenRouter lanes get the empty `ANTHROPIC_API_KEY` its guide requires, so DeepSeek and MiniMax environments are unchanged. -- Panel diversity counts the lab behind the model. An OpenRouter lane counts as its model ID's namespace, and `anthropic`, `openai`, `x-ai`, `deepseek`, and `minimax` match the direct providers. -- Runner fixes found by the OpenRouter dry run, which apply to every lane: Claude Code 2.1.289's 401 result ("Failed to authenticate", `"api_error_status":401`) now classifies as `unauthenticated` instead of `child-failed`, and `usage.output_tokens_details.thinking_tokens` is recorded as `reasoningTokens`. -- `scripts/probe-openrouter.sh` runs the #7 route battery (V6 in `docs/LANES.md`). It spends real credit, so it is not part of `check.sh` or CI. -- Design and flow diagrams: [docs/plans/2026-10-05-openrouter-gateway.md](docs/plans/2026-10-05-openrouter-gateway.md). +pstack-flex is a separate distribution of open-pstack. The plugin remains `pstack`; install it as `pstack@pstack-flex`. -## Unreleased: project model sheets and GPT-6.1 Sol +- Configurable model families and defaults, with direct CLI and DeepSeek, MiniMax, and OpenRouter gateway lanes. Assigned models fail as named dropouts instead of falling back. +- Global or private project model sheets, with role assignments, requested efforts, and live probes before setup writes. +- Shared configuration parsing and harness metadata, plus `pstack-context` for inspecting the parent configuration. Tracked in [#3](https://github.com/thisguymartin/pstack-flex/issues/3), [#34](https://github.com/thisguymartin/pstack-flex/issues/34), and [#35](https://github.com/thisguymartin/pstack-flex/issues/35). +- Opt-in skill invocation in Claude Code and Codex. +- OpenCode as a parent and external lane provider, in beta. Installed OpenCode verification remains required by the [live gate](docs/LIVE-GATE.md); headless CI use is unsupported. +- Optional external lane journal for the separate [psf-monitor](https://github.com/thisguymartin/psf-monitor) plugin. +- `intake` and `diff-behavior` skills for issue briefs and observable branch comparisons. +- Documentation reduced to installation, current runtime contracts, provenance, and the live gate. -- `setup-pstack` asks on every run whether to configure the global sheet or a project sheet. A project sheet lives at `/.claude/pstack-models.md` (Claude Code) or `/.codex/pstack-models.md` (Codex), starts from the global assignments, and is listed in `.git/info/exclude` so it never reaches a commit. It needs no CLAUDE.md include and no AGENTS.md block. -- `provider-dispatch.md` gains a "Sheet scope" section: the parent reads the project path once before dispatch, an existing project sheet replaces the global sheet whole, and a project without one uses the global sheet. The two are never merged per role. -- Add the `sol-6.1` stock family, `codex:gpt-6.1-sol@high`, which Codex CLI 0.160.0 lists as its latest coding model. It takes the first-run `feature, refactoring`, `bug-fix`, `perf-issue`, and `hillclimb` roles. `sol-6` stays selectable; existing sheets keep their rows until setup reassigns them. -- `architect runners` gets its own first-run default, `codex:gpt-6-astra@high, claude:fable@max`, from a new "Default architect panel" line. The other three panel roles keep the four-lane default. -- The matrix test and the static invariant check cover the new row, the architect line, and the scope contract. +## Source release -## Unreleased: pstack-flex becomes its own distribution - -- The marketplace is now `pstack-flex` (was `open-pstack`) in both the Claude Code and Codex marketplace files. Install with `pstack@pstack-flex`. The plugin keeps the name `pstack`, so skill names such as `pstack:poteto-mode` are unchanged. Manifests, package names, and docs point at `thisguymartin/pstack-flex`; attribution to open-pstack, pstack-claude, and Cursor pstack stays in README and NOTICE. -- New skills: `intake` ([#10](https://github.com/thisguymartin/pstack-flex/issues/10)) turns GitHub issues into ready-to-run poteto-mode briefs with a playbook, an observable exit condition, a verification plan, and a worktree; it is read-only and parks briefs with open product questions. `diff-behavior` ([#15](https://github.com/thisguymartin/pstack-flex/issues/15)) runs the same scenarios on trunk and head through `swarm`, normalizes, and classifies every difference as intended, unintended, or noise. Neither changes an upstream skill body; wiring them into poteto-mode and the multi-phase-plan regression lane is a follow-up. -- Fixes: `codex-tools.md` now says the default panel runs four lanes across three providers (it said four providers). `docs/LANES.md` replaces the stale claim that OpenRouter needs a local translator with OpenRouter's documented Claude Code connection and the probes from [#7](https://github.com/thisguymartin/pstack-flex/issues/7). - -## Unreleased: lane journal - -- `pstack-runner` writes an opt-in lane journal (start record with the head of the prompt, stdout as it arrives, receipt copy) under `~/.pstack-flex/lanes/` while that directory exists. `--label` names a lane. A journal failure never changes a lane's receipt, exit status, or output. [psf-monitor](https://github.com/thisguymartin/psf-monitor) reads the journal to show lanes while they run; the agent monitor that first shipped here moved there. Tracked in [pstack-flex #23](https://github.com/thisguymartin/pstack-flex/issues/23). - -## Unreleased: merge open-pstack 1.5.0 (Cursor pstack 0.15.5) - -- Merge open-pstack 1.5.0, which syncs Cursor pstack 0.15.2 to 0.15.5: code-ready rounds and owner authority in the autopilot playbooks, Swarm SHA and method briefs, Architect reading `architect runners`, decision-trail `start` rows and the non-truncating `log.sh`, retired-role handling in setup, and the upstream prompt cuts. The upstream exclusions in `UPSTREAM.md` carry over. -- Take upstream's Grok and Opus matrix rows: Grok pins `grok-4.7`, and Opus defaults to `max`. The default panel becomes `claude:fable@max, codex:gpt-6-astra@high, grok:grok-4.7@xhigh, claude:opus@max`. Existing sheets keep their rows until setup reassigns them. -- Keep the fork's first-run roles. Upstream's three-model Opus, Sol, and Grok panel and its Grok and Opus solo-role defaults are not applied. GPT-6 Sol, Luna, Fable, and the four-lane panel stay as in the GPT-6 entry below. -- `setup-pstack` takes upstream's step layout: the role question moves to step 2, and step 4 only collects efforts. The flex families, GPT-6 guidance, and panel-diversity rule are unchanged. -- Not merged: upstream's `docs/plans/2026-10-01-triage-and-roadmap.md`, which plans open-pstack's own issue queue. - -## Unreleased: pstack runs only on request in Claude Code - -- The `SessionStart` hook no longer routes every non-trivial engineering task into `poteto-mode`. `hooks/session-start-context.md` is now an opt-in gate: Claude invokes a `pstack:*` skill only when the user types `/pstack:`, names pstack or one of its skills in the request, or keeps a standing CLAUDE.md or AGENTS.md instruction for it. A pstack skill the user started keeps routing to the skills it needs for that task, and dispatched subagents follow their dispatch prompt. -- The gate also covers skills whose own descriptions ask for automatic use, such as `unslop` and `typescript-best-practices`, without changing those upstream descriptions or adding `disable-model-invocation`, which would block `poteto-mode` from invoking them. -- Why: the mandate sent ordinary tasks through the full poteto-mode pipeline (playbook and principle reads, `how`, `architect` panels, external Codex delegation, `interrogate`), which made simple work slow on Claude Code. Codex runs the same hook (observed on Codex 0.157.1, which injected the old mandate as a developer message), so the gate applies there too. Users who want the old always-on routing add one standing instruction to CLAUDE.md. Tracked in [pstack-flex #18](https://github.com/thisguymartin/pstack-flex/issues/18). - -## Unreleased: GPT-6 Codex families become stock - -- Move `astra`, `sol-6`, and `luna` from the additional matrix into the stock model matrix. The first-run sheet now assigns GPT-6 Sol high to `feature, refactoring`, `bug-fix`, `perf-issue`, and `hillclimb`; Luna high to `how explorer` and `swarm workers`; and Astra high in place of GPT-5.6 Sol on every panel. `codex:gpt-5.6-sol` stays selectable; existing sheets are untouched until a role is reassigned in setup. -- `provider-dispatch.md` gains a "Default panel" section that is the single source for the four panel lanes. `setup-pstack`, `arena`, `architect`, and `interrogate` copy it; the static invariant reads that line instead of deriving the panel from every matrix row, and the solo-code invariant checks the `sol-6` row. -- The matrix test asserts seven stock rows, the GPT-6 descriptors in the first-run sheet, and the panel contract. `UPSTREAM-FLEX.md` records the new permanent conflict surface on upstream syncs. Tracked in [pstack-flex #17](https://github.com/thisguymartin/pstack-flex/issues/17). - -## Unreleased: multiple gateway model choices - -- Add DeepSeek V4 Pro and MiniMax M3.1 Flash Preview as independent setup families alongside existing Flash and M3 choices. Keep provider-owned routing and existing descriptors. -- Validate unique model families and provider/model pairs instead of requiring one row per provider; cover both parent routes and substituted-model rejection. -- Document preview Token Plan access, model-specific thinking semantics, and required installed live validation. Tracked in [pstack-flex #5](https://github.com/thisguymartin/pstack-flex/issues/5). - -This port applies the Cursor → Claude Code substitutions in skill bodies. Earlier drafts left them flagged; this revision resolves them. A later pass added a Codex build that shares the same skills; see [Codex port](#codex-port) below. - -## pstack-flex (unreleased) — gateway lanes and optional families - -Fork of open-pstack v1.4.1. Additive changes, all in port-owned files: - -- Runner: new gateway providers `deepseek` and `minimax` (`runner/flex-providers.ts`). Each spawns the stock `claude` binary with the exact claude argv, plus injected environment: the lab's Anthropic-compatible endpoint, `ANTHROPIC_AUTH_TOKEN` from `DEEPSEEK_API_KEY`/`MINIMAX_API_KEY`, model pins, and an isolated `CLAUDE_CONFIG_DIR` (`~/.pstack-flex/`). Inherited `ANTHROPIC_*` values are deleted before injection so a parent's credentials or endpoint never bleed into a gateway child. -- OAuth-leak guard: a gateway lane refuses to start (in-process, `unauthenticated` receipt, exit 77) when its API key variable is missing or when its config dir carries a claude.ai OAuth credentials file, so a claude.ai login can never be pointed at a third-party endpoint. -- Gateway preflight is `claude --version`; the one-shot invocation is the real auth test. Gateway receipts force `costUsd` to null (the CLI prices at Anthropic rates) and match served models case-insensitively, falling back to `modelEvidence: "pinned-argv"` like Codex. -- `provider-dispatch.md`: new additive "Flex model matrix" section, extended route table, gateway preflight semantics, and the panel-diversity rule (arena runners and interrogate reviewers span at least two providers unless the operator explicitly confirms otherwise). The stock model matrix is byte-unchanged. -- Optional GPT-6 families: the additional model matrix declares `codex:gpt-6-astra`, `codex:gpt-6-sol`, and `codex:gpt-6-luna` with default effort `high` and selectable `low`, `medium`, `high`, `xhigh`, and `max`. Setup can assign and probe each for `architect runners` or another configurable role. Codex parents use native `spawn_agent`; Claude Code parents use the external Codex runner. GPT-6 Sol has its own `sol-6` family. Stock models and first-run role assignments stay unchanged. -- `setup-pstack`: role assignments are selected first, and only assigned families get effort questions and probes; there is no requirement to assign every matrix family (mirrors upstream PR #73 / issue #72). The first-run sheet, its stock quad, and the fail-closed write rules are unchanged. -- Tests: the model-matrix contract gains a flex-matrix section check cross-validated against the runner's gateway specs; runner, commands, parse-output, and CLI tests cover env injection, the guard, cost nulling, and case-insensitive verification. Nothing in the suite performs network I/O. - -## 1.5.0 syncs to Cursor pstack 0.15.5 - -Open Pstack 1.5.0 tracks Cursor pstack 0.15.5 at `12d587dfb20741cafc376c42c696c5f6e2a64487`. The first-run panel is now three models: `claude:opus@max`, `codex:gpt-5.6-sol@max`, and `grok:grok-4.7@xhigh`. Opus max takes judgment and prose, hardest tasks, and the How explainer. Grok 4.7 xhigh takes feature and refactoring work, the How explorer, and Swarm workers. Bug-fix, perf-issue, and hillclimb stay on Sol max. Why and Reflect stay on `inherit-parent`. Fable stays in the model matrix with its native agents, but no first-run role uses it. Setup asks about roles first, then asks efforts for and probes only the assigned families (the ordering follows PR #73 by @arjitj2). It drops and lists retired-role rows such as `how critics`. - -Imported: the operator-neutral wording, tick status only on change, code-ready rounds with two or more audit lanes and one fix-forward, rebases at the code-ready report and at merge prep, `children.tsv`, the owner Babysit exception, countersign as approval, Architect reading `architect runners`, Swarm SHA and method briefs, Shipping build-noise lane reuse, decision-trail `start` rows and supersede-only corrections, the non-truncating `log.sh`, and the #419 cuts. Not applied: the five exclusions listed in `UPSTREAM.md`, the Cursor manifest, and `docs/guide/`. Existing sheets are not rewritten. Delete role lines and rerun setup to take the new defaults. A `grok:grok-4.6` row keeps running until setup asks for its replacement. With no sheet, the new defaults apply on update. - -## 1.4.1 syncs to Cursor pstack 0.15.1 - -Open Pstack 1.4.1 tracks Cursor pstack 0.15.1 at `f8abeddd1862dc73704e3d719dd73df0d51b8c71`. Poteto-mode now requires each claim to include its evidence or a measured, inferred, or guess label in the same sentence. Agents also run any check they can run themselves instead of handing that check to the user. No playbook, model, runtime, or dependency changed. - -## 1.4.0 syncs to Cursor pstack 0.15.0 - -Open Pstack tracks Cursor pstack 0.15.0 at `71ed0d1076fec562c1b74ee353121a8d00f75382`. The shared catalog contains 54 skills, including 23 principles. This sync imports the skill density and punctuation passes, the 361,140-byte logo, and the verbatim upstream README. How, Why, and Teach take the shorter explanation guidance. Reflect runs only on explicit invocation. Poteto-mode no longer requires reading the whole principle index as the first todo, but still requires reading any applied leaf and citing it truthfully. Unslop adds the mannered-prose and over-compression rules, preserving stable rule numbers. Technical-writing proposes new abstract-metaphor offenders and replacements without editing Unslop automatically. - -How's critique mode and `how critics` role are retired, and its two review references are deleted. Architect and Investigation use explain-only How. Setup now renders 15 role rows. An old 16-row sheet's `how critics` row has no consumer in ordinary dispatch. Setup reports it as an unknown role before any probe or write. Remove that row before rerunning the normal validated setup flow. On Codex, the editable sheet and bounded AGENTS block must agree. Invalid state or a failed probe leaves both unchanged. - -Two new leaves carry narrow local correctness corrections, also reflected in their descriptions and Poteto-mode's index. `principle-test-behavior-not-implementation` warns conditionally about assertions that miss relevant behavior. `toBeDefined`, `toBeTruthy`, `toBeInstanceOf`, and `toBeGreaterThan(0)` do fail on `undefined`, and useful negative-path, prompt/configuration, and relational contract checks remain valid. `principle-attack-the-premise` limits the actor census and reassignment prescription to imbalance problems. An even census is evidence against an asymmetry hypothesis, not proof that a shared premise is correct. Later syncs take upstream's equivalent corrections if they land. - -Opening a PR adopts the "briefing, not the lab notebook" guidance and links detailed evidence. The "about 40 lines" squash-body cutoff is omitted. The port's required installed-version, user-action, and observed-result evidence sections remain unchanged. The PR 44 shipping rules, parent-owned provider dispatch, Sol defaults for bug-fix, perf-issue, and hillclimb, and no-fallback/no-implicit-timeout contracts remain in place. The Autopilot chooser and Multi-phase plan reference remain intact. - -Existing exclusions remain: `make-bot-ui`, Benny automations, Cursor-only guide and sticky-mode content, Cursor-only solo-model defaults, invocation-blocking flags on How, Why, Unslop, and TypeScript best practices, and the unsupported Claude manifest logo field. Those four skills stay user-invocable and model-invocable. The principle leaves keep `user-invocable: false`, which hides them from the slash menu but leaves model invocation open. The Cursor manifest is not imported. Watcher and orchestrator directories, package metadata, and lockfile have no upstream changes in this range. The unrelated Grok Voice plugin remains outside this sync. - -## 1.3.0 syncs to Cursor pstack 0.14.7 - -Open Pstack now tracks Cursor pstack 0.14.7 at `efa2a531985e0a8084d36ff3cf87233be8a9f34b`. - -**Forge-neutral pull request playbooks.** Shipping, Babysit, Autopilot-full, Autopilot-stack, Opening a PR, and Multi-phase plan use GitHub CLI by default and Origin when its CLI can resolve the repository. Same-repository base chains and fork stacks land from the bottom one pull request at a time. Stable patch IDs decide when a rewritten head needs a new code verdict. Owners open pull requests early, follow repository draft rules until required evidence exists, keep an uncommitted decision trail, and compare load-bearing behavior and performance at trunk and head. Graphite remains only in the unchanged Orchestrate frontier tooling. - -Factory panel review hardened the upstream landing flow. Shipping freezes the stack and clears every pre-existing auto-merge request and native merge-queue entry before verification. On GitHub, it confirms that both `autoMergeRequest` and the separate `mergeQueueEntry` are null. It repeats the check for the current bottom pull request and every descendant before any branch or base mutation. Both autopilots apply the same rule before rewriting an existing pull request, and Opening a PR applies it before an existing child is rebased, force-pushed, or retargeted. Autopilot-stack delivers each bottom pull request through Shipping instead of offering merge-when-ready for descendants. Same-repository heads use a base-branch chain. Fork heads keep local parent ancestry, target trunk in the base repository, and use a frozen bottom-to-top list because a fork-only parent branch cannot be a pull request base. Autopilot-full preserves the same parent ancestry in its short private-stack exception. Owners open pull requests early but leave them draft when repository instructions require live evidence before readiness. - -Fork-safe rebases resolve the base fetch remote separately from the head push remote. The playbooks resolve the head push URL before fetching stack parents, worktree refreshes, and live-lane heads. Rebase and publish steps use that exact URL, binding an explicit force-with-lease to the SHA captured before the rewrite. Each verdict records its head, patch base, and stable patch ID separately from the current landing head and base. A patch-equivalent rewrite keeps the verdict, but every merge guard uses the current landing head. After a parent is squash-merged, an explicit `git rebase --onto` moves only the child's commits from its recorded old parent to current trunk. The rebase command terminates option parsing before passing the short branch name. Git then updates the branch instead of leaving a detached HEAD. Fork pull requests use GitHub's pull-request API with the validated head repository name, which also supports organization-owned forks. Every GitHub pull request command names the canonical base repository. The installed-plugin watcher receives its owner, repository, and pull request number explicitly. - -Values returned by the forge or repository stay shell data. The imported pull request playbooks pass them as quoted arguments instead of pasting contributor-controlled branch names or other forge values into shell source. - -Immediate merges require a server-enforced expected-head guard. The playbook verifies `baseRefName` immediately before the GitHub merge and monitors it until merge. GitHub has no expected-base guard, so Shipping uses the supported head guard and reports a safety failure if the pull request lands in another base. Server-side auto-merge also requires a SHA-scoped independent-verdict check. Without that check, the agent watches until it can issue the guarded immediate merge. Every observed head or base change disarms every pending request and returns to verification. An unarmed frontier returns to the merge step when it becomes ready. Every terminal non-passing required check ends the watch while auto-merge or a merge-queue entry is pending. `UNSTABLE` remains nonterminal by itself. A stale or conflicted frontier returns to the guarded rebase flow. - -Origin merge commands omit GitHub's unsupported `--squash` flag and stop when Origin cannot provide atomic expected-head and expected-base guards. Both autopilot playbooks use the same captured-SHA lease rule. Autopilot-full routes owner merges through Shipping's current-head flow. - -**TypeScript boundary guidance.** The TypeScript skill scopes Claude Code automatic loading to `.ts` and `.tsx` files, prefers repository-owned runtime schemas over hand-written property guards, and derives types from those schemas with helpers such as `z.infer`. Codex continues to invoke the same shared skill by name. - -**Codex presentation.** The exact upstream `assets/logo.png` now ships in the shared plugin. The Codex manifest exposes it through `interface.logo`. The Claude manifest does not add a logo field because Claude Code has no schema for it. - -**No-op upstream revisions.** Upstream moved several Fable references to a newer revision slug. Open Pstack already stores rolling `claude:fable` and `claude:opus` aliases, so those edits require no port change. - -**Upstream-only exclusions.** `make-bot-ui` from `799151d` and `6fecddb` is not ported because it is built entirely from Cursor routine, webhook, and UI primitives. The `disable-model-invocation` additions from `73f8be4` are not applied to `how`, `why`, `unslop`, or `typescript-best-practices`; the flag would break poteto-mode's named invocation path on Claude Code. The `23a56e2` defaults that move `bug-fix`, `perf-issue`, and `hillclimb` from Sol to Fable are not applied because Fable costs much more for these frequent delegated code roles; all three stay on `codex:gpt-5.6-sol@max`. The Claude manifest logo field from `efa2a53` is also omitted because Claude Code has no schema for it. - -## 1.2.1 keeps Fable and Opus on their latest Claude revisions - -Open Pstack now stores `claude:fable@` and `claude:opus@` in its model matrix, role defaults, and generated setup sheets. Claude Code resolves those aliases to the latest family revision. Native Claude agents and external runner calls pass the same aliases. - -Runtime dispatch normalizes the provider-qualified versioned Fable and Opus descriptors written by earlier releases before it chooses a route, so an installed sheet stops pinning as soon as the plugin updates. It does not write user files and reports that setup should persist the migration. `setup-pstack` applies the same rule before validation, preserves roles, lane order, and effort selections, and writes only after the existing probes and confirmation. The runner rejects any version pin that reaches its boundary, records both the requested alias and Claude's concrete reported revision, and verifies that the report belongs to the requested family. Static checks reject new active version pins. - -## 1.2.0 adds verified multi-PR plans, earlier runtime diagnostics, and shared review-bot triage - -Plans with several stages now use one checklist instead of an overview and separate files for each stage. It has one ordered section for every pull request and keeps all ten ways of testing the real product, unit tests, live and performance proof, checks for how changes work together, merge rules, and supporting details in one place. A Node-based checker with no extra dependencies rejects missing or out-of-order sections, fake screenshots, empty definitions of success, incomplete performance proof, incorrectly written review checks, unsupported punctuation, and incorrect command use. Claude Code and Codex use the same installed skill and checker through their existing parent-controlled setup. If a provider fails, it is identified by name and treated as a dropout. No backup provider or hidden time limit was added. - -The shared startup script now checks for Node before using features that only Bun provides. If the startup script or watcher is run with Node, it prints one clear message and exits cleanly instead of failing later because `Bun` does not exist. The normal Bun behavior and every existing test for runners, watchers, and orchestrators are unchanged. CI uses Node 22.23.2 so this behavior is always checked against the same known version. - -The separate Babysit skill now follows the same three-choice rule as poteto-mode Babysit: fix the problem, dismiss it, or ask what to do. Both versions point to one official Bugbot policy. A check of the packaged files confirms that both links work, only one copy of the policy exists, all three choices are present, and the listed situations default to asking, so the two versions cannot quietly become different. - -## 1.1.0 adds selectable requested effort to setup-pstack - -`setup-pstack` asks one requested effort per frontier family (`low`, `medium`, `high`, `xhigh`, `max`) instead of probing a fixed default quartet. `provider-dispatch.md` owns the model matrix: family, upstream choice, provider, model, first-run default effort, selectable efforts, and Claude-native agent stem. Setup loads the current sheet first, folds mixed per-family efforts with an explicit operator choice, probes only the four requested pairs, and writes nothing on a failed probe. A rerun rewrites that family's `@effort` suffix on every assigned role and leaves customized role-to-family lanes in place. First-run defaults remain Fable `max`, Sol `max`, Grok `xhigh`, and Opus `xhigh`. - -Claude-native dispatch ships Fable and Opus agents at all five efforts. Names are `pstack--`. `pstack-fable-max` and `pstack-opus-xhigh` stay. The runner already mapped all five efforts; tests now cover `low`, `medium`, and `high` for each provider. Evidence remains requested effort and route. There is no runtime resolver, same-provider external fallback, implicit timeout, or second configuration source. - -## 1.0.2 retries transient Grok authentication preflights - -`pstack-runner` now waits five seconds and retries `grok models` once when Grok's first preflight would be classified as unauthenticated. This handles the CLI's brief contradictory output during self-update, when it can print an unauthenticated banner while exiting 0 and listing the requested model. A second failure remains terminal with exit 77. The retry shares the existing absolute deadline and cancellation latch, the receipt keeps evidence from both attempts, and model execution still runs at most once. - -## 1.0.1 removes duplicate workflow entries - -Claude Code and Codex both load the native `plugins/pstack/skills/` tree. Codex 0.149.0 also converts each `plugins/pstack/commands/*.md` file into a generated `.codex-plugin/migrated-command-skills/source-command-*/SKILL.md`. The 31 same-named command trampolines therefore doubled Codex's workflow inventory. Claude's component inventory also registered both layers, although Claude Desktop visually merged the duplicate names. - -This release deletes all 31 command trampolines. The native skill tree is now the only workflow source. Claude Code still supports both model-initiated `Skill` tool calls and user `/pstack:` invocations through each native `SKILL.md`. Codex loads that same tree without generated source-command copies. The 21 `principle-*` leaves declare `user-invocable: false`; Claude hides them, while Codex 0.149.0 currently ignores that picker metadata ([#8](https://github.com/ericlitman/open-pstack/issues/8)). - -`tests/skill-collision-repro.sh` rejects any legacy command layer, checks the requested principle visibility metadata, compares the version across `UPSTREAM.md` and the three manifests, and preserves the default model-quad check. Its behavioral mode builds a one-skill Claude fixture and proves that both a model-initiated `Skill` tool call and a user `/testplug:foo` invocation reach `SKILL-RAN`. - -## 1.0.0 — establish open-pstack - -`ericlitman/open-pstack` becomes the canonical cross-harness distribution. Its imported baseline is `053ed78732e3b71826933170eafe7f7782dda844`, synchronized to Cursor pstack v0.14.2 at `46125561306434d8a1d7745d540d8932ab0cd2a2`. The repository preserves the existing history and attribution while moving marketplace identity, links, and the private tooling package to open-pstack. - -Claude Code and Codex continue to share one skill tree. The parent-owned Claude/Codex/Grok provider dispatch, explicit failure receipts, no-fallback rule, and lack of an implicit runtime timeout are unchanged. `UPSTREAM.md` records the review-first sync procedure, and ordinary pull requests run the Bun tests, typecheck, manifest parse, and static maintenance invariants in GitHub Actions. Live provider behavior remains a release gate because CI cannot substitute for subscribed CLI execution. - -## 0.9.12 — restore the upstream frontier panel across Claude Code and Codex - -The v0.14.2 sync kept pstack's workflow structure but replaced upstream's Fable 5 / GPT-5.6 Sol / Grok 4.6 / Opus 5 panel with Claude-only choices. This release restores the upstream frontier panel from either supported parent harness without adding another orchestrator. - -**Parent-owned dispatch.** Model configuration now uses explicit `:@` descriptors. Claude Code keeps Fable and Opus native and launches Sol and Grok externally. Codex keeps Sol native and launches Fable, Opus, and Grok externally. The top-level parent resolves that table once; children do not inspect inherited environment markers, select providers, or reroute themselves. Same-provider external calls are rejected. - -**Deterministic external runner.** `skills/poteto-mode/scripts/runner/pstack-runner` accepts an already-resolved parent, provider, model, effort, access mode, prompt file, cwd, output file, receipt file, and optional timeout. It preflights the assigned subscribed CLI, invokes it once with provider-native read-only or isolated-write controls, disables nested agent orchestration, restricts provider tools, reserves concurrent outputs exclusively, and writes a structured receipt with model proof, elapsed time, usage, cost when exposed, and bounded failure evidence. Claude uses project-only settings and an explicit tool list. Grok uses its kernel sandbox and an explicit tool list, and its terminal Messages result is persisted without progress narration. Codex receipts distinguish a provider report from an exact pinned argv instead of fabricating a reported model. Missing CLIs, authentication failures, unavailable models, explicit timeouts, cancellations, catchable post-reservation failures, non-zero exits, malformed output, and model mismatches are loud dropouts. There is no fallback. Parent harnesses launch it as resumable background work because Claude's foreground Bash path has a ten-minute ceiling. The runner and preflight impose no arbitrary default deadline; `--timeout` is opt-in only when the user or task provides a real external bound, and that bound starts at wrapper entry, remains one end-to-end deadline rather than a fresh allowance per child, and is armed in runtime-safe chunks without shortening long bounds. The runner is self-contained and does not enter the shared script bootstrap/re-exec path. A run-scoped cancellation latch owns SIGINT and SIGTERM from reservation through the terminal receipt, wakes the active direct child wait to send the first signal once, tolerates repeated delivery while reaping it, cancels inherited-pipe reads after child exit, keeps child exit codes distinct from launcher exit status, records only signals it actually sent to the child, clears losing timers, removes the empty output reservation, and preserves a durable terminal receipt; retries use fresh attempt paths. - -**Native Claude effort.** `pstack-fable-max` and `pstack-opus-xhigh` agent definitions pin model plus effort for Claude-native lanes and mechanically deny nested Agent/Task dispatch. Codex passes model and `reasoning_effort` directly to `spawn_agent`. - -**Workflow integration.** One `provider-dispatch.md` reference owns the route contract. `setup-pstack`, poteto-mode, every code-writing playbook, and the How, Why, Reflect, Arena, Swarm, Architect, and Interrogate skills resolve configured roles through it. The default code roles and four-provider panel match upstream v0.14.2. MCP-dependent Why and Reflect roles stay native through `inherit-parent`, because an external provider process does not inherit the parent's live MCP tool surface. The Codex mapping treats an unavailable native lane as a named dropout instead of collapsing the panel, and the README documents the external CLI requirements. - -**Tests.** The vendored Bun suite covers exact command construction, terminal provider output parsing, multi-model usage envelopes, auth/model preflight, honest provider-report versus pinned-argv proof, missing CLI, unavailable model, authentication/model classification, wrapper-entry and shared explicit deadlines, no-timeout delayed work, runtime-safe very-long deadlines, no-spawn after preflight exhaustion, inherited-pipe deadline/cancellation, truthful receipts for post-exit and externally signalled children, bootstrap-free isolated launch, terminal post-reservation failures, prompt retained-handle exit after a losing long timer, preflight and model cancellation, repeated-signal child reaping with a race-free fixture handshake, fresh-path retry, same-provider rejection, identity cleanup, exclusive paths, and simultaneous same-provider lanes. `tests/skill-collision-repro.sh` compares the provider-qualified panel descriptors across the setup sheet, four panel skills, and canonical dispatch reference. Live Claude-parent and Codex-parent behavioral probes are a release gate; unit tests alone are insufficient. - -## 0.9.11 — sync to upstream v0.14.2 - -Catches the port up with upstream `cursor/plugins/pstack` from `3fe2823` (v0.11.3) to `4612556` (v0.14.2). Skills 48 → 52, commands 27 → 31, subagents 1 → 2, plus a vendored `scripts/` tree under `poteto-mode/`. - -**New skills.** - -- `swarm` (upstream `b79f8ca`, `91dd7b7`): fan out N parallel workers over slices or races, drain them, return one report. Upstream spawns Cursor cloud agents; Claude Code has no remote worker environment, so the port spawns local background subagents and takes isolation from a worktree or a per-worker output directory. The worker default follows the port's single-role default (`claude-opus-4-8`) and reads `swarm workers` from `~/.claude/pstack-models.md`. -- `no-comments` plus the `comment-sicko` subagent (#185): a comment-stripping review pass. `Task` → `Agent`, and upstream's agent name `Comment Sicko` becomes `comment-sicko` because a `subagent_type` with a space is not addressable. -- `technical-writing` (#185): the layered Diátaxis / Google developer style / STE / Global English standard. No Cursor primitives, copied as-is. -- `bro` (#187): restate the last message in plain language. Copied as-is. - -Per the 0.9.8 invariant, all four drop upstream's skill-side `disable-model-invocation: true` and get command trampolines that carry it instead. - -**New playbooks** (#185, #187), under `poteto-mode/playbooks/`: `babysit` (drive a PR or stack to merge-ready), `shipping` (verify each PR independently, then land the contiguous verified run), `orchestrate` (a standing multi-day program under one coordinator), `autopilot-full` and `autopilot-stack` (one owner per PR, swarm-verified), and `worktree-cleanup` (safety-gated disk reclamation). They share the new `references/bugbot-triage.md` rubric, which the poteto-mode review-bot trigger now points at. - -Substitutions in the six: Cursor cloud agents become local background subagents isolated by worktree; `control-cli` / `control-ui` become the `run` / `verify` built-ins; `Task` becomes `Agent`; `AskQuestion` becomes `AskUserQuestion`; the Cursor agent store becomes `~/.claude/orchestrate//`, which outlives the session the way a multi-day program's store has to; a Cursor restart becomes a session restart; the Cursor dashboard becomes the background task list. Cursor's `/goal` has no Claude Code equivalent, so the autopilots keep the program objective in the standing orders and the todolist, and their audit tick re-reads the playbook from the installed plugin instead of `git show origin/main:pstack/...`. Graphite (`gt`) is not Cursor-specific and stays. - -**Babysit, skill versus playbook.** Upstream v0.14.0 stopped routing PR-status requests to Cursor's built-in babysit and gave poteto-mode its own playbook. The port's bundled `babysit` skill (the 0.9.2 analog of that built-in) stays as the standalone `/babysit` entry point; inside poteto-mode the playbook supersedes it, and both files say so. - -**Vendored scripts.** `poteto-mode/scripts/` carries upstream's `watch-pr` (the PR watcher the Babysit and Shipping playbooks poll), `orch` (the orchestrate store CLI), `worktree-audit.sh`, and the bun bootstrap. Three edits: `worktree-audit.sh` reads `~/.claude/projects/` — scanning the whole projects tree, since a session run inside a worktree gets its own encoded directory — instead of `~/.cursor/projects//agent-transcripts`; it warns when `jq` or `rg` is missing, because their absence silently blanks the PR and LAST_CHAT columns and downgrades an in-use worktree to `safe` in the one playbook that deletes user state; and the private workspace package is renamed `@pstack-claude/poteto-mode-tools` in `package.json` and `bun.lock`. `bun` joins `gh` as a documented system dependency, `gt` only for the stack playbooks and `jq`/`rg` only for the worktree audit. `node_modules/` under the scripts dir is gitignored. - -**Content refinements over the port's existing translation.** `architect` gains design-it-twice, the `design-red-flags.md` screen, and the interface-depth comparison (#175), with the matching rationale-template and runner-prompt edits. Nine `principle-*` leaves, `unslop`, `typescript-best-practices` (plus 69 new lines of `references/patterns.md`), and `lead-judgment.md` take upstream's HEAD bodies verbatim; their bodies were byte-identical to upstream `3fe2823` beforehand, so the port keeps only its own frontmatter. `interrogate` moves its reviewer defaults into a labelled A/B/C/D table (#167). `create-verification-skill` points at the new `feature-map-example/` and names the four required H2s (#178). `automate-me` learns that mode skills can live in a personal category directory (#187). `opening-a-pr` takes upstream's title, description, readiness, and babysit rewrite (#185, #238), and `autonomous-run` takes mid-run-discovery ownership (#170). - -**Model configuration.** Upstream's slug bumps (#165, #166, #169, #210: fable / sol / grok / Opus 5) are not ported — the port keeps its own Claude quad (`claude-opus-5`, `claude-fable-5`, `claude-opus-4-6`, `claude-sonnet-5`) and its `claude-opus-4-8` single-role default. What is ported is the structure: the `inherit-parent` / `auto` aliases (#163), which on Claude Code mean omitting `model` on the `Agent` call so the role runs on the parent session's model, and the config-source-first phrasing (#167) in `arena`, `interrogate`, and `swarm`. `setup-pstack` gains the aliases, the `swarm workers` row, and the alias-aware validation rules. - -**Deliberately not ported.** `docs/guide/` (the ten-chapter tutorial and its six screenshots, 2.3 MB) teaches pstack through Cursor's UI, sticky mode, and cloud agents, and ships no skill content; README links it upstream. `is_background: true` on `poteto-agent` is Cursor agent frontmatter with no Claude Code key. Sticky mode and Benny remain out, unchanged from the earlier reviews. - -**Test fix.** The quad invariant in `tests/skill-collision-repro.sh` had been searching the panel skills for `claude-sonnet-4-6`, a slug the 0.9.10 panel swap removed, so the check failed on `main` for every skill. It now derives the anchor slug from the canonical `arena runners` row and reads `interrogate`'s quad from its new reviewer table. - -**Verified.** 52 skills / 31 commands / 2 subagents; three manifests parse at 0.9.11. Static invariants pass, including the repaired quad check. The vendored scripts pass `bun install --frozen-lockfile`, `bun test orch watch-pr` (52 tests), and `bun run typecheck` from their ported location. Not yet live-verified in a Claude Code or Codex session. - -## 0.9.10 — sync to upstream v0.11.3 - -Catches the port up with upstream `cursor/plugins/pstack` from `0452e08` (v0.10.0) to `3fe2823` (v0.11.3). Skill count 44 → 48, commands 24 → 27. - -**New skills.** - -- `teach` (#153): composes the `how` and `why` skills into one plain explanation. Command-paired public skill; platform note for the parallel dispatch and image-gen tool. -- `principle-model-the-domain` (#147): encode the domain in a structure instead of scattered conditionals. Verbatim upstream prose, `user-invocable: false` per the 0.9.9 principle convention, woven into `poteto-mode`'s data-shape trigger and Architecture index, and into the `feature` and `refactoring` playbooks. -- `create-verification-skill` and `maintain-verification-skill` (#150, #151): generate and maintain a persistent, repo-tailored project verification skill plus feature map. A different layer from Claude's built-in `run`/`verify` per-session drivers, which they complement. Translation: `.cursor/skills/` → `.claude/skills/`, drop `disable-model-invocation` (command-paired), add command trampolines and platform notes. - -**Content refinements (#155, #156), applied over the port's existing Cursor→Claude translation.** The perf playbook's eight strategy families; `interrogate` "four-model" → "multi-model"; the `hillclimb` ground-the-workload-first rewrite; rationale one-liners across five playbooks; `typescript-best-practices` real-tests and structured-telemetry rows; the `why` databricks source `SHOW TABLES` note. - -**Model strategy (#143, #156).** `poteto-mode` now tiers code delegates by difficulty — hardest changes to the strongest judgment model (`claude-fable-5`) or the strongest instruction follower, trivial edits to a fast code model, everything else `claude-opus-4-8`. `setup-pstack` gains the configurable `arena cross-judge pool` row. The upstream `grok`/`gpt` slugs stay substituted with `claude-*`. - -**Deliberately not ported.** Sticky mode (#144) is Cursor-only frontmatter (`mode`/`icon`/`color`/`reminder`) with no Claude Code equivalent; the port's 0.9.5 SessionStart hook already auto-fires `poteto-mode` with the same non-trivial/trivial/opt-out logic. Benny (#137) remains out, per the earlier review. - -## 0.9.9 — principle leaves hide from the slash menu - -0.9.8 kept `disable-model-invocation: true` on the 20 `principle-*` leaves, on the reasoning that they have no command and are read by path from `poteto-mode`. But that flag only blocks *model* invocation. It does not hide a skill from the user `/` menu, so all 20 surfaced as bare `/principle-*` slash commands in every session (confirmed across projects on the desktop app). They are internal references; users should never invoke them. - -Fix: swap the flag for `user-invocable: false` on all 20 leaves. Per the [Claude Code skills docs](https://code.claude.com/docs/en/skills.md), `user-invocable: false` hides a skill from the `/` menu and controls menu visibility only, not Skill-tool or file access — so `poteto-mode` reading each leaf by path (`../principle-/SKILL.md`, the mechanism the leaves have always used) is untouched. The two flags are mutually exclusive: setting both leaves a skill neither user- nor model-invocable, so this is a swap, not an addition. The visible consequence is that the leaves become model-auto-invocable on description match — the same standing the 12 command-paired skills took in 0.9.8, and immaterial to the by-path reference the leaves actually rely on. The invariant is now: every command carries `disable-model-invocation: true`, no command-paired skill carries it, and every `principle-*` leaf carries `user-invocable: false`. - -## 0.9.8 — command-paired skills drop `disable-model-invocation` - -The 0.9.7 fix put `disable-model-invocation: true` on all 24 command trampolines so the Skill tool resolves a colliding name to the skill. But 12 of those skills carried the same flag in their own frontmatter (present since the initial port), and the flag on a **skill** makes the Skill tool refuse the invocation outright. Net effect: the 0.9.5 SessionStart mandate ("invoke `pstack:poteto-mode` with the Skill tool") was refused every session, five of the six direct-entry skills it lists (`poteto-mode`, `tdd`, `architect`, `arena`, `interrogate`; only `how` and `why` were unflagged) were model-unreachable, and every user-typed `/pstack:` for a flagged skill expanded to a trampoline body the model then couldn't follow. 0.9.7 didn't cause the skill-side flags, but it surfaced them: before it, the same calls died in the trampoline loop instead. - -Fix: remove the flag from the 12 command-paired skills that carried it (`architect`, `arena`, `automate-me`, `blast-radius`, `figure-it-out`, `interrogate`, `poteto-mode`, `recall`, `reflect`, `show-me-your-work`, `tdd`, `thermo-nuclear-code-quality-review`). The resulting invariant is symmetric: every command carries the flag, no command-paired skill does. The `principle-*` leaves keep it — they have no commands and are deliberately read by path from `poteto-mode`. Model auto-invocation on description match is now possible for the 12; that is the 0.9.5 design intent, and the same standing the other 12 always-unflagged skills (`how`, `why`, `babysit`, `deslop`, …) already had. - -`tests/skill-collision-repro.sh` gains the mirrored static invariant (no skill with a same-named command may carry the flag) and a fourth behavioral leg preserving the repro: with the flag on the skill, the Skill tool refuses the invocation even though the command no longer shadows it. - -## 0.9.7 — command trampolines no longer shadow their skills - -Every user-facing skill ships with a same-named `commands/.md` trampoline whose body is "Invoke the `` skill and follow it." On Claude Code, the Skill tool resolves a colliding name to the **command**, not the skill — so a model-initiated invoke of `pstack:` got the trampoline back, which told it to invoke the skill, which resolved to the trampoline again. Mutual recursion; the real `SKILL.md` never loaded. This hit every model-side entry path, including the 0.9.5 SessionStart mandate (whose whole job is telling the model to invoke `pstack:poteto-mode`), and it made each name appear twice in the model's skill list. Observed in desktop-app sessions (inline `--plugin-dir` loading, same path as the 0.9.3 entry); reproduced on CLI 2.1.195 with a minimal two-artifact plugin. - -Fix: all 24 command files now carry `disable-model-invocation: true` — the flag the `principle-*` leaf skills already use. Verified on 2.1.195 with the same minimal plugin: the Skill tool then resolves the colliding name to the skill (its `SKILL.md` is what gets injected), while a user-typed `/plugin:` still runs the command trampoline, whose "invoke the skill" body now lands on the skill instead of looping. Command bodies are unchanged, so the Codex prompts path (which reads `description` frontmatter and the filename, and ignores keys it doesn't know — the established `name`-key precedent) is untouched. Whether a full Codex plugin install still surfaces the stubs in its picker with the flag present is unverified; if it hides them, the skills themselves remain the primary Codex surface and are invocable by name. Incidental finding from the same repro, recorded for future use: `${CLAUDE_PLUGIN_ROOT}` **is** substituted inside command markdown bodies on 2.1.195, not just in hooks and MCP configs. - -The repro is preserved as `tests/skill-collision-repro.sh` (manual; needs the `claude` CLI, makes three haiku calls). It checks the static invariant (every command carries the flag) and the three behavioral legs: command wins the collision without the flag, skill wins with it, user-typed `/command` still runs. The first leg is a precedence detector — if it ever fails, upstream changed the undocumented resolution order and the flag should be re-evaluated, not the fix declared broken. - -## 0.9.6 — hook hardening and duplication trims (thermo-nuclear review) - -A strict maintainability review of the 0.9.3–0.9.5 range drove these: - -- `hooks/session-start` collapsed from 43 lines to 3: SessionStart hook stdout reaches context directly (per the hooks docs; verified end-to-end on 2.1.197), so the JSON envelope, the `escape_for_json` pass, and the Cursor/Copilot platform branches — dead code here, since `hooks.json` is the only registration — are gone. This also removes a verified failure-path bug: the old `cat ... 2>&1 || echo` fallback was additive, silently injecting raw `cat` stderr plus the fallback string into session context when the context file was unreadable; now a missing file fails the hook cleanly and injects nothing. The script is no longer adapted from superpowers (NOTICE updated; `run-hook.cmd` remains near-verbatim and attributed). -- Panel-quad enumeration trimmed from the `poteto-mode` meta-files (`SKILL.md`, `references/plan.md`, `references/codex-tools.md`) — the slugs now live only in the four panel skills and the `setup-pstack` sheet, with a grep-identical rule added to Maintenance. This drift class already bit once (0.9.4 fixed a three-reviewers-vs-"four different models" mismatch). -- README's desktop-app `dependency-unsatisfied` narrative deduplicated to a two-sentence summary linking the CHANGES 0.9.3 entry. - -## Upstream review through `0452e08` (v0.10.0), 2026-07-01 - -One upstream pstack commit landed after the `e46364b` sync: `0452e08` adds the dormant `automations/benny/` pack (Slack issue triage plus reproduce-and-fix, built on Cursor's event-triggered automations) and bumps upstream to 0.10.0. Deliberately not ported — rationale and revisit criteria in README → What's deliberately not ported. `cursor-team-kit` has no commits since the sync point (its latest, `679fdaf`, 2026-05-28, predates `e46364b`). The port's skill tree was current with upstream HEAD as of this note; the later 0.9.10 sync carries it forward to v0.11.3 (see above). - -## 0.9.5 — poteto-mode auto-fires via SessionStart hook - -`plugins/pstack/hooks/` is new. `hooks.json` registers a `SessionStart` hook (matcher `startup|clear|compact`) that injects `hooks/session-start-context.md` (~0.3k tokens) as additional context — the same mechanism superpowers uses to auto-load its skill-use mandate. The injected block routes any non-trivial engineering task into `pstack:poteto-mode` before the first response, lists the direct-entry skills, tells dispatched subagents to ignore it, and defers to explicit user instructions. The full poteto-mode skill still loads only on invoke. `run-hook.cmd` (cross-platform polyglot) and the JSON-emission pattern in `session-start` are adapted from superpowers (MIT; see NOTICE.md and LICENSE-superpowers). Codex is unaffected — it has no plugin hook runtime; invoke poteto-mode by name there. - -## 0.9.4 — Sonnet 5 joins the default panels - -The multi-model panels (`arena` runners, `architect` runners, `interrogate` reviewers, `how` critics) grow from a triple to a quad: `claude-opus-4-8`, `claude-sonnet-5`, `claude-opus-4-6`, `claude-sonnet-4-6` — both generations in each of two tiers. This also restores upstream's four-way `interrogate` split; the port had been running three reviewers under a "four different models" description. `setup-pstack` adds Sonnet 5 (`claude-sonnet-5`) to the available-family enumeration and to the four panel rows of its default sheet. Single-model delegation defaults stay `claude-opus-4-8`. Touched: `arena`, `architect`, `interrogate`, `how`, `setup-pstack`, `poteto-mode` (`SKILL.md`, `references/plan.md`, `references/codex-tools.md`), and the README substitution-table panel row. The historical Cursor→Claude mapping rows (`composer-2.5-fast`, `gpt-5.x`) are unchanged — they record what the 0.9.2 sync substituted, not current defaults. - -## 0.9.3 — dependency declaration removed - -`plugin.json` no longer declares `dependencies: [{ "name": "plugin-dev", "marketplace": "claude-plugins-official" }]`, and `marketplace.json` drops the matching `allowCrossMarketplaceDependenciesOn`. The Claude Code desktop app passes every enabled plugin to the CLI as a session-only `--plugin-dir`, which strips marketplace identity (`pstack@inline`); a cross-marketplace dependency can never resolve in that mode, and the loader disables the entire plugin with `dependency-unsatisfied`. Result: pstack loaded in the CLI and the VS Code extension but silently vanished from desktop-app sessions. `optional: true` on a dependency entry passes `claude plugin validate` but is not honored by the loader (tested on 2.1.197). `plugin-dev` is now a documented manual install (README → Dependencies); skill bodies still route skill-authoring to `plugin-dev:skill-development` when it is present. - -## Codex port - -pstack also ships as a Codex plugin. The skill bodies are not forked or regenerated. The same `skills/` tree serves both runtimes. One mapping file does the Claude-to-Codex translation. That single-mapping-file spine is the same one the official `superpowers` plugin ships for Codex. - -pstack diverges from superpowers in one respect, and it is deliberate. superpowers writes its skill prose in tool-neutral language ("dispatch a subagent"), so no skill names a runtime tool and no per-skill note is needed. pstack instead keeps the upstream Claude-native prose intact, to stay in lockstep with upstream sync, and adds a one-line Platform note to each skill that names a Claude primitive. The note points at the mapping. Rewriting 44 upstream skills into neutral language would fork them from upstream and was rejected for that reason. - -**Added.** - -- `plugins/pstack/.codex-plugin/plugin.json` is the Codex plugin manifest (`skills: ./skills/`), with key-parity to the `superpowers` Codex manifest. -- `.agents/plugins/marketplace.json` is the Codex marketplace manifest at the repo root, sourcing `./plugins/pstack` the way the Claude `.claude-plugin/marketplace.json` does. -- `plugins/pstack/skills/poteto-mode/references/codex-tools.md` is the single Claude to Codex map. It covers tool actions (`Agent` becomes `spawn_agent` / `wait_agent` / `close_agent`, `AskUserQuestion` becomes plain text, the todolist becomes `update_plan`), the `multi_agent` config flag, subagent policy (Codex has no `poteto-agent` type, so dispatch a `spawn_agent` told to read `poteto-mode` first), model slugs (`claude-*` becomes your configured Codex models), the Claude built-ins pstack names (`run`, `verify`, `loop`, `plugin-dev:skill-development`), and the instructions file (`AGENTS.md`). - -**Platform notes (pointer-only edits).** - -- `skills/poteto-mode/SKILL.md` gained a "Platform Adaptation" section pointing at the mapping. -- `skills/{architect,arena,automate-me,babysit,how,interrogate,reflect,why}/SKILL.md` each gained a one-line Platform note, since each names a Claude tool, a `claude-*` slug, or a Claude built-in. The pure-prose skills (the `principle-*` set, `tdd`, `figure-it-out`, and the cursor-team-kit imports) needed nothing. -- `skills/setup-pstack/SKILL.md` gained a Codex branch. It writes `~/.codex/pstack-models.md` referenced from `~/.codex/AGENTS.md`, using Codex slugs instead of `claude-*`. - -**Commands.** The 24 `commands/*.md` files are Codex-compatible as written, no rewrite needed. Codex command discovery reads the `description` frontmatter and the filename and ignores the extra `name` key, and each body (`Invoke the skill and follow it`) is a valid Codex prompt. They surface as slash commands once the full plugin is installed in Codex. For the symlink-based install, drop the same files into `~/.codex/prompts/` for loose `/name` shortcuts alongside the symlinked skills. - -**Deliberately not ported.** - -- `agents/poteto-agent.md`. Codex has no `subagent_type`, so ad-hoc subagents are dispatched via `spawn_agent` told to read `poteto-mode` first. The mapping covers this. - -**Verified.** Codex discovers the skills and namespaces them under `pstack` (`pstack:poteto-mode` and so on) in a live session. Mapping resolution mid-task and `spawn_agent` fan-out follow the `superpowers` pattern and are worth confirming per session. - -**Maintenance.** The open-pstack version string lives in `plugins/pstack/.claude-plugin/plugin.json`, `.claude-plugin/marketplace.json`, `plugins/pstack/.codex-plugin/plugin.json`, and the current-version row in `UPSTREAM.md`. A version bump must update all four. `tests/skill-collision-repro.sh` checks that they match. `.agents/plugins/marketplace.json` carries no version field. The canonical default panel is the `## Default panel` line in `provider-dispatch.md`. It is copied into the four panel rows of the setup-pstack first-run sheet and into `arena`, `architect`, and `interrogate`. Keep those copies grep-identical when models change. The static test reads the panel from that line, and `model-matrix.test.ts` pins its descriptors and checks each against the matrix defaults. After a sync that touches `skills/poteto-mode/scripts/`, run `bun install --frozen-lockfile`, `bun run test`, and `bun run typecheck` from that directory. `hooks/session-start-context.md` names only `poteto-mode` as the default entry. Re-verify it if that skill is renamed. The package must not contain a `commands/` layer. Claude Code and Codex load the native `skills/` tree directly, and a command layer duplicates that inventory. The 23 `principle-*` leaves carry `user-invocable: false` to request exclusion from the user picker while `poteto-mode` reads them by path. Claude honors the metadata; Codex 0.149.0 currently does not ([#8](https://github.com/ericlitman/open-pstack/issues/8)). They must not carry `disable-model-invocation`, which would make them unreachable to the model. Re-run the behavioral mode of `tests/skill-collision-repro.sh` after Claude Code upgrades to check both model-initiated and user-initiated native skill invocation. - -## 0.9.2 sync (against upstream `e46364b`) - -Upstream pstack jumped from `0.1.0` → `0.9.2` between syncs. 30+ commits, including 11 new files. - -**New pstack-native skills/playbooks pulled in (Cursor refs in them re-substituted on the way in):** - -- `skills/blast-radius/` — find what a change could break beyond the diff. -- `skills/recall/` — reconstruct recent working context. Cursor transcript path (`~/.cursor/projects//agent-transcripts//.jsonl`) rewritten to Claude Code path (`~/.claude/projects//.jsonl`). -- `skills/setup-pstack/` — model-per-role configuration. Substantially rewritten: original wrote `~/.cursor/rules/pstack-models.mdc` (Cursor's `.mdc` always-applied-rule feature, no Claude Code analog). Replacement writes `~/.claude/pstack-models.md` and instructs the user to add an `@~/.claude/pstack-models.md` include to `~/.claude/CLAUDE.md` so the override sheet loads each session. -- `skills/principle-build-the-lever/`, `skills/principle-sequence-verifiable-units/` — new principles. -- `skills/poteto-mode/playbooks/{hillclimb,pause-safely,refactoring,session-pickup,trace-forensics}.md` — new playbooks. -- `skills/interrogate/references/code-quality-review.md` — new interrogate reference. - -**Re-applied substitutions across changed + new content:** - -- Bulk pass through 28 files via Python regex covering all entries in the substitution table above. -- Targeted fixes for variants the bulk pass missed: - - `recall/SKILL.md` line 15 — Cursor transcript path rewrite. - - `why/SKILL.md` line 100 — MCP discovery wording variant. - - `poteto-mode/SKILL.md` lines 22–25 — `cursor-team-kit` qualifiers removed; Bugbot triage refs to `babysit`. - - `reflect/SKILL.md` lines 37, 45, 49 — readonly/agent-mode language; `Task` → `Agent`. - - `poteto-mode/playbooks/session-pickup.md` line 7 — `agent-transcripts/` path. - - `poteto-agent.md` description — `generalPurpose` → `general-purpose`. -- Bumped Opus references from `claude-opus-4-7` to `claude-opus-4-8` (current Claude family head). -- Multi-model panels (`arena`, `architect`, `interrogate`, `how` critics, and the `setup-pstack` defaults) had a duplicate `claude-sonnet-4-6` in the third slot. Replaced one with `claude-opus-4-6` so the panel runs three distinct models (`claude-opus-4-8`, `claude-opus-4-6`, `claude-sonnet-4-6`) instead of two — cross-generation diversity inside the opus tier where cross-vendor diversity isn't available. -- All single-subagent delegation defaults bumped from `claude-sonnet-4-6` to `claude-opus-4-8`: `bug-fix`, `feature`, `perf-issue`, `refactoring`, `hillclimb` (the five poteto-mode code-writing playbooks); `how-explorer`, `why-investigators`, `reflect-tooling` (the three multi-subagent dispatches that run the same model in parallel rather than a diverse panel). Setup-pstack override sheet updated to match. Meta-defaults in `poteto-mode/SKILL.md` and `plan.md` rephrased: "default `claude-opus-4-8` for code-writing delegations" replaces the old "claude-sonnet-4-6 for code" wording. Sonnet now appears only in the diverse 3-model panels. - -**Command stubs added:** `commands/blast-radius.md`, `commands/recall.md`, `commands/setup-pstack.md`. - -**Manifest changes:** - -- `plugins/pstack/.claude-plugin/plugin.json` — version `0.1.0` → `0.9.2`; added `displayName: "pstack (Claude Code port)"`. -- `.claude-plugin/marketplace.json` — plugin entry version bumped to `0.9.2`. - -**Team-kit imports:** unchanged. The upstream diff showed only `verify-this` (which we didn't import) changed in `cursor-team-kit/skills/`. - -**`babysit` skill:** unchanged. Locally authored; not affected by upstream sync. - ---- - -## Substitution table - -| Cursor primitive | Replaced with | Notes | -| --- | --- | --- | -| `Task` tool | `Agent` tool | Claude Code's `Agent` tool is the equivalent. | -| `subagent_type: generalPurpose` | `subagent_type: "general-purpose"` | Kebab-case in Claude Code. | -| `subagent_type: "poteto-agent"` | `subagent_type: "poteto-agent"` | Unchanged — this plugin ships that agent. | -| `readonly: true` / `readonly: false` | (dropped; rewritten as "pick a subagent_type that retains MCP access") | Claude Code controls tool/MCP access via subagent_type, not a per-call readonly flag. | -| `AskQuestion` | `AskUserQuestion` | Tool rename; semantics match. | -| Cursor `/loop` (built-in) | Claude Code `loop` skill | 1:1 replacement; available as a built-in skill. | -| Cursor `/babysit` (built-in) | This plugin's `babysit` skill | New Claude Code analog at `skills/babysit/` wrapping `gh` + `loop`. | -| Cursor `/create-skill` (built-in) | `plugin-dev:skill-development` skill | Claude Code's authoring guidance for SKILL.md. | -| `cursor-team-kit` `/deslop` | This plugin's `deslop` skill | Ported in (only team-kit skill imported). | -| `cursor-team-kit` `control-cli` | `run` skill (Claude Code built-in) | Drives CLIs/TUIs. | -| `cursor-team-kit` `control-ui` | `verify` skill (Claude Code built-in, VS Code extension) | Drives UIs (browser/Electron). | -| `~/.cursor/projects/*/` transcripts | `~/.claude/projects//*.jsonl` | `` is the workspace's working directory with `/` → `-`. | -| Cursor `agent-transcripts/` dir | `~/.claude/projects//` | Same as above. | -| `.cursor/skills/`, `~/.cursor/skills/`, `~/.cursor/plugins/` | `.claude/skills/`, `~/.claude/skills/`, `~/.claude/plugins/` | Path-only translation. | -| Cursor `mcps/` directory | Tool list at top of system prompt (`mcp____` prefixed entries), or `.mcp.json`, or `claude mcp list` | Discovery surface differs. | -| Model: `composer-2.5-fast` | `claude-sonnet-4-6` | Fast workhorse Claude. | -| Model: `claude-opus-4-X-thinking-xhigh` | `claude-opus-4-8` (with note "extended thinking" where it appeared in a table) | Claude Code uses model IDs without the Cursor UI suffix; extended thinking is a separate knob. Originally substituted to `4-7`, then bumped to `4-8` to match the current Claude family. | -| Model: `gpt-5.3-codex-high-fast`, `gpt-5.5-high-fast` | `claude-sonnet-4-6`, `claude-haiku-4-5` | Within Claude Code, cross-vendor diversity isn't native. Skills that need a harsher pass now route to the bundled `thermo-nuclear-code-quality-review` skill (imported from cursor-team-kit) as the escape hatch. Different style of pressure (strict maintainability rubric), not vendor diversity. | - -## New / imported files - -- `skills/babysit/SKILL.md` — Claude Code analog of Cursor's `/babysit`. Wraps `gh pr view` / `gh pr checks` / `gh run view --log-failed` plus the `loop` skill for pacing. Provenance: independently authored; workflow informed by Cursor's public `/babysit` behavior. Not a copy of Cursor's closed-source implementation. -- `commands/babysit.md` — slash command routing to the babysit skill. -- `skills/thermo-nuclear-code-quality-review/SKILL.md` — imported verbatim from `cursor-team-kit`. Used as the harsher-critique escape hatch in `arena`, `interrogate`, `architect`, and `how` (replaces the Cursor-original cross-vendor bridge). -- `commands/thermo-nuclear-code-quality-review.md` — slash command stub. -- `skills/make-pr-easy-to-review/`, `skills/fix-ci/`, `skills/fix-merge-conflicts/`, `skills/get-pr-comments/`, `skills/what-did-i-get-done/` — five more skills imported verbatim from `cursor-team-kit`. Audited for Cursor-specific refs; none found, so no rewiring needed. They use only `gh` and `git` primitives. -- `commands/make-pr-easy-to-review.md`, `commands/fix-ci.md`, `commands/fix-merge-conflicts.md`, `commands/get-pr-comments.md`, `commands/what-did-i-get-done.md` — slash command stubs. -- `.claude-plugin/marketplace.json` — marketplace manifest so the repo is installable via `/plugin marketplace add michael-denyer/pstack-claude`. Declares `allowCrossMarketplaceDependenciesOn: ["claude-plugins-official"]` so the cross-marketplace dependency on `plugin-dev` resolves at install time. -- `plugin.json` `dependencies` — declares `plugin-dev` (from `claude-plugins-official` marketplace) as a required dependency, since the rewiring routes skill-authoring tasks to `plugin-dev:skill-development`. - -## Per-skill changes applied - -### `skills/poteto-mode/SKILL.md` - -- Triggers section: `create-skill` → `plugin-dev:skill-development`; `deslop` "from `cursor-team-kit`" qualifier dropped; `control-cli`/`control-ui` line replaced with `run`/`verify` driver guidance; `Cursor's built-in **babysit**` → this plugin's `babysit`. -- Subagents section: `Task` → `Agent`; `composer-2.5-fast` → `claude-sonnet-4-6`; `claude-opus-4-8-thinking-xhigh` → `claude-opus-4-8`; "agent mode (readonly strips MCP)" → "full tool access (do not pick a subagent_type that strips MCP)". - -### `skills/poteto-mode/references/plan.md` - -- `AskQuestion` → `AskUserQuestion`. -- `generalPurpose` → `"general-purpose"`; built-in `plan` subagent_type → Claude Code's built-in `Plan` agent; both model slugs updated. -- `create-skill` → `plugin-dev:skill-development`. -- `control-ui` / `control-cli` lines replaced with `verify` / `run` driver skills. -- "Cursor's built-in **babysit** skill" → "the **babysit** skill". - -### `skills/poteto-mode/playbooks/` - -- `authoring-a-skill.md`: `create-skill` → `plugin-dev:skill-development`. -- `autonomous-run.md`: "Cursor's `/loop` command (a built-in, not a pstack skill)" → "Claude Code's `loop` skill (built-in)". -- `bug-fix.md`, `feature.md`, `perf-issue.md`: `composer-2.5-fast` → `claude-sonnet-4-6`; "control skill" → "driver skill (`run` for CLIs/TUIs, `verify` for UIs)". -- `eval.md`: `agent-transcripts/` + `~/.cursor/projects/*/` → `~/.claude/projects//*.jsonl`. -- `opening-a-pr.md`: `Task` → `Agent`; "Cursor's built-in **babysit** skill" → "the **babysit** skill". -- `prototype.md`, `runtime-forensics.md`, `visual-parity.md`: "control skill" → "driver skill" with `run`/`verify` explicit. - -### `skills/automate-me/SKILL.md` - -- Description and body: `create-skill` (6 places) → `plugin-dev:skill-development`. -- `AskQuestion` (2 places) → `AskUserQuestion`. -- `.cursor/skills/` / `~/.cursor/skills/` → `.claude/skills/` / `~/.claude/skills/`. -- `agent-transcripts/` + `~/.cursor/projects/*/` → `~/.claude/projects//*.jsonl`. - -### `skills/reflect/SKILL.md` + `references/*.md` - -- Transcript paths → `~/.claude/projects//*.jsonl`. -- `Task` → `Agent` (in SKILL.md and all three reviewer references). -- `generalPurpose` → `"general-purpose"`; `readonly: false` + "agent mode" → "pick a subagent_type that retains MCP access". -- Model slugs updated (`composer-2.5-fast` → `claude-sonnet-4-6`; `claude-opus-4-8-thinking-xhigh` → `claude-opus-4-8`). -- `create-skill` (3 routing rules) → `plugin-dev:skill-development`. -- Reference files: `.cursor/skills/`, `~/.cursor/skills/`, `~/.cursor/plugins/` → `.claude/...`, `~/.claude/...`. - -### `skills/why/SKILL.md` - -- MCP discovery: Cursor environment / `mcps/` directory → Claude Code tool list / `.mcp.json` / `claude mcp list`. -- `generalPurpose` → `"general-purpose"`; readonly/agent-mode language → "pick a subagent_type that retains MCP access". -- Model slugs updated. - -### `skills/how/SKILL.md` - -- `generalPurpose` → `"general-purpose"` (all 4 occurrences). -- `composer-2.5-fast` → `claude-sonnet-4-6` (replace_all). -- `claude-opus-4-8-thinking-xhigh` → `claude-opus-4-8` (replace_all for inline; table cell updated separately). -- Critic model table: GPT slugs → Claude family; added note about bridging to `/gsd-review` for cross-vendor critique. -- `readonly: true` lines dropped from subagent config blocks. - -### `skills/interrogate/SKILL.md` - -- `Task tool` → `Agent` tool. -- Reviewer model table: `claude-opus-4-8-thinking-xhigh` / `gpt-5.3-codex-high-fast` / `gpt-5.5-high-fast` / `composer-2.5-fast` → Claude family variants. -- `generalPurpose` → `"general-purpose"`; `readonly: true` dropped. -- Added cross-vendor-bridge note (`/gsd-review`). - -### `skills/arena/SKILL.md` - -- Default 3 runners: GPT/composer slugs → Claude family. Added cross-vendor-bridge note. - -### `skills/architect/SKILL.md` - -- Phase B runner slugs: GPT/composer → Claude family. Added cross-vendor-bridge note. - -### `skills/show-me-your-work/SKILL.md` - -- Transcript audit path: `agent-transcripts/` + `~/.cursor/projects/*/` → `~/.claude/projects//*.jsonl`. - -## Deliberately not changed - -- **`claude-opus-4-8` model ID.** Already a valid Claude model; no edit needed beyond stripping the Cursor `-thinking-xhigh` UI suffix. Extended thinking is configured separately, not as a model variant. -- **`/loop`, `/deslop`, `/babysit` slash references.** These all resolve in Claude Code now (`loop` is a built-in skill; `deslop` and `babysit` ship in this plugin). -- **`run_in_background: true`.** Claude Code's `Agent` tool supports this — kept as-is. -- **"currently open files, recent edits, the cursor location"** in `why/SKILL.md` (line 59). "Cursor location" here means editor cursor (caret position), not the IDE; generic phrasing, no edit. -- **`poteto-agent` subagent ID.** Plugin ships this agent; references stay. -- **Cursor's `/create-skill` writing style guidance referenced indirectly.** Pointed at `plugin-dev:skill-development` which covers the same ground in Claude Code. If you want stricter parity, also install Anthropic's `superpowers:writing-skills` skill. - -## Forking note - -This port now diverges from upstream pstack content. To track upstream: - -```bash -# diff against the pinned commit -diff -ru /tmp/pstack-src/pstack/skills/ skills/ # caveats: ignores the babysit/ and deslop/ dirs -``` - -If you want a clean re-port (e.g. when upstream releases v0.2.0), the rebuild recipe is: - -1. Copy upstream skills verbatim. -2. Re-apply the substitution table above (most of it is mechanical find/replace). -3. Re-add `skills/babysit/`, `commands/babysit.md`, and the cursor-team-kit `deslop` import. - -## Provenance - -- Upstream pstack: [cursor/plugins/pstack @ e46364b](https://github.com/cursor/plugins/tree/e46364b8be46000b7df0f260550cd712afbb8d36/pstack) — MIT, (c) 2026 Lauren Tan. -- Upstream deslop: [cursor/plugins/cursor-team-kit/skills/deslop @ e46364b](https://github.com/cursor/plugins/tree/e46364b8be46000b7df0f260550cd712afbb8d36/cursor-team-kit/skills/deslop) — MIT, (c) 2026 Cursor. -- babysit: independently authored; workflow informed by Cursor's public `/babysit` behavior — no code or prose copied. -- Inspected for prior-art decisions: [v1truv1us/ai-eng-system](https://github.com/v1truv1us/ai-eng-system) (namespaces pstack under `pstack/` but keeps Cursor refs intact); [Evan-Kim2028/agent-fleet](https://github.com/Evan-Kim2028/agent-fleet) (vendors pstack under `base-kit/pstack/`, same posture). +The latest imported source is [open-pstack v1.5.0](https://github.com/ericlitman/open-pstack/releases/tag/v1.5.0), which tracks Cursor pstack 0.15.5. Exact pins, substitutions, and sync procedure live in [UPSTREAM.md](UPSTREAM.md). Earlier change history is in Git. diff --git a/NOTICE.md b/NOTICE.md index a1c62890..0980f60c 100644 --- a/NOTICE.md +++ b/NOTICE.md @@ -1,64 +1,24 @@ -# NOTICE +# Attribution and licenses -This plugin is a port of upstream MIT-licensed work. All upstream copyright notices and license terms are preserved. The open-pstack history begins from `michael-denyer/pstack-claude` through proven import commit `053ed78732e3b71826933170eafe7f7782dda844`. +pstack-flex is a fork of [ericlitman/open-pstack](https://github.com/ericlitman/open-pstack), originally forked at v1.4.1, commit `de67e6b40511814171e5e4c8ad7af3b79f07c9ee`. open-pstack ports [Lauren Tan's pstack](https://github.com/cursor/plugins/tree/main/pstack) and retains the earlier [Michael Denyer pstack-claude](https://github.com/michael-denyer/pstack-claude) history through import commit `053ed78732e3b71826933170eafe7f7782dda844`. -## pstack-flex provenance +The latest imported open-pstack release is v1.5.0 at `77a91fd6f75483b971fa5cca4a88f1337f6099dd`, tracking Cursor pstack 0.15.5 at `12d587dfb20741cafc376c42c696c5f6e2a64487`. [UPSTREAM.md](UPSTREAM.md) records current adaptations and sync ownership. Git preserves the individual import commits. -This repository, **pstack-flex** (Martin Patino), is a fork of [ericlitman/open-pstack](https://github.com/ericlitman/open-pstack) at v1.4.1 (`de67e6b40511814171e5e4c8ad7af3b79f07c9ee`), which ports [Lauren Tan's pstack](https://github.com/cursor/plugins/tree/main/pstack) (Cursor) to Claude Code and Codex. Provenance chain: pstack-flex <- ericlitman/open-pstack <- cursor/plugins/pstack. All licenses remain MIT; every upstream license and notice file is preserved. The flex gateway providers, optional families, docs, and tests are (c) 2026 Martin Patino, MIT, and are inventoried in [UPSTREAM-FLEX.md](UPSTREAM-FLEX.md). +## Sources -## Upstream sources +| Component | Source | Copyright | License file | +| --- | --- | --- | --- | +| Shared pstack skills, principles, playbooks, agents, scripts, and assets | [Cursor pstack at the current content pin](https://github.com/cursor/plugins/tree/12d587dfb20741cafc376c42c696c5f6e2a64487/pstack) | 2026 Lauren Tan | [LICENSE](LICENSE) | +| `deslop`, `thermo-nuclear-code-quality-review`, `make-pr-easy-to-review`, `fix-ci`, `fix-merge-conflicts`, `get-pr-comments`, and `what-did-i-get-done` skills | [Cursor Team Kit at the import pin](https://github.com/cursor/plugins/tree/e46364b8be46000b7df0f260550cd712afbb8d36/cursor-team-kit/skills) | 2026 Cursor | [LICENSE-cursor-team-kit](LICENSE-cursor-team-kit) | +| `plugins/pstack/hooks/run-hook.cmd` | [Superpowers in the official Claude plugin repository](https://github.com/anthropics/claude-plugins-official/tree/main/plugins/superpowers), imported at 6.1.0, originally obra/superpowers | 2025 Jesse Vincent | [LICENSE-superpowers](LICENSE-superpowers) | +| pstack-flex modifications and additions | [thisguymartin/pstack-flex](https://github.com/thisguymartin/pstack-flex) | 2026 Martin Patino | [LICENSE](LICENSE) | -| Component | Upstream | Copyright | License | License file | -| --- | --- | --- | --- | --- | -| `plugins/pstack/skills/poteto-mode/`, `plugins/pstack/skills/architect/`, `plugins/pstack/skills/arena/`, `plugins/pstack/skills/automate-me/`, `plugins/pstack/skills/figure-it-out/`, `plugins/pstack/skills/how/`, `plugins/pstack/skills/interrogate/`, `plugins/pstack/skills/reflect/`, `plugins/pstack/skills/show-me-your-work/`, `plugins/pstack/skills/tdd/`, `plugins/pstack/skills/typescript-best-practices/`, `plugins/pstack/skills/unslop/`, `plugins/pstack/skills/why/`, `plugins/pstack/skills/principle-*/`, `plugins/pstack/agents/poteto-agent.md` | [cursor/plugins/pstack @ e46364b](https://github.com/cursor/plugins/tree/e46364b8be46000b7df0f260550cd712afbb8d36/pstack) | (c) 2026 Lauren Tan | MIT | [LICENSE](LICENSE) | -| `plugins/pstack/skills/deslop/` | [cursor/plugins/cursor-team-kit/skills/deslop @ e46364b](https://github.com/cursor/plugins/tree/e46364b8be46000b7df0f260550cd712afbb8d36/cursor-team-kit/skills/deslop) | (c) 2026 Cursor | MIT | [LICENSE-cursor-team-kit](LICENSE-cursor-team-kit) | -| `plugins/pstack/skills/thermo-nuclear-code-quality-review/` | [cursor/plugins/cursor-team-kit/skills/thermo-nuclear-code-quality-review @ e46364b](https://github.com/cursor/plugins/tree/e46364b8be46000b7df0f260550cd712afbb8d36/cursor-team-kit/skills/thermo-nuclear-code-quality-review) | (c) 2026 Cursor | MIT | [LICENSE-cursor-team-kit](LICENSE-cursor-team-kit) | -| `plugins/pstack/skills/make-pr-easy-to-review/` | [cursor/plugins/cursor-team-kit/skills/make-pr-easy-to-review @ e46364b](https://github.com/cursor/plugins/tree/e46364b8be46000b7df0f260550cd712afbb8d36/cursor-team-kit/skills/make-pr-easy-to-review) | (c) 2026 Cursor | MIT | [LICENSE-cursor-team-kit](LICENSE-cursor-team-kit) | -| `plugins/pstack/skills/fix-ci/` | [cursor/plugins/cursor-team-kit/skills/fix-ci @ e46364b](https://github.com/cursor/plugins/tree/e46364b8be46000b7df0f260550cd712afbb8d36/cursor-team-kit/skills/fix-ci) | (c) 2026 Cursor | MIT | [LICENSE-cursor-team-kit](LICENSE-cursor-team-kit) | -| `plugins/pstack/skills/fix-merge-conflicts/` | [cursor/plugins/cursor-team-kit/skills/fix-merge-conflicts @ e46364b](https://github.com/cursor/plugins/tree/e46364b8be46000b7df0f260550cd712afbb8d36/cursor-team-kit/skills/fix-merge-conflicts) | (c) 2026 Cursor | MIT | [LICENSE-cursor-team-kit](LICENSE-cursor-team-kit) | -| `plugins/pstack/skills/get-pr-comments/` | [cursor/plugins/cursor-team-kit/skills/get-pr-comments @ e46364b](https://github.com/cursor/plugins/tree/e46364b8be46000b7df0f260550cd712afbb8d36/cursor-team-kit/skills/get-pr-comments) | (c) 2026 Cursor | MIT | [LICENSE-cursor-team-kit](LICENSE-cursor-team-kit) | -| `plugins/pstack/hooks/run-hook.cmd` (near-verbatim) | [anthropics/claude-plugins-official → superpowers @ 6.1.0](https://github.com/anthropics/claude-plugins-official/tree/main/plugins/superpowers) (originally obra/superpowers) | (c) 2025 Jesse Vincent | MIT | [LICENSE-superpowers](LICENSE-superpowers) | -| `plugins/pstack/skills/what-did-i-get-done/` | [cursor/plugins/cursor-team-kit/skills/what-did-i-get-done @ e46364b](https://github.com/cursor/plugins/tree/e46364b8be46000b7df0f260550cd712afbb8d36/cursor-team-kit/skills/what-did-i-get-done) | (c) 2026 Cursor | MIT | [LICENSE-cursor-team-kit](LICENSE-cursor-team-kit) | -| `plugins/pstack/skills/teach/`, `plugins/pstack/skills/principle-model-the-domain/`, `plugins/pstack/skills/create-verification-skill/`, `plugins/pstack/skills/maintain-verification-skill/` (v0.11.3 additions) | [cursor/plugins/pstack @ 3fe2823](https://github.com/cursor/plugins/tree/3fe2823ce17c1656c222d4b7c59d3f82fbf20143/pstack) | (c) 2026 Lauren Tan | MIT | [LICENSE](LICENSE) | -| `plugins/pstack/skills/{swarm,no-comments,technical-writing,bro}/`, `plugins/pstack/agents/comment-sicko.md`, `plugins/pstack/skills/poteto-mode/playbooks/{babysit,shipping,orchestrate,autopilot-full,autopilot-stack,worktree-cleanup,multi-phase-plan}.md`, `plugins/pstack/skills/poteto-mode/references/bugbot-triage.md`, `plugins/pstack/skills/poteto-mode/scripts/`, `plugins/pstack/skills/architect/references/design-red-flags.md`, `plugins/pstack/skills/create-verification-skill/references/feature-map-example/` (v0.14.2 additions, v0.14.3 checklist) | [cursor/plugins/pstack @ bdf7aa3](https://github.com/cursor/plugins/tree/bdf7aa355337897f167153e05069aca505dae17c/pstack) | (c) 2026 Lauren Tan | MIT | [LICENSE](LICENSE) | -| `plugins/pstack/skills/poteto-mode/playbooks/{shipping,babysit,autopilot-full,autopilot-stack,opening-a-pr,multi-phase-plan}.md`, `plugins/pstack/skills/poteto-mode/references/bugbot-triage.md`, `plugins/pstack/skills/poteto-mode/SKILL.md`, `plugins/pstack/skills/typescript-best-practices/{SKILL.md,references/patterns.md}`, `plugins/pstack/assets/logo.png` (v0.14.6 and v0.14.7 changes) | [cursor/plugins/pstack @ efa2a53](https://github.com/cursor/plugins/tree/efa2a531985e0a8084d36ff3cf87233be8a9f34b/pstack) | (c) 2026 Lauren Tan | MIT | [LICENSE](LICENSE) | -| `plugins/pstack/skills/` (0.15.0 prose changes and the new `principle-attack-the-premise` and `principle-test-behavior-not-implementation` leaves), `plugins/pstack/assets/logo.png`, `README-UPSTREAM.md` | [cursor/plugins/pstack @ 71ed0d1](https://github.com/cursor/plugins/tree/71ed0d1076fec562c1b74ee353121a8d00f75382/pstack) | (c) 2026 Lauren Tan | MIT | [LICENSE](LICENSE) | -| `plugins/pstack/skills/poteto-mode/SKILL.md` (0.15.1 reply-writing evidence rule) | [cursor/plugins/pstack @ f8abedd](https://github.com/cursor/plugins/tree/f8abeddd1862dc73704e3d719dd73df0d51b8c71/pstack) | (c) 2026 Lauren Tan | MIT | [LICENSE](LICENSE) | -| `plugins/pstack/skills/` (0.15.2 to 0.15.5 changes: Grok 4.7 and Opus max matrix rows, code-ready rounds, owner authority, prompt cuts, decision-trail runs, `show-me-your-work/scripts/log.sh`), `README-UPSTREAM.md` | [cursor/plugins/pstack @ 12d587d](https://github.com/cursor/plugins/tree/12d587dfb20741cafc376c42c696c5f6e2a64487/pstack) | (c) 2026 Lauren Tan | MIT | [LICENSE](LICENSE) | +All listed licenses are MIT. The distribution preserves their notices and license terms. -## What changed in the port +## Adaptations and original work -The port is editorial, not mechanical. See [CHANGES.md](CHANGES.md) for the full per-skill audit of substitutions applied. +open-pstack adds Claude Code and Codex packaging, harness translation, provider dispatch, native Claude agent definitions, tests, and hook integration. Its bundled `babysit` skill is independently authored from Cursor's public workflow behavior, rather than copied from Cursor's built-in implementation. -Summary of structural changes: +pstack-flex changes model configuration and defaults, adds gateway and OpenCode execution, scopes model sheets, makes skill invocation opt-in, and adds a lane journal. It also adds `intake` and `diff-behavior`. These are maintained adaptations; upstream-derived skill bodies are not claimed to be verbatim copies. -- Plugin content lives at `plugins/pstack/` (with its own `.claude-plugin/plugin.json`). The repo root holds `.claude-plugin/marketplace.json` and the LICENSE / NOTICE / README / CHANGES docs. -- `.claude-plugin/marketplace.json` added at repo root so the repo is installable via `/plugin marketplace add`. The marketplace's single plugin entry sources from `./plugins/pstack`. -- The native `plugins/pstack/skills/` tree is the only user-facing workflow surface. Claude Code and Codex invoke those skills directly. -- Seven skills imported from `cursor-team-kit`: `deslop`, `thermo-nuclear-code-quality-review`, `make-pr-easy-to-review`, `fix-ci`, `fix-merge-conflicts`, `get-pr-comments`, `what-did-i-get-done`. All copied verbatim — no rewiring needed. -- `plugins/pstack/skills/babysit/` is independently authored as the Claude Code analog of Cursor's `/babysit` built-in. It has no upstream pstack equivalent; its workflow is informed by Cursor's public `/babysit` behavior. No code or prose was copied from any source. -- `plugins/pstack/skills/poteto-mode/scripts/` is vendored from upstream (`watch-pr`, `orch`, `bootstrap.ts`, `worktree-audit.sh`, `package.json`, `bun.lock`) with these port edits: `worktree-audit.sh` reads `~/.claude/projects/` instead of Cursor's transcript directory and warns when `jq` or `rg` is missing (their absence silently blanks the columns the prune decision reads), the private workspace package is named `@open-pstack/poteto-mode-tools`, `bootstrap.ts` rejects Node before it reads Bun-only APIs, and `package.json` includes the port-authored tests in `bun run test`. `check-plan.mjs` is the Cursor 0.14.3 checker adapted for the shared Claude Code and Codex skeleton. `bootstrap.test.ts` and `check-plan.test.ts` are authored for this port. -- `plugins/pstack/agents/comment-sicko.md` is upstream's `Comment Sicko` agent, renamed to `comment-sicko` so the name works as a Claude Code `subagent_type`. The body is verbatim. -- Claude-native Fable and Opus lanes are port-authored agent definitions. They select the rolling family alias plus requested effort for every selectable Claude-native pair in the provider-dispatch model matrix. -- A Codex build shares the same `skills/` tree. It adds `plugins/pstack/.codex-plugin/plugin.json`, a root `.agents/plugins/marketplace.json`, and `plugins/pstack/skills/poteto-mode/references/codex-tools.md` (the Claude-to-Codex tool, model, and built-in map), plus a one-line Platform note in the skills that name a Claude primitive. The skill content itself is unchanged. See [CHANGES.md](CHANGES.md#codex-port). - -## Modifications - -Per the MIT license, modifications are permitted. Skill bodies have been edited to substitute Cursor-specific primitives with their Claude Code equivalents (the full substitution table is in [CHANGES.md](CHANGES.md)). All upstream copyright notices in source files (where present) are preserved. - -Files authored for this port (not derived from upstream): - -- `plugins/pstack/.claude-plugin/plugin.json` -- `.claude-plugin/marketplace.json` (repo root) -- `plugins/pstack/.codex-plugin/plugin.json` -- `.agents/plugins/marketplace.json` (repo root) -- `plugins/pstack/skills/poteto-mode/references/codex-tools.md` -- `plugins/pstack/skills/poteto-mode/scripts/bootstrap.test.ts` -- `plugins/pstack/skills/poteto-mode/scripts/check-plan.test.ts` -- `plugins/pstack/skills/babysit/SKILL.md` (independently authored; workflow informed by Cursor's public `/babysit` behavior) -- `plugins/pstack/agents/pstack-fable-*.md` and `plugins/pstack/agents/pstack-opus-*.md` (Claude-native frontier lanes at each selectable effort) -- `plugins/pstack/hooks/hooks.json`, `plugins/pstack/hooks/session-start`, and `plugins/pstack/hooks/session-start-context.md` (the SessionStart hook and its opt-in gate) -- `NOTICE.md` (this file) -- `README.md` -- `CHANGES.md` -- `LICENSE-cursor-team-kit` (copied verbatim from upstream cursor-team-kit MIT) +Retain these notices and all three license files when redistributing the plugin. diff --git a/README-UPSTREAM.md b/README-UPSTREAM.md deleted file mode 100644 index b7cdef8a..00000000 --- a/README-UPSTREAM.md +++ /dev/null @@ -1,261 +0,0 @@ -# pstack - -i'm [poteto](https://x.com/poteto). i'm not a president or ceo, but i've worked with millions of lines of code at Meta, Netflix, and Cursor. i'm also on the react core team where i help build and maintain react compiler. - -there's a growing sense that ai writes too much slop code. i agree. i don't want to ship like a team of twenty slop artists. throughput without quality is not a goal i aspire to. if you want to go fast, go deep first. - -**pstack is my answer.** these are the same skills i use everyday to ship high quality code at Cursor. this turns cursor into a real engineering team. the goal is not to maximize loc, in fact it's the opposite. pstack helps you write less, but higher quality code. - -**pstack gives you fearless parallelism.** when you can go deep on one agent and trust it to write good, verifiable code, you can truly parallelize with confidence. start multiple agents up with `poteto-mode` and trust that they'll apply rigorous engineering principles to their work. - -**cursor gives you the best of all worlds.** every frontier model has its strengths and weaknesses. use any model with pstack. in fact, many of my skills use multi-model workflows to take advantage of each model's unique strengths. - -fork it. improve it. make it yours. PRs are welcome! - -## install - -```bash -/add-plugin pstack -``` - -## get started - -two steps: - -1. run [`/setup-pstack`](./skills/setup-pstack/SKILL.md), pick a reasoning budget, and choose which models you want. -2. use [`/poteto-mode`](./skills/poteto-mode/SKILL.md) whenever you're doing anything that requires rigor. - -new here? the [pstack guide](./docs/guide/README.md) walks you through a first real task, from setup and prompting through verification and overnight runs. - -that's it. the other skills are situational; the mode skill uses them for you as needed. out of the box the mode splits work by model strength: code delegates (feature, refactoring, bug fix, perf, hillclimb) go to grok, while the hardest changes, prose, and judgment go to opus 5.5. the default panel is opus 5.5 / sol / grok. [`/setup-pstack`](./skills/setup-pstack/SKILL.md) changes any of it. - -## usage - -use [`/poteto-mode`](./skills/poteto-mode/SKILL.md) at the start of a task. it reads your request, picks from a set of playbooks, and runs the other skills as the steps need them. - -### just use [`/poteto-mode`](./skills/poteto-mode/SKILL.md) - -this skill is the main shortcut. i use it whenever i need the agent to do rigorous engineering work. it comes with twenty-three playbooks: - -``` -/poteto-mode this pr has a subtle bug where the scroll drifts every 750ms even when idle. repro -first, then fix and verify. -``` - -``` -/poteto-mode i'm going to bed. land the stack even if ci flakes. i want everything merged by -morning. -``` - -
-the twenty-three playbooks - -| playbook | for | -|---|---| -| [investigation](./skills/poteto-mode/playbooks/investigation.md) | a read-only question. how does x work, why was y built this way, are we sure. | -| [bug fix](./skills/poteto-mode/playbooks/bug-fix.md) | reproduce a defect, root-cause it, and fix with runtime evidence. | -| [perf](./skills/poteto-mode/playbooks/perf-issue.md) | trace a measured slowness and improve it against a baseline. | -| [hillclimb](./skills/poteto-mode/playbooks/hillclimb.md) | sustained, scientific improvement of one metric against a target, looping hypotheses with before/after measurement and one commit per accepted win. | -| [runtime forensics](./skills/poteto-mode/playbooks/runtime-forensics.md) | diagnose a live symptom (leak, idle-cpu spin, glitch) from instrumentation. | -| [trace forensics](./skills/poteto-mode/playbooks/trace-forensics.md) | diagnose a captured profiling artifact (cpuprofile, trace, spindump, heap snapshot). | -| [feature](./skills/poteto-mode/playbooks/feature.md) | new or changed behavior, built from a named data shape. | -| [refactoring](./skills/poteto-mode/playbooks/refactoring.md) | a behavior-preserving change to structure or shape. | -| [prototype](./skills/poteto-mode/playbooks/prototype.md) | a throwaway sketch to make a design or behavioral decision cheaply, or to settle an empirical fork by observing it. | -| [visual parity](./skills/poteto-mode/playbooks/visual-parity.md) | pixel-exact ui equivalence between two implementations. | -| [authoring a skill](./skills/poteto-mode/playbooks/authoring-a-skill.md) | writing or editing a SKILL.md. | -| [eval](./skills/poteto-mode/playbooks/eval.md) | test how a skill or prompt change affects agent behavior, blinded. | -| [babysit](./skills/poteto-mode/playbooks/babysit.md) | drive a pr or a stack to merge-ready: conflicts, review threads, ci. | -| [shipping](./skills/poteto-mode/playbooks/shipping.md) | independently verify a green stack, then land the contiguous verified run bottom-up through github by default or origin when available. | -| [autonomous run](./skills/poteto-mode/playbooks/autonomous-run.md) | drive a long task to completion without stopping. | -| [orchestrate](./skills/poteto-mode/playbooks/orchestrate.md) | a standing project handed to one coordinator chat: multi-day, many stacked prs, fleets of subagents. | -| [autopilot-full](./skills/poteto-mode/playbooks/autopilot-full.md) | run independent prs to merged with one owner per pr and a root swarm verdict on each round, from the code-ready head on. | -| [autopilot-stack](./skills/poteto-mode/playbooks/autopilot-stack.md) | build and verify one linear base-branch stack for the operator to review and land. | -| [session pickup](./skills/poteto-mode/playbooks/session-pickup.md) | resume or take over a prior agent's in-flight work. | -| [pause safely](./skills/poteto-mode/playbooks/pause-safely.md) | suspend in-flight work cleanly so it can be resumed later. | -| [multi-phase plan](./skills/poteto-mode/playbooks/multi-phase-plan.md) | work that spans phases or stacked PRs. | -| [worktree cleanup](./skills/poteto-mode/playbooks/worktree-cleanup.md) | reclaim disk by pruning merged or abandoned worktrees and stale ios simulators, safety-gated. | -| [opening a pr](./skills/poteto-mode/playbooks/opening-a-pr.md) | open a ready pr from small ordered commits with a conventional commits title and a briefing-style body. invoked at the end of every other playbook. | - -
- - - -when invoked it: - -1. matches your task to a [playbook](./skills/poteto-mode/playbooks/) and opens a todo list whose first items are its steps, copied in verbatim. -2. routes to the other skills as the steps fire. -3. writes unslopped replies framed for the consumer and the maintainer. - -the full rules and playbooks live in [`skills/poteto-mode/SKILL.md`](./skills/poteto-mode/SKILL.md). - -[`/poteto-mode`](./skills/poteto-mode/SKILL.md) is also a sticky mode: once entered it stays on across turns, applying itself when a playbook matches or the task needs rigor and staying out of the way otherwise. opt out any time by saying so. - -[`/poteto-mode`](./skills/poteto-mode/SKILL.md) works extremely well with cursor's `/loop` command. you can make cursor work for many hours without sacrificing rigor. - -## skills - -[`/poteto-mode`](./skills/poteto-mode/SKILL.md) runs most of these for you when a step needs them (`how`, `why`, `architect`, `arena`, `swarm`, `interrogate`, `unslop`, `no-comments`, `technical-writing`, `tdd`, and the principles). the table below is for when you want one directly: - -``` -/how do we cancel runs? do we have an n+1 when we look up every run to cancel? -``` - -``` -/interrogate review this pr. -``` - -
-all skills - -| skill | use it when | -|---|---| -| [`/poteto-mode`](./skills/poteto-mode/SKILL.md) | default entry point for any non-trivial task. | -| [`/how`](./skills/how/SKILL.md) | you want a walkthrough of how a subsystem works. | -| [`/why`](./skills/why/SKILL.md) | you want to know why something was built this way. discovers available MCPs at run time and queries each evidence category in parallel (source control, issue tracker, long-form docs, real-time chat, infra observability, error tracking, analytics warehouse). | -| [`/recall`](./skills/recall/SKILL.md) | you're starting or resuming work and want your recent context on a topic rebuilt from your own chat history and the shared record, handed back as a tight current-state brief. | -| [`/blast-radius`](./skills/blast-radius/SKILL.md) | you have a small-looking change and want to know what else it could break, with the one fact it's safe because of proven by running code, not asserted. | -| [`/architect`](./skills/architect/SKILL.md) | you're about to write code that crosses a function boundary and want the caller's usage, types, and module shape settled first. | -| [`/arena`](./skills/arena/SKILL.md) | you want N parallel attempts at the same thing, then to grab the best parts of each. | -| [`/swarm`](./skills/swarm/SKILL.md) | you want N parallel workers across different slices or races, then one aggregated report. | -| [`/interrogate`](./skills/interrogate/SKILL.md) | you have a diff and want several different models to try to break it, including a strict code-quality lens. | -| [`/automate-me`](./skills/automate-me/SKILL.md) | you want your own `-mode` skill, drafted from how you've actually worked. | -| [`/make-bot-ui`](./skills/make-bot-ui/SKILL.md) | you want a page or dashboard whose buttons wake a Grok Bot over a webhook, including the sender-key handoff and Tailscale. | -| [`/setup-pstack`](./skills/setup-pstack/SKILL.md) | you want to pick which models pstack uses per role. detects your models and writes a config rule. | -| [`/reflect`](./skills/reflect/SKILL.md) | a long task landed and you want the recipe captured as a skill edit. | -| [`/teach`](./skills/teach/SKILL.md) | you want to actually understand a change or subsystem, not just have it summarized. runs how + why and weaves one plain explanation, built up diagram by diagram. | -| [`/tdd`](./skills/tdd/SKILL.md) | you're fixing a bug and there's a cheap local test path. write the failing test first, then the fix. | -| [`/no-comments`](./skills/no-comments/SKILL.md) | strip comments before review; spawns Comment Sicko, fixes accepted findings, offers encodings for claimed constraints. | -| [`/typescript-best-practices`](./skills/typescript-best-practices/SKILL.md) | you're reading or editing typescript. grounds the type-system-discipline principle in syntax. | -| [`/figure-it-out`](./skills/figure-it-out/SKILL.md) | no bundled playbook fits. designs a rigorous, auditable playbook for the task. | -| [`/show-me-your-work`](./skills/show-me-your-work/SKILL.md) | you want a reviewable decision trail. logs decisions to a tsv you can commit. | -| [`/create-verification-skill`](./skills/create-verification-skill/SKILL.md) | your project has no scripted way to prove app behavior. generates a project-local verify skill with a feature map, for any language or platform. | -| [`/maintain-verification-skill`](./skills/maintain-verification-skill/SKILL.md) | your verify skill's feature map has drifted from the app. source wave + one live pass, at most one PR of proven corrections. | -| [`/unslop`](./skills/unslop/SKILL.md) | you're cleaning up writing. removes AI tells. | -| [`/bro`](./skills/bro/SKILL.md) | you want the last message restated in plain human language, no jargon. | -| [`/technical-writing`](./skills/technical-writing/SKILL.md) | layered doc standard (Diátaxis + Google developer style + STE + Global English) for docs, RFCs, readmes, PR descriptions, commit messages. | - -
- - - -### examples - -mostly i type [`/poteto-mode`](./skills/poteto-mode/SKILL.md) at the start of a task and let it route to a playbook. the other skills fire as the steps need them. a few i reach for directly. - - -
-all the examples - -``` -bug fix: /poteto-mode this pr has a subtle bug where the scroll drifts every 750ms even - when idle. repro first, then fix and verify. -perf: /poteto-mode a big list takes a second or two to load even though we virtualize. - run a cpu trace and tell me why. -feature: /poteto-mode build a small feature behind a feature flag. verify it really works. -prototype: /poteto-mode build two prototypes of the markdown renderer so we can compare. - spawn an agent for each. -multi-phase: /poteto-mode open source these skills as a plugin. nothing internal leaks, work - in a temp dir, show me the dependency graph first. -overnight run: /poteto-mode i'm going to bed. land the stack even if ci flakes. i want - everything merged by morning. -babysit: /poteto-mode check on pr 123. anything outstanding? -visual parity: /poteto-mode the row spacing is too tall when this flag is on. the second image - is correct. repro and fix until it matches. -figure it out: /poteto-mode i'm stepping away. migrate every caller from the synchronous store - to the new async one, keeping behavior identical. i want to trust it was done - right when i'm back. -how: /how do we cancel runs? do we have an n+1 when we look up every run to cancel? -why: /why is this feature flag not on yet? -architect: design this instrumentation to be high signal with no false positives. /architect - this first. -arena: /arena take my prompt to the arena verbatim. i want to compare their proposals - with yours. -swarm: /swarm check every package under packages/ against its check.sh. one worker per - package. one report. -interrogate: /interrogate review this pr. -tdd: /tdd implement -unslop: can we unslop and tighten the new changes? -reflect: /reflect that took too long. capture what we learned so the next run doesn't - repeat it. -show-me-your-work: /show-me-your-work keep a decision trail i can review when i'm back. -automate-me: /automate-me -``` - -
- -## the `poteto-agent` and Comment Sicko subagents - -pstack also ships a subagent that runs my style end to end. spawn it from a parent agent via [`subagent_type: "poteto-agent"`](./agents/poteto-agent.md). it reads `poteto-mode` in full, including its inline principles index, before doing any work. substituting `generalPurpose` skips that read and drifts. - -[`/poteto-mode`](./skills/poteto-mode/SKILL.md) and [`subagent_type: "poteto-agent"`](./agents/poteto-agent.md) route through the same wrapper. - -pstack also ships [Comment Sicko](./agents/comment-sicko.md), a read-only comment reviewer available as `subagent_type: "Comment Sicko"`. usually invoke it through [`/no-comments`](./skills/no-comments/SKILL.md), not directly. - -## principles - -twenty-three short skills, one principle each. `poteto-mode` indexes them inline and reads that index at task start. the standalone files are there so other skills can reference a principle by name, and so the index can point at the full rule for each. - -
-all twenty-three principles - -| principle | group | rule | -|---|---|---| -| [laziness-protocol](./skills/principle-laziness-protocol/SKILL.md) | core | Bias toward deletion and the smallest change that solves the problem. | -| [foundational-thinking](./skills/principle-foundational-thinking/SKILL.md) | core | Apply before writing logic: choosing core types and data structures, sequencing scaffold-vs-feature work, asking what concurrent actors share. Get the data structures right so downstream code becomes obvious. | -| [redesign-from-first-principles](./skills/principle-redesign-from-first-principles/SKILL.md) | core | Redesign as if the requirement had been a foundational assumption from day one, instead of bolting it on. | -| [attack-the-premise](./skills/principle-attack-the-premise/SKILL.md) | core | Apply when two or more fixes that share one premise have failed the same gate. Take a census of which actors hold the imbalance before the next fix, then question the premise instead of writing another fix that assumes it. | -| [subtract-before-you-add](./skills/principle-subtract-before-you-add/SKILL.md) | core | Remove dead weight, redundant validators, and stub references first, then build on the simpler base. | -| [minimize-reader-load](./skills/principle-minimize-reader-load/SKILL.md) | core | Count layers between question and answer, and hidden state in the reader's head; collapse one-caller wrappers and shrink mutable scope. | -| [outcome-oriented-execution](./skills/principle-outcome-oriented-execution/SKILL.md) | core | Apply during planned rewrites and migrations with explicit phase boundaries. Converge on the target architecture; don't preserve smooth intermediate states with throwaway compatibility code. | -| [experience-first](./skills/principle-experience-first/SKILL.md) | core | Choose user delight over implementation convenience; ship fewer polished features over more rough ones. | -| [exhaust-the-design-space](./skills/principle-exhaust-the-design-space/SKILL.md) | core | Build 2-3 competing prototypes and compare side by side before committing. | -| [build-the-lever](./skills/principle-build-the-lever/SKILL.md) | core | Apply to any non-trivial work, not just bulk work: edits, migrations, analyses, checks. Build the tool that does it or proves it (codemod, script, generator, or a skill your subagents follow) instead of working by hand. The tool is the artifact a reviewer can rerun. | -| [model-the-domain](./skills/principle-model-the-domain/SKILL.md) | architecture | Encode the domain in a structure instead of scattered conditionals. | -| [boundary-discipline](./skills/principle-boundary-discipline/SKILL.md) | architecture | Concentrate guards at system boundaries (CLI, config, network, external APIs); trust internal types and keep business logic in pure functions. | -| [type-system-discipline](./skills/principle-type-system-discipline/SKILL.md) | architecture | Make illegal states unrepresentable, brand semantic primitives, parse external data at boundaries, refuse to lie to the compiler, exhaust variants, derive from authoritative schemas. | -| [make-operations-idempotent](./skills/principle-make-operations-idempotent/SKILL.md) | architecture | Converge to the same end state regardless of partial prior runs. | -| [migrate-callers-then-delete-legacy-apis](./skills/principle-migrate-callers-then-delete-legacy-apis/SKILL.md) | architecture | Migrate callers and delete the old API in the same wave instead of preserving compatibility layers. | -| [separate-before-serializing-shared-state](./skills/principle-separate-before-serializing-shared-state/SKILL.md) | architecture | Eliminate the sharing first; serialize structurally only when one shared writer is a real invariant. | -| [prove-it-works](./skills/principle-prove-it-works/SKILL.md) | verification | Apply after completing a task, before declaring done. Verify against the real artifact (run the feature, read the actual value, inspect the diff), not a proxy, self-report, or 'it compiles.'. | -| [fix-root-causes](./skills/principle-fix-root-causes/SKILL.md) | verification | Trace each symptom to its root cause and fix it there; reproduce first, ask why until you reach it, resist nil-check guards that silence crashes. | -| [sequence-verifiable-units](./skills/principle-sequence-verifiable-units/SKILL.md) | verification | Apply to multi-step work (sweeps, migrations, runs of similar edits) and to how you stack commits and PRs. Break work into small units that each end in a verifiable state, check each before the next, and order delivery so the sequence proves itself to a reviewer. | -| [test-behavior-not-implementation](./skills/principle-test-behavior-not-implementation/SKILL.md) | verification | Apply when you write, change, or keep a test. Call the code the way its users do and assert the result they observe against a literal expected value. If the test would still pass when every imported function returns undefined, rewrite the assertion or delete the test. | -| [guard-the-context-window](./skills/principle-guard-the-context-window/SKILL.md) | delegation | Route bulk to subagents; keep summaries in the main thread, not raw payloads. | -| [never-block-on-the-human](./skills/principle-never-block-on-the-human/SKILL.md) | delegation | Proceed, present the result, let the human course-correct after the fact; reserve confirmation for irreversible actions. | -| [encode-lessons-in-structure](./skills/principle-encode-lessons-in-structure/SKILL.md) | meta | Encode the rule as a lint, metadata flag, runtime check, or script instead of more text. | - -
- -## not shipped here - -a few things `poteto-mode` references but doesn't bundle: - -- `/deslop` and the `deslop` skill ship in the `cursor-team-kit` plugin. -- `control-cli` (for CLIs and TUIs) and `control-ui` (for browser, Electron, web) ship in `cursor-team-kit` too. -- `/create-skill` is a cursor built-in. cursor also ships a built-in `/babysit`; inside `poteto-mode`, the [babysit playbook](./skills/poteto-mode/playbooks/babysit.md) supersedes it for pr-status requests. - -install `cursor-team-kit` alongside pstack if you want the full set. - -## why are there no planning skills? - -cursor already has a great plan mode which works great with pstack. but personally, i don't believe in planning. the best spec is code. if you do want to make a plan, [`/poteto-mode`](./skills/poteto-mode/SKILL.md) covers it, but it's not a default. - -## make it yours - -`poteto-mode` is my style. you may not want exactly that. - -type [`/automate-me`](./skills/automate-me/SKILL.md). it mines your recent transcripts, drafts a `-mode` skill from how you've actually worked, and routes through pstack underneath. you keep pstack as the base and end up with your own routing skill alongside `poteto-mode`. - -models are configurable too. type [`/setup-pstack`](./skills/setup-pstack/SKILL.md). it detects the models you have access to and writes a small always-applied rule mapping each role (code, judgment, the review panels) to a model. every skill reads it and falls back to sensible defaults when the rule is absent, so you override only what you want. - -a rule written before 0.15.3 pins the old default models. delete those role lines, or delete the file, then run `/setup-pstack` again. a rerun keeps any role whose model differs from the default. - -## automations - -pstack also ships a dormant [benny automation pack](./automations/benny/). benny triages slack issue reports, then reproduces and fixes confirmed bugs with real ui evidence. its files are not registered as slash skills. - -to set it up, point cursor at [`FOR_AGENTS.md`](./automations/benny/FOR_AGENTS.md). setup copies the pack into the target repository at `.cursor/automations/benny/`, enables pstack there for shared skills, and keeps user configuration outside the copied pack. - -## license - -MIT diff --git a/README.md b/README.md index 14cc4ef3..490d81c0 100644 --- a/README.md +++ b/README.md @@ -1,84 +1,19 @@ # pstack-flex [![CI](https://github.com/thisguymartin/pstack-flex/actions/workflows/ci.yml/badge.svg)](https://github.com/thisguymartin/pstack-flex/actions/workflows/ci.yml) -[![Based on open-pstack v1.5.0](https://img.shields.io/badge/based%20on-open--pstack%20v1.5.0-blue)](https://github.com/ericlitman/open-pstack/releases/tag/v1.5.0) [![MIT license](https://img.shields.io/github/license/thisguymartin/pstack-flex)](LICENSE) -**pstack-flex runs [Lauren Tan (@poteto)](https://x.com/poteto)'s [pstack](https://github.com/cursor/plugins/tree/main/pstack) in Claude Code and Codex on the models you actually have.** It is built on [ericlitman/open-pstack](https://github.com/ericlitman/open-pstack), which translates pstack's Cursor-specific parts for Claude Code and Codex. open-pstack assumes four frontier subscriptions. pstack-flex keeps its skills and workflows, changes which models setup accepts and how they are reached, and adds its own skills. The agent monitor lives in its own repository, [psf-monitor](https://github.com/thisguymartin/psf-monitor). +pstack-flex is a portable version of [Lauren Tan's pstack](https://github.com/cursor/plugins/tree/main/pstack): one shared skill and workflow system, with adapters for coding agents and model providers. It builds on [open-pstack](https://github.com/ericlitman/open-pstack). Claude Code and Codex are supported; OpenCode is beta and has not passed the installed live gate. -Lauren built pstack from the skills she uses to ship code at Cursor. In a [55-minute interview with Denis Labelle](https://x.com/DenisLabelle/status/2091337807939706928), she says that she shipped 1,000 pull requests in one month after steadily improving how her agents work and verify their results. - -> If you want to go fast, go deep first. - -If Cursor is your main coding environment, use [Lauren's original pstack](https://github.com/cursor/plugins/tree/main/pstack). If you hold all four subscriptions and want the closest translation, use [open-pstack](https://github.com/ericlitman/open-pstack). If you want to pick your own models, pay per token where it makes sense, or run with no subscription at all, use this repository. - -## What pstack does - -pstack is a plugin for coding agents. It is not a new model or a hosted service. It gives your agent engineering rules, step-by-step workflows for different kinds of work, focused skills, and small local tools. - -The normal entry point is `poteto-mode`. You give it a task in plain language. It then: - -- reads the task and chooses a workflow that fits; -- learns how the current system works before changing it; -- compares designs when the choice matters; -- favors small, simple changes over extra machinery; -- asks several models to challenge important decisions when useful; -- runs the code and checks real behavior instead of stopping at "the tests pass"; and -- carries the work through review, continuous integration (CI), and a ready-to-merge pull request when asked. - -![How pstack routes a task through focused skills, real-app proof, and a review-ready pull request](assets/pstack-workflow.png) - -pstack does not ask you to trust an agent on day one. It helps the agent leave evidence you can inspect. Start with supervised work. Let it run more work in parallel only after its checks have earned that trust in your own repositories. - -## The models - -Every pstack role (who writes code, who explores, who sits on a review panel) maps to one family. A family is one `(provider, model)` pair with its own requested effort and its own live probe in setup. These are the families pstack-flex ships: - -| Family | Descriptor at default effort | Needs | First-run role | -| --- | --- | --- | --- | -| `fable` | `claude:fable@max` | Claude Code login | judgment, prose, explanation, hardest tasks, panels | -| `opus` | `claude:opus@max` | Claude Code login | panels | -| `astra` | `codex:gpt-6-astra@high` | Codex (ChatGPT) login | panels | -| `sol-6.1` | `codex:gpt-6.1-sol@high` | Codex (ChatGPT) login | feature, refactoring, bug-fix, perf-issue, hillclimb | -| `sol-6` | `codex:gpt-6-sol@high` | Codex (ChatGPT) login | none; selectable | -| `luna` | `codex:gpt-6-luna@high` | Codex (ChatGPT) login | how explorer, swarm workers | -| `sol` | `codex:gpt-5.6-sol@max` | Codex (ChatGPT) login | none; selectable | -| `grok` | `grok:grok-4.7@xhigh` | Grok CLI login | panels | -| `deepseek` | `deepseek:deepseek-flash@high` | `DEEPSEEK_API_KEY` | none; selectable | -| `deepseek-pro` | `deepseek:deepseek-v4-pro@high` | `DEEPSEEK_API_KEY` | none; selectable | -| `minimax` | `minimax:MiniMax-M3@high` | `MINIMAX_API_KEY` | none; selectable | -| `minimax-preview` | `minimax:MiniMax-M3.1-Flash-Preview@high` | `MINIMAX_API_KEY` (Token Plan) | none; selectable | -| `openrouter` | `openrouter:/@high`, any OpenRouter model | `OPENROUTER_API_KEY` | none; selectable | - -The default review panel is `claude:fable@max, codex:gpt-6-astra@high, grok:grok-4.7@xhigh, claude:opus@max`: four lanes across three providers. Architect sketches default to `codex:gpt-6-astra@high, claude:fable@max`. Any family can take any role. Panels must span at least two providers, and two models from one provider count as one, because the adversarial signal comes from model diversity. An OpenRouter lane counts as the lab that made its model. - -The DeepSeek, MiniMax, and OpenRouter lanes run the stock `claude` binary against an Anthropic-compatible endpoint with that provider's key, in an isolated config directory, with inherited Anthropic routing stripped. A lane refuses to start if it finds a claude.ai login in that directory, so a subscription credential can never reach a third-party endpoint. Their receipts keep real token usage but set `costUsd` to null (Claude Code prices at Anthropic rates); the price table is in [docs/LANES.md](docs/LANES.md). Anthropic does not support pointing Claude Code at non-Anthropic endpoints; use synthetic data for gateway testing and keep keys in your local environment. Through OpenRouter, a role can use any model in its catalog, such as `openrouter:moonshotai/kimi-k3@high`; setup's live probe on that model is the only gate. - -### How a role becomes a lane - -```mermaid -flowchart LR - S["pstack-models.md
role -> provider:model@effort"] --> P["Parent harness
(Claude Code or Codex)"] - P -->|"parent's own provider"| N["Native subagent
Agent / spawn_agent"] - P -->|"any other provider"| R["pstack-runner
one process per lane"] - R --> C1["codex CLI"] - R --> C2["grok CLI"] - R --> C3["claude CLI + env
DeepSeek, MiniMax, or OpenRouter endpoint"] - N --> O["Output + receipt
model, effort, tokens, status"] - C1 --> O - C2 --> O - C3 --> O -``` - -The parent resolves every route once, before fan-out. Children never detect the harness or pick a model. A lane that cannot start drops out with a named receipt; nothing substitutes a weaker model or invents a timeout. +Give `poteto-mode` a task. It chooses a playbook, investigates the current system, settles the design, makes the change, and verifies real behavior. Skills such as `architect`, `arena`, and `interrogate` can compare work across models. You choose each role's models in a local model sheet. ## Install -You need a current Claude Code or Codex installation and [Bun](https://bun.sh) for the lane runner. Sign in only to the CLIs whose plans you have (Claude Code, Codex, Grok), and export `DEEPSEEK_API_KEY`, `MINIMAX_API_KEY`, or `OPENROUTER_API_KEY` in the shell that starts your session for the gateway lanes. Any subset works, down to a zero-subscription setup on two keys. +Install [Bun](https://bun.sh) for the external lane runner. Install and authenticate only the provider CLIs you assign to roles. Gateway lanes read `DEEPSEEK_API_KEY`, `MINIMAX_API_KEY`, or `OPENROUTER_API_KEY` from the session's environment. Use synthetic data for gateway testing. ### Claude Code -Run these commands inside Claude Code: +Run inside Claude Code: ```text /plugin marketplace add thisguymartin/pstack-flex @@ -88,118 +23,78 @@ Run these commands inside Claude Code: ### Codex -Run these commands in your shell: +Run in your shell: ```shell codex plugin marketplace add thisguymartin/pstack-flex --ref main codex plugin add pstack@pstack-flex ``` -Turn on Codex subagents in `~/.codex/config.toml` so pstack can compare work in parallel: +Enable subagents in your Codex config, normally `~/.codex/config.toml`: ```toml [features] multi_agent = true ``` -Start a new Codex task after installation so it can discover the new skills and setting. - -## Get started +Start a new task after installation. The marketplace is `pstack-flex`; the plugin and skill namespace remain `pstack`. If an older installation uses `pstack@open-pstack`, check its source repository before removing it. Keep only one active `pstack` installation. -### 1. Set up the models +### OpenCode beta -In Claude Code, run: +Clone the repository: -```text -/pstack:setup-pstack +```shell +git clone https://github.com/thisguymartin/pstack-flex ~/src/pstack-flex ``` -In Codex, ask: +Add the checkout to your [OpenCode config](https://opencode.ai/docs/config/), normally `~/.config/opencode/opencode.jsonc`: -```text -Use pstack:setup-pstack to configure pstack. +```jsonc +{ + "skills": { "paths": ["~/src/pstack-flex/plugins/pstack/skills"] }, + "instructions": ["~/src/pstack-flex/plugins/pstack/hooks/session-start-context.md"] +} ``` -Setup is assignment-first. It shows the role map, asks which roles to change, asks one effort per assigned family, probes only those families with a real one-turn run, and writes nothing until every probe passes and you confirm. A fresh run proposes the defaults in the table above. An existing sheet keeps its assignments until you change a named role. - -Setup first asks which scope to configure. The global sheet lives at `~/.claude/pstack-models.md` (Claude Code) or `~/.codex/pstack-models.md` (Codex). A project sheet lives at `.claude/pstack-models.md` or `.codex/pstack-models.md` in the repository root. It replaces the global sheet for that project, stays out of git through `.git/info/exclude`, and starts from your global assignments. A project without one uses the global sheet; delete the project sheet to go back. Change either sheet by rerunning setup rather than editing it by hand, so every choice is probed before it is saved. - -A model sheet from an earlier release keeps its panel. To take the new defaults, delete those role lines and run setup again; setup fills missing roles from the defaults. A `grok:grok-4.6` entry keeps running until the next setup run asks you to replace it. +OpenCode loads skills under bare names such as `poteto-mode`. Every provider-qualified role uses the external runner. Only `inherit-parent` and `auto` use its native `task` tool. OpenCode writer lanes can edit their worktree but cannot run shell commands, Git commands, tests, or builds. A task requiring those tools must report the unsupported requirement. Headless CI use is unsupported. -### 2. Use poteto-mode - -Start any task that needs careful engineering with `poteto-mode`. +## Configure and run In Claude Code: ```text -/pstack:poteto-mode Add saved filters to search. Keep the design simple, verify it in the real app, and open a pull request. +/pstack:setup-pstack +/pstack:poteto-mode Add saved filters to search. Verify it in the app. ``` In Codex: ```text -Use pstack:poteto-mode. Add saved filters to search. Keep the design simple, verify it in the real app, and open a pull request. +Use pstack:setup-pstack to configure pstack. +Use pstack:poteto-mode. Add saved filters to search. Verify it in the app. ``` -For that feature, poteto-mode should first understand how search works today. It should decide how the data should be represented before writing code, implement the smallest complete version, run the feature the way a user would, review the result, and prepare the pull request. - -That is the main workflow. The other skills are there when poteto-mode needs them or when you want to call one directly. **[docs/USAGE.md](docs/USAGE.md)** is the longer walkthrough: three setup configurations (full frontier, hybrid saver, zero-subscription), copy-paste examples for the daily skills, how to read receipts, and troubleshooting. - -### 3. Watch your agents (optional) - -The agent monitor is a separate plugin, [psf-monitor](https://github.com/thisguymartin/psf-monitor). Install it next to pstack to watch each pstack session, the agents it spawned, and the external lanes pstack launched, live, and to cancel a running lane from the page. - -## Useful skills - -| Skill | Use it when | -| --- | --- | -| `how` | You want a clear explanation of how part of the system works. | -| `why` | You want evidence for why the system was built that way. | -| `architect` | A change crosses a function or module boundary and the design needs to be settled first. | -| `arena` | You want several complete attempts, followed by a comparison of their best parts. | -| `interrogate` | You want different models to try to break a design or diff. | -| `create-verification-skill` | Your project has no repeatable way for an agent to prove real behavior. | -| `maintain-verification-skill` | The project's verification instructions no longer match the product. | -| `babysit` | A pull request needs CI failures and review comments handled until it is ready. | -| `reflect` | A hard task is finished and its lessons should improve the next run. | -| `intake` | You have GitHub issues and want each turned into a ready-to-run brief with a playbook, an observable exit condition, and a worktree. | -| `diff-behavior` | You want to know what a change did from the outside: the same scenarios on trunk and head, with every unclaimed difference flagged. | - -Plugin skills include `pstack:` in their name. In Claude Code, invoke a native skill such as `/pstack:architect`. In Codex, ask for the skill, such as `Use pstack:architect for this design.` See the [technical reference](docs/reference.md) for the full list. - -## Cost - -Some workflows use one model. `architect`, `arena`, and `interrogate` run several in parallel. Subscription lanes spend that CLI's plan; gateway lanes bill per token on the lab's account. The cost playbook in [docs/USAGE.md](docs/USAGE.md#cost-playbook) shows where the gateway lanes pay off: high-volume code-writing roles on DeepSeek Flash, long-context reading on MiniMax M3, and one frontier lane plus two gateway lanes for a three-provider panel at a fraction of three subscriptions. Keep `judgment and prose` and `hardest tasks` on your strongest lane; they are the last roles to economize. pstack-flex never replaces a failed model with a cheaper one; a lane that fails is reported, not swapped. - -## Claude Code and Codex - -Both apps read the same pstack skills. Only the way they start those skills and models is different. - -| | Claude Code | Codex | -| --- | --- | --- | -| Start poteto-mode | Run `/pstack:poteto-mode` or ask for pstack by name. A small startup instruction keeps Claude from starting pstack skills on its own. | Ask for `pstack:poteto-mode` by name. Codex runs the same startup instruction, so pstack skills also wait for a request there. | -| Runs inside the app | Claude models stay inside Claude Code. | The Codex families stay inside Codex. | -| Other models | The Codex families and Grok run through their signed-in command-line tools. | Claude and Grok run through their signed-in command-line tools. | -| Gateway models | DeepSeek, MiniMax, and OpenRouter always run through the external runner with an isolated config directory, never as a native agent. | Same. | -| Skills and workflows | Shared with Codex. | Shared with Claude Code. | - -Grok, DeepSeek, MiniMax, and OpenRouter models can take part in a multi-model review. You cannot use any of them as the main app running pstack. +In OpenCode beta, ask for `setup-pstack`, then `poteto-mode` by their bare names. -## Upstream +Setup asks for global or project scope, role assignments, and one effort per assigned model family. It probes each assigned family and writes only after all probes pass and you confirm. Existing assignments stay until you change them. See the [model matrix and dispatch contract](plugins/pstack/skills/poteto-mode/references/provider-dispatch.md) for supported descriptors and defaults. -Lauren's [pstack guide](https://github.com/cursor/plugins/tree/main/pstack/docs/guide) walks through a real task, verification, and longer unattended runs. It uses Cursor's interface, but the ideas are the same. Use the translated skill invocations above in Claude Code or Codex. +Each app has its own global sheet. A project sheet replaces that app's global sheet for the whole repository, including its worktrees. See [sheet paths and overrides](docs/reference.md#model-sheets). Rerun setup to change assignments with a live probe. -This repository tracks two upstreams. [UPSTREAM.md](UPSTREAM.md) records the Cursor pstack commit open-pstack imported (0.15.5 at [`12d587d`](https://github.com/cursor/plugins/commit/12d587dfb20741cafc376c42c696c5f6e2a64487)) and how new pstack releases are brought over. [UPSTREAM-FLEX.md](UPSTREAM-FLEX.md) records the open-pstack fork point (v1.4.1), the last merged release (v1.5.0), which files this fork owns, and the merge procedure. The fork keeps every upstream skill body as-is except the default model descriptors; its own changes are the model matrix, the first-run sheet, the gateway providers in the runner, setup's assignment-first flow, and the docs. +pstack runs when you name it or keep a standing instruction for it. Once started, it can invoke the skills its workflow needs. A failed lane becomes a named dropout. The runner never substitutes a weaker model or adds an implicit timeout. External lanes use the selected CLI account or gateway key and can incur usage charges. -Also kept here: [the original README](README-UPSTREAM.md), unchanged; [the technical reference](docs/reference.md) for every skill and harness detail; [the change record](CHANGES.md); and [the attribution record](NOTICE.md). +## Find the right reference -## Contributing +- [Technical reference](docs/reference.md) covers configuration, runtime boundaries, receipts, and adding a harness. +- [Provider dispatch](plugins/pstack/skills/poteto-mode/references/provider-dispatch.md) owns model choices and lane execution. +- [Harness tools](plugins/pstack/skills/poteto-mode/references/codex-tools.md) maps shared workflows to each app's tools. +- [Upstream contract](UPSTREAM.md) records source pins, local ownership, and sync rules. +- [Live gate](docs/LIVE-GATE.md) defines installed verification before merge or release. +- [psf-monitor](https://github.com/thisguymartin/psf-monitor) is a separate optional plugin for watching external lanes. -Fixes for Claude Code or Codex, new lanes, and help bringing over new pstack releases are welcome. Search this repository's [GitHub Issues](https://github.com/thisguymartin/pstack-flex/issues) before opening a new one. For changes to upstream-derived content, explain why the change belongs here instead of in open-pstack or Lauren's original project. +## Contribute -Read [UPSTREAM.md](UPSTREAM.md) and [UPSTREAM-FLEX.md](UPSTREAM-FLEX.md) before changing content brought over from either upstream. Pull requests must keep one shared skill tree for Claude Code and Codex and pass the repository's tests, type checks, plugin validation, and static checks. Nothing merges until the exact candidate is installed and the changed behavior passes a live test from the real user surface in every affected harness; the [pull request template](.github/pull_request_template.md) records that evidence, and a PR without it stays a draft. Run `bash scripts/check.sh` for the local checks and follow [docs/LIVE-GATE.md](docs/LIVE-GATE.md) for the live test. Adding a gateway provider has its own checklist in [docs/LANES.md](docs/LANES.md#adding-a-gateway-provider). +Track durable work in this repository's [GitHub Issues](https://github.com/thisguymartin/pstack-flex/issues). Read [UPSTREAM.md](UPSTREAM.md) before editing upstream-derived content. Keep one shared skill tree, with tool translation and provider routing at their existing boundaries. -## License +Run `bash scripts/check.sh` before opening a PR. Nothing merges, tags, releases, or rolls out until the exact candidate passes from the real user surface in every affected harness. Record that evidence in the [PR template](.github/pull_request_template.md). A PR without it stays a draft. -MIT. pstack was created by Lauren Tan. open-pstack builds on Michael Denyer's [pstack-claude](https://github.com/michael-denyer/pstack-claude) port and includes attributed MIT-licensed work from Cursor Team Kit and Superpowers. pstack-flex started as a fork of open-pstack and is maintained as its own distribution. See [NOTICE.md](NOTICE.md) and the preserved license files for details. +MIT. pstack was created by Lauren Tan. The distribution also includes work from pstack-claude, Cursor Team Kit, and Superpowers. [NOTICE.md](NOTICE.md) records attribution; [CHANGES.md](CHANGES.md) records the fork's current changes. diff --git a/UPSTREAM-FLEX.md b/UPSTREAM-FLEX.md deleted file mode 100644 index 7140521b..00000000 --- a/UPSTREAM-FLEX.md +++ /dev/null @@ -1,62 +0,0 @@ -# Flex fork synchronization - -pstack-flex layers on top of open-pstack's own upstream tracking. Two sync relationships exist: - -1. `cursor/plugins/pstack` -> `ericlitman/open-pstack` — documented in [UPSTREAM.md](UPSTREAM.md), unchanged by this fork. -2. `ericlitman/open-pstack` -> `thisguymartin/pstack-flex` — this document. - -## Fork point - -| Source | Value | -| --- | --- | -| Repository | `https://github.com/ericlitman/open-pstack.git` | -| Tag | `v1.4.1` | -| Commit | `de67e6b40511814171e5e4c8ad7af3b79f07c9ee` | -| Tracks Cursor pstack | `0.15.1` (`f8abedd`) | - -## Last merge - -| Source | Value | -| --- | --- | -| Tag | `v1.5.0` | -| Commit | `77a91fd6f75483b971fa5cca4a88f1337f6099dd` | -| Tracks Cursor pstack | `0.15.5` (`12d587d`) | - -Not merged: open-pstack's `docs/plans/2026-10-01-triage-and-roadmap.md`, which plans open-pstack's own issue queue. Later upstream edits to that file come back as modify/delete conflicts; resolve them by deleting the file. - -The fork keeps full upstream history. The `upstream` remote points at ericlitman/open-pstack. - -## What the fork owns - -All flex changes are additive and live in port-owned files so upstream merges stay cheap: - -- `plugins/pstack/skills/poteto-mode/scripts/runner/flex-providers.ts` and `flex-providers.test.ts` (new) -- Gateway-provider hooks in `runner/{types,commands,run,parse-output,cli}.ts` and their tests -- The four GPT-6 rows in the stock model matrix, the "Default panel", "Default architect panel", and "Sheet scope" sections, the "Flex model matrix" section, and the route-table columns in `references/provider-dispatch.md` -- The first-run sheet in `skills/setup-pstack/SKILL.md` and the default descriptors named in `arena`, `architect`, `interrogate`, `how`, `swarm`, and the `feature`, `refactoring`, `bug-fix`, `perf-issue`, and `hillclimb` playbooks -- The assignment-first restructure of `skills/setup-pstack/SKILL.md`, and its project-or-global scope question, project sheet paths, and `.git/info/exclude` write -- `docs/LANES.md`, this file, the README fork section, and the NOTICE/LICENSE/CHANGES additions -- `plugins/pstack/hooks/session-start-context.md`, which the fork rewrote from open-pstack's auto-fire mandate into an opt-in gate, and the docs lines that describe it -- The `intake` and `diff-behavior` skills: `plugins/pstack/skills/intake/` and `plugins/pstack/skills/diff-behavior/`. Upstream has no equivalent, so they never conflict. -- The OpenRouter gateway: its `GATEWAY_SPECS` row and `openRouterModelRefusal` in `runner/flex-providers.ts`, the router check in `run.ts` `validateOptions`, the exact-match branch in `parse-output.ts` `reportedModelMatches`, the open `openrouter` row and its paragraph in the flex section, the lab rule in the panel-diversity paragraph, and the OpenRouter lines in `skills/setup-pstack/SKILL.md`. -- The lane journal: `runner/flex-journal.ts` and its test (new), its call sites in `runner/{types,run,cli}.ts` (an optional stdout callback on the model run, the journal opened after output reservation and finished on both return paths, and the `--label` flag), the journal tests in `run.test.ts`, and the pstack-flex paragraph after the invocation block in `references/provider-dispatch.md`. On a sync, keep these call sites; they change no receipt or exit status. - -Every upstream skill body is byte-unchanged except for the default-descriptor mentions listed above. Since [#17](https://github.com/thisguymartin/pstack-flex/issues/17), the stock matrix carries three fork-owned GPT-6 rows and the first-run sheet uses them, so those two surfaces conflict on every upstream sync and are resolved by hand: keep the fork's rows and defaults, take upstream's wording for everything else. - -## Merge procedure - -```shell -git fetch upstream -git switch -c merge-rehearsal -git merge --no-ff --no-commit upstream/main -# inspect, resolve, run the full local gate, then merge for real or abort -``` - -Expected conflict surface on future upstream releases: - -- `plugins/pstack/skills/setup-pstack/SKILL.md` — since 1.5.0 the fork uses upstream's step layout (role question in step 2, efforts in step 4). Take upstream's wording; keep the flex families, the GPT-6 guidance, the panel-diversity rule, and the fork's first-run sheet. -- `plugins/pstack/skills/poteto-mode/scripts/runner/model-matrix.test.ts` and `tests/skill-collision-repro.sh` — upstream reads its three-lane panel from the setup sheet; the fork reads its four-lane panel from the "Default panel" line. Take upstream's structural changes; keep the fork's panel source, GPT-6 rows, and flex-matrix checks. -- `plugins/pstack/hooks/session-start-context.md` — keep the fork's opt-in gate. If upstream changes its mandate, port only edits that still make sense for an opt-in gate, such as a renamed entry skill. -- `plugins/pstack/skills/poteto-mode/references/provider-dispatch.md` — the four upstream rows and their prose are upstream's; the GPT-6 rows, the "Default panel" section, and the flex section are fork-owned. Upstream's own panel changes land in the "Default panel" line only if the fork wants them. - -After every merge: run the full local gate (`bun install --frozen-lockfile`, `bun run test`, `bun run typecheck`, manifest JSON parse, `PSTACK_STATIC_ONLY=1 bash tests/skill-collision-repro.sh`), then record the installed version, action, and observed result for each affected harness in the pull request before tagging. diff --git a/UPSTREAM.md b/UPSTREAM.md index d6c051f9..f4812308 100644 --- a/UPSTREAM.md +++ b/UPSTREAM.md @@ -1,39 +1,76 @@ -# Upstream synchronization +# Upstream contract -open-pstack tracks [Cursor's pstack](https://github.com/cursor/plugins/tree/main/pstack) while adapting Cursor-specific primitives for Claude Code and Codex. +The content source is [Cursor's pstack](https://github.com/cursor/plugins/tree/main/pstack). [open-pstack](https://github.com/ericlitman/open-pstack) ports that content to Claude Code and Codex. pstack-flex keeps their history and owns the model configuration, additional providers, and harness adaptations below. -## Current sync point +## Source pins -| Source | Value | +| Relationship | Version | Commit | +| --- | --- | --- | +| Cursor content imported by open-pstack | pstack 0.15.5 | `12d587dfb20741cafc376c42c696c5f6e2a64487` | +| Latest open-pstack merge | v1.5.0 | `77a91fd6f75483b971fa5cca4a88f1337f6099dd` | +| pstack-flex fork point | open-pstack v1.4.1 | `de67e6b40511814171e5e4c8ad7af3b79f07c9ee` | + +Cursor, open-pstack, and pstack-flex versions identify separate distributions. The plugin manifests currently retain version `1.5.0`; use the installed commit to identify an unreleased candidate. [NOTICE.md](NOTICE.md) preserves attribution and license sources. + +## Ownership boundaries + +There is one `plugins/pstack/skills/` tree for Claude Code, Codex, and OpenCode beta. Keep workflow and principle changes shared. Translate tool names in `poteto-mode/references/codex-tools.md`, whose path stays stable for upstream pointers. Put provider models, dispatch rules, and receipts in `poteto-mode/references/provider-dispatch.md`. The parent resolves routes once; children never detect or reroute themselves. + +The fork owns these adaptations: + +| Area | Files or contract | +| --- | --- | +| Harness metadata and configuration | `poteto-mode/scripts/harnesses.ts`, `configuration.ts`, and `pstack-context`; scope paths, descriptor parsing, and explicit parent context | +| Model choices | Fork families and default panels in `provider-dispatch.md`; setup's first-run sheet and matching role defaults in skills and playbooks | +| Setup | `setup-pstack/SKILL.md`; scope selection, assignment-first questions, family probes, confirmation, and transactional writes | +| Provider execution | `poteto-mode/scripts/runner/`; gateway specs, OpenCode lanes, isolation, model evidence, and tests | +| Tool translation | `poteto-mode/references/codex-tools.md`; Claude-native workflows mapped to Codex and OpenCode tools | +| Invocation | `hooks/session-start-context.md`; pstack skills start on request or a standing instruction | +| Lane journal | `runner/flex-journal.ts` and call sites; optional journal without changing lane output or receipts | +| Fork skills | `skills/intake/` and `skills/diff-behavior/` | +| Distribution | Marketplace and plugin manifests, packaging checks, docs, and fork attribution | + +Paths beginning with `poteto-mode/` are relative to `plugins/pstack/skills/`. Runtime configuration work is tracked in [#3](https://github.com/thisguymartin/pstack-flex/issues/3), project scope in [#34](https://github.com/thisguymartin/pstack-flex/issues/34), and harness mapping in [#35](https://github.com/thisguymartin/pstack-flex/issues/35). [PR #33](https://github.com/thisguymartin/pstack-flex/pull/33) introduced the OpenCode parent and lane work. + +## Current substitutions + +These substitutions replace Cursor-specific operations without creating separate skill copies: + +| Cursor assumption | Shared port | | --- | --- | -| Repository | `https://github.com/cursor/plugins.git` | -| Path | `pstack/` | -| Commit | `12d587dfb20741cafc376c42c696c5f6e2a64487` | -| Upstream version | `0.15.5` | -| open-pstack version | `1.5.0` | +| `Task`, `generalPurpose`, and `readonly` | Parent-native subagents or an external lane, with explicit access mode and isolated writers | +| `AskQuestion` | The parent app's question tool | +| Built-in `loop`, `babysit`, and `create-skill` | Parent built-ins or bundled skills through the tool mapping | +| `control-cli` and `control-ui` | Parent CLI and browser drivers through the tool mapping | +| Cursor transcripts, skills, and model rules | Parent-specific paths; `pstack-models.md` for configured role assignments | +| Cursor MCP directory | Parent tool discovery and installed MCP configuration | +| Cloud agents | Local subagents or CLI processes, with worktrees for writers | +| Standing goal and orchestrator store | Standing instructions and the parent-specific durable run store | +| Upstream model defaults | Fork defaults from `provider-dispatch.md`, copied into setup and consuming roles | -The table above is the current Cursor sync point. Open Pstack 1.5.0 imports this 0.15.5 sync. `README-UPSTREAM.md` preserves the upstream pstack README verbatim. `CHANGES.md` and `NOTICE.md` describe the adaptations and provenance. +Same-provider Claude and Codex descriptors stay native. OpenCode beta uses the external runner for every qualified descriptor and native `task` only for `inherit-parent` or `auto`. Its writer lanes cannot run shell commands or tests. An unsupported task requirement must be reported explicitly. -## Upstream-only exclusions +## Deliberate exclusions -- Commits `799151d` and `6fecddb` add and relocate `make-bot-ui`. It depends on Cursor routines, webhook events, and UI primitives that Claude Code and Codex do not share. -- Four `disable-model-invocation: true` lines from `73f8be4` are not applied to `how`, `why`, `unslop`, or `typescript-best-practices`. Poteto-mode invokes those skills by name, and the flag blocks that route on Claude Code. -- The default-model hunks for `bug-fix`, `perf-issue`, and `hillclimb` from `23a56e2`, `889ec4b`, and `70b2dc8` are not applied. Those frequent code-writing roles stay on a Codex model (`codex:gpt-6.1-sol@high` in pstack-flex). -- `5bf2b15`'s setup budget question, its `# budget` line, and its step down to a lower detected effort are not applied. Setup already asks one requested effort per assigned family, and the step-down would silently lower a requested effort. -- `12d587d`'s rule that reruns a rejected configured entry on its family default or the closest valid slug is not applied. An unavailable model stays a named dropout per `provider-dispatch.md`. -- The expected-runtime column in `70b2dc8`'s `children.tsv` and its expected-runtime stuck test are not applied. A lane is stuck only on affirmative failure evidence. -- The explicit Grok, Opus, and Sol defaults for the Why and Reflect roles are not applied. Those roles stay on `inherit-parent` because the external runner omits the parent's MCP servers. -- The Claude manifest does not take the logo field from `efa2a53` because Claude Code has no schema for it. The shared asset is exposed through the Codex manifest instead. +- Cursor's `automations/benny/`, `make-bot-ui`, and sticky-mode metadata require Cursor's event or UI runtime. The Cursor usage guide stays [upstream](https://github.com/cursor/plugins/tree/main/pstack/docs/guide). +- Upstream planning documents describe its own issue queue. Do not import them or retain completed plans here; durable work belongs in this fork's GitHub Issues. +- `disable-model-invocation: true` from `73f8be4` is omitted for `how`, `why`, `unslop`, and `typescript-best-practices` because it blocks workflow invocation on Claude Code. +- Solo code defaults from `23a56e2`, `889ec4b`, and `70b2dc8` stay on the fork's configured Codex default. Why and Reflect defaults remain `inherit-parent` to retain parent MCP access. +- Setup's budget question and effort reduction from `5bf2b15` are omitted. Requested effort is never silently reduced. +- Configured-model fallback from `12d587d` is omitted. Unavailable models remain named dropouts. +- Expected-runtime stuck detection from `70b2dc8` is omitted. A stuck lane requires affirmative failure evidence; there is no implicit timeout. +- The logo field from `efa2a53` stays out of the Claude manifest, which has no matching schema field. The Codex manifest exposes the asset. -## Check for changes +## Sync changes -The repository already names Cursor's repository as the `cursor` remote in the maintainer checkout. A fresh clone can add it once: +For a fresh clone, add the source remotes once: ```shell git remote add cursor https://github.com/cursor/plugins.git +git remote add upstream https://github.com/ericlitman/open-pstack.git ``` -Fetch and inspect only commits that touched pstack after the recorded sync point: +Inspect Cursor content changes after the recorded pin: ```shell git fetch cursor main @@ -41,15 +78,18 @@ git log --oneline 12d587dfb20741cafc376c42c696c5f6e2a64487..cursor/main -- pstac git diff --stat 12d587dfb20741cafc376c42c696c5f6e2a64487..cursor/main -- pstack ``` -No output means the tracked pstack tree has not changed. This comparison does not need a polling service or generated mirror branch. +No output means the tracked content tree has not changed. Inspect the open-pstack merge separately: -## Incorporate a change +```shell +git fetch upstream +git log --oneline 77a91fd6f75483b971fa5cca4a88f1337f6099dd..upstream/main +``` -1. Create or update a GitHub issue in `thisguymartin/pstack-flex` and branch from current `main`. -2. Read each upstream pstack commit in order. Bring over its intent and content, then apply only the Claude Code and Codex substitutions documented in `CHANGES.md`. -3. Keep one shared `plugins/pstack/skills/` tree. Put harness translation in the existing `codex-tools.md` and provider routing in `provider-dispatch.md`; do not fork a skill per harness. -4. Update the commit and version in this file, the affected provenance rows in `NOTICE.md`, and `README-UPSTREAM.md` when upstream changes it. -5. Run CI-equivalent checks locally, then run the installed Claude Code and Codex behavioral lanes required by the changed surface. Unit tests alone are not a release gate. -6. Merge the reviewed PR before tagging the next open-pstack release. +1. Create or update a GitHub issue and branch from this fork's current `main`. +2. Read upstream commits in order. Bring over their intent while retaining the boundaries and exclusions above. +3. Reconcile setup, model defaults, dispatch, hook instructions, and their invariant tests by hand. Do not treat those files as unchanged upstream bodies. +4. Keep omitted upstream plans deleted. Update the source pins and attribution when imported sources change. +5. Run `bash scripts/check.sh`, install the exact candidate, and follow [the live gate](docs/LIVE-GATE.md) in every affected harness. +6. Record installed evidence in the PR. Merge before tagging only after the live gate passes. -Cursor's version and open-pstack's version are independent. Cursor's version identifies the imported content; open-pstack's version identifies the cross-harness distribution. +There is no generated README snapshot or parallel skill tree to refresh. Git retains prior content and the complete import history. diff --git a/docs/LANES.md b/docs/LANES.md deleted file mode 100644 index 36b6b7d8..00000000 --- a/docs/LANES.md +++ /dev/null @@ -1,175 +0,0 @@ -# Lanes: models, providers, and cost control - -pstack-flex's reason to exist: you choose which models run and what they cost. This document covers the lane concepts, the gateway environment reference, prices, the zero-subscription walkthrough, and the safety rules. - -Prices and endpoints below were verified 2026-09-25 and drift. Re-verify against each provider's own docs before relying on a number. - -## Lane kinds - -| Kind | Lanes | Auth | Billing | Route | -| --- | --- | --- | --- | --- | -| Subscription | `claude:fable`, `claude:opus`, `codex:gpt-6-astra`, `codex:gpt-6.1-sol`, `codex:gpt-6-sol`, `codex:gpt-6-luna`, `codex:gpt-5.6-sol`, `grok:grok-4.7` | each CLI's own login | that CLI's plan | native or external per the route table | -| Gateway (flex) | DeepSeek Flash / V4 Pro; MiniMax M3 / M3.1 Flash Preview; any OpenRouter model | API key in the environment | provider billing; preview requires Token Plan | always the external runner | - -A gateway lane is the stock `claude` binary env-pointed at the lab's Anthropic-compatible endpoint. There is no custom agent loop and no separate harness: the same runner that spawns Codex and Grok lanes spawns gateway lanes with injected environment. Each provider documents this Claude Code setup itself (DeepSeek: `deepseek-ai/awesome-deepseek-agent`, `docs/claude_code.md`; MiniMax: platform.minimax.io, Claude Code guide; OpenRouter: [Claude Code integration](https://openrouter.ai/docs/guides/guides/claude-code-integration)). - -## GPT-6 Codex families - -Three of the four stock GPT-6 Codex families carry the first-run defaults. They use the same ChatGPT login as `codex:gpt-5.6-sol`: - -| Family | Descriptor at default requested effort | First-run roles | Codex's own description | -| --- | --- | --- | --- | -| astra | `codex:gpt-6-astra@high` | every panel (`arena runners`, `arena cross-judge pool`, `interrogate reviewers`) and `architect runners`, where it pairs with Fable | Frontier tier for the most demanding work | -| sol-6.1 | `codex:gpt-6.1-sol@high` | `feature, refactoring`, `bug-fix`, `perf-issue`, `hillclimb` | Latest workhorse model for coding and everyday work | -| sol-6 | `codex:gpt-6-sol@high` | none; selectable | Previous generation workhorse model | -| luna | `codex:gpt-6-luna@high` | `how explorer`, `swarm workers` | Fast, low-cost tier for easier tasks | - -A fresh `/setup-pstack` run proposes these. An existing sheet keeps its assignments until you change a named role in setup; `codex:gpt-5.6-sol` remains a selectable family for that. The `sol-6.1`, `sol-6`, and `sol` families are separate, so each keeps its own effort. All Codex families count as one provider for panel diversity, so Astra plus GPT-6 Sol does not satisfy the two-provider rule; the default panel spans Claude, Codex, and Grok. The route matches Sol: native `spawn_agent` in a Codex parent, and the external runner (`codex exec`) in a Claude Code parent. Codex also lists an `ultra` effort for Astra, GPT-6.1 Sol, and GPT-6 Sol. It is outside the pstack effort universe and is not selectable. The descriptions and effort lists come from the Codex CLI 0.160.0 model list, checked 2026-10-04. - -## Multiple models per provider - -The flex matrix now includes four independently assignable model families: - -| Family | Descriptor at default requested effort | Selection guidance | -| --- | --- | --- | -| deepseek | `deepseek:deepseek-flash@high` | Existing everyday option | -| deepseek-pro | `deepseek:deepseek-v4-pro@high` | Candidate for difficult debugging, architecture, and review | -| minimax | `minimax:MiniMax-M3@high` | Existing MiniMax option | -| minimax-preview | `minimax:MiniMax-M3.1-Flash-Preview@high` | Preview coding option with tunable thinking | - -These are choices, not automatic replacements or a performance ranking. Existing sheets keep their assignments. In `/setup-pstack`, assign named roles to the desired model family; efforts and probes are independent per model, even for models sharing a key. Two models from one provider count as one provider for panel diversity. No runtime routing change or new configuration file is needed. - -As of 2026-09-27, [MiniMax's model guide](https://platform.minimax.io/docs/guides/models-intro) restricts M3.1 Flash Preview to Token Plan and MiniMax Code. For gateway access, supply the eligible Token Plan key as `MINIMAX_API_KEY`; the live probe must confirm entitlement. It is not a zero-subscription option. A working M3 call does not establish preview access. - -[MiniMax's Anthropic API](https://platform.minimax.io/docs/api-reference/text-anthropic-api) documents always-on thinking for the preview and `output_config.effort` from `low` to `max`. Higher effort increases thinking latency; the matrix proposes `high`, while the API defaults to `max` when omitted. M3 defaults to thinking off at the API and needs adaptive thinking to enable it. Its requested effort flag is not evidence of the preview's depth controls. Verify the installed CLI forwards the intended parameters; receipts prove requested effort, not hidden applied depth. [DeepSeek documents V4 Pro through its Anthropic endpoint](https://api-docs.deepseek.com/guides/anthropic_api). - -Before recommending a fastest or strongest default, compare the same synthetic coding tasks for correctness, completion time, tool-call reliability, token usage, and actual provider billing. Preview pricing and plan limits must be checked against the active plan rather than inferred from M3 rates. - -## OpenRouter: any model, one key - -One `OPENROUTER_API_KEY` reaches every model in [OpenRouter's catalog](https://openrouter.ai/models). There is no allowlist. Name any model ID with its namespace, and setup's live probe on that exact model is the gate: - -```text -openrouter:moonshotai/kimi-k3@high -openrouter:z-ai/glm-5.3@xhigh -openrouter:google/gemini-3.8-flash@low -``` - -`curl -s https://openrouter.ai/api/v1/models` lists the IDs without a key. Each distinct model ID is its own family with its own effort and probe. - -- **One refusal.** The runner refuses OpenRouter's own routers (`openrouter/auto`, `openrouter/free`, and the rest of the `openrouter/` namespace). They choose the model server-side, which would hide which model ran. Every model they could choose is reachable by its own ID. -- **Tools are required.** A lane is a Claude Code agent, so the model must support tool calls. Setup's OpenRouter probe reads its marker from a file, so a model without tool support fails there, not in a real lane. -- **What OpenRouter guarantees.** OpenRouter guarantees Claude Code only with Anthropic's first-party models. Other models work as far as their probe shows; [gateway-model-probes.md](gateway-model-probes.md) records the route evidence across labs. -- **Effort.** OpenRouter maps the requested effort onto each model's reasoning controls. A receipt proves the request, not the applied depth. -- **Model proof.** The reported model must match the requested ID exactly, apart from case. A sibling such as `z-ai/glm-5.3-air` fails a `z-ai/glm-5.3` lane. OpenRouter may fail over between hosts serving the same model; it does not swap the model unless you ask it to through a router or fallback list, which pstack never sends. -- **Context.** Claude Code does not know a third-party model's context window. When a model's window is small, a long lane can fail as it fills; `OPENROUTER_MAX_CONTEXT_TOKENS` sets the cap for every OpenRouter lane. -- **Panel diversity counts labs.** An OpenRouter lane counts as its model ID's namespace. `anthropic`, `openai`, `x-ai`, `deepseek`, and `minimax` match the `claude`, `codex`, `grok`, `deepseek`, and `minimax` providers. `openrouter:deepseek/deepseek-v4-pro` plus `deepseek:deepseek-flash` is one provider. -- **A rejected key is slow to fail.** Claude Code 2.1.289 retries a 401 for about three minutes before it gives up. The receipt then says `unauthenticated`. -- **Privacy and spend live on OpenRouter's dashboard.** Turn off data collection or require zero-data-retention hosts, and set a credit limit on the key. OpenRouter forwards each prompt to whichever host serves the model. - -## Gateway environment reference - -Set by you: - -| Variable | Required | Meaning | -| --- | --- | --- | -| `DEEPSEEK_API_KEY` / `MINIMAX_API_KEY` / `OPENROUTER_API_KEY` | yes, per lane | the lab's API key; the lane refuses to start without it | -| `DEEPSEEK_BASE_URL` / `MINIMAX_BASE_URL` / `OPENROUTER_BASE_URL` | no | endpoint override; defaults are in the flex model matrix. OpenRouter's must end in `/api`, not the `/api/v1` other tools use | -| `PSTACK_FLEX_DEEPSEEK_CONFIG_DIR` / `PSTACK_FLEX_MINIMAX_CONFIG_DIR` / `PSTACK_FLEX_OPENROUTER_CONFIG_DIR` | no | config-dir override; default `~/.pstack-flex/` | -| `DEEPSEEK_MAX_CONTEXT_TOKENS` / `MINIMAX_MAX_CONTEXT_TOKENS` / `OPENROUTER_MAX_CONTEXT_TOKENS` | no | context-cap override for the claude CLI | - -Injected by the runner at spawn time (never written to disk, never in receipts): `ANTHROPIC_BASE_URL`, `ANTHROPIC_AUTH_TOKEN`, the model pins (`ANTHROPIC_MODEL`, the opus/sonnet/haiku alias defaults, `CLAUDE_CODE_SUBAGENT_MODEL`), `CLAUDE_CODE_ATTRIBUTION_HEADER=0`, `CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC=1`, `CLAUDE_CONFIG_DIR`, and for OpenRouter an empty `ANTHROPIC_API_KEY`, as its guide requires. The runner first removes inherited `ANTHROPIC_*` values and Claude Code cloud-provider flags from the parent session. - -## Storing keys - -Keys reach a lane through the environment only; the runner never writes them to disk, receipts, or sheets. So key hygiene is entirely about how your shell gets them. Do not put raw keys in dotfiles or committed `.env` files. - -Recommended: your OS keychain, loaded on demand. - -- **macOS** (built in, encrypted at rest, unlocks with login): - - ```zsh - # once per key — prompts for the value, nothing lands in shell history - security add-generic-password -a "$USER" -s pstack-deepseek -w - security add-generic-password -a "$USER" -s pstack-minimax -w - security add-generic-password -a "$USER" -s pstack-openrouter -w - - # in .zshrc: a function, not an export — keys enter env only when called - pstack-keys() { - export DEEPSEEK_API_KEY=$(security find-generic-password -a "$USER" -s pstack-deepseek -w) - export MINIMAX_API_KEY=$(security find-generic-password -a "$USER" -s pstack-minimax -w) - export OPENROUTER_API_KEY=$(security find-generic-password -a "$USER" -s pstack-openrouter -w) - } - ``` - -- **Linux**: `pass` (GPG-encrypted, git-syncable) or `secret-tool` (libsecret) with the same load-on-demand function shape. -- **1Password CLI**: `op run --env-file=.env.tpl -- claude` injects the keys at process start with biometric unlock and exports nothing into the shell permanently. -- **direnv**: fine for per-project scoping (gateway lanes are per-project opt-in anyway), but a raw `.envrc` is plaintext — have it call the keychain instead of holding the key. - -Honest threat model: encryption at rest protects against dotfile repos, backups, and file theft. Once a key is in process env, any process running as your user can read it — the same exposure your CLI OAuth credential files already have. Keychain storage plus two ops controls is the right amount: **set spend caps on the DeepSeek, MiniMax, and OpenRouter dashboards** (on OpenRouter, a credit limit on the key) (the real blast-radius limiter) and rotate keys if a machine is ever compromised. - -## Prices (verified 2026-09-25 — re-check before budgeting) - -| Lane | Price per million tokens | Notes | -| --- | --- | --- | -| DeepSeek V4.1-Flash (`deepseek-flash`) | $0.30 in / $1.20 out peak; $0.15 / $0.60 off-peak; cache hits near-free | Off-peak windows: 01:00-04:00 and 06:00-10:00 UTC on weekdays. The discount is automatic on DeepSeek's side; pstack-flex surfaces the window but never delays your work to hit it. MIT open weights. | -| DeepSeek V4-Pro | $1.32 / $3.96 peak; half off-peak | Stronger model for hard lanes; assign it per role if wanted. | -| MiniMax M3 (`MiniMax-M3`) | $0.30 / $1.20 at up to 512K input; higher above | 1M context. Custom community model license (irrelevant for API use). | -| OpenRouter (any model) | the serving provider's price, passed through with no markup | OpenRouter charges 5.5% when you buy credits by card ([FAQ](https://openrouter.ai/docs/faq), checked 2026-10-05). Each model's page lists its price. | -| Claude / Codex / Grok subscription lanes | plan-dependent | Billed by each provider's plan, not per token here. | - -Gateway receipts always report `costUsd: null`: the claude CLI computes `total_cost_usd` at Anthropic list prices, which would be fiction for third-party traffic. Token usage in receipts is real — multiply it by the table above. - -## Zero-subscription walkthrough - -Goal: run poteto-mode and its panels with no Claude, ChatGPT, or Grok plan — only two API keys. The Claude Code binary is a free download; a subscription is only needed to reach Anthropic's servers. - -1. Install the claude CLI, Bun, and this plugin as usual. Do not run `claude login` anywhere in this setup. -2. Export `DEEPSEEK_API_KEY` and `MINIMAX_API_KEY`. -3. Make the parent session itself a DeepSeek session — same mechanism as a lane, applied to your interactive shell: - - ```shell - export ANTHROPIC_BASE_URL="" - export ANTHROPIC_AUTH_TOKEN="$DEEPSEEK_API_KEY" - export ANTHROPIC_MODEL="deepseek-flash" - export CLAUDE_CODE_SUBAGENT_MODEL="deepseek-flash" - export CLAUDE_CONFIG_DIR="$HOME/.pstack-flex/parent-deepseek" - claude - ``` - - Native `claude:*` lanes spawned by this parent inherit the endpoint, so the fable/opus role slots ride DeepSeek too. -4. Run `/setup-pstack`. Assign roles across the `deepseek` and `minimax` families (a `budget-duo` style panel), skip the unassigned stock families, and let the probes confirm both endpoints. -5. Panels keep real diversity: DeepSeek and MiniMax are two distinct providers, which satisfies the two-provider panel rule without any override. - -Quality note: this trades peak capability for cost control. The hardest-task role on a frontier subscription lane is a config choice you can add later without touching anything else. - -## Safety and policy - -- **Unsupported, not prohibited.** Anthropic's docs state that routing Claude Code to non-Claude models through gateways is not supported. No terms clause or enforcement against pointing the unmodified binary at a third-party endpoint was found (2026-09-25), but a CLI update can break compatibility without notice. Pin the claude CLI version on machines that depend on gateway lanes and bump it deliberately. -- **Never a claude.ai login on a gateway path.** Do not run `claude login` or `claude setup-token` inside any `~/.pstack-flex/` config dir. The runner enforces this: a gateway lane refuses to start when its config dir carries an OAuth credentials file. Caveat: on macOS the CLI may store credentials in the Keychain where the file check cannot see them — the rule above is the real defense; the check is a backstop. -- **Privacy: gateway lanes are opt-in per project.** Do not send client or customer code to third-party providers by default. Keep sensitive repositories on subscription lanes, and enable gateway lanes deliberately, per project. -- **No runner fallback.** A failed gateway lane is a named dropout receipt. A reported model mismatch fails the lane. If the endpoint reports no model, the receipt says `modelVerified: false` and `modelEvidence: "pinned-argv"`; this cannot prove which model the gateway served. Confirm supported model slugs during the live probe. - -## Optional lanes - -- **Local via Ollama (planned).** Ollama serves an Anthropic-compatible API since v0.14, so a `local` gateway provider pointed at it is the natural next lane: full compute control, zero per-token cost, your hardware. Not wired in yet. - -## Adding a gateway provider - -Any lab that serves an Anthropic-compatible `/v1/messages` endpoint can become a gateway lane. The runner, parser, and preflight branch on `isGatewayProvider`, so no `switch` needs a new case. - -1. Add the provider name to `GATEWAY_PROVIDERS` in `plugins/pstack/skills/poteto-mode/scripts/runner/types.ts`. -2. Add its row to `GATEWAY_SPECS` in `runner/flex-providers.ts`: API key variable, base URL default, override variables, and context-window default. Typecheck fails until this row exists. -3. Add its row to the "Flex model matrix" in `plugins/pstack/skills/poteto-mode/references/provider-dispatch.md`. `model-matrix.test.ts` fails until the key variable and base URL match the spec. -4. Add its probe row to the table in `plugins/pstack/skills/setup-pstack/SKILL.md`, its variables to the gateway environment reference above, and its prices to the price table. -5. Run the live validation checklist below for the new lane before merging. - -## Live validation checklist (before merge or rollout, real keys, never in CI) - -- V1: one DeepSeek probe through the runner (`--provider deepseek --model deepseek-flash --effort high`, read-only). Expect a `complete` receipt with `costUsd: null`; record the `reportedModel` string and confirm the base-URL default against DeepSeek's current guide; confirm `--effort` is accepted end-to-end. -- V2: same for MiniMax (`MiniMax-M3`); record the served-model casing. -- New-model gate: install the exact candidate and run `/setup-pstack` from both real Claude Code and Codex surfaces. Select Flash plus Pro and M3 plus Preview, verify independent efforts and probes, then run a read-only mixed panel. Record installed version/commit, surface, action, requested model/effort, served model, and observed result. Verify a failed preview entitlement probe leaves the sheet unchanged and does not select M3. A fake CLI regression test is not this gate. -- V3: run `claude auth status --json` inside a fresh flex config dir with `ANTHROPIC_AUTH_TOKEN` set and record the output here. On macOS, confirm whether `claude login` under an explicit `CLAUDE_CONFIG_DIR` writes `.credentials.json` or the Keychain. -- V4: the zero-subscription walkthrough above, end to end, on a machine with no stored provider logins. -- V5: OAuth guard live: `claude login` inside a scratch flex config dir, run a lane, confirm the refusal receipt, then delete that login. -- V6: OpenRouter route battery ([#7](https://github.com/thisguymartin/pstack-flex/issues/7)), scripted as `OPENROUTER_API_KEY=... bash scripts/probe-openrouter.sh [model ...]`, on models from at least four labs, read-only, synthetic workspace. For each: the exact ID answers; a tool call reads a file marker that is not in the prompt; a second turn uses that result; reasoning tokens differ between `low` and `high`; the receipt's reported model matches OpenRouter's activity log. Capture OpenRouter's real errors for a wrong ID, a bad key, no credits, and a model without tools. Record everything in [gateway-model-probes.md](gateway-model-probes.md). diff --git a/docs/LIVE-GATE.md b/docs/LIVE-GATE.md index 3592c5a8..0723e365 100644 --- a/docs/LIVE-GATE.md +++ b/docs/LIVE-GATE.md @@ -1,72 +1,72 @@ # Live gate -AGENTS.md says nothing merges, tags, releases, or rolls out until the exact candidate is installed and the changed behavior passes a live test from the real user surface in every affected harness. Unit tests, validators, and source reading do not count. This page is how to run that test and where to record it. +Nothing merges, tags, releases, or rolls out until the exact candidate is installed and the changed behavior passes from the real user surface in every affected harness. Unit tests, validators, source inspection, and agent self-reports do not satisfy this gate. A PR without installed evidence stays a draft. -There are two halves: +## Run the local gate -1. **Local gate.** One command, runs anywhere: `bash scripts/check.sh`. It installs both Bun packages, runs their tests and strict typechecks, parses every manifest, runs the static invariants, and runs `claude plugin validate` on the marketplace and both plugins when the `claude` CLI is present. -2. **Live gate.** You, in a real Claude Code and a real Codex session, with the candidate installed. Steps below. +Run `bash scripts/check.sh`. It installs the Bun package, runs tests and strict typechecks, parses manifests, and checks static invariants. It also runs Claude plugin validation when the CLI is installed. Record any skipped check rather than calling it passed. -## 1. Install the exact candidate +## Install the exact candidate -Push the branch first, so both harnesses install the same commit. +Publish the candidate branch so each app installs the same commit. Clone that branch for local marketplace or skill-path installation: -Claude Code: check out the branch, then add that checkout as the marketplace (inside a session): +```shell +git clone -b https://github.com/thisguymartin/pstack-flex ~/src/pstack-flex-candidate +git -C ~/src/pstack-flex-candidate rev-parse HEAD +``` + +In Claude Code, add that checkout and reload: ```text -! git clone -b https://github.com/thisguymartin/pstack-flex ~/src/pstack-flex-candidate /plugin marketplace add ~/src/pstack-flex-candidate /plugin install pstack@pstack-flex /reload-plugins ``` -Codex (shell): +For Codex, install the candidate branch from the shell: ```shell codex plugin marketplace add thisguymartin/pstack-flex --ref codex plugin add pstack@pstack-flex ``` -Start a new session in each harness afterwards. Record the installed version: the `pstack` version from the plugin list, plus `claude --version` and `codex --version`. +For OpenCode beta, point `skills.paths` and the opt-in `instructions` entry at the candidate checkout using the [README configuration](../README.md#opencode-beta). Keep the beta qualification until its installed behavior passes. -If you had open-pstack or an older pstack-flex installed under the `open-pstack` marketplace name, remove it first so only one plugin named `pstack` is active. +Start a fresh session in each affected app. Check the active plugin's source repository and installed commit; a matching version alone is insufficient. Keep only one active `pstack` installation. Record the plugin version or checkout commit, plus the app's CLI version. -## 2. Run the checks for what changed +## Exercise the changed behavior -Run the rows that match the change. A change that touches the runner or the model sheet runs rows A to C in both harnesses. +Use a scratch repository with synthetic data. Run the rows affected by the change. Runner, configuration, or setup changes require setup, scope, and dispatch checks in every affected parent. -| Row | Action | Pass when | +| Change | Action | Required observation | | --- | --- | --- | -| A. Opt-in gate | In a fresh session, ask for a two-line fix without naming pstack. Then ask again with "Use pstack for this." | The first request runs no `pstack:` skill. The second enters `pstack:poteto-mode`. | -| B. Setup | Run `/pstack:setup-pstack` (Claude Code) or `Use pstack:setup-pstack.` (Codex). Keep defaults or change one role. | Every assigned family probes `complete`, the sheet is written to `~/.claude/pstack-models.md` or `~/.codex/pstack-models.md`, and a failed probe writes nothing. | -| C. Mixed panel | `Use pstack:interrogate on the last commit.` | Each configured reviewer returns, external lanes write receipts with `status: complete`, and any missing CLI shows as a named dropout, not a substitute. | -| D. Gateway lane | With `DEEPSEEK_API_KEY`, `MINIMAX_API_KEY`, or `OPENROUTER_API_KEY` exported, assign one role to that family in setup (for OpenRouter, any model ID you name) and run it once. | The receipt shows `status: complete`, the requested model, and `costUsd: null`. | -| E. Lane journal | `mkdir -p ~/.pstack-flex/lanes`, run one external lane (for example an interrogate with a Codex reviewer from Claude Code), then `rm -rf ~/.pstack-flex/lanes`. With [psf-monitor](https://github.com/thisguymartin/psf-monitor) installed, watch the lane on its page instead. | While it runs, the lane's directory holds `lane.json` and a growing `stream.jsonl`; after it ends, `receipt.json` matches the runner's receipt. With the directory removed, the next lane writes nothing there and its receipt is unchanged. | -| F. intake | In a scratch repo with a real issue: `Use pstack:intake for #.` | A brief appears under `.pstack/intake/`, with one playbook, an observable exit condition, and open questions when the issue is vague. No product code changes. | -| G. diff-behavior | In a scratch repo, make a branch that changes one scenario on purpose and one by accident. `Use pstack:diff-behavior on this branch.` | The report lists the accidental change as unintended and the deliberate one as intended, with evidence for both sides. | +| Opt-in instruction | Request a small fix without naming pstack, then repeat with "Use pstack for this." | Only the second request enters `poteto-mode`. | +| Setup | Run `setup-pstack`, change one role and its effort, then repeat with an invalid model. | Assigned families probe successfully before writes. The invalid probe leaves sheets and integrations unchanged. An unchanged rerun preserves bytes. | +| Sheet scope | Configure different global and project roles, run from the primary checkout and a linked worktree, then inspect `pstack-context`. | Both worktrees select the same project sheet. A separate repository without one uses global scope. Config directory overrides select the expected paths. | +| Mixed panel | Run `interrogate` with qualified native and external roles, plus an inherited role. | Each route matches the context. External receipts name requested models and efforts. Missing providers become named dropouts without substitutions. | +| Gateway lane | Assign a gateway model in setup and run a task that reads a synthetic file marker absent from its prompt. | A tool call reads the marker. The receipt completes with the requested model and `costUsd: null`. Bad auth or model IDs fail without changing the sheet. | +| OpenCode beta parent | Run setup and a panel from a real OpenCode session with qualified and inherited roles. | Qualified roles run externally with parent `opencode`; inherited roles use native `task`. Scope and instructions update once. | +| OpenCode beta lane | Run a read-only lane and a writer lane, then request an unsupported effort and a task requiring tests. | Permissions hold. The invalid effort fails. The writer reports tests as unsupported and does not claim to run them. | +| Background and cancellation | Launch a lane that outlives the parent's foreground shell limit, then send SIGTERM to a throwaway lane. | The first returns a complete receipt after that limit. The cancelled lane writes a cancelled receipt. | +| Lane journal | Run with a temporary `PSTACK_FLEX_LANES_DIR` directory, then repeat with that directory absent. | The directory contains start metadata, growing stdout, and the matching receipt. Without it, the lane still completes without journal files. | +| `intake` | Run against a scratch repository's real issue. | A brief under `.pstack/intake/` names a playbook, observable exit condition, and unresolved questions. Product code stays unchanged. | +| `diff-behavior` | Compare branches with one intended and one accidental observable change. | The report identifies both changes with evidence from each branch. | -## 3. Record the evidence +New models also need a probe and a role run from each affected app. Gateway protocol changes need a file tool call, a second turn using its result, and explicit bad-key and bad-model results. `scripts/probe-openrouter.sh` supplies the OpenRouter route battery and spends real credit; it is outside CI. Record results in the tracking issue and PR rather than a separate probe archive. -Paste this into the pull request under "Live evidence", one block per harness: +## Record evidence in the PR + +Use one block per affected harness: ```text -Harness: Claude Code | Codex -Installed: pstack @ -Row : -Observed: -Result: pass | fail +Harness: +Installed: pstack @ +Surface: +Action: +Observed: +Result: pass | fail | not run ``` -A pull request without this stays a draft. - -## Outstanding live tests - -These merged without an installed live test. Clear them with one session per harness on current `main`, then record the results in a tracking issue and tick the rows. +Record unsupported requirements and failures directly. Do not convert a runner-only result into proof of parent behavior. Keep the PR draft until every required installed test passes. -| PR | Change | Rows to run | Status | -| --- | --- | --- | --- | -| #16 | GPT-6 families and first-run defaults | B, C | not run | -| #19, #20 | Opt-in gate (Claude Code and Codex) | A | not run | -| #22 | open-pstack 1.5.0 merge, setup step order, grok-4.7 pin | B, C | not run | -| #24, #25 | Lane journal (the monitor itself moved to psf-monitor in #28) | E | checkout build only, not installed | -| #26 | Rename to pstack-flex, intake, diff-behavior, doc fixes | A, B, F, G | not run | +Outstanding installed checks from earlier merged work are tracked in [#36](https://github.com/thisguymartin/pstack-flex/issues/36). Each candidate still needs its own evidence. diff --git a/docs/USAGE.md b/docs/USAGE.md deleted file mode 100644 index a6fb8301..00000000 --- a/docs/USAGE.md +++ /dev/null @@ -1,271 +0,0 @@ -# Using pstack-flex - -The walkthrough: what this plugin is, how work flows through it, how to set it up on the models you actually have, and copy-paste examples for the skills you will use daily. Lane mechanics and pricing live in [LANES.md](LANES.md); the fork's delta over upstream is in [UPSTREAM-FLEX.md](../UPSTREAM-FLEX.md). - -## What this is - -pstack is a plugin of engineering skills, playbooks, and small local tools for coding agents — not a model, not a service. You hand `poteto-mode` a task; it matches the task to a playbook, works the steps, and leaves evidence (diffs, runs, receipts) you can inspect instead of asking for trust. Its sharpest edge is multi-model adversarial review: several different model families challenge important work, because the adversarial signal comes from model diversity, not assigned personas. - -pstack-flex adds one thing on top: **you choose the models and the compute**. Any subset of families works, and two open labs — DeepSeek and MiniMax — are first-class lanes on plain API keys, down to a zero-subscription setup. With one OpenRouter key, a role can use any model in OpenRouter's catalog. - -If you also use my [thisguyskills](https://github.com/thisguymartin/skills) collection: that repo decides **what** to build (shaping, spec, Linear, handoff) and its handoff ends with "Use `pstack:poteto-mode`" — which is exactly where this repo picks up. - -## The big picture - -```mermaid -flowchart TD - T([Your task]) --> P["/pstack:poteto-mode"] - P --> PB[Playbook match
feature, bug-fix, refactoring, perf, ...] - PB --> S[Skills fire per step
how, tdd, interrogate, arena, ...] - S --> F{Lane fan-out} - F --> N1["claude:fable / claude:opus
native Agent (Claude sub)"] - F --> N2["codex:gpt-6-astra / gpt-6-sol / gpt-6-luna
native or codex CLI (ChatGPT sub)"] - F --> N3["grok:grok-4.7
grok CLI (Grok sub)"] - F --> G1["deepseek:deepseek-flash
runner + env -> DeepSeek API (key)"] - F --> G2["minimax:MiniMax-M3
runner + env -> MiniMax API (key)"] - F --> G3["openrouter:any/model
runner + env -> OpenRouter API (key)"] - N1 --> R[Outputs + receipts] - N2 --> R - N3 --> R - G1 --> R - G2 --> R - G3 --> R - R --> V[Verification: run it, judge it,
cross-model consensus] - V --> PR([Review-ready PR]) -``` - -Every lane is a real agent process with tools and file access. The parent harness (your Claude Code or Codex session) resolves the route once; children never pick their own models. - -## Install - -The marketplace is named `pstack-flex` and carries the `pstack` plugin. If you have Eric Litman's open-pstack installed, remove it first: both ship a plugin named `pstack`, so their skills would share the `pstack:` names. - -Claude Code: - -```text -/plugin marketplace add thisguymartin/pstack-flex -/plugin install pstack@pstack-flex -/reload-plugins -``` - -Codex: - -```shell -codex plugin marketplace add thisguymartin/pstack-flex --ref main -codex plugin add pstack@pstack-flex -``` - -Plus [Bun](https://bun.sh) for the lane runner, and `multi_agent = true` under `[features]` in `~/.codex/config.toml` if Codex is your parent. Sign in only to the CLIs whose subscriptions you actually have — missing families are fine now. - -## Keys for the gateway lanes - -DeepSeek, MiniMax, and OpenRouter have no login flow here; their lanes read an API key from your environment at spawn time. The runner never writes keys to disk or receipts, so the only question is how the env gets populated. Don't paste keys into `.zshrc` — store them encrypted and load on demand. macOS Keychain, built in and free: - -```zsh -# once: store each key (prompts for the value, nothing in shell history) -security add-generic-password -a "$USER" -s pstack-deepseek -w -security add-generic-password -a "$USER" -s pstack-minimax -w -security add-generic-password -a "$USER" -s pstack-openrouter -w - -# in .zshrc: a function, not an export — keys enter env only when you call it -pstack-keys() { - export DEEPSEEK_API_KEY=$(security find-generic-password -a "$USER" -s pstack-deepseek -w) - export MINIMAX_API_KEY=$(security find-generic-password -a "$USER" -s pstack-minimax -w) - export OPENROUTER_API_KEY=$(security find-generic-password -a "$USER" -s pstack-openrouter -w) -} -``` - -Daily flow: `pstack-keys -> claude -> /pstack:poteto-mode`. Alternatives, the threat model, and the spend-cap advice are in [LANES.md](LANES.md#storing-keys). Set spend caps on each provider dashboard (on OpenRouter, a credit limit on the key); that is the real blast-radius control. - -## First-time setup: /setup-pstack - -```text -/pstack:setup-pstack -``` - -(Codex: `Use pstack:setup-pstack to configure pstack.`) - -Setup is assignment-first: pick which roles run on which families, answer one effort question per **assigned** family, and only assigned families get probed. Unassigned families are skipped, not errors. Every probe is a real one-turn run — a failed probe writes nothing. Three configurations that make sense: - -**A. Full frontier** (Claude + ChatGPT + Grok subs) — accept the defaults. GPT-6.1 Sol writes code, Luna explores and verifies, and the panel spans three providers: - -```text -feature, refactoring: codex:gpt-6.1-sol@high -bug-fix: codex:gpt-6.1-sol@high -how explorer: codex:gpt-6-luna@high -swarm workers: codex:gpt-6-luna@high -arena runners: claude:fable@max, codex:gpt-6-astra@high, grok:grok-4.7@xhigh, claude:opus@max -``` - -**B. Hybrid saver** (Claude sub + two API keys) — frontier judgment, cheap volume: - -```text -feature, refactoring: deepseek:deepseek-flash@high -bug-fix: deepseek:deepseek-flash@high -judgment and prose: claude:fable@max -hardest tasks: claude:fable@max -swarm workers: deepseek:deepseek-flash@high -arena runners: claude:fable@max, deepseek:deepseek-flash@high, minimax:MiniMax-M3@high -interrogate reviewers: claude:fable@max, deepseek:deepseek-flash@high, minimax:MiniMax-M3@high -``` - -**C. Zero-subscription budget duo** (nothing but two keys) — start your parent session env-pointed at DeepSeek (walkthrough in [LANES.md](LANES.md#zero-subscription-walkthrough)), then assign everything across the two flex families: - -```text -arena runners: deepseek:deepseek-flash@high, minimax:MiniMax-M3@high -interrogate reviewers: deepseek:deepseek-flash@high, minimax:MiniMax-M3@high -``` - -Two labs are two distinct families, so panels keep real diversity without any override. A single-provider panel needs your explicit confirmation — by design. - -## Daily driving: the skills, with examples - -**poteto-mode** — the default entry point for any real task. pstack runs only when you ask for it, so start it by name. Once started, it stays sticky across turns and pairs well with long autonomous sessions. - -```text -/pstack:poteto-mode - -Take ENG-142: saved reports lose their date-range filter after rename. -Repro is in the issue. Fix it, prove it in the running app, and prep the PR. -``` - -**interrogate** — multi-model review of a decision, design, or diff. Reviewers come from different families; the parent sorts their findings. - -```text -/pstack:interrogate - -Review this migration plan in docs/plans/report-store.md. Attack the -premise, the rollout order, and anything that loses data on rollback. -``` - -```mermaid -flowchart LR - Q[Decision or diff] --> A[Reviewer A
family 1] - Q --> B[Reviewer B
family 2] - Q --> C[Reviewer C
family 3] - A --> S[Parent synthesizes] - B --> S - C --> S - S --> O["consensus (2+ models) -> act on
lone findings -> consider
disagreements -> resolve explicitly"] -``` - -**arena** — N parallel attempts at the same task, an independent cross-judge, then graft the best parts onto a base. - -```text -/pstack:arena - -Implement the rate limiter from the spec in docs/spec.md. Run the -configured arena panel and keep the winner's tests regardless of base. -``` - -```mermaid -flowchart LR - T[Task] --> C1[Candidate 1] - T --> C2[Candidate 2] - T --> C3[Candidate 3] - C1 --> J[Cross-judge
different provider] - C2 --> J - C3 --> J - J --> G[Pick base + graft
best pieces] -``` - -**swarm** — same-shaped work fanned across N workers, one combined report. Good for sweeps: "apply this codemod across packages," "audit every endpoint for X." - -```text -/pstack:swarm - -Audit every handler under src/api/ for missing input validation. -One worker per file group, combined findings ranked by severity. -``` - -**architect** — competing designs from different families, scored by a judge on yet another family, before any code. - -```text -/pstack:architect - -Design the offline sync layer: local-first edits, conflict policy, -and migration from the current always-online store. -``` - -Worth knowing by name: `how` (explain how something works before touching it), `why` (root-cause an incident with your MCP context), `tdd`, `unslop` (de-slop prose and code), `fix-ci`, `babysit` (drive a PR to green). The 23 `principle-*` leaves are loaded by poteto-mode as needed — you rarely invoke them directly. - -## What actually happens on a gateway lane - -No new harness. The same runner that launches Codex and Grok lanes spawns the stock `claude` binary with swapped environment: - -```mermaid -sequenceDiagram - participant P as Parent session - participant R as pstack-runner - participant C as claude -p (subprocess) - participant D as DeepSeek / MiniMax / OpenRouter API - P->>R: lane: deepseek:deepseek-flash@high - R->>R: guard: DEEPSEEK_API_KEY set?
config dir free of OAuth creds? - Note over R: refusal = unauthenticated receipt,
no subprocess ever spawned - R->>C: spawn with ANTHROPIC_BASE_URL,
ANTHROPIC_AUTH_TOKEN, isolated CLAUDE_CONFIG_DIR - C->>D: every model request in the agent loop - D-->>C: completions - C-->>R: JSON result - R-->>P: output file + receipt -``` - -The guard order matters: key check and OAuth check happen in-process **before** anything runs, so a claude.ai login can never be pointed at a third-party endpoint. Inherited `ANTHROPIC_*` values from your parent session are stripped before injection. - -## Reading receipts - -Every external lane writes a JSON receipt next to its output. The fields that matter: - -| Field | Meaning | -| --- | --- | -| `status` | `complete`, or a named dropout (`unauthenticated`, `unavailable-cli`, `timed-out`, ...) | -| `modelVerified` + `modelEvidence` | `provider-report` = the endpoint echoed the requested model (case-insensitive for gateways). `pinned-argv` = it didn't, but the argv pinned it — normal for Codex and sometimes gateways | -| `usage` | real token counts — trust these | -| `costUsd` | real for claude/grok subscription lanes; **always `null` on gateway lanes** (the CLI would price at Anthropic rates). Multiply `usage` by the [LANES.md](LANES.md) table instead | - -## Watching your agents - -The agent monitor moved to its own plugin, [psf-monitor](https://github.com/thisguymartin/psf-monitor). It draws each pstack session, the agents it spawned, and the external lanes pstack launched as a live graph, and can cancel a running lane. - -pstack's side is the lane journal. While `~/.pstack-flex/lanes/` exists, `pstack-runner` records each external lane's start, its output as it streams, and a copy of its receipt there. psf-monitor creates that directory when it starts. Delete the directory to stop journaling. A journal failure never changes a lane's receipt, exit status, or output. - -## Cost playbook - -- High-volume code-writing roles (`feature`, `bug-fix`, `swarm workers`) -> `deepseek:deepseek-flash` — cheapest tokens, near-free cache hits, and half price in the off-peak window. -- Long-context research and big-repo reading -> `minimax:MiniMax-M3` — 1M context. -- `judgment and prose` and `hardest tasks` -> your best frontier lane if you have one; this is the last role to economize. -- Panels: one frontier + two flex lanes gets you three-family diversity at a fraction of three subscriptions. - -## Troubleshooting - -| Symptom | Meaning | Fix | -| --- | --- | --- | -| Receipt `unauthenticated`, exit 77, "KEY is not set" | lane env missing | run your `pstack-keys` function (or export the key) in the shell that starts the parent | -| Receipt `unauthenticated`, "OAuth credentials found" | a claude.ai login sits in the lane's config dir | that's the leak guard working; remove the login from `~/.pstack-flex/` — never `claude login` there | -| Receipt `unauthenticated` after the model ran | the endpoint rejected the key (401) | check the key and the base URL against the provider's current guide | -| Exit 69 `unavailable-cli` | the `claude` binary isn't on PATH for the runner | install it or fix PATH | -| A panel ran with fewer lanes than configured | a lane dropped out with a named receipt | read that receipt; pstack proceeds N-1 and never silently substitutes a model | -| Everything gateway broke after a claude CLI update | Anthropic doesn't support third-party endpoints; compatibility can shift | pin the CLI version on machines that depend on gateway lanes; see [LANES.md](LANES.md#safety-and-policy) | - -## The GPT-6 Codex models - -`astra` (`codex:gpt-6-astra@high`), `sol-6.1` (`codex:gpt-6.1-sol@high`), and `luna` (`codex:gpt-6-luna@high`) are stock families and the first-run defaults: Astra on every panel and, with Fable, on architect sketches; GPT-6.1 Sol on the solo code-writing roles; Luna on exploration and swarm work. They need only your Codex login. Each gets its own effort question and live probe. See [GPT-6 Codex families](LANES.md#gpt-6-codex-families). - -A sheet written before this release keeps its assignments. To move a role, run `/setup-pstack` and name it; every role you do not change keeps its descriptor, and `codex:gpt-5.6-sol` stays selectable. For example, this row keeps GPT-5.6 Sol on bug fixes while the rest of the sheet takes the new defaults: - -```text -bug-fix: codex:gpt-5.6-sol@max -``` - -## Selecting the additional gateway models - -Run `/setup-pstack` and assign `deepseek-pro` (`deepseek:deepseek-v4-pro@high`) or `minimax-preview` (`minimax:MiniMax-M3.1-Flash-Preview@high`) to named roles. Existing `deepseek` and `minimax` choices remain available. Each model has its own effort selection and live probe. MiniMax preview requires an eligible Token Plan key in `MINIMAX_API_KEY`; see [model choices and thinking controls](LANES.md#multiple-models-per-provider). No existing assignment changes until setup succeeds and you confirm the rendered sheet. - -## Using any model through OpenRouter - -Export `OPENROUTER_API_KEY`, run `/setup-pstack`, and give a role any model ID from [OpenRouter's catalog](https://openrouter.ai/models), written with its namespace: - -```text -interrogate reviewers: claude:fable@max, openrouter:moonshotai/kimi-k3@high, openrouter:z-ai/glm-5.3@high -``` - -There is no list to pick from. Setup probes the exact model you name, with a marker the model has to read from a file, and writes nothing if the probe fails. The only refused IDs are OpenRouter's own routers (`openrouter/auto`, `openrouter/free`), because they choose the model for you. Panel diversity counts the lab behind the model, so the sheet above spans three providers. Turn off data collection on OpenRouter's privacy settings, or require zero-data-retention hosts, before sending real code. See [LANES.md](LANES.md#openrouter-any-model-one-key). diff --git a/docs/gateway-model-probes.md b/docs/gateway-model-probes.md deleted file mode 100644 index 138ddfec..00000000 --- a/docs/gateway-model-probes.md +++ /dev/null @@ -1,57 +0,0 @@ -# Gateway model probe evidence - -Tracking: [issue #5](https://github.com/thisguymartin/pstack-flex/issues/5). - -## Candidate and scope - -- Source candidate: branch `flex/multiple-gateway-models`, based on `92dc0bc`, with uncommitted implementation changes. -- Implementation diff SHA-256 before this evidence file: `84b4c3ee7fc66902532e1d457048a487e6a63629b31d2d214e52585f72863c75`. -- Packaged version: 1.4.1. This candidate has not been installed as a plugin. -- Actual parent: Codex session, invoking the candidate's external runner with `--parent codex`. -- CLI: Claude Code 2.1.283. -- Each probe used `--effort high`, read-only mode, a separate empty synthetic workspace and isolated Claude configuration, and a synthetic text file. No repository or customer data was used in the prompt. -- Keys were supplied through hidden terminal input, injected into child environments, and were not included in commands, this repository, or evidence below. - -## Observed results - -Each probe exited 0, returned the exact requested marker, recorded `modelVerified: true` with `modelEvidence: provider-report`, and retained `costUsd: null`. - -| Requested model | Reported model | Elapsed milliseconds | Receipt status | -| --- | --- | --- | --- | -| `deepseek-flash` | `deepseek-flash` | 2684 | `complete` | -| `deepseek-v4-pro` | `deepseek-v4-pro` | 7737 | `complete` | -| `MiniMax-M3` | `MiniMax-M3` | 11416 | `complete` | -| `MiniMax-M3.1-Flash-Preview` | `MiniMax-M3.1-Flash-Preview` | 5090 | `complete` | - -These single short probes establish authentication, model selection, and successful completion through the runner. They do not rank coding quality or speed, prove hidden reasoning depth, or verify CLI request-body effort forwarding. The prompt included the expected marker, so completion does not independently prove a file tool was used. - -## Remaining release gate - -Install the exact candidate and run setup from both real Claude Code and Codex user surfaces. Verify independent model effort choices, per-model probes, mixed-provider panels, saved-sheet readback, and unchanged configuration on failed access. Record installed version, surface, action, and observed result before merge or rollout. Changing only the runner's `--parent` flag would not satisfy this gate. - -## OpenRouter - -Tracking: [issue #7](https://github.com/thisguymartin/pstack-flex/issues/7) and draft PR [#31](https://github.com/thisguymartin/pstack-flex/pull/31). The battery is `scripts/probe-openrouter.sh` (V6 in [LANES.md](LANES.md#live-validation-checklist-before-merge-or-rollout-real-keys-never-in-ci)). - -### Dry run with an invalid key (2026-10-05) - -- Candidate: branch `feat/openrouter-gateway` at `f6bad90`, run from source, not installed. Parent: a Claude Code session, runner invoked with `--parent claude`. CLI: Claude Code 2.1.289. -- Action: `OPENROUTER_API_KEY=sk-or-v1-invalid-dryrun bash scripts/probe-openrouter.sh z-ai/glm-5.3`. It was stopped after the three per-model lanes. - -| Case | Effort | Exit | Elapsed ms | Claude Code result | -| --- | --- | --- | --- | --- | -| chain | high | 1 | 189180 | `api_error_status: 401`, `"Failed to authenticate. API Error: 401 User not found."`, `duration_api_ms: 0` | -| low | low | 1 | 182891 | same | -| high | high | 1 | 178841 | same | - -Findings: - -- The wrong key reached OpenRouter as the bearer token and was refused with 401. No other credential was used. -- Claude Code retries the 401 for about three minutes before it exits. -- The runner labelled these lanes `child-failed`, because its pattern matched "authentication" but not "Failed to authenticate". Fixed on the branch: Claude Code's 401 result now classifies as `unauthenticated`. -- Claude Code logs `[claude-code:unrecognized_model]` for the OpenRouter ID and still sends the request. -- Claude Code reports `usage.output_tokens_details.thinking_tokens`. The runner now records it as `reasoningTokens`, which the battery's effort check reads. - -### Route battery with a real key - -Pending. diff --git a/docs/plans/2026-10-05-openrouter-gateway.md b/docs/plans/2026-10-05-openrouter-gateway.md deleted file mode 100644 index 90b58127..00000000 --- a/docs/plans/2026-10-05-openrouter-gateway.md +++ /dev/null @@ -1,174 +0,0 @@ -# Plan: OpenRouter gateway lanes, any model (issue #7) - -Status (2026-10-05): Steps 1, 3, and 4 are implemented on `feat/openrouter-gateway`. Step 2 (live route probes) and the live gate are pending. - -## Context - -Goal: assign any pstack role to **any model OpenRouter serves** through one OpenRouter key, and know that the model you picked is the one that ran. There is no allowlist. The setup probe on the model you pick is the only gate. If the model can't run, the lane fails loudly and nothing silently swaps to another model. - -**Which harness? None new.** An OpenRouter lane is the stock Claude Code CLI (`claude -p`), pointed at OpenRouter's Anthropic-compatible endpoint through environment variables. DeepSeek and MiniMax already run this way (the "gateway" path in `pstack-runner`). The parent harness stays Claude Code or Codex. No proxy, no OpenCode, no custom agent loop, no new runner. - -**One exclusion: OpenRouter's own router IDs** (`openrouter/auto`, `openrouter/free`, and the rest of the `openrouter/` namespace). They pick the model for you, which breaks the project rule against automatic model selection and fallback. Every model a router could pick is reachable directly by its own ID, so no model is lost. - -## Diagram 1: what runs where - -``` -Parent harness: Claude Code or Codex (unchanged) - model sheet → arena runners: claude:fable@max, openrouter:z-ai/glm-5.3@high - │ argv: --provider openrouter --model z-ai/glm-5.3 --effort high --mode read-only - ▼ -pstack-runner (existing gateway path) - 1 validate any namespaced OpenRouter ID; only openrouter/* routers refused - 2 guard OPENROUTER_API_KEY set? ~/.pstack-flex/openrouter free of claude.ai OAuth? - 3 env strip inherited ANTHROPIC_* → inject OpenRouter URL, token, model pins - 4 preflight claude --version - │ spawn - ▼ -Child harness: stock `claude -p` (Claude Code CLI, headless) - --model z-ai/glm-5.3 --effort high --permission-mode plan --output-format json - ANTHROPIC_BASE_URL=https://openrouter.ai/api ANTHROPIC_AUTH_TOKEN=$OPENROUTER_API_KEY - ANTHROPIC_API_KEY="" CLAUDE_CONFIG_DIR=~/.pstack-flex/openrouter - │ Anthropic Messages API - ▼ -OpenRouter /api/v1/messages (normalizes --effort into each model's reasoning setting) - account: no data collection (or ZDR-only hosts), credit limit on the key - may fail over between hosts of the SAME model; never swaps the model - ├──▶ Google ├──▶ Z.ai ├──▶ Moonshot ├──▶ Qwen ├──▶ … any of ~460 models - ▼ -pstack-runner - 5 parse claude JSON → text + token usage; costUsd = null - 6 prove reported model == requested (exact) → else the lane fails, no fallback - 7 receipt → parent drains it like any other lane -``` - -## Diagram 2: picking a model in /setup-pstack - -``` -operator: "use moonshotai/kimi-k3 for interrogate reviewers" - │ - ▼ -descriptor openrouter:moonshotai/kimi-k3@high - (each distinct OpenRouter model = its own family: own effort, own probe) - │ - ▼ -probe runner, read-only, one turn; the marker is in a FILE, not the prompt - → proves the model answers AND makes a tool call - │ - ├─ fail ─▶ named error, sheet unchanged - │ wrong ID / no endpoints → unavailable-model - │ bad key → unauthenticated - │ no credits / no tools → child-failed, with OpenRouter's message - ▼ pass -diversity lab = the ID's namespace (moonshotai); anthropic/openai/x-ai/deepseek/minimax - count as the same lab as claude/codex/grok/deepseek/minimax lanes - │ - ▼ -confirm → sheet written -``` - -## Operator flow (after it ships) - -1. Store the key in the keychain; `pstack-keys` exports `OPENROUTER_API_KEY` (same pattern as the DeepSeek/MiniMax keys in `docs/LANES.md`). -2. OpenRouter dashboard: turn off data collection (or require ZDR), set a credit limit on the key. -3. `/setup-pstack` → name any OpenRouter model ID for any role → choose an effort → the probe runs → confirm. - -## Implementation - -Branch `feat/openrouter-gateway` from `main`. All edits stay in port-only files. Upstream-derived skill bodies (arena, interrogate, poteto-mode `SKILL.md`) stay byte-unchanged, per `UPSTREAM-FLEX.md`. - -### Step 1: the provider (code + unit tests) - -All files are under `plugins/pstack/skills/poteto-mode/scripts/runner/`. - -- `types.ts`: add `"openrouter"` to `GATEWAY_PROVIDERS`. The guard, preflight, env, parser, and model proof all branch on `isGatewayProvider`, so no new `switch` case is needed. -- `flex-providers.ts`: add a `GATEWAY_SPECS` row: - - `OPENROUTER_API_KEY` - - `https://openrouter.ai/api` - - `OPENROUTER_BASE_URL` - - `PSTACK_FLEX_OPENROUTER_CONFIG_DIR` - - no context default - - `OPENROUTER_MAX_CONTEXT_TOKENS` - - Add one spec field so only OpenRouter gets `ANTHROPIC_API_KEY=""`. `run.ts` already strips the inherited key, so this changes "unset" to "empty" for OpenRouter alone, and the DeepSeek/MiniMax env stays byte-identical. -- `run.ts` `validateOptions`: for `openrouter`, require a namespaced ID (`/`) and refuse the `openrouter/` namespace. Nothing else is filtered; the probe decides. -- `parse-output.ts` `reportedModelMatches`: use exact, case-insensitive matching for `openrouter`. The current gateway rule also accepts `${wanted}-…`, so `z-ai/glm-5.3-air` would wrongly pass for `z-ai/glm-5.3`. -- Tests: - - `flex-providers.test.ts`: the spec, the env map, and the empty key only for OpenRouter. - - `run.test.ts` `gateway lanes`: missing key, OAuth refusal, a namespaced ID through argv and receipt, a near-miss reported model failing, and `openrouter/auto` refused. - - `parse-output.test.ts`: exact-match cases. - - `commands.test.ts` and `cli.test.ts`: provider lists. - -### Step 2: prove the route on a spread of models (needs your key; costs cents) - -This is evidence that the path works for non-Anthropic models in general. It is not a list. - -Run the full #7 battery through the Step 1 runner on 5 models from different labs: `anthropic/claude-sonnet-5.5` (the control OpenRouter guarantees), `google/gemini-3.8-flash`, `moonshotai/kimi-k3`, `z-ai/glm-5.3`, `qwen/qwen3.8-max-0902`. Each run is read-only, in a synthetic workspace. - -The battery: -- the exact ID answers; -- a tool call reads a file marker that is not in the prompt; -- a second turn uses that result (thinking blocks survive); -- reasoning tokens differ between low and high; -- the receipt's reported model matches what OpenRouter's activity log shows was served. - -Also capture: -- OpenRouter's real failure strings: a wrong ID, 402 insufficient credits, a bad key, and a model without tool support; -- one `~` rolling alias and one `:free` variant, to learn what the receipt reports for each; -- empty versus unset `ANTHROPIC_API_KEY`. - -Write the results to `docs/gateway-model-probes.md` and to #7. Then: -- teach `run.ts` `unavailableStatus` the captured strings (for example, "no endpoints found" → `unavailable-model`), with fixtures from the real output; -- if a `~` alias or `:free` variant reports a different string than requested, add one documented translation rule in `reportedModelMatches`, or refuse that form with a clear message if no exact rule is possible. - -### Step 3: the contracts - -- `references/provider-dispatch.md`: - - One **open** row in the flex matrix: provider `openrouter`, model ``, default `high`, selectable `low`…`max`, key `OPENROUTER_API_KEY`, URL `https://openrouter.ai/api`. - - A rule that each distinct OpenRouter model is its own family. - - The descriptor grammar: split at the first `:` and the last `@`. - - Panel diversity counts labs: for `openrouter` lanes, the lab is the ID's namespace, and `anthropic`, `openai`, `x-ai`, `deepseek`, `minimax` equal the `claude`, `codex`, `grok`, `deepseek`, `minimax` providers. - - Add `openrouter` to the `--provider` list, the gateway paragraph, the override variables, and the route table. -- `setup-pstack/SKILL.md`: - - accept any OpenRouter ID for any role, mapping it to the open row; - - an OpenRouter probe row that puts the marker in a file; - - the lab-based diversity wording; - - a privacy and credit-limit disclosure before the first OpenRouter probe; - - the description line. -- `references/codex-tools.md`: the diversity lines. -- `runner/model-matrix.test.ts`: - - the open row's placeholder allowed only for `openrouter`, plus gateway ordering; - - no `openrouter:` in the first-run sheet; - - both diversity-string assertions; - - "Different models sharing a provider count as one provider" stays true for direct providers. - -### Step 4: docs - -- `docs/LANES.md`: replace "OpenRouter (not shipped)". - - Add OpenRouter to the lane-kind table, the env reference (including the empty key), prices (no token markup, 5.5% fee on credit purchases), and privacy. - - Note that `OPENROUTER_BASE_URL` must end in `/api`, not `/api/v1`. - - Note that models with less than 200K context can fail on long lanes; `OPENROUTER_MAX_CONTEXT_TOKENS` caps it. -- `README.md`: the gateway table and the lane diagram. -- `docs/USAGE.md`: keys, an example sheet with OpenRouter models, and the gateway sequence diagram. -- `docs/LIVE-GATE.md`: row D gets `OPENROUTER_API_KEY`. -- `UPSTREAM-FLEX.md`: list the open row and the lab rule under what the fork owns. -- `CHANGES.md`: an Unreleased entry. - -## Verification - -1. `bash scripts/check.sh`: Bun tests, strict typecheck, manifests, static invariants, `claude plugin validate`. -2. Step 2 evidence, committed in `docs/gateway-model-probes.md`. -3. Live gate (`docs/LIVE-GATE.md`, rows A–C in both harnesses plus row D): install the exact candidate. - - From **Claude Code** and from **Codex**, run `/setup-pstack`, type in an OpenRouter model that is **not** among the Step 2 models, and watch the probe pass and the sheet get written. - - Run a read-only mixed panel (for example `claude:fable` + `openrouter:z-ai/glm-5.3`) and check the receipts: `complete`, `provider-report`, `costUsd: null`. - - Check that a wrong ID, `openrouter/auto`, and a missing key each fail without changing the sheet. - - Record one evidence block per harness in the PR. The PR stays a draft until this is done. - -## Follow-ups (separate issues, not this PR) - -- **psf-monitor:** the Setup page's lane regex (`src/setup.ts`) rejects `/`, so OpenRouter roles would disappear from that view. -- **Credential leak risk on all gateways:** `CLAUDE_CODE_OAUTH_TOKEN` is not stripped from gateway children (`flex-providers.ts` `GATEWAY_INHERITED_CONFLICTS`, `run.ts` `CLAUDE_IDENTITY`). A parent's claude.ai token could reach a third-party endpoint. It affects DeepSeek/MiniMax too, so it gets its own fix and live gate. - -## Out of scope - -OpenRouter presets, the auto router or model fallback lists, real cost from OpenRouter billing, per-model automatic context caps, OpenRouter as the parent session (works today via the zero-subscription env walkthrough, docs only), OpenCode ([open-pstack#68](https://github.com/ericlitman/open-pstack/issues/68)), the registry redesign ([open-pstack#103](https://github.com/ericlitman/open-pstack/issues/103)). diff --git a/docs/plans/upstream-0.15.0.md b/docs/plans/upstream-0.15.0.md deleted file mode 100644 index b4e157f4..00000000 --- a/docs/plans/upstream-0.15.0.md +++ /dev/null @@ -1,113 +0,0 @@ -# Sync upstream pstack 0.15.0 into open-pstack - -Plan prepared September 8, 2026 for [GitHub issue #61](https://github.com/ericlitman/open-pstack/issues/61). Implemented in [PR #60](https://github.com/ericlitman/open-pstack/pull/60). Fable approved the revised plan with `Ship`. - -The update should import the four pstack commits after the last recorded sync, remove upstream's retired How critic workflow, and keep the existing Claude Code and Codex adaptations. Use one shared skill tree and the existing routing boundaries. This is one update PR with three reviewable commits, not a new synchronization framework. - -| Compared tree | Pinned revision | Version | -| --- | --- | --- | -| open-pstack main | `56bfd14418fa733e34d98f714f357d28788470e3` | 1.3.0 | -| Recorded Cursor sync | `efa2a531985e0a8084d36ff3cf87233be8a9f34b` | 0.14.7 | -| Target pstack tree | `71ed0d1076fec562c1b74ee353121a8d00f75382` | 0.15.0 | - -The target is the latest commit that changes `pstack/`. Cursor's repository head at inspection was `2b8ae2ee306f823d54879d3da7f8496b73c31d5d`, which adds another plugin and does not change this target tree. The original local checkout was behind main, so it was not used as the port baseline. The merged 1.3.0 tree and tag exist even though GitHub's latest published release still reports 1.2.1. This plan compares source trees, not installed caches or release-page labels. - -| Upstream commit | Change | Decision | -| --- | --- | --- | -| `7314f72` / [PR 309](https://github.com/cursor/plugins/pull/309) | Reduce logo to 361,140 bytes | Take the exact asset. Keep each distribution's manifest schema. | -| `e8d856f` / [PR 329](https://github.com/cursor/plugins/pull/329) | Skill density pass, retired How critics, two new principles | Port behavioral intent and deletions. Preserve existing platform substitutions. Apply two narrow correctness edits described below. | -| `d7cde2b` / [PR 331](https://github.com/cursor/plugins/pull/331) | Punctuation pass | Apply only upstream-changed prose. Do not run a repository-wide punctuation rewrite. | -| `71ed0d1` / [PR 333](https://github.com/cursor/plugins/pull/333) | 0.15.0 manifest and catalog corrections | Update provenance and actual port counts. Do not copy the Cursor manifest. | - -The rerunnable audit reports 96 changed files, including 40 `SKILL.md` files. There are 92 modifications, two additions, and two deletions. Of those paths, 24 match the old upstream blobs, 64 already diverge in the port, two are new upstream files, and six need distribution-specific treatment. These are exact file comparisons, not claims that every divergent file conflicts. No upstream runtime script changes in this range. - -Run from the repository root after fetching both remotes: - -```sh -git fetch origin -git fetch cursor main -python3 scripts/upstream-audit.py \ - --port 56bfd14418fa733e34d98f714f357d28788470e3 \ - --upstream 71ed0d1076fec562c1b74ee353121a8d00f75382 > /tmp/pstack-0.15.0-audit.json -``` - -The dependency-free script reads committed Git objects. It maps upstream skills, agents, assets, and the verbatim README mirror; lists every changed blob and port-only file; and labels unmapped documentation/manifests for review. It performs no fetch, checkout mutation, patch application, or automatic approval. Repeated runs against these SHAs produced byte-identical output. All 96 reported paths matched an independent `git diff --name-only --no-renames` check, and four representative classifications were checked by hand. - -## Changes to bring over - -| Area | Implementation decision | -| --- | --- | -| `how` and callers | Adopt explain-only How. Remove critique mode, `references/critic-prompt.md`, and `references/critique-rubric.md`. Remove How-critique routing from Architect and Investigation. Architectural challenge remains available through Interrogate and Architect's own review. Do not add a compatibility alias. | -| Model configuration | Remove `how critics` from setup's generated sheet and from tests and active documentation that require it. The documented role map becomes 15 rows. Preserve the remaining role assignments, panel order, provider descriptors, and per-family effort controls. | -| Existing model sheets | Normal dispatch uses the invoked skill's requested role, so the leftover `how critics` row has no consumer and cannot launch a critic. The runner accepts assigned argv and does not read the sheet. Whole-sheet validation belongs to setup, where the retired row becomes unknown. Document removing that row before setup accepts the sheet, then use the normal validated setup flow. For Codex the editable sheet and bounded AGENTS block must agree. A failed probe or invalid sheet must leave both unchanged. No silent rewrite, runtime validator, or migration mechanism. | -| `why` and `teach` | Adopt the shorter instructions while preserving evidence gathering, source citations, contradictions, unknowns, and the distinction between code behavior and historical intent. Keep Why and Reflect's MCP-dependent work on the parent-native route. | -| `reflect` | Adopt explicit invocation only. Do not trigger a reflection pass automatically after a task, failure, or correction. | -| `poteto-mode` | Remove the mandatory first todo to read the entire principle index. Keep applied-leaf reads and truthful citations. Register both new principles. Preserve the Feature throughput checkpoint and all existing implementation/review gates. | -| `unslop`, `technical-writing`, PR playbooks | Adopt shorter prose, removal of the Adding soul advice, and the new mannered-prose and over-compression guidance. Keep upstream's stable rule numbering. Use concise PR briefs and short squash messages. Propose changes to an offender skill without editing it automatically. | -| PR evidence | Keep the required exact-candidate installed-behavior evidence in the PR template. Link detailed logs and measurements from the concise description. Do not treat upstream's shorter PR-body guidance as permission to omit the installed version, user action, observed result, or evidence required before readiness. | -| Other touched skills/playbooks | Import the upstream hunk intent, including retained cross-references. Review deletions for lost rules, not just prose size. The Autopilot chooser rule restored in upstream PR 329 must remain reachable from Multi-phase plan. | - -The How deletion touches more than the skill directory. Inspect `plugins/pstack/skills/setup-pstack/SKILL.md`, `plugins/pstack/skills/poteto-mode/scripts/runner/model-matrix.test.ts`, `tests/skill-collision-repro.sh`, `docs/reference.md`, and the Architect/Investigation callers. Remove only the How-specific critic expectations. Preserve all remaining multi-model panel assertions. - -At port `56bfd14`, the precise removal points are `setup-pstack/SKILL.md:99`, `runner/model-matrix.test.ts:29,44`, `tests/skill-collision-repro.sh:107-125`, `architect/SKILL.md:24`, `poteto-mode/SKILL.md:90`, `poteto-mode/playbooks/investigation.md:7`, and `docs/reference.md:171,196`. There is no dedicated How-critic agent to delete; the shared Fable/Opus agent definitions remain in use. - -Clarify in the existing `provider-dispatch.md` that model rows configure roles a skill actually uses and cannot create a workflow. No special-case dispatch implementation is needed. The current runner contract is visible in `runner/cli.ts:59-108`, `runner/types.ts:11-22`, and `runner/run.ts:486-536`; none reads the model sheet. Selection belongs to the parent under `provider-dispatch.md:32,45-52`. Release notes must distinguish an unused row during ordinary dispatch from the setup-time unknown-role diagnostic. - -Add `principle-attack-the-premise` and `principle-test-behavior-not-implementation` to the shared skills tree and catalogs. Use the port's existing `user-invocable: false` frontmatter convention so the principles stay model-readable. Do not copy upstream's `disable-model-invocation: true` onto them. - -Two correctness adjustments need explicit provenance in `CHANGES.md`: - -- The testing principle labels several assertions as passing when imported functions return `undefined`. A Bun probe confirmed that `toBeDefined`, `toBeTruthy`, `toBeInstanceOf`, and `toBeGreaterThan(0)` fail on `undefined`. Change the categorical list heading and examples into conditional warnings about tests that fail to observe the relevant behavior. Preserve useful negative-path tests and relational contract checks. Do not turn this sync into a test-suite rewrite or delete the port's prompt/configuration contract checks merely because they inspect text. -- The premise principle assumes repeated failures come from an imbalance among actors and says an even census rules out the premise. Scope the census and reassignment prescription to imbalance problems. An even census is evidence against that asymmetry hypothesis, not proof that any shared premise is correct. Keep the instruction to question a premise after repeated failed fixes. - -These edits correct specific false generalizations. They do not establish a separate house style for the principles or justify rewriting unrelated upstream text. - -Both corrections are recorded in `CHANGES.md` with their reasons. At execution the upstream proposal, its disposition tracking, and the per-sync reassessment procedure were dropped as extra process; a later sync takes upstream's equivalent correction if one lands and deletes the local delta. - -## Port boundaries to preserve - -- Keep one shared `plugins/pstack/skills/` tree. Tool translation stays in `poteto-mode/references/codex-tools.md`; model routing stays in `provider-dispatch.md`. Do not add another abstraction or per-harness skill fork. -- The parent resolves provider, model, effort, and access mode once. Children never detect or choose a route. Native versus external execution, receipts, cancellation, named dropouts, and no fallback or implicit timeout remain unchanged. -- Retain rolling Fable/Opus aliases, existing selectable efforts, and the Sol defaults for `bug-fix`, `perf-issue`, and `hillclimb`. Upstream model-slug prose must not override user configuration. -- Preserve Claude/Codex transcript discovery, MCP access, local worktree isolation, tool mapping, namespaced skill resolution, and installed-script paths. No Cursor login, event runtime, cloud VM, `control-ui`/`control-cli`, built-in babysitter, or Cursor filesystem path becomes a requirement. -- Preserve the forge-neutral shipping changes from [open-pstack PR 44](https://github.com/ericlitman/open-pstack/pull/44): independent verdicts, queue disarming, captured-SHA leases, expected-head protection, correct fork remotes, and bottom-first landing. Apply small prose edits to these adapted files instead of replacing them. -- Preserve the existing shipping assertions in `tests/skill-collision-repro.sh:212-323`. They already check disarming before verification/mutation, queue-entry removal, captured-SHA leases, independent verdicts, expected-head merges, fork remotes, and bottom-first landing. The unmodified baseline passed during this planning task. After importing prose, keep these checks passing. If wording changes require an assertion update, retain the same invariant and prove that removing the protected instruction still makes the check fail. Do not add another shipping-test framework or weaken checks to make a copy pass. -- Keep the existing exclusions in `UPSTREAM.md`: `make-bot-ui`, the invocation-blocking flags on `how`/`why`/`unslop`/`typescript-best-practices`, Cursor-only model-default hunks, and the unsupported Claude logo field. The older Benny automation pack remains excluded. The unrelated Grok Voice plugin is outside `pstack/`. -- Preserve the port-only skills, agents, runner, watcher, orchestrator, package/lockfile, and four distribution manifests except for deliberate release metadata and the How-role test edits. Do not import a command-wrapper layer. -- Upstream's stronger "Do not add guards" wording in Fix Root Causes applies to symptom-hiding workarounds. It does not override external-input validation required by Boundary Discipline or justify removing runner validation in this sync. - -[GitHub issue 36](https://github.com/ericlitman/open-pstack/issues/36) still proposes importing `make-bot-ui`; later merged PR 44 and current `UPSTREAM.md` explicitly exclude it. This update follows the later merged decision and the user's request to omit Cursor-specific capabilities. It does not reopen or implement that issue. Other open setup/provider and How/Why enhancement issues remain separate work, not prerequisites or additions to this sync. - -## Execution order - -1. **Retire the removed workflow and add the two leaves.** Start from refreshed main, rerun the audit, and bind the work to this ticket. Port the semantic changes, remove dead How references and its role, register the new leaves, and add the two narrow correctness adjustments. Update only the affected role/catalog assertions. Run the affected static/model-map checks before continuing. -2. **Port the remaining upstream prose and asset.** Read upstream commits in order, apply the three-way comparison to adapted files, keep prior release safeguards, and replace the logo with the exact upstream bytes. Confirm no Cursor-dependent instruction was introduced. Do not mass-format scripts, rewrite tests, or change runtime provider behavior. -3. **Record the sync and verify the release candidate.** Set `UPSTREAM.md` to `71ed0d1` / 0.15.0, copy `README-UPSTREAM.md` verbatim, and update `NOTICE.md`, `CHANGES.md`, README, reference documentation, counts, and versioned manifests. The expected catalog is 54 shared skills and 23 principles. Verify the actual tree before writing counts. Keep Cursor's version independent of open-pstack's. Choose the next port release from the then-current tags and explicitly document the removed mode/role; do not promise an old configuration still works. - -Commit the audit tool with the plan so the implementer can rerun it. The implementation can remain one PR because it adds no runtime architecture. Keep that PR draft until both affected applications pass the installed-candidate checks below. Merge only the reviewed candidate, then tag/release and read back the actual published state. This planning task does not perform that implementation or release. - -## Acceptance and verification - -- Every upstream delta path is accounted for as imported, adapted, deleted, or excluded with a reason. Recheck against current main at execution time and keep new upstream arrivals outside the pinned target unless deliberately re-reviewed. -- No active path invokes How critics or requires its role. Both obsolete reference files are gone. Setup renders 15 valid rows; the other role families and efforts remain unchanged. Test upgrading with the old 16-row sheet without rerunning setup: How launches no critics, and another retained role still dispatches as configured. Invoking setup on that same stale sheet yields an actionable diagnostic before probes or writes. A corrected sheet passes the usual probes and readback in both applications. -- Both new principles load through normal model invocation, follow the port's visibility convention, and use the corrected statements. A deliberately incorrect function makes its behavior test fail. A useful negative-path test remains valid. A repeated failure shared evenly by all actors does not incorrectly terminate premise investigation. -- From the exact installed candidate in fresh Claude Code and Codex sessions, run How on simple and complex fixtures. Obtain grounded explanations using the configured explorer/explainer routes, with no critic fan-out. Check Why's source-backed response and named gaps on a fixed fixture, and Teach's concise combination of How/Why results. Inspect transcripts and actual outputs, not just self-reported success. -- Confirm Reflect does not start after an ordinary completed task, and does run when explicitly invoked. Check a PR-writing fixture gives a short review brief with required installed-candidate evidence and linked details. Check the retained Autopilot chooser resolves. -- Demonstrate named `how`, `why`, `unslop`, and `typescript-best-practices` invocation still works. Confirm `make-bot-ui` remains absent. Verify both new principles are discoverable to the model and the port-only skills remain packaged. -- Before merge, compare every installed plugin file to the exact candidate and record candidate identity, application/version, invoked action, and observed result. Existing runtime tests and focused native/external routing canaries must still demonstrate no silent provider substitution or implicit deadline. Full setup across both applications already exercises the selected routes. -- Run the repository gates from `.github/workflows/ci.yml`: `bun install --frozen-lockfile`, `bun run test`, and `bun run typecheck` in `plugins/pstack/skills/poteto-mode/scripts`; parse all four JSON manifests; run `PSTACK_STATIC_ONLY=1 bash tests/skill-collision-repro.sh`; run Claude plugin validation; and run `git diff --check`. Do not run Claude's marketplace validator on the Codex manifest. -- Compare `README-UPSTREAM.md` and the logo bytes against the pinned upstream objects. Require the asset to be below 512 KiB. Check all live links and catalog counts affected by the deletion/additions. -- Treat upstream's reported token reduction/evals as upstream evidence only. Record port text-size changes and run focused behavioral fixtures in both applications; do not claim upstream's token or latency numbers for this port. No new benchmark system is needed. - -## Sources and review record - -- [Recorded 0.14.7 sync contract](https://github.com/ericlitman/open-pstack/blob/56bfd14418fa733e34d98f714f357d28788470e3/UPSTREAM.md) -- [Prior sync and its installed-candidate evidence](https://github.com/ericlitman/open-pstack/pull/44) -- [Exact upstream comparison](https://github.com/cursor/plugins/compare/efa2a531985e0a8084d36ff3cf87233be8a9f34b...71ed0d1076fec562c1b74ee353121a8d00f75382) -- [Upstream testing principle](https://github.com/cursor/plugins/blob/71ed0d1076fec562c1b74ee353121a8d00f75382/pstack/skills/principle-test-behavior-not-implementation/SKILL.md) -- [Upstream premise principle](https://github.com/cursor/plugins/blob/71ed0d1076fec562c1b74ee353121a8d00f75382/pstack/skills/principle-attack-the-premise/SKILL.md) -- Comparison tool: `scripts/upstream-audit.py`. Evidence was generated at `/Users/ericlitman/projects/pstack/evidence/upstream-0.15.0/`. -- GitHub issue [#61](https://github.com/ericlitman/open-pstack/issues/61) is the tracker for this sync, per `AGENTS.md`. -- Fable first returned `Fix`. The revised plan ties the two correctness edits to an upstream proposal, distinguishes ordinary dispatch from setup validation, and identifies the shipping assertions that already exist and pass. -- Fable's final verdict is `Ship`. The reviewed approach removes obsolete behavior without a migration or shim, retains existing safeguards, and uses a small read-only audit instead of a sync engine. -- Remaining review risk: a rule in one of the 64 adapted files could disappear during the prose import without a focused fixture covering it. The per-file adaptation review and installed-candidate checks above remain required; path accounting alone does not prove semantic preservation. diff --git a/docs/reference.md b/docs/reference.md index b4abc367..0835c9f0 100644 --- a/docs/reference.md +++ b/docs/reference.md @@ -1,237 +1,129 @@ -# pstack-flex technical reference +# Technical reference -This page contains the full skill, dependency, runtime, and porting reference. For the plain-English introduction and quick start, see the [main README](../README.md). +This page is for maintainers and users diagnosing configuration or lane execution. Installation and first use are in the [README](../README.md). Model matrices and execution details live in [provider-dispatch.md](../plugins/pstack/skills/poteto-mode/references/provider-dispatch.md). -[Poteto](https://x.com/poteto)'s [pstack](https://github.com/cursor/plugins/tree/main/pstack), adapted to run in Claude Code and Codex without Cursor. One shared skill tree serves both harnesses; Grok remains available as a model-provider lane. Version 1.5.0 is synced to Cursor pstack v0.15.5 at `12d587dfb20741cafc376c42c696c5f6e2a64487`. See [UPSTREAM.md](../UPSTREAM.md) for the exact sync contract. +## Shared workflow and runtime -Original by Lauren Tan. This distribution builds on Michael Denyer's [pstack-claude](https://github.com/michael-denyer/pstack-claude) port and retains its history and MIT attribution. It imports seven MIT-licensed skills from [cursor-team-kit](https://github.com/cursor/plugins/tree/main/cursor-team-kit): `deslop`, `thermo-nuclear-code-quality-review`, `make-pr-easy-to-review`, `fix-ci`, `fix-merge-conflicts`, `get-pr-comments`, `what-did-i-get-done`. +`plugins/pstack/skills/` is the sole workflow tree. Claude Code and Codex expose namespaced skills such as `pstack:poteto-mode`; OpenCode beta loads bare names such as `poteto-mode`. The SessionStart instruction keeps invocation opt-in. A started workflow can invoke the skills it needs. -> if you want to go fast, go deep first. pstack helps you write less, but higher quality code. rigorous agent workflows you can parallelize with confidence. +The runtime separates these contracts: -This is not a verbatim copy. Skill bodies have been edited so every Cursor-specific primitive resolves to its Claude Code or Codex equivalent — see [Differences from upstream](#differences-from-upstream) for the full list. The exhaustive per-skill audit lives in [CHANGES.md](../CHANGES.md); license attribution lives in [NOTICE.md](../NOTICE.md); the upstream README is preserved verbatim at [README-UPSTREAM.md](../README-UPSTREAM.md). +| Source | Owns | +| --- | --- | +| `poteto-mode/scripts/harnesses.ts` | Flat harness rows for native provider, config paths, integration method, identity, session, and launch capability | +| `poteto-mode/scripts/configuration.ts` | Sheet scope paths, reads, descriptor parsing, normalization, and route resolution | +| `poteto-mode/scripts/pstack-context` | Read-only inspection of explicit parent configuration | +| `setup-pstack/SKILL.md` | Role questions, effort selection, live probes, confirmation, and transactional writes | +| `poteto-mode/references/codex-tools.md` | Shared workflow operations mapped to parent tools and built-ins | +| `poteto-mode/references/provider-dispatch.md` | Model families, default panels, native and external execution, isolation, and receipts | +| `poteto-mode/scripts/runner/` | CLI invocation, provider environment, output parsing, receipt files, and lane journal | -## Install +Paths in this table are relative to `plugins/pstack/skills/`. The mapping file retains its historical `codex-tools.md` path so upstream references resolve. -### Claude Code +## Model sheets -This repo ships as a Claude Code marketplace containing one plugin (`pstack`). +A model sheet maps each role to descriptors of the form `provider:model@effort`. Parsing splits at the first colon and the last `@`, so provider model IDs may contain slashes. `inherit-parent` and `auto` use the parent session's model and effort. -```text -/plugin marketplace add thisguymartin/pstack-flex -/plugin install pstack@pstack-flex -/reload-plugins -``` +Each parent has a global sheet and a private project sheet: -pstack runs only when you ask for it. A `SessionStart` hook on startup, `/clear`, and post-compact injects a short opt-in gate: Claude invokes a `pstack:*` skill only when you type `/pstack:`, name pstack or one of its skills in the request, or keep a standing instruction for it in CLAUDE.md. Once you start a pstack skill, it keeps routing to the other pstack skills it needs for that task. Subagents that pstack dispatches follow their dispatch prompt instead of the gate. To route every non-trivial engineering task into `poteto-mode` again, add a standing instruction such as `Use pstack:poteto-mode for non-trivial engineering tasks.` to `~/.claude/CLAUDE.md` or a project CLAUDE.md. +| Parent | Global sheet | Project sheet | Global override | +| --- | --- | --- | --- | +| Claude Code | `~/.claude/pstack-models.md` | `/.claude/pstack-models.md` | `CLAUDE_CONFIG_DIR` | +| Codex | `~/.codex/pstack-models.md` | `/.codex/pstack-models.md` | `CODEX_HOME` | +| OpenCode beta | `~/.config/opencode/pstack-models.md` | `/.opencode/pstack-models.md` | `$XDG_CONFIG_HOME/opencode` | -### Codex +`XDG_CONFIG_HOME` defaults to `~/.config`. Claude and Codex overrides replace their global directories. The project root is the primary checkout resolved from Git's common directory; linked worktrees share that project's sheet. Without a repository root, configuration uses the global sheet. -The same plugin carries a `.codex-plugin/plugin.json` manifest and a root `.agents/plugins/marketplace.json`. Install it through the Codex marketplace: +An existing project sheet replaces the global sheet as a whole. Roles are never merged across sheets. Setup starts a new project sheet from the global assignments and excludes it through the repository's `.git/info/exclude`. Project configuration needs no committed instruction file. -```shell -codex plugin marketplace add thisguymartin/pstack-flex --ref main -codex plugin add pstack@pstack-flex -``` +Global setup integrates the sheet into the parent's instructions using its existing method. Claude Code uses an include, Codex uses a mirrored block, and OpenCode beta uses its config's `instructions` array. Setup owns the probe and write transaction; context inspection writes nothing. -Codex discovers the plugin skills under the `pstack` namespace, so they list as `pstack:poteto-mode`, `pstack:tdd`, and so on. The namespace comes from `plugins/pstack/.codex-plugin/plugin.json`. To enable the multi-model and parallel-subagent skills (`interrogate`, `arena`, `how`, `why`, `reflect`, `architect`), turn on subagents in `~/.codex/config.toml`: +Known versioned Claude Fable and Opus descriptors normalize to rolling aliases in memory. Setup rewrites stale persisted values only after its probes and confirmation. Other invalid descriptors remain errors. Supported families, efforts, and panel diversity rules have one reference in [provider dispatch](../plugins/pstack/skills/poteto-mode/references/provider-dispatch.md). -```toml -[features] -multi_agent = true -``` +## Inspect the active context -For local plugin development, you can clone the repository and link its skills directly: +The bundled executable requires an explicit parent and accepts the task's directory: ```shell -git clone https://github.com/thisguymartin/pstack-flex -cd pstack-flex -for s in plugins/pstack/skills/*/; do ln -s "$PWD/$s" ~/.agents/skills/"$(basename "$s")"; done +plugins/pstack/skills/poteto-mode/scripts/pstack-context --parent codex --cwd "$PWD" ``` -The marketplace install is the normal user path. Direct links are only for testing a checkout before publishing it. Remove the linked skill directories when the test is over. - -## Layout - -```text -. -├── .claude-plugin/marketplace.json # Claude Code marketplace manifest (repo root) -├── .agents/plugins/marketplace.json # Codex marketplace manifest (repo root) -├── plugins/pstack/ # the pstack plugin -│ ├── .claude-plugin/plugin.json # Claude Code manifest -│ ├── .codex-plugin/plugin.json # Codex manifest (skills: ./skills/) -│ ├── skills/ # 56 skills shared by Claude Code and Codex -│ │ ├── poteto-mode/references/{codex-tools,provider-dispatch}.md # tool + provider routing -│ │ └── poteto-mode/scripts/ # bun/bash/node tooling: watch-pr, orch, runner, check-plan.mjs, worktree-audit.sh -│ ├── hooks/ # SessionStart opt-in gate: pstack runs only on request (Claude Code and Codex) -│ └── agents/ # Claude subagents, including native Fable and Opus lanes at each selectable effort -├── tests/skill-collision-repro.sh # native-skill package invariants and Claude invocation checks -├── LICENSE # pstack upstream MIT -├── LICENSE-cursor-team-kit # cursor-team-kit upstream MIT -├── LICENSE-superpowers # superpowers upstream MIT (hook runner) -├── NOTICE.md # attribution table -├── UPSTREAM.md # current Cursor sync point and update procedure -├── CHANGES.md # per-skill substitution audit -├── README.md # plain-English introduction and quick start -└── docs/reference.md # this technical reference -``` +Accepted parents are `claude`, `codex`, and `opencode`. The JSON result contains: -Plugin-internal `skills//` path references in the docs below are relative to `plugins/pstack/`. +| Field | Meaning | +| --- | --- | +| `harness` | The parent's metadata row | +| `paths` | `globalSheet`, `projectSheet`, `excludeFile`, and `integrationTargets` | +| `activeSheet` | The selected sheet | +| `roles` | Parsed role descriptors, normalized aliases, native or external routes, capabilities, and known model labs | -## Running on Codex +The command reads configuration without probing providers, writing sheets, changing integrations, or detecting a parent from inherited environment markers. The caller supplies the parent once. -The Codex build shares one `skills/` tree with the Claude Code build. Nothing is forked or generated. Two narrow references keep runtime translation separate: `codex-tools.md` maps harness primitives and `provider-dispatch.md` maps model providers. pstack otherwise keeps the upstream Claude-native prose and adds a one-line Platform note to each skill that names a Claude primitive, so the port stays in lockstep with upstream sync. +Setup uses `--paths-only` before choosing a scope so a malformed sheet can be repaired without blocking path inspection. -- **Skill invocation.** Codex loads `SKILL.md` natively. There is no `Skill` tool. You invoke a skill by name (ask for it, or pick `pstack:poteto-mode` from the list). -- **Package surface.** The native `skills/` tree is the only workflow source. The plugin ships no `commands/` layer and does not link prompts into `~/.codex/prompts/`. Codex would migrate such files into duplicate source-command skills while loading the native skill tree. The 23 `principle-*` leaves declare `user-invocable: false`. Claude keeps them out of its user picker; Codex 0.149.0 currently shows them despite that metadata ([open-pstack #8](https://github.com/ericlitman/open-pstack/issues/8)). -- **Tool and built-in mapping.** Claude tool names and built-in skills resolve through [`codex-tools.md`](../plugins/pstack/skills/poteto-mode/references/codex-tools.md). Model execution resolves separately through [`provider-dispatch.md`](../plugins/pstack/skills/poteto-mode/references/provider-dispatch.md), so Codex can keep Sol native while invoking Claude and Grok externally. -- **Subagents.** The `Agent` tool maps to Codex `spawn_agent` / `wait_agent`, enabled by `multi_agent = true`. Parallel fan-out is multiple `spawn_agent` calls in one turn. If the native Codex lane is unavailable, record that lane as a dropout; external Claude and Grok lanes still run, and no provider is silently substituted. There is no `poteto-agent` subagent type on Codex; route ad-hoc subagents by dispatching a `spawn_agent` told to read `poteto-mode` first. -- **Opt-in.** Codex runs the plugin's `hooks/` SessionStart hook and adds the opt-in gate to each session as a developer message (observed on Codex 0.157.1). Codex records trust for the hook in `~/.codex/config.toml` under `hooks.state`. Enter `pstack:poteto-mode` by name, or add a standing instruction to `~/.codex/AGENTS.md` if you want every non-trivial task routed into it. After a plugin update, run `codex plugin marketplace upgrade pstack-flex` so the installed copy carries the current gate. -- **Models.** `/setup-pstack` writes provider-qualified descriptors and asks one requested effort per assigned family (`low`, `medium`, `high`, `xhigh`, `max`). It asks first whether to configure the global sheet or a private project sheet, which replaces the global one for that repository. The first-run panel is Fable max, GPT-6 Astra high, Grok 4.7 xhigh, and Opus max; architect sketches use GPT-6 Astra high and Fable max. Fable and Opus use Claude's rolling aliases. Runtime dispatch normalizes older versioned descriptors in memory, so an installed sheet stops pinning immediately. A setup rerun persists that migration while keeping each role's family and effort. The GPT-6 Astra, GPT-6.1 Sol, GPT-6 Sol, and Luna Codex families are stock: GPT-6.1 Sol high carries `feature, refactoring`, `bug-fix`, `perf-issue`, and `hillclimb`; Luna high carries `how explorer` and `swarm workers`; Astra high sits on every panel. GPT-6 Sol and GPT-5.6 Sol remain selectable families. In Codex, every Codex family uses native `spawn_agent`; Claude and Grok use the deterministic external runner. In Claude Code, Fable and Opus use native agents; the Codex families and Grok use the runner. Children never detect the parent or reroute themselves. The solo code roles stay on a Codex model instead of upstream's Fable default because it costs less for these frequent delegated code roles. +An aggregator's route does not establish the model's maker. `lab: null` requires verification before counting panel diversity; it never counts as an additional lab. -Verified upstream in fresh installed open-pstack Claude Code and Codex sessions (pstack-flex's own live-test record is in [LIVE-GATE.md](LIVE-GATE.md)): the user-facing skills are discovered and namespaced under `pstack`; both parents fan out the configured panel through the documented native/external route table, retain long-running handles without a default timeout, and cross-judge only after every candidate is terminal. The `principle-*` leaves remain available for `poteto-mode` to read by path. Claude honors their `user-invocable: false` metadata; Codex 0.149.0 does not ([open-pstack #8](https://github.com/ericlitman/open-pstack/issues/8)). +## Native and external lanes -## Dependencies +The parent reads the active sheet and resolves each lane before fan-out. Every child receives its provider, model, effort, access mode, working directory, prompt, and output location. Children do not reroute or launch nested models. -Nothing is declared in `plugin.json`. Install the one companion plugin yourself: +Claude descriptors use native agents in Claude Code. Codex descriptors use native subagents in Codex. Other descriptors use `pstack-runner`. OpenCode beta sends every provider-qualified descriptor through the runner; its native `task` tool handles only `inherit-parent` and `auto`. -- **`plugin-dev`** (from the `claude-plugins-official` marketplace) — the rewiring routes skill-authoring tasks (in `automate-me`, `reflect`, `poteto-mode`) to the `plugin-dev:skill-development` skill: +External lanes receive complete tasks directly, without an intermediary model. Each writer gets its own worktree. Each run gets unique output paths. A missing CLI, rejected model, authentication error, or child failure produces a named dropout. The runner adds no implicit timeout and substitutes no model. - ```shell - /plugin marketplace add anthropics/claude-plugins-official - /plugin install plugin-dev@claude-plugins-official - ``` +Launcher arguments, receipt schema, status codes, and cancellation are defined in [provider dispatch](../plugins/pstack/skills/poteto-mode/references/provider-dispatch.md#external-lanes). A receipt distinguishes the requested model from the provider's reported model. `modelEvidence: "pinned-argv"` proves the requested invocation but does not prove which model answered. - Until 0.9.2 this was a `dependencies` entry in `plugin.json`. The desktop app's `--plugin-dir` load mode can never resolve cross-marketplace dependencies and hard-disables the whole plugin, so 0.9.3 removed the declaration — full mechanism in the 0.9.3 entry of [CHANGES.md](../CHANGES.md). Without `plugin-dev` installed, only the skill-authoring routes degrade; everything else works. +## Gateway and OpenCode limits -Not declared as deps, but referenced in skill bodies: +DeepSeek, MiniMax, and OpenRouter lanes run the stock Claude CLI against a provider endpoint. The runner strips inherited Anthropic routing, injects the chosen endpoint and key, and uses an isolated config directory. Its OAuth-file guard checks `.credentials.json`; it cannot detect macOS Keychain credentials. Do not authenticate a Claude subscription inside a gateway config directory. -- **`run`, `verify`, `loop`** — Claude Code CLI built-ins (ship with the binary, always available). -- **`gh` (GitHub CLI).** This is the default forge for every stack playbook and a system-level requirement of the standalone `babysit` skill. Install it with [`brew install gh`](https://cli.github.com) and authenticate with `gh auth login`. If Origin's `origin` CLI is installed and can resolve the repository, the stack playbooks use it instead. Only the Orchestrate playbook and its `scripts/orch` frontier tooling still require `gt`. -- **`bun`** — runs the vendored `skills/poteto-mode/scripts/` tooling (`watch-pr`, `orch`, `runner`). Install via [`brew install oven-sh/bun/bun`](https://bun.sh). `bootstrap.ts` installs dependencies for `watch-pr` and `orch`; the runner uses only Bun and Node built-ins, so it launches directly without an install/re-exec layer. -- **`node`** — runs `skills/poteto-mode/scripts/check-plan.mjs`. The checker uses only Node built-ins and does not need Bun. -- **Claude Code, Codex, and Grok Build CLIs** — the external runner uses the assigned subscribed CLI directly. Install and authenticate only the providers present in your model sheet. Same-provider work stays native; the runner refuses it. -- **`jq` and `rg` (ripgrep)** — only for `scripts/worktree-audit.sh` (the Worktree cleanup playbook). Without them the audit still runs but blanks its PR and LAST_CHAT columns, so it warns on stderr rather than returning a table that looks complete. +Enable third-party routing per project, including through OpenCode. Keep customer data out of those lanes and use synthetic data for testing. Gateway keys stay in the local environment. Gateway receipts keep usage but set `costUsd` to `null`; the Claude CLI's Anthropic cost estimate is unsuitable for another provider. This repository maintains no pricing table. -No third-party plugins. The harsher-critique escape hatch lives in the bundled `thermo-nuclear-code-quality-review` skill (imported from cursor-team-kit), not in an external plugin. +OpenCode support is beta. Runner-level checks do not establish installed OpenCode parent support. The real-session gate remains required, and headless CI use is unsupported. -## Skills +OpenCode lanes use deny-first permissions. Each invocation uses a fresh agent name so ambient `plan` and `build` permissions cannot merge into its access policy. Read-only lanes can inspect files. Writers can also edit files inside their worktree. They cannot run shell commands, Git commands, tests, or builds. A lane that requires those tools must report the unsupported requirement rather than claim verification. OpenCode receipts use `pinned-argv` evidence and `costUsd: null`. -The table uses the short upstream names. Claude Code exposes each native skill with a `/pstack:` prefix, such as `/pstack:poteto-mode`. In Codex, ask for the namespaced skill, such as `pstack:poteto-mode`. +## Optional lane journal -| skill | use it when | -| --- | --- | -| `/poteto-mode` | default entry point for any non-trivial task | -| `/how` | walk through how a subsystem works | -| `/why` | investigate why something was built this way (parallel multi-MCP evidence) | -| `/architect` | settle types and module shape before writing code that crosses a function boundary | -| `/arena` | run N parallel attempts at the same task and pick the best parts | -| `/interrogate` | have several different models try to break a diff | -| `/automate-me` | draft your own personal -mode skill from recent transcripts | -| `/reflect` | capture a long task's lessons as a skill edit | -| `/tdd` | fix a bug by writing the failing test first, then the fix | -| `/typescript-best-practices` | ground type-system discipline in TypeScript syntax | -| `/teach` | understand a change or subsystem for real: `how` + `why` woven into one plain explanation | -| `/swarm` | fan out N parallel workers across slices or races, then one aggregated report | -| `/technical-writing` | write docs, RFCs, readmes, PR descriptions, and commit messages to one layered standard | -| `/bro` | restate the last message in plain human language, no jargon | -| `/figure-it-out` | design a rigorous, auditable playbook for a task no bundled playbook fits | -| `/show-me-your-work` | log decisions to a reviewable tsv decision trail | -| `/blast-radius` | find what a change could break beyond the diff and prove safety by running code | -| `/intake` | turn GitHub issues into ready-to-run poteto-mode briefs with a playbook, exit condition, and worktree (pstack-flex) | -| `/diff-behavior` | run the same scenarios on trunk and head and classify every observable difference as intended, unintended, or noise (pstack-flex) | -| `/recall` | catch up on recent working context from chat history, live state, and the shared record | -| `/setup-pstack` | configure pstack per-role model choices and per-family requested effort | -| `/unslop` | clean up writing by removing AI tells | -| `/no-comments` | strip comments before review via the `comment-sicko` subagent, then fix what it finds | -| `/create-verification-skill` | generate a project-local verification skill and feature map | -| `/maintain-verification-skill` | re-sync a drifted verification skill and its feature map | -| `/deslop` | deslop a diff before commit | -| `/babysit` | monitor an open PR, fix CI/comments, keep it merge-ready | -| `/thermo-nuclear-code-quality-review` | extremely strict maintainability audit | -| `/make-pr-easy-to-review` | clean noisy history and improve PR description before review | -| `/fix-ci` | find failing PR checks, inspect logs, apply focused fixes | -| `/fix-merge-conflicts` | non-interactively resolve merge conflicts, validate, finalize | -| `/get-pr-comments` | fetch and summarize review comments from the active PR | -| `/what-did-i-get-done` | summarize authored commits over a user-chosen period | - -## Lane journal - -`pstack-runner` writes an opt-in journal under `~/.pstack-flex/lanes/` (or `PSTACK_FLEX_LANES_DIR`) while that directory exists. Each lane gets one directory holding `lane.json` (schema version 1: provider, model, effort, mode, label, the first 300 characters of the prompt, runner pid, parent harness and session), `stream.jsonl` (stdout as it arrives), and `receipt.json` (a copy of the receipt). Sending SIGTERM to the runner pid cancels the lane and writes a `cancelled` receipt. A journal failure never changes a lane's receipt, exit status, or output. - -[psf-monitor](https://github.com/thisguymartin/psf-monitor) reads this journal to show lanes while they run. It keeps its own copy of the schema, so a change to `lane.json` needs a matching psf-monitor change. - -## Subagents - -`poteto-agent` ships unchanged. Spawn from a parent with `subagent_type: "poteto-agent"`. - -`comment-sicko` is the read-only comment reviewer the `no-comments` skill spawns. Upstream names it `Comment Sicko`; the port renames it to `comment-sicko` so the name is a valid `subagent_type`. Invoke it through `/no-comments`, not directly. - -Fable and Opus each ship at `low`, `medium`, `high`, `xhigh`, and `max`. Names are `pstack--`. `pstack-fable-max` and `pstack-opus-xhigh` remain. Each file selects the rolling family alias and requested effort, runs in the background, and denies nested Agent/Task dispatch. pstack dispatches them from provider-qualified descriptors; they are not user-facing workflows. - -## Differences from upstream - -The port is editorial, not mechanical. Anywhere upstream pstack assumed Cursor-specific primitives, this port substitutes the Claude Code equivalent so refs actually resolve. Two prior ports ([v1truv1us/ai-eng-system](https://github.com/v1truv1us/ai-eng-system), [Evan-Kim2028/agent-fleet](https://github.com/Evan-Kim2028/agent-fleet)) stop at namespacing — they vendor pstack under `pstack/` and leave the Cursor refs intact. This port does the content surgery. - -### What's added - -- **`skills/babysit/`** — Claude Code analog of Cursor's closed-source `/babysit` built-in. Wraps `gh pr view` / `gh pr checks` / `gh run view --log-failed` plus the `loop` skill for pacing. Independently authored; workflow informed by Cursor's public `/babysit` behavior — not a copy of Cursor's implementation. Since the v0.14.2 sync, poteto-mode routes PR-status requests to the ported `playbooks/babysit.md` instead, and this skill is the standalone `/babysit` entry point. -- **`skills/deslop/`** — imported verbatim from `cursor-team-kit`. Cleans AI tells out of diffs before commit. -- **`skills/thermo-nuclear-code-quality-review/`** — imported verbatim from `cursor-team-kit`. -- **`skills/make-pr-easy-to-review/`** — imported verbatim from `cursor-team-kit`. Composes with `opening-a-pr` and `babysit`. -- **`skills/fix-ci/`** — imported verbatim from `cursor-team-kit`. Narrower CI-fix primitive that `babysit` can route to. -- **`skills/fix-merge-conflicts/`** — imported verbatim from `cursor-team-kit`. Pairs with `babysit` step 5. -- **`skills/get-pr-comments/`** — imported verbatim from `cursor-team-kit`. Primitive for `babysit` step 4 and `reflect`. -- **`skills/what-did-i-get-done/`** — imported verbatim from `cursor-team-kit`. Commit summary over a chosen period. - -### What's substituted in skill bodies - -| Upstream (Cursor) | This port (Claude Code) | +The runner journals lanes while `~/.pstack-flex/lanes/` exists, or while the directory named by `PSTACK_FLEX_LANES_DIR` exists. Each lane directory holds `lane.json`, streamed stdout in `stream.jsonl`, and `receipt.json`. + +The start record includes provider, model, effort, mode, label, the first 300 prompt characters, runner PID, and parent identity. A journal error never changes a lane's output, receipt, or exit status. Sending SIGTERM to the runner cancels the lane and produces a cancelled receipt. + +The separate [psf-monitor](https://github.com/thisguymartin/psf-monitor) reads this journal. A journal schema change also needs a matching monitor change. + +## Dependencies + +The runner and orchestration tools use Bun. Git supplies worktree isolation. GitHub workflow skills use `gh`. `check-plan.mjs` uses Node built-ins. `worktree-audit.sh` uses `jq` and `rg` and warns when missing tools leave its PR or last-chat columns blank. + +Install and authenticate only CLIs assigned in the active model sheet. Native Codex fan-out requires `multi_agent = true`. CLI and browser verification use the parent tools named in the [harness mapping](../plugins/pstack/skills/poteto-mode/references/codex-tools.md). + +## Add a harness + +The extension points are the existing metadata, configuration, and mapping boundaries. A new harness needs: + +1. A row in `harnesses.ts` naming its native provider, paths, config override, integration, launch capability, and identity or session variables. +2. Configuration tests covering its paths, linked worktrees, and global fallback. The shared resolver reads the metadata row. +3. A tool map in `codex-tools.md` for invocation, questions, subagents, background drain, cancellation, and verification. +4. Tests for runner parent validation and child environment isolation, both derived from the row. Add a provider adapter only if the harness also supplies external model lanes. +5. An integration recipe if its instruction format differs from the existing three. Keep setup's shared probes, confirmation, and write transaction. +6. Static invariants and installed evidence from the new app's real user interface under [LIVE-GATE.md](LIVE-GATE.md). + +Unsupported tools need an explicit requirement failure. They do not justify per-harness skill trees, child routing, compatibility layers, implicit timeouts, or weaker-model fallback. + +## Diagnose a failed run + +The active context and receipt identify different failure points: + +| Symptom | Relevant evidence | | --- | --- | -| `Task` tool, `subagent_type: generalPurpose`, `readonly: false/true` | `Agent` tool with model selection, requested effort, and `disallowedTools`; access mode is assigned by the parent, with writers isolated in worktrees | -| `AskQuestion` tool | `AskUserQuestion` tool | -| Cursor's built-in `/loop` | Claude Code's built-in `loop` skill | -| Cursor's built-in `/babysit` | `babysit` skill bundled in this plugin. From v0.14.0 upstream routes PR-status requests inside poteto-mode to `playbooks/babysit.md` instead; the port does the same, and `/babysit` stays the standalone entry point | -| Cursor's built-in `/create-skill` | `plugin-dev:skill-development` skill | -| `cursor-team-kit` `control-cli` (CLI/TUI driver) | Claude Code's `run` skill | -| `cursor-team-kit` `control-ui` (browser/Electron driver) | Claude Code's `verify` skill | -| Transcripts at `~/.cursor/projects/*/` or `agent-transcripts/` | `~/.claude/projects//*.jsonl` (where `` is the workspace cwd with `/` → `-`) | -| Skill paths `.cursor/skills/`, `~/.cursor/plugins/` | `.claude/skills/`, `~/.claude/plugins/` | -| MCP discovery via Cursor's `mcps/` directory | Tool list at top of system prompt (`mcp____` entries), or `.mcp.json`, or `claude mcp list` | -| Cursor cloud agents (`environment: "cloud"`, `cloud_base_branch`) | Local background subagents (`run_in_background: true`), isolated by git worktree | -| Cursor's `/goal` (standing objective across turns) | The program objective written into the run's standing orders and restated in the todolist | -| The Cursor agent store (path in the system prompt) | `~/.claude/orchestrate//`, which survives the session restarts a multi-day program expects | -| Model rule `~/.cursor/rules/pstack-models.mdc` | Override sheet `~/.claude/pstack-models.md`, included from `CLAUDE.md` | -| Multi-model panels (arena, architect, interrogate) | Provider dispatch owns the default panel: `claude:fable@max`, `codex:gpt-6-astra@high`, `grok:grok-4.7@xhigh`, `claude:opus@max`, and the architect default: `codex:gpt-6-astra@high`, `claude:fable@max`. Same-provider lanes stay native; external lanes use the bundled runner. | - -### Cross-vendor dispatch - -The earlier port collapsed panels to Claude-only models. The bundled runner restores upstream's cross-provider judgment signal without adding a daemon or model-router service. Claude Code shells out to Codex and Grok; Codex shells out to Claude and Grok. The top-level parent chooses every route and each external process receives a complete task directly, so there is no supervising model invocation and no child-side harness detection. - -### What's deliberately kept - -- The `poteto-agent` subagent ID and all references to it. -- `run_in_background: true` on Agent calls (Claude Code supports it). -- `/loop`, `/deslop`, `/babysit` slash references in skill bodies — they all resolve in Claude Code now. -- The principle/playbook structure and upstream principle prose, except the local correctness edits in `principle-attack-the-premise` and `principle-test-behavior-not-implementation`. - -### What's deliberately not ported - -- **`automations/benny/`** (upstream `0452e08`, the only pstack change between `e46364b` and v0.10.0) — a dormant Slack issue-triage and reproduce-and-fix automation pack built on Cursor's event-triggered automations. It registers no slash skills even upstream, so excluding it changes nothing about the ported plugin's behavior. Porting it would require Cursor's event-trigger runtime, Slack, and tracker plumbing that Open Pstack does not provide. -- **`docs/guide/`** (upstream `02c03a9`, `0b7ef5b`, `424829e`) — the ten-chapter usage tutorial and its six screenshots (2.3 MB). It teaches pstack through Cursor's UI, sticky mode, and cloud agents, so a faithful port would be a rewrite rather than a sync, and none of it ships as skill content. Read it upstream at [cursor/plugins/pstack/docs/guide](https://github.com/cursor/plugins/tree/main/pstack/docs/guide); the concepts map through the substitution table above. -- **`make-bot-ui`** (upstream `799151d`, relocated by `6fecddb`) uses Cursor routines, webhook events, hosted bot state, and Cursor UI primitives that have no shared Claude Code and Codex mapping. A provider-specific rewrite would be a separate feature, not an upstream sync. -- **Solo code defaults** (upstream `23a56e2` moved them to Fable; `889ec4b` and `70b2dc8` move them to Grok 4.7). pstack-flex keeps `bug-fix`, `perf-issue`, and `hillclimb` on a Codex model (`codex:gpt-6.1-sol@high`). -- **Sticky mode** (upstream `#144`) — Cursor-only `mode`/`icon`/`color`/`reminder` frontmatter with no Claude Code equivalent. The port's 0.9.5 SessionStart hook was the analog. pstack-flex turned that hook into an opt-in gate, so start `poteto-mode` by name. -- **`is_background: true` on `poteto-agent`** (upstream `99559f2`) — Cursor names this key differently. Claude-native frontier definitions use `background: true`; ad-hoc `poteto-agent` calls remain background dispatches at the call site. -- **`cursor-team-kit` beyond the seven imported skills** — the rest either duplicate Claude Code built-ins (`verify-this` → the `verify` skill and built-in verification discipline; `check-compiler-errors` → LSP diagnostics; `control-cli`/`control-ui` → `run`/`verify`, already the substitution targets) or overlap skills this port ships (`loop-on-ci`, `review-and-ship`, `weekly-review` vs `babysit`, `fix-ci`, `make-pr-easy-to-review`, `what-did-i-get-done`). `pr-review-canvas` is Cursor-UI-specific. - -### Forking note - -Editing skill bodies forks this from upstream. Re-syncing to a future pstack release means re-applying the substitution table. The full re-port recipe is in [CHANGES.md](../CHANGES.md). - -## License - -MIT. Three upstream LICENSE files are preserved: - -- [LICENSE](../LICENSE) — pstack (Lauren Tan) -- [LICENSE-cursor-team-kit](../LICENSE-cursor-team-kit) — Cursor (covers the `deslop` and `thermo-nuclear-code-quality-review` skills) -- [LICENSE-superpowers](../LICENSE-superpowers) — superpowers, Jesse Vincent (covers the vendored `hooks/run-hook.cmd`) +| Wrong role assignment | `pstack-context` shows the active sheet and parsed role. Project scope replaces global scope. | +| Role never starts | The selected native route or external dropout identifies the unavailable tool or model. | +| Model differs from the request | The receipt's requested model, reported model, and evidence distinguish invocation from provider proof. | +| Writer cannot verify | The access contract identifies allowed tools. OpenCode test and build requirements are unsupported. | +| Setup fails | Failed probes leave the previous sheet and integration intact. | + +## Maintainer checks + +`bash scripts/check.sh` runs Bun tests, strict typechecks, manifest parsing, static invariants, and Claude plugin validation when that CLI is installed. Local checks are one part of verification. [LIVE-GATE.md](LIVE-GATE.md) defines the installed behavior required before merge or release. + +Source pins, substitutions, and sync procedure are in [UPSTREAM.md](../UPSTREAM.md). Copyright and license sources are in [NOTICE.md](../NOTICE.md). Durable work belongs in this repository's GitHub Issues. diff --git a/plugins/pstack/.claude-plugin/plugin.json b/plugins/pstack/.claude-plugin/plugin.json index 61c3b97c..9c8ae07b 100644 --- a/plugins/pstack/.claude-plugin/plugin.json +++ b/plugins/pstack/.claude-plugin/plugin.json @@ -2,7 +2,7 @@ "name": "pstack", "displayName": "pstack", "version": "1.5.0", - "description": "if you want to go fast, go deep first. pstack helps you write less, but higher quality code. rigorous agent workflows you can parallelize with confidence. Ported from cursor/plugins/pstack for Claude Code and Codex.", + "description": "Shared pstack engineering workflows with coding-agent harness adapters and configurable model lanes.", "author": { "name": "Lauren Tan" }, diff --git a/plugins/pstack/.codex-plugin/plugin.json b/plugins/pstack/.codex-plugin/plugin.json index d87d532b..6d93c654 100644 --- a/plugins/pstack/.codex-plugin/plugin.json +++ b/plugins/pstack/.codex-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "pstack", "version": "1.5.0", - "description": "if you want to go fast, go deep first. pstack helps you write less, but higher quality code. rigorous agent workflows you can parallelize with confidence. Codex port of the Claude Code plugin; skills are shared, tool names resolve via skills/poteto-mode/references/codex-tools.md.", + "description": "Shared pstack engineering workflows with coding-agent harness adapters and configurable model lanes.", "author": { "name": "Lauren Tan" }, @@ -21,8 +21,8 @@ "interface": { "logo": "./assets/logo.png", "displayName": "pstack", - "shortDescription": "Rigorous, parallelizable agent workflows: go deep first, write less, verify everything", - "longDescription": "pstack guides agent work through poteto-mode's principles, parallel design exploration (architect/arena), adversarial multi-model review (interrogate), root-cause debugging, prose deslopping, and verified delivery. Skills are shared with the Claude Code build; on Codex, tool names resolve via the codex-tools mapping.", + "shortDescription": "Shared engineering workflows with configurable model lanes", + "longDescription": "pstack-flex shares pstack skills across coding-agent harnesses. poteto-mode guides investigation, design, implementation, and verification; architect and interrogate compare configured model lanes.", "developerName": "Lauren Tan (original), pstack-flex maintainers", "category": "Developer Tools", "capabilities": [ diff --git a/plugins/pstack/hooks/session-start-context.md b/plugins/pstack/hooks/session-start-context.md index 766054ed..0b8977c2 100644 --- a/plugins/pstack/hooks/session-start-context.md +++ b/plugins/pstack/hooks/session-start-context.md @@ -1,13 +1,15 @@ pstack is installed, and it runs only when the user asks for it. -Invoke a `pstack:*` skill only when one of these holds: +Invoke a pstack skill only when one of these holds: -- The user typed `/pstack:`, or the current request names a pstack skill, pstack, or poteto-mode. A request for pstack that names no skill enters through `pstack:poteto-mode`. +- The user typed `/pstack:`, or the current request names a pstack skill, pstack, or poteto-mode. A request for pstack that names no skill enters through `poteto-mode`. - A standing user instruction in CLAUDE.md or AGENTS.md asks for pstack on this kind of task. - A pstack skill the user started this session is still working on the current task, and it routes to another pstack skill or principle leaf. Otherwise, do the task directly and invoke no pstack skill. This applies even when a pstack skill's description says to always apply it, to apply it to certain files, or to apply it in a situation that matches the task. +Use the installed skill name. Some harnesses qualify it as `pstack:`; others expose `` directly. + If a pstack skill dispatched you as a subagent, follow your dispatch prompt instead of this block. diff --git a/plugins/pstack/skills/poteto-mode/SKILL.md b/plugins/pstack/skills/poteto-mode/SKILL.md index 8cd59fd4..bc427ca4 100644 --- a/plugins/pstack/skills/poteto-mode/SKILL.md +++ b/plugins/pstack/skills/poteto-mode/SKILL.md @@ -7,7 +7,7 @@ description: poteto's agent style for concise, detailed responses, deliberate su ## Platform Adaptation -These skills share one tree across Claude Code and Codex. Read [`references/provider-dispatch.md`](references/provider-dispatch.md) whenever a configured role launches. It defines the provider-qualified model descriptors, native/external route table, launcher, isolation, receipts, and dropout policy. Children never choose routes. When a skill names a Claude tool or built-in skill (`run`, `verify`, `plugin-dev:skill-development`), read [`references/codex-tools.md`](references/codex-tools.md) for the Codex equivalent. +These skills share one tree across supported harnesses. Read [`references/provider-dispatch.md`](references/provider-dispatch.md) whenever a configured role launches. It defines the provider-qualified model descriptors, native/external route table, launcher, isolation, receipts, and dropout policy. Children never choose routes. When a skill names a Claude tool or built-in skill (`run`, `verify`, `plugin-dev:skill-development`), read [`references/codex-tools.md`](references/codex-tools.md) for the current harness adapter. ## Non-negotiables diff --git a/plugins/pstack/skills/poteto-mode/references/codex-tools.md b/plugins/pstack/skills/poteto-mode/references/codex-tools.md index 6260c0ce..8c154ab8 100644 --- a/plugins/pstack/skills/poteto-mode/references/codex-tools.md +++ b/plugins/pstack/skills/poteto-mode/references/codex-tools.md @@ -1,64 +1,66 @@ -# Codex tool mapping for pstack +# Harness integration -pstack skills retain Claude Code tool language (`Skill`, `Agent`, `AskUserQuestion`) in shared prose. On Codex the files are the same; only those tool names resolve differently. Model execution is not translated here. Read [`provider-dispatch.md`](provider-dispatch.md) for the parent-owned Claude/Codex/Grok route table and provider-qualified descriptors. +Shared skills retain upstream tool names. This file is the harness adapter boundary; its path remains stable for upstream skill references. Model providers are separate. Read [provider dispatch](provider-dispatch.md) for models and receipts. -## Tool actions +The parent matches the current session and tools to the exact `id` in [harnesses.ts](../scripts/harnesses.ts) once. Do not infer it inside children. Run `scripts/pstack-context --parent --cwd ` under the installed poteto-mode skill before the first configured dispatch. It reads the selected model sheet. Keep that context for the run. It never spawns a lane or writes configuration. -| pstack / Claude action | Codex equivalent | -|------------------------|------------------| -| Read a file | `shell` (`cat`, `head`, `tail`) | -| Create / edit / delete a file | `apply_patch` | -| Run a shell command | `shell` | -| Search file contents / find files | `shell` (`rg`, `grep`, `find`, `ls`) | -| Fetch a URL | `shell` with `curl` / `wget` | -| Search the web | `web_search` | -| Invoke a skill (the `Skill` tool, `/command`) | Skills load natively. Follow the instructions presented. | -| `paths` frontmatter scopes automatic loading | Claude Code only. On Codex, invoke `pstack:typescript-best-practices` by name. | -| Dispatch a subagent (the `Agent`/`Task` tool) | `spawn_agent` | -| Dispatch N parallel subagents in one turn | N `spawn_agent` calls in one response | -| Wait for a subagent result | `wait_agent` | -| Free a finished subagent slot | `close_agent` | -| Track tasks (the todolist / `TodoWrite`) | `update_plan` | -| Ask the human a fixed-choice question (`AskUserQuestion`) | Ask in plain text and let the user answer. Codex has no structured-choice tool. | +The output includes global/project paths, integration targets, the selected sheet, normalized role descriptors, routes, model labs, and lane capabilities. Report stale `normalizedFrom` values once; setup persists them only after probes and confirmation. Model-family validation uses provider dispatch's matrices. For a task that requires shell execution, add `--role '' --require-shell`; a lane lacking that capability fails explicitly. An inherited lane uses the actual session's available tools. -Subagent dispatch needs `multi_agent` enabled. Add to `~/.codex/config.toml`: +When a role has no configured row, use the calling skill's defaults with the same routing and capability rules. `--role` checks only persisted assignments. -```toml -[features] -multi_agent = true -``` +`lab: null` means the descriptor does not establish a lab. OpenCode aggregators can serve several labs; verify the model's maker before counting panel diversity. Never count an unknown aggregator as an additional lab. -Without it, the native Codex lane is a named dropout. Independent external lanes still run, and the parent records the reduced provider count. Never collapse a panel into a sequential single-model pass. +## Tools -## Subagent policy +| Shared action | Claude Code | Codex | OpenCode | +| --- | --- | --- | --- | +| Read | `Read` | file or shell tool | `read` | +| Write/edit | `Write`, `Edit` | `apply_patch` | `write`, `edit` | +| Shell | `Bash` | persistent exec session | `bash` | +| Search files | `Grep`, `Glob` | `rg` | `grep`, `glob` | +| Fetch/search web | `WebFetch`, `WebSearch` | available web tools | `webfetch`, `websearch` | +| Load skill | `Skill`, `/pstack:` | load `pstack:` | `skill` with `` | +| Subagent | `Agent` | `spawn_agent` | `task` with `general` | +| Wait | retained task handle | `wait_agent` | task response | +| Tasks | `TodoWrite` | available plan tool | `todowrite` | +| Ask user | `AskUserQuestion` | available question tool or plain text | `question` | -poteto-mode's Subagents section sets Claude-specific defaults (`subagent_type: "poteto-agent"`, `run_in_background: true`). On Codex: +`paths` frontmatter is a Claude loading feature. Elsewhere invoke the named skill explicitly. An unavailable native subagent is a named dropout; never replace it with an external or weaker model silently. -- There is no `poteto-agent` subagent type. Route an ad-hoc subagent through poteto-mode's style by dispatching a `spawn_agent` whose instructions tell it to read the `poteto-mode` skill in full first. -- `spawn_agent` calls already run concurrently with your turn, so `run_in_background: true` has no separate flag. Issue the dispatch and continue. -- There is no `comment-sicko` subagent type either. The **no-comments** skill spawns it on Claude Code; on Codex dispatch a `spawn_agent` whose instructions tell it to read `agents/comment-sicko.md` in full first. -- Claude Code runs every subagent on this machine, so the **swarm** skill's workers and the fan-out playbooks (`orchestrate`, `autopilot-full`, `autopilot-stack`) isolate writers with worktrees. The same holds on Codex. -- Keep the rest of the policy unchanged. Pass file pointers not inlined context, give each worker its own worktree or branch when they write, review every subagent's diff yourself. +## Native dispatch -## Models and providers +Use the route returned for the configured descriptor. `inherit-parent` and `auto` always request the session's current model and effort. An explicit model is native only when its provider matches the adapter's `nativeProvider`. -Do not replace every configured entry with a Codex model. `/setup-pstack` writes portable descriptors such as `claude:fable@max`, `codex:gpt-5.6-sol@max`, and `grok:grok-4.7@xhigh`. In a Codex parent, only `codex:*` is native. Route Claude and Grok descriptors through the external launcher exactly as `provider-dispatch.md` specifies. The current default panel runs four lanes across three providers (Fable and Opus are both Claude) and contains no older GPT or Claude substitute. pstack-flex gateway descriptors (`deepseek:*`, `minimax:*`, `openrouter:*`) also always route through the external launcher in a Codex parent; they are never `spawn_agent` lanes. +- Claude Code uses `Agent`. Match the descriptor's `(provider, model)` to the matrix's native agent stem, then use `pstack--`. Those definitions set the rolling model alias, effort, and background execution. Ad-hoc inherited work uses `poteto-agent`. +- Codex uses `spawn_agent` with the selected `model` and `reasoning_effort`. Enable `multi_agent` in the Codex feature configuration. There is no `poteto-agent` type; inherited helpers read poteto-mode, and comment reviewers read `agents/comment-sicko.md`. Spawn calls already run concurrently. +- OpenCode's `task` cannot select a model per call. It serves only inherited roles. Its qualified model descriptors use the external runner, even when their provider is `opencode`. A `general` task reads the relevant agent instruction file. -## Claude built-in skills pstack references +Pass the full task, access mode, grounding paths, and unique output location. Give writers dedicated worktrees. Launch independent lanes together, retain handles, then drain every lane before judging. -Some triggers name skills that ship with Claude Code, not pstack. They do not exist on Codex. Substitute the behavior: +## External launch -| Claude built-in named in pstack | On Codex | -|---------------------------------|----------| -| `run` (drive a CLI/TUI to see a change work) | Run the app yourself via `shell` and observe the real output. | -| `verify` (drive a UI to confirm a fix) | Drive the UI with whatever automation you have, or hand the user a concrete manual check. Do not claim done without observing the artifact. | -| `plugin-dev:skill-development` (Claude's SKILL.md authoring guidance) | Follow your platform's skill-authoring guidance; the `writing-skills` skill if present. Keep `name` + `description` frontmatter and progressive disclosure. | -| `loop` (recurring/self-paced re-invocation, used by `babysit`) | Codex has no `loop` skill. Re-run the step yourself on a cadence, or use a Codex scheduled task if available. | +Use the adapter's `launch` recipe. The runner's own timeout is absent unless the user or task supplies a deadline. -## Vendored scripts +- `task` uses Claude's Bash tool with `run_in_background: true`. Retain its task ID. A foreground call has a ten-minute ceiling. +- `session` uses Codex's persistent exec session. Retain its session ID and poll that handle. +- `detached` uses OpenCode's shell. Its foreground calls default to two minutes and have no background handle. Start `nohup pstack-runner … >"" 2>&1 & echo $!` and retain the PID. Drain with short receipt checks. `kill -0 ` checks liveness; `kill -TERM ` requests cancellation. Session abort does not reach a detached lane. -`skills/poteto-mode/scripts/` ships the `watch-pr` PR watcher, the `orch` store CLI, `worktree-audit.sh`, and `runner/pstack-runner`. They are plain bun and bash, so they run the same on Codex; invoke them through `shell`. The external runner additionally needs the assigned `claude`, `codex`, or `grok` executable already authenticated. It rejects a Codex provider when Codex is the parent because that lane belongs on native `spawn_agent`. The other scripts need `bun`, `gh`, (for stack work) `gt`, and (for `worktree-audit.sh`) `jq` and `rg`. `worktree-audit.sh` reads Claude Code transcripts under `~/.claude/projects/`; point it at your runtime's transcript directory instead when you run it elsewhere. +A receipt is terminal only after it contains valid complete JSON. Empty reserved files are still running. Do not judge while another lane is writing. -## Instructions file +## Configuration integration -Where a pstack skill says "your instructions file", on Codex that is `AGENTS.md` (project root, plus `~/.codex/AGENTS.md` global). On Claude Code it is `CLAUDE.md`. +Paths and integration type come from `pstack-context`, including `CLAUDE_CONFIG_DIR`, `CODEX_HOME`, and `XDG_CONFIG_HOME` overrides. Shared scope resolution is in [configuration.ts](../scripts/configuration.ts). Setup asks which scope to edit every run; dispatch uses the existing project sheet in preference to global, without merging roles. + +Project setup writes the sheet and lists its repository-relative path in the common git directory's `info/exclude`. It does not change global instructions. Global setup applies the selected integration recipe: + +- `include` adds one `@` line to the target instructions file. Leave unrelated lines alone. +- `mirror-block` copies the exact sheet bytes between `` and `` in the target instructions file. Replace that block on rerun. If neither marker exists, append one block. Refuse an unmatched, duplicate, or reversed marker. +- `instructions-array` adds the absolute sheet path once to `instructions` in the first existing integration target. Preserve JSONC comments, unrelated keys, and entries. If no target exists, create the JSON target with `$schema` and that array. Refuse malformed config. Do not create a global `AGENTS.md`, which would shadow OpenCode's Claude-instruction fallback. + +Before changing either target, snapshot its bytes. Write only after exact model probes and user confirmation, read back both targets, and restore both on failure. An unchanged rerun leaves bytes unchanged. These recipes are the only harness-specific write rules; setup owns the common transaction. + +## Built-ins and local state + +Where upstream names `run`, drive the CLI yourself. Where it names `verify`, drive the UI with available tools and observe the artifact. Where it names `plugin-dev:skill-development`, use the current harness's skill-authoring guidance. Where it names `loop`, use an available recurring task or rerun the step at the specified cadence. + +Use the current harness's consumed instructions file for standing rules. Transcript paths and native agent stores are host state, not lane-provider configuration. `worktree-audit.sh` reads Claude transcripts; its chat column is unavailable on other hosts. Do not create a Claude transcript directory to compensate. diff --git a/plugins/pstack/skills/poteto-mode/references/provider-dispatch.md b/plugins/pstack/skills/poteto-mode/references/provider-dispatch.md index 407f44eb..ef94af18 100644 --- a/plugins/pstack/skills/poteto-mode/references/provider-dispatch.md +++ b/plugins/pstack/skills/poteto-mode/references/provider-dispatch.md @@ -43,18 +43,9 @@ This line is the single source for the architect default. `setup-pstack`'s first ## Sheet scope -pstack-flex addition. A model sheet is either global or project-scoped. +A model sheet is either global or project-scoped. Run the installed `scripts/pstack-context --parent --cwd ` once before the first configured dispatch. It returns paths and normalized descriptors using the shared [configuration parser](../scripts/configuration.ts) and [harness table](../scripts/harnesses.ts). -| Parent | Global sheet | Project sheet | -|---|---|---| -| Claude Code | `~/.claude/pstack-models.md` | `/.claude/pstack-models.md` | -| Codex | `~/.codex/pstack-models.md` | `/.codex/pstack-models.md` | - -The project root is the top level of the repository's primary checkout, so every worktree of one repository shares one project sheet. Read it as the parent directory of `git rev-parse --path-format=absolute --git-common-dir`. Outside a git repository there is no project sheet. - -Before the first configured role launches in a run, the parent reads its project sheet path once. If the file exists, it is the model sheet for the whole run and replaces the global sheet, including a global sheet already loaded into context. If the file does not exist, the global sheet applies. Never merge the two role by role: every sheet carries every documented role, so one sheet always answers. Say which sheet is in use when reporting a panel. - -A project sheet is private to the machine. `setup-pstack` writes it, lists it in `.git/info/exclude`, and never commits it. Deleting the file returns the project to the global sheet. +An existing project sheet replaces the global sheet for the entire run. Never merge the two role by role. Worktrees share the primary checkout's project sheet; outside git only global scope exists. Setup keeps a project sheet private through the common git directory's `info/exclude`. Deleting it restores global scope. Report the selected sheet with the panel. ## Flex model matrix @@ -74,12 +65,26 @@ The `openrouter` row is open. Its Model cell stands for any model ID in OpenRout MiniMax preview requires Token Plan access; set `MINIMAX_API_KEY` to the eligible subscription key. A pay-as-you-go key is not proof of preview access. The preview always thinks and supports `low` through `max`; do not disable thinking. M3 thinking is off by default at the API and requires adaptive thinking to enable it; its effort flag does not imply preview-style depth control. Selectable efforts are runner requests, not a claim that every provider applies five distinct reasoning levels. Verify CLI forwarding and model access with live probes. Sources: [MiniMax models](https://platform.minimax.io/docs/guides/models-intro), [MiniMax thinking controls](https://platform.minimax.io/docs/api-reference/text-anthropic-api), [DeepSeek Anthropic compatibility](https://api-docs.deepseek.com/guides/anthropic_api) (checked 2026-09-27). -Flex lanes have no Claude-native agent stem and always take the external runner in both parents. The base URL is a documented default; override it with `DEEPSEEK_BASE_URL`, `MINIMAX_BASE_URL`, or `OPENROUTER_BASE_URL` (OpenRouter's must end in `/api`, not `/api/v1`), and confirm it against the provider's current Claude Code guide during setup's live probe. The config dir defaults to `~/.pstack-flex/` (override: `PSTACK_FLEX__CONFIG_DIR`). Secrets stay in the environment: nothing in the sheet, the receipts, or this repository carries a key. +Flex lanes have no Claude-native agent stem and always take the external runner under every harness. The base URL is a documented default; override it with `DEEPSEEK_BASE_URL`, `MINIMAX_BASE_URL`, or `OPENROUTER_BASE_URL` (OpenRouter's must end in `/api`, not `/api/v1`), and confirm it against the provider's current Claude Code guide during setup's live probe. The config dir defaults to `~/.pstack-flex/` (override: `PSTACK_FLEX__CONFIG_DIR`). Secrets stay in the environment: nothing in the sheet, the receipts, or this repository carries a key. -Gateway receipt semantics differ from stock claude lanes in two documented ways. `costUsd` is always `null`: the claude CLI prices `total_cost_usd` at Anthropic rates, which would be fiction for third-party traffic; real prices live in [LANES.md](../../../../../docs/LANES.md), and token usage in the receipt stays accurate. Model verification accepts a case-insensitive matching provider report. An OpenRouter report must otherwise match exactly: any catalog model can be requested, so a prefix rule would accept a sibling such as `z-ai/glm-5.3-air` for `z-ai/glm-5.3`. A mismatched report fails the lane. When the endpoint reports no model, the receipt uses `modelEvidence: "pinned-argv"` and `modelVerified: false`. +Gateway receipt semantics differ from stock claude lanes in two documented ways. `costUsd` is always `null`: the claude CLI prices `total_cost_usd` at Anthropic rates, which would be fiction for third-party traffic; billing comes from the provider, and token usage in the receipt stays accurate. Model verification accepts a case-insensitive matching provider report. An OpenRouter report must otherwise match exactly: any catalog model can be requested, so a prefix rule would accept a sibling such as `z-ai/glm-5.3-air` for `z-ai/glm-5.3`. A mismatched report fails the lane. When the endpoint reports no model, the receipt uses `modelEvidence: "pinned-argv"` and `modelVerified: false`. Panel diversity rule (pstack-flex): `arena runners` and `interrogate reviewers` must span at least two distinct providers. DeepSeek plus MiniMax satisfies it. For this rule a lane's provider is the lab that made the model, not the route that reaches it. An `openrouter` lane counts as its model ID's namespace, and the namespaces `anthropic`, `openai`, `x-ai`, `deepseek`, and `minimax` count as the `claude`, `codex`, `grok`, `deepseek`, and `minimax` providers. So `openrouter:deepseek/deepseek-v4-pro` plus `deepseek:deepseek-flash` is one provider, and `openrouter:google/gemini-3.8-flash` plus `openrouter:z-ai/glm-5.3` is two. A single-provider panel is written only after the operator explicitly confirms the reduced diversity during setup, and the setup report records that confirmation. The adversarial signal comes from model diversity, so treat the override as an exception, not a configuration style. +## OpenCode lanes + +pstack-flex addition, beta: verified at the runner level, not yet from a real session. An `opencode` lane runs `opencode run` headless on any model OpenCode can reach, with OpenCode's own credentials: a provider login (`opencode providers login`) or the provider's key variable, such as `OPENROUTER_API_KEY`. It needs neither the `claude` nor the `codex` CLI, so an OpenRouter key and OpenCode are enough for a multi-model panel. + +| Family | Provider | Model | Default effort | Selectable efforts | Credentials | +|---|---|---|---|---|---| +| opencode | opencode | | high | low medium high xhigh max | OpenCode's own | + +The row is open. Its Model cell stands for any ID that `opencode models` lists, written as OpenCode's `/`, such as `opencode:openrouter/z-ai/glm-5.3@high`. Split a descriptor at the first `:` and the last `@`. Each distinct model ID is its own family with its own requested effort and probe, and setup's live probe is the gate. The effort becomes OpenCode's `--variant`, and each model offers its own variants (`opencode models --verbose`). OpenCode silently ignores a variant the model does not offer, so the launcher's preflight refuses that effort as `unavailable-model` and names the offered ones. A model that offers no variants cannot take a pstack effort and cannot run as a lane. An `opencode` lane routes through the launcher under every parent, OpenCode's included, because OpenCode's native `task` subagent cannot choose a model. + +The lane runs with `--pure`, Claude Code compatibility off, and a private in-memory OpenCode database (`OPENCODE_DB=:memory:`). OpenCode 1.18.31 can lock a shared database during parallel startup. Each invocation uses a fresh agent name with its own deny-first permissions; named `plan` and `build` agents would deep-merge ambient permissions. Read-only lanes can read, list, glob, and grep. Writers can also edit files inside their worktree. Subagents, skills, web access, MCP tools, and shell commands are denied. OpenCode has no shell sandbox, and even Git reads can execute external helpers, so these lanes cannot run Git commands, tests, or builds. Report such requirements as unsupported. Receipts use `modelEvidence: "pinned-argv"` because OpenCode's JSON does not name the answering model, and `costUsd: null` because its cost is a catalog estimate. + +For panel diversity, an `opencode` lane counts as the lab that made its model. For `openrouter//` the lab is the namespace, with the same equivalences as OpenRouter lanes. For another OpenCode provider, verify the model's maker before counting it. For example, `opencode/glm-5.3` is from `z-ai`, but the aggregator name alone does not prove that. A context with `lab: null` contributes no additional lab until verified. + ## Read-time normalization Normalize configured descriptors before matching them to the matrix or choosing a route. If a provider-qualified Claude model starts with `claude-fable-` or `claude-opus-` and its remaining revision contains only digits and hyphens, replace that model component in memory with `fable` or `opus`. Preserve provider, effort, role, and lane order. Use only the normalized descriptor for native dispatch or runner argv. Never pass the versioned predecessor to Claude. @@ -92,23 +97,13 @@ This read-time rule makes an older installed sheet use the latest family revisio The top-level harness resolves the route once. A child receives an assigned provider, model, effort, access mode, prompt, working directory, and output path. A child never detects the harness, chooses a provider, or launches another model. Environment markers may corroborate the top-level harness before fan-out, but nested processes inherit parent markers and must not use them for routing. -| Parent | `claude:*` | `codex:*` | `grok:*` | `deepseek:*` | `minimax:*` | `openrouter:*` | -|---|---|---|---|---|---|---| -| Claude Code | native `Agent` | external runner | external runner | external runner | external runner | external runner | -| Codex | external runner | native `spawn_agent` | external runner | external runner | external runner | external runner | - -Flex gateway descriptors are never native, even under a Claude Code parent: the gateway lane must run in its own process with injected endpoint, token, and isolated config dir, which the parent's native `Agent` primitive cannot provide. +The harness table declares `nativeProvider`. A qualified descriptor takes the native route only when its execution provider matches that field. All other descriptors take the external runner. An adapter with no native provider uses external execution for every qualified descriptor. The runner validates an already-selected external route; it does not detect or choose the parent. -`inherit-parent` and `auto` remain aliases. They use the parent's current model and effort through its native subagent primitive. In a panel they still consume one lane, but they reduce provider diversity; say so in the synthesis record. +`inherit-parent` and `auto` use the parent's current model and effort through its native subagent primitive. They still count as a lane and reduce model diversity. Read [harness integration](codex-tools.md) for the selected adapter's tools, native invocation, configuration wiring, and retained launch handles. Gateway lanes stay external because native agents cannot receive their isolated endpoint and credentials. ## Native lanes -Native dispatch avoids a second CLI startup and its base context. - -- Claude Code: match the descriptor's `(provider, model)` to one model-matrix row, then dispatch it through `pstack--` using that row's Claude-native agent stem and the descriptor's effort. Those definitions select the rolling model alias, requested effort, and `background: true`. `pstack-fable-max` and `pstack-opus-xhigh` remain in that set. Pass the complete task, grounding paths, access mode, and unique output location in the `Agent` prompt. Retain the task handle and drain it only after fan-out. -- Codex: call `spawn_agent` with the descriptor's model and `reasoning_effort`, the complete task, grounding paths, access mode, and unique output location. Use an isolated worktree for a writer. Codex subagents already run concurrently. - -Do not send a same-provider descriptor to the external runner. It rejects that call because the native route is cheaper and already available. +Use the adapter's native dispatch recipe with the requested model, effort, access mode, full task, grounding paths, and unique output location. The model matrix's native agent stem supplies agent definitions where required. Never send a native descriptor to the external runner, and never reinterpret an external descriptor as a native model slug. ## External lanes @@ -116,8 +111,8 @@ The launcher lives at `skills/poteto-mode/scripts/runner/pstack-runner` under th ```text pstack-runner \ - --parent \ - --provider \ + --parent \ + --provider \ --model \ --effort \ --mode \ @@ -138,10 +133,7 @@ Grok authentication preflight has one bounded retry. If the first `grok models` The parent tool sandbox still governs whether a subscribed child CLI can reach its credentials and network. Run setup's live probe from the actual parent profile. A blocked external CLI is a loud dropout, not a reason to elevate permissions or substitute a model silently. -The parent invocation must itself be resumable background work: - -- Claude Code: call the launcher through a Bash tool invocation with `run_in_background: true` and retain its task ID. A foreground Bash tool call has an automatic ten-minute ceiling even when the runner's own timeout is longer. Shelling out with `&` and losing the task handle is not equivalent. -- Codex: run the launcher in a persistent exec session that returns a session ID, then wait or poll that handle. Do not hold one foreground tool call open for the model's full runtime. +Use the selected adapter's retained background launch recipe in [harness integration](codex-tools.md). Foreground tool limits belong to that adapter; they never become a lane timeout. Start the background process, continue launching the other lanes, then drain their handles. Native and external lanes belong in the same fan-out phase. @@ -157,7 +149,7 @@ Success requires all of these: 1. Exit status `0`. 2. Receipt status `complete`. -3. Either `modelVerified: true` with `modelEvidence: "provider-report"`, or a Codex receipt with `reportedModel: null`, `modelVerified: false`, and `modelEvidence: "pinned-argv"`, or a gateway (`deepseek`/`minimax`/`openrouter`) receipt with `modelVerified: false` and `modelEvidence: "pinned-argv"` when the endpoint does not echo the requested slug. For Claude's `fable` and `opus` aliases, the concrete provider report must belong to the requested family. Codex 0.149.0 accepts the exact `--model` argument but does not report the served model in its JSONL stream. Gateway reports match case-insensitively because third-party endpoints are inconsistent about slug casing; an OpenRouter report must otherwise match the requested ID exactly. +3. Either `modelVerified: true` with `modelEvidence: "provider-report"`, or a Codex receipt with `reportedModel: null`, `modelVerified: false`, and `modelEvidence: "pinned-argv"`, or a gateway (`deepseek`/`minimax`/`openrouter`) receipt with `modelVerified: false` and `modelEvidence: "pinned-argv"` when the endpoint does not echo the requested slug, or an `opencode` receipt with `reportedModel: null`, `modelVerified: false`, and `modelEvidence: "pinned-argv"`. For Claude's `fable` and `opus` aliases, the concrete provider report must belong to the requested family. Codex 0.149.0 accepts the exact `--model` argument but does not report the served model in its JSONL stream. Gateway reports match case-insensitively because third-party endpoints are inconsistent about slug casing; an OpenRouter report must otherwise match the requested ID exactly. 4. A non-empty output file. The receipt also carries elapsed time, token usage when the CLI exposes it, and cost when available. Keep it with the arena or review artifacts so parent-harness comparisons are evidence-based. diff --git a/plugins/pstack/skills/poteto-mode/scripts/configuration.test.ts b/plugins/pstack/skills/poteto-mode/scripts/configuration.test.ts new file mode 100644 index 00000000..65760322 --- /dev/null +++ b/plugins/pstack/skills/poteto-mode/scripts/configuration.test.ts @@ -0,0 +1,143 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import { execFileSync } from "node:child_process"; +import { mkdtempSync, mkdirSync, readFileSync, rmSync, realpathSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { configurationPaths, modelLab, parseDescriptor, parseModelSheet, readConfiguration } from "./configuration.ts"; +import { HARNESSES, laneRoute } from "./harnesses.ts"; + +const fixtures: string[] = []; +afterEach(() => { for (const path of fixtures.splice(0)) rmSync(path, { recursive: true, force: true }); }); + +function fixture() { + const root = realpathSync(mkdtempSync(join(tmpdir(), "pstack-configuration-"))); + fixtures.push(root); + const cwd = join(root, "repo"); + mkdirSync(cwd); + execFileSync("git", ["init", "--quiet", cwd]); + return { root, cwd, env: { HOME: join(root, "home") } }; +} + +function write(path: string, text: string) { + mkdirSync(join(path, ".."), { recursive: true }); + writeFileSync(path, text); +} + +describe("configuration", () => { + it("parses qualified IDs without losing catalog punctuation or lane order", () => { + const sheet = parseModelSheet("# pstack model configuration\ninterrogate reviewers: openrouter:z-ai/glm-5.3:free@high, opencode:openrouter/~google/gemini-flash-latest@max, auto\n"); + expect(sheet["interrogate reviewers"].map((lane) => lane.descriptor)).toEqual([ + "openrouter:z-ai/glm-5.3:free@high", "opencode:openrouter/~google/gemini-flash-latest@max", "auto", + ]); + expect(parseDescriptor("claude:claude-fable-5-3@max")).toMatchObject({ + model: "fable", effort: "max", normalizedFrom: "claude:claude-fable-5-3@max", + }); + }); + + it("refuses malformed descriptors, model routers, duplicate and empty roles", () => { + for (const value of ["fable", "claude:@high", "codex:gpt-6-sol@lowish", "future:model@high", "openrouter:openrouter/auto@high", "opencode:bare-model@high", "opencode:openrouter/openrouter/auto@high"]) { + expect(() => parseDescriptor(value)).toThrow(); + } + expect(() => parseModelSheet("how explorer: auto\nhow explorer: inherit-parent")).toThrow("duplicate"); + expect(() => parseModelSheet("how explorer: ")).toThrow("empty"); + expect(() => parseModelSheet("# no assignments")).toThrow("no role"); + }); + + it("separates execution routes from model labs", () => { + const pairs = [ + ["claude:fable@max", "opencode:anthropic/fable@max", "claude"], + ["codex:gpt-6-sol@high", "opencode:openrouter/openai/gpt-6-sol@high", "codex"], + ["deepseek:deepseek-flash@high", "openrouter:deepseek/deepseek-v4-pro@high", "deepseek"], + ]; + for (const [direct, routed, lab] of pairs) { + expect(modelLab(parseDescriptor(direct))).toBe(lab); + expect(modelLab(parseDescriptor(routed))).toBe(lab); + } + for (const descriptor of ["opencode:opencode/glm-5.3@high", "opencode:amazon-bedrock/claude-fable@high"]) { + expect(modelLab(parseDescriptor(descriptor))).toBeNull(); + } + expect(modelLab(parseDescriptor("opencode:openrouter/~google/gemini-flash-latest@max"))).toBe("google"); + expect(laneRoute("claude", "claude")).toBe("native"); + expect(laneRoute("codex", "codex")).toBe("native"); + expect(laneRoute("codex", "claude")).toBe("external"); + expect(laneRoute("opencode", "opencode")).toBe("external"); + expect(laneRoute("opencode", "codex")).toBe("external"); + for (const harness of HARNESSES) expect(laneRoute(harness.id, null)).toBe("native"); + }); + + it("uses host directories and their overrides without changing project paths", () => { + const { cwd, env, root } = fixture(); + for (const harness of HARNESSES) { + const paths = configurationPaths(harness.id, cwd, env); + expect(paths.globalSheet).toBe(join(env.HOME, harness.configDirectory, "pstack-models.md")); + expect(paths.projectSheet).toBe(join(cwd, harness.projectDirectory, "pstack-models.md")); + } + for (const [parent, key] of [["claude", "CLAUDE_CONFIG_DIR"], ["codex", "CODEX_HOME"]] as const) { + const paths = configurationPaths(parent, cwd, { ...env, [key]: join(root, "override") }); + expect(paths.globalSheet).toBe(join(root, "override/pstack-models.md")); + } + expect(configurationPaths("opencode", cwd, { ...env, XDG_CONFIG_HOME: join(root, "xdg") }).globalSheet) + .toBe(join(root, "xdg/opencode/pstack-models.md")); + }); + + it("selects one project sheet, returns to global after deletion, and never merges roles", () => { + const { cwd, env } = fixture(); + for (const harness of HARNESSES) { + const paths = configurationPaths(harness.id, cwd, env); + if (paths.projectSheet === null) throw new Error("fixture must have a project sheet"); + write(paths.globalSheet, "how explorer: codex:gpt-6-luna@high\njudgment and prose: claude:fable@max\n"); + write(paths.projectSheet, "how explorer: opencode:openrouter/z-ai/glm-5.3@high\n"); + const selected = readConfiguration(harness.id, cwd, env); + expect(selected.activeSheet).toBe(paths.projectSheet); + expect(Object.keys(selected.roles)).toEqual(["how explorer"]); + expect(selected.roles["how explorer"][0].route).toBe("external"); + expect(selected.roles["how explorer"][0].capabilities?.shell).toBe(false); + expect(readFileSync(paths.globalSheet, "utf8")).toContain("judgment and prose:"); + rmSync(paths.projectSheet); + expect(readConfiguration(harness.id, cwd, env).activeSheet).toBe(paths.globalSheet); + } + }); + + it("shares primary checkout configuration with linked worktrees and uses global outside git", () => { + const { cwd, env, root } = fixture(); + execFileSync("git", ["-c", "user.name=Fixture", "-c", "user.email=fixture@example.invalid", "commit", "--allow-empty", "-qm", "fixture"], { cwd }); + const linked = join(root, "linked"); + execFileSync("git", ["worktree", "add", "--detach", linked], { cwd, stdio: "ignore" }); + for (const harness of HARNESSES) { + expect(configurationPaths(harness.id, linked, env)).toEqual(configurationPaths(harness.id, cwd, env)); + expect(configurationPaths(harness.id, root, env).projectSheet).toBeNull(); + } + }); + + it("refuses an invalid working directory instead of selecting global configuration", () => { + const { cwd, env } = fixture(); + expect(() => configurationPaths("codex", join(cwd, "missing"), env)).toThrow(); + const file = join(cwd, "file"); + write(file, "synthetic fixture"); + expect(() => configurationPaths("codex", file, env)).toThrow("cwd is not a directory"); + }); + + it("fails explicitly when a configured lane cannot provide required shell access", () => { + const { cwd, env } = fixture(); + const paths = configurationPaths("opencode", cwd, env); + write(paths.globalSheet, "feature, refactoring: opencode:openrouter/z-ai/glm-5.3@high\n"); + const cli = join(import.meta.dir, "pstack-context"); + const run = Bun.spawnSync([process.execPath, cli, "--parent", "opencode", "--cwd", cwd, "--role", "feature, refactoring", "--require-shell"], { env: { PATH: process.env.PATH, ...env } }); + expect(run.exitCode).toBe(64); + expect(run.stderr.toString()).toContain("does not support shell execution"); + expect(readFileSync(paths.globalSheet, "utf8")).toBe("feature, refactoring: opencode:openrouter/z-ai/glm-5.3@high\n"); + }); + + it("lets setup inspect paths before repairing an invalid project sheet", () => { + const { cwd, env } = fixture(); + const paths = configurationPaths("codex", cwd, env); + if (paths.projectSheet === null) throw new Error("fixture must have a project sheet"); + write(paths.projectSheet, "how explorer: invalid-model\n"); + const cli = join(import.meta.dir, "pstack-context"); + const run = Bun.spawnSync([process.execPath, cli, "--parent", "codex", "--cwd", cwd, "--paths-only"], { env: { PATH: process.env.PATH, ...env } }); + expect(run.exitCode).toBe(0); + expect(JSON.parse(run.stdout.toString()).paths.projectSheet).toBe(paths.projectSheet); + expect(readFileSync(paths.projectSheet, "utf8")).toBe("how explorer: invalid-model\n"); + expect(() => readConfiguration("codex", cwd, env)).toThrow("invalid model descriptor"); + }); +}); diff --git a/plugins/pstack/skills/poteto-mode/scripts/configuration.ts b/plugins/pstack/skills/poteto-mode/scripts/configuration.ts new file mode 100644 index 00000000..5026e9fa --- /dev/null +++ b/plugins/pstack/skills/poteto-mode/scripts/configuration.ts @@ -0,0 +1,115 @@ +import { existsSync, readFileSync, statSync } from "node:fs"; +import { homedir } from "node:os"; +import { dirname, join, resolve } from "node:path"; +import { execFileSync } from "node:child_process"; +import { harnessAdapter, laneRoute, type Parent } from "./harnesses.ts"; +import { versionedClaudeAlias } from "./runner/model-aliases.ts"; +import { modelRefusal } from "./runner/model-refusal.ts"; +import { EFFORTS, PROVIDERS, laneCapabilities, type Effort, type Provider } from "./runner/types.ts"; + +export type ModelChoice = + | { readonly kind: "inherit"; readonly descriptor: "inherit-parent" | "auto" } + | { + readonly kind: "model"; + readonly descriptor: string; + readonly provider: Provider; + readonly model: string; + readonly effort: Effort; + readonly normalizedFrom: string | null; + }; + +export function parseDescriptor(input: string): ModelChoice { + const value = input.trim(); + if (value === "inherit-parent" || value === "auto") return { kind: "inherit", descriptor: value }; + const colon = value.indexOf(":"); + const at = value.lastIndexOf("@"); + const provider = PROVIDERS.find((entry) => entry === value.slice(0, colon)); + const effort = EFFORTS.find((entry) => entry === value.slice(at + 1)); + let model = value.slice(colon + 1, at); + if (colon < 1 || at <= colon + 1 || provider === undefined || effort === undefined || /[\s,]/.test(model)) { + throw new Error(`invalid model descriptor: ${input}`); + } + const alias = provider === "claude" ? versionedClaudeAlias(model) : null; + if (alias !== null) model = alias; + const refusal = modelRefusal(provider, model); + if (refusal !== null) throw new Error(refusal); + return { + kind: "model", provider, model, effort, + descriptor: `${provider}:${model}@${effort}`, + normalizedFrom: alias === null ? null : value, + }; +} + +export function parseModelSheet(text: string): Record { + const roles: Record = {}; + for (const line of text.split(/\r?\n/)) { + const match = /^([a-z][a-z ,_-]*):\s*(.*)$/.exec(line); + if (match === null) continue; + const [, role, entries] = match; + if (Object.hasOwn(roles, role)) throw new Error(`duplicate model role: ${role}`); + if (entries.trim() === "") throw new Error(`empty model role: ${role}`); + roles[role] = entries.split(",").map(parseDescriptor); + } + if (Object.keys(roles).length === 0) throw new Error("model sheet has no role assignments"); + return roles; +} + +const LAB_ALIASES: Readonly> = { + anthropic: "claude", openai: "codex", "x-ai": "grok", +}; + +export function modelLab(choice: ModelChoice): string | null { + if (choice.kind === "inherit") return null; + if (choice.provider === "openrouter" || choice.provider === "opencode") { + const parts = choice.model.split("/"); + if (choice.provider === "opencode" && parts[0] !== "openrouter") { + return LAB_ALIASES[parts[0]] ?? (parts[0] === "deepseek" || parts[0] === "minimax" ? parts[0] : null); + } + const namespace = (choice.provider === "opencode" && parts[0] === "openrouter" ? parts[1] : parts[0]).replace(/^~/, ""); + return LAB_ALIASES[namespace] ?? namespace; + } + return choice.provider; +} + +export function configurationPaths(parent: Parent, cwd: string, env: NodeJS.ProcessEnv = process.env) { + if (!statSync(cwd).isDirectory()) throw new Error(`cwd is not a directory: ${cwd}`); + const harness = harnessAdapter(parent); + const homeDirectory = env.HOME || homedir(); + const override = env[harness.configDirectoryVariable]; + const configDirectory = override?.trim() + ? join(resolve(override), harness.configDirectorySuffix) + : join(homeDirectory, harness.configDirectory); + let commonDirectory: string | null; + try { + commonDirectory = execFileSync("git", ["rev-parse", "--path-format=absolute", "--git-common-dir"], { + cwd, encoding: "utf8", stdio: ["ignore", "pipe", "pipe"], env: { ...process.env, LC_ALL: "C" }, + }).trim(); + } catch (error) { + if (error instanceof Error && "stderr" in error && String(error.stderr).includes("not a git repository")) { + commonDirectory = null; + } else { + throw error; + } + } + return { + globalSheet: join(configDirectory, "pstack-models.md"), + projectSheet: commonDirectory === null ? null : join(dirname(commonDirectory), harness.projectDirectory, "pstack-models.md"), + excludeFile: commonDirectory === null ? null : join(commonDirectory, "info/exclude"), + integrationTargets: harness.integrationFiles.map((file) => join(configDirectory, file)), + }; +} + +export function readConfiguration(parent: Parent, cwd: string, env: NodeJS.ProcessEnv = process.env) { + const harness = harnessAdapter(parent); + const paths = configurationPaths(parent, cwd, env); + const activeSheet = paths.projectSheet !== null && existsSync(paths.projectSheet) + ? paths.projectSheet : existsSync(paths.globalSheet) ? paths.globalSheet : null; + const choices = activeSheet === null ? {} : parseModelSheet(readFileSync(activeSheet, "utf8")); + const roles = Object.fromEntries(Object.entries(choices).map(([role, lanes]) => [ + role, lanes.map((choice) => ({ + ...choice, route: laneRoute(parent, choice.kind === "inherit" ? null : choice.provider), lab: modelLab(choice), + capabilities: choice.kind === "inherit" ? null : laneCapabilities(choice.provider), + })), + ])); + return { harness, paths, activeSheet, roles }; +} diff --git a/plugins/pstack/skills/poteto-mode/scripts/context.ts b/plugins/pstack/skills/poteto-mode/scripts/context.ts new file mode 100644 index 00000000..f6235285 --- /dev/null +++ b/plugins/pstack/skills/poteto-mode/scripts/context.ts @@ -0,0 +1,37 @@ +import { parseArgs } from "node:util"; +import { PARENTS, harnessAdapter } from "./harnesses.ts"; +import { configurationPaths, readConfiguration } from "./configuration.ts"; + +export function main(args: string[]): number { + try { + const { values } = parseArgs({ args, options: { + parent: { type: "string" }, cwd: { type: "string" }, role: { type: "string" }, + "require-shell": { type: "boolean" }, + "paths-only": { type: "boolean" }, + } }); + const parent = PARENTS.find((entry) => entry === values.parent); + if (parent === undefined) throw new Error(`--parent must be one of ${PARENTS.join(", ")}`); + const cwd = values.cwd ?? process.cwd(); + if (values["paths-only"]) { + if (values.role !== undefined || values["require-shell"]) throw new Error("--paths-only cannot check a role"); + process.stdout.write(`${JSON.stringify({ harness: harnessAdapter(parent), paths: configurationPaths(parent, cwd) }, null, 2)}\n`); + return 0; + } + const context = readConfiguration(parent, cwd); + if (values["require-shell"] && values.role === undefined) throw new Error("--require-shell needs --role"); + if (values.role !== undefined) { + const lanes = context.roles[values.role]; + if (lanes === undefined) throw new Error(`model sheet has no role: ${values.role}`); + for (const lane of lanes) { + if (values["require-shell"] && lane.capabilities?.shell === false) { + throw new Error(`${lane.descriptor} does not support shell execution; ${values.role} requires it`); + } + } + } + process.stdout.write(`${JSON.stringify(context, null, 2)}\n`); + return 0; + } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`); + return 64; + } +} diff --git a/plugins/pstack/skills/poteto-mode/scripts/harnesses.ts b/plugins/pstack/skills/poteto-mode/scripts/harnesses.ts new file mode 100644 index 00000000..2146ce56 --- /dev/null +++ b/plugins/pstack/skills/poteto-mode/scripts/harnesses.ts @@ -0,0 +1,88 @@ +import type { Provider } from "./runner/types.ts"; + +interface Harness { + readonly id: string; + readonly nativeProvider: Provider | null; + readonly configDirectory: string; + readonly configDirectoryVariable: string; + readonly configDirectorySuffix: string; + readonly projectDirectory: string; + readonly integration: "include" | "mirror-block" | "instructions-array"; + readonly integrationFiles: readonly string[]; + readonly launch: "task" | "session" | "detached"; + readonly sessionVariable: string | null; + readonly identityVariables: readonly string[]; +} + +export const HARNESSES = [ + { + id: "claude", + nativeProvider: "claude", + configDirectory: ".claude", + configDirectoryVariable: "CLAUDE_CONFIG_DIR", + configDirectorySuffix: "", + projectDirectory: ".claude", + integration: "include", + integrationFiles: ["CLAUDE.md"], + launch: "task", + sessionVariable: "CLAUDE_CODE_SESSION_ID", + identityVariables: [ + "CLAUDECODE", "CLAUDE_CODE_CHILD_SESSION", "CLAUDE_CODE_SESSION_ID", + "CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS", + ], + }, + { + id: "codex", + nativeProvider: "codex", + configDirectory: ".codex", + configDirectoryVariable: "CODEX_HOME", + configDirectorySuffix: "", + projectDirectory: ".codex", + integration: "mirror-block", + integrationFiles: ["AGENTS.md"], + launch: "session", + sessionVariable: "CODEX_THREAD_ID", + identityVariables: [ + "CODEX_THREAD_ID", "CODEX_SESSION_ID", "CODEX_CI", "CODEX_SHELL", + "CODEX_SANDBOX", "CODEX_SANDBOX_NETWORK_DISABLED", + "CODEX_INTERNAL_ORIGINATOR_OVERRIDE", + ], + }, + { + id: "opencode", + nativeProvider: null, + configDirectory: ".config/opencode", + configDirectoryVariable: "XDG_CONFIG_HOME", + configDirectorySuffix: "opencode", + projectDirectory: ".opencode", + integration: "instructions-array", + integrationFiles: ["opencode.jsonc", "opencode.json"], + launch: "detached", + sessionVariable: null, + identityVariables: ["OPENCODE", "OPENCODE_PID", "AGENT"], + }, +] as const satisfies readonly Harness[]; + +export type Parent = (typeof HARNESSES)[number]["id"]; +export const PARENTS = HARNESSES.map((harness) => harness.id); + +export function harnessAdapter(parent: Parent) { + const adapter = HARNESSES.find((harness) => harness.id === parent); + if (adapter === undefined) throw new Error(`unsupported harness: ${parent}`); + return adapter; +} + +export function laneRoute(parent: Parent, provider: Provider | null): "native" | "external" { + return provider === null || harnessAdapter(parent).nativeProvider === provider + ? "native" + : "external"; +} + +export function withoutParentIdentity(provider: Provider, source: NodeJS.ProcessEnv): NodeJS.ProcessEnv { + const result = { ...source }; + for (const harness of HARNESSES) { + if (harness.nativeProvider === provider) continue; + for (const key of harness.identityVariables) delete result[key]; + } + return result; +} diff --git a/plugins/pstack/skills/poteto-mode/scripts/package.json b/plugins/pstack/skills/poteto-mode/scripts/package.json index a63c436a..34bd7f4c 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/package.json +++ b/plugins/pstack/skills/poteto-mode/scripts/package.json @@ -3,8 +3,8 @@ "private": true, "type": "module", "scripts": { - "test": "\"$npm_execpath\" test --parallel bootstrap orch watch-pr runner check-plan", - "typecheck": "tsc --project watch-pr/tsconfig.json --noEmit --strict && tsc --project runner/tsconfig.json --noEmit --strict && tsc --project check-plan.tsconfig.json --noEmit --strict" + "test": "\"$npm_execpath\" test --parallel bootstrap orch watch-pr runner check-plan configuration", + "typecheck": "tsc --project watch-pr/tsconfig.json --noEmit --strict && tsc --project tsconfig.json --noEmit --strict && tsc --project check-plan.tsconfig.json --noEmit --strict" }, "dependencies": { "commander": "14.0.0" diff --git a/plugins/pstack/skills/poteto-mode/scripts/pstack-context b/plugins/pstack/skills/poteto-mode/scripts/pstack-context new file mode 100755 index 00000000..36b1540a --- /dev/null +++ b/plugins/pstack/skills/poteto-mode/scripts/pstack-context @@ -0,0 +1,3 @@ +#!/usr/bin/env bun +const { main } = await import("./context.ts"); +process.exitCode = main(process.argv.slice(2)); diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/cli.test.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/cli.test.ts index 53ec3ee9..dbdffa67 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/cli.test.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/cli.test.ts @@ -69,6 +69,36 @@ describe("runner CLI parsing", () => { expect(parsed?.model).toBe("moonshotai/kimi-k3"); }); + it("passes an OpenCode model with its provider through unchanged", () => { + const parsed = parseArgs( + argv().map((value, index, all) => + all[index - 1] === "--provider" + ? "opencode" + : all[index - 1] === "--model" + ? "openrouter/z-ai/glm-5.3" + : value + ) + ); + expect(parsed?.provider).toBe("opencode"); + expect(parsed?.model).toBe("openrouter/z-ai/glm-5.3"); + }); + + it("accepts OpenCode as a parent", () => { + const parsed = parseArgs( + argv().map((value, index, all) => + all[index - 1] === "--parent" ? "opencode" : value + ) + ); + expect(parsed?.parent).toBe("opencode"); + expect(() => + parseArgs( + argv().map((value, index, all) => + all[index - 1] === "--parent" ? "cursor" : value + ) + ) + ).toThrow("claude, codex, opencode"); + }); + it("takes an optional display label and nothing else from it", () => { expect(parseArgs(argv())?.label).toBeUndefined(); expect(parseArgs(argv(["--label", " arena cross-judge "]))?.label).toBe("arena cross-judge"); diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/cli.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/cli.ts index ac20bdf7..e7235b1e 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/cli.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/cli.ts @@ -13,7 +13,7 @@ import { UsageError, } from "./types.ts"; -const HELP = `Usage: pstack-runner --parent --provider <${PROVIDERS.join("|")}> \\ +const HELP = `Usage: pstack-runner --parent <${PARENTS.join("|")}> --provider <${PROVIDERS.join("|")}> \\ --model --effort --mode \\ --prompt --cwd --output --receipt [--timeout ] \\ [--label ] diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/commands.test.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/commands.test.ts index e6b15e72..4f340e26 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/commands.test.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/commands.test.ts @@ -164,13 +164,47 @@ describe("invocationCommand", () => { it("preflights gateway lanes with a version probe, not an auth check", () => { for (const provider of ["deepseek", "minimax", "openrouter"] as const) { - const spec = preflightCommand(provider); + const spec = preflightCommand(provider, "model"); expect(spec.command).toBe("claude"); expect(spec.args).toEqual(["--version"]); expect(spec.stdin).toBe("none"); } }); + it("runs an opencode lane headless with the effort as its variant", () => { + const model = "openrouter/z-ai/glm-5.3"; + const readOnly = invocationCommand(options({ provider: "opencode", model, effort: "high" })); + const writer = invocationCommand( + options({ provider: "opencode", model, effort: "max", mode: "isolated-write" }) + ); + expect(readOnly).toEqual({ + command: "opencode", + args: [ + "run", + "--model", model, + "--variant", "high", + "--agent", expect.stringMatching(/^pstack-lane-/), + "--format", "json", + "--dir", options().cwd, + "--pure", + ], + stdin: "prompt", + environment: { OPENCODE_CONFIG_CONTENT: expect.any(String) }, + }); + expect(writer.args).toEqual(expect.arrayContaining(["--variant", "max", "--agent", expect.stringMatching(/^pstack-lane-/)])); + for (const invocation of [readOnly, writer]) { + const agent = invocation.args[invocation.args.indexOf("--agent") + 1]; + const config = invocation.environment?.OPENCODE_CONFIG_CONTENT; + if (config === undefined) throw new Error("OpenCode invocation must carry its agent config"); + expect(Object.keys(JSON.parse(config).agent)).toEqual([agent]); + } + expect(preflightCommand("opencode", model)).toEqual({ + command: "opencode", + args: ["models", "openrouter", "--verbose", "--pure"], + stdin: "none", + }); + }); + it("covers low, medium, and high for every external provider", () => { const cases = [ { diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/commands.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/commands.ts index e61951d6..723e6bd5 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/commands.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/commands.ts @@ -5,14 +5,16 @@ import type { RunnerOptions, } from "./types.ts"; import { isGatewayProvider } from "./types.ts"; +import { openCodeLane, openCodeProviderId } from "./opencode-lane.ts"; export interface CommandSpec { readonly command: string; readonly args: readonly string[]; readonly stdin: "prompt" | "none"; + readonly environment?: NodeJS.ProcessEnv; } -export function preflightCommand(provider: Provider): CommandSpec { +export function preflightCommand(provider: Provider, model: string): CommandSpec { if (isGatewayProvider(provider)) { // Gateway lanes run the claude binary with token auth against a // third-party endpoint. `claude auth status` semantics under token @@ -36,6 +38,12 @@ export function preflightCommand(provider: Provider): CommandSpec { }; case "grok": return { command: "grok", args: ["models"], stdin: "none" }; + case "opencode": + return { + command: "opencode", + args: ["models", openCodeProviderId(model), "--verbose", "--pure"], + stdin: "none", + }; } } @@ -162,5 +170,28 @@ export function invocationCommand(options: RunnerOptions): CommandSpec { ], stdin: "none", }; + case "opencode": { + const lane = openCodeLane(options.mode); + // `--dir` matters: OpenCode takes its root from an inherited PWD otherwise. + return { + command: "opencode", + args: [ + "run", + "--model", + options.model, + "--variant", + options.effort, + "--agent", + lane.agent, + "--format", + "json", + "--dir", + options.cwd, + "--pure", + ], + stdin: "prompt", + environment: lane.environment, + }; + } } } diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/flex-journal.test.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/flex-journal.test.ts index 99e80295..7f2bd919 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/flex-journal.test.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/flex-journal.test.ts @@ -46,6 +46,21 @@ describe("openLaneJournal", () => { expect(heads[1]).toHaveLength(300); }); + it("records no parent session for OpenCode, which exposes none", () => { + const root = join(scratch, "lanes"); + mkdirSync(root); + openLaneJournal({ ...options, parent: "opencode" }, Date.now(), { + [LANES_DIR_VAR]: root, + CODEX_THREAD_ID: "unrelated-thread", + OPENCODE_PID: "4242", + }); + const [lane] = readdirSync(root); + expect(JSON.parse(readFileSync(join(root, lane!, "lane.json"), "utf8"))).toMatchObject({ + parent: "opencode", + parentSessionId: null, + }); + }); + it("defaults under the user's pstack-flex directory and honors an override", () => { expect(lanesRoot({})).toEndWith(join(".pstack-flex", "lanes")); expect(lanesRoot({ [LANES_DIR_VAR]: "/elsewhere" })).toBe("/elsewhere"); diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/flex-journal.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/flex-journal.ts index ce440f8b..381f7ff6 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/flex-journal.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/flex-journal.ts @@ -1,23 +1,13 @@ import { randomBytes } from "node:crypto"; -import { closeSync, mkdirSync, openSync, readFileSync, renameSync, writeFileSync, writeSync } from "node:fs"; +import { closeSync, existsSync, mkdirSync, openSync, readFileSync, renameSync, writeFileSync, writeSync } from "node:fs"; import { homedir } from "node:os"; import { join } from "node:path"; +import { harnessAdapter } from "../harnesses.ts"; import type { AccessMode, Effort, Parent, Provider, RunnerOptions, RunnerReceipt } from "./types.ts"; -// pstack-flex addition. An opt-in journal of each external lane, so the agent -// monitor can show a lane while it runs. Journaling is on only when the lanes -// directory exists; psf-monitor's `journal on` creates it. A journal failure -// never changes the lane's receipt, exit code, or output. - export const LANES_DIR_VAR = "PSTACK_FLEX_LANES_DIR"; const PROMPT_HEAD_CHARS = 300; -// The parent harness's own session id, inherited by the runner from the tool that launched it. -const PARENT_SESSION_VAR: Record = { - claude: "CLAUDE_CODE_SESSION_ID", - codex: "CODEX_THREAD_ID", -}; - export interface LaneRecord { readonly schemaVersion: 1; readonly laneId: string; @@ -73,13 +63,15 @@ export function openLaneJournal( started: number, env: NodeJS.ProcessEnv = process.env ): LaneTap { + const root = lanesRoot(env); + if (!existsSync(root)) return OFF; const laneId = `${started.toString(36)}-${process.pid.toString(36)}-${randomBytes(3).toString("hex")}`; - const dir = join(lanesRoot(env), laneId); + const dir = join(root, laneId); let descriptor: number | null = null; try { - // Not recursive: a missing lanes directory means journaling is off. mkdirSync(dir, { mode: 0o700 }); - const session = env[PARENT_SESSION_VAR[options.parent]]; + const sessionVar = harnessAdapter(options.parent).sessionVariable; + const session = sessionVar === null ? undefined : env[sessionVar]; const record: LaneRecord = { schemaVersion: 1, laneId, @@ -111,7 +103,6 @@ export function openLaneJournal( try { closeSync(open); } catch { - // Already closed; nothing to recover. } open = null; }; @@ -122,7 +113,6 @@ export function openLaneJournal( let written = 0; while (written < chunk.length) written += writeSync(open, chunk, written); } catch { - // A full disk or revoked directory stops the journal, never the lane. close(); } }, @@ -131,7 +121,6 @@ export function openLaneJournal( try { writeAtomic(join(dir, "receipt.json"), receipt); } catch { - // The canonical receipt is already written; the journal copy is a convenience. } }, }; diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/flex-providers.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/flex-providers.ts index 7dccf69c..d71b7990 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/flex-providers.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/flex-providers.ts @@ -3,11 +3,6 @@ import { homedir } from "node:os"; import { join } from "node:path"; import type { GatewayProvider } from "./types.ts"; -// pstack-flex addition. Gateway providers run the stock `claude` binary -// against a third-party Anthropic-compatible endpoint. Everything a lane -// needs is injected as environment at spawn time; secrets come from the -// operator's environment and are never written to disk or receipts. - export interface GatewaySpec { readonly apiKeyVar: string; readonly baseUrlDefault: string; @@ -124,8 +119,6 @@ export interface GatewayRefusal { readonly evidence: string; } -// Runs in-process before any subprocess is spawned, so no request can leave -// the machine first. Refusals surface as `unauthenticated` receipts. export function gatewayGuard( provider: GatewayProvider, source: NodeJS.ProcessEnv = process.env diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/model-matrix.test.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/model-matrix.test.ts index 0ed6cad4..4f1c981e 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/model-matrix.test.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/model-matrix.test.ts @@ -343,7 +343,6 @@ describe("model matrix", () => { expect(row.defaultEffort).toBe("high"); expect(row.selectableEfforts).toEqual([...EFFORTS]); expect(row.claudeNativeAgentStem).toBeNull(); - // sol-6 stays selectable; sol-6.1 took its first-run roles. if (row.family !== "sol-6") { expect(sheet).toContain(defaultDescriptor(row)); } @@ -351,7 +350,6 @@ describe("model matrix", () => { expect(new Set(rows.map((row) => row.family)).size).toBe(rows.length); expect(new Set(rows.map((row) => `${row.provider}:${row.model}`)).size) .toBe(rows.length); - // Solo code roles ride the sol-6.1 row; exploration and swarm ride luna. const sol6 = defaultDescriptor(rows.find((row) => row.family === "sol-6.1")!); const luna = defaultDescriptor(rows.find((row) => row.family === "luna")!); for (const role of ["feature, refactoring", "bug-fix", "perf-issue", "hillclimb"]) { @@ -360,30 +358,7 @@ describe("model matrix", () => { for (const role of ["how explorer", "swarm workers"]) { expect(sheet).toContain(`${role}: ${luna}\n`); } - expect(setup).toContain("Its model matrices (stock and flex)"); - expect(setup).toContain("Read the model matrices, stock and flex."); - expect(setup).toContain("any stock or flex matrix family"); - expect(setup).toContain("Offer every stock family, including Astra, GPT-6.1 Sol, GPT-6 Sol, and Luna, when changing `architect runners`"); - expect(setup).toContain("Read each model, proposed effort, and selectable efforts from its row."); - expect(setup).toContain("outside the stock and flex matrix families"); - expect(setup).toContain( - "| Astra | Astra matrix row + selected effort | external runner | native `spawn_agent` |" - ); - expect(setup).toContain( - "| GPT-6.1 Sol | sol-6.1 matrix row + selected effort | external runner | native `spawn_agent` |" - ); - expect(setup).toContain( - "| GPT-6 Sol | sol-6 matrix row + selected effort | external runner | native `spawn_agent` |" - ); - expect(setup).toContain( - "| Luna | Luna matrix row + selected effort | external runner | native `spawn_agent` |" - ); - expect(setup).toContain("each assigned Codex family gets a native `spawn_agent` probe"); - expect(setup).not.toContain("additional matrix"); - expect(dispatch).not.toContain("## Additional model matrix"); - expect(dispatch).toContain( - "These Codex families use native `spawn_agent` under a Codex parent and the external Codex runner under a Claude Code parent." - ); + }); it("owns the default panel: four lanes, three providers, matrix default efforts", () => { @@ -415,31 +390,6 @@ describe("model matrix", () => { ); }); - it("scopes a sheet to the project or globally, and always asks which", () => { - const scopeStart = dispatch.indexOf("## Sheet scope"); - const scopeEnd = dispatch.indexOf("## Flex model matrix"); - expect(scopeStart).toBeGreaterThan(-1); - expect(scopeEnd).toBeGreaterThan(scopeStart); - const scope = dispatch.slice(scopeStart, scopeEnd); - for (const path of [ - "`~/.claude/pstack-models.md`", - "`/.claude/pstack-models.md`", - "`~/.codex/pstack-models.md`", - "`/.codex/pstack-models.md`", - ]) { - expect(scope).toContain(path); - expect(setup).toContain(path); - } - expect(scope).toContain("replaces the global sheet"); - expect(scope).toContain("If the file does not exist, the global sheet applies."); - expect(scope).toContain("Never merge the two role by role"); - expect(setup).toContain("### 1. Establish the parent and scope"); - expect(setup).toContain("Ask every run; never infer the scope"); - expect(setup).toContain("load the global sheet instead as the starting assignments"); - expect(setup).toContain("`.git/info/exclude`"); - expect(setup).toContain("Never add the sheet to a tracked `.gitignore`, stage it, or commit it."); - }); - it("passes each GPT-6 family's selected model and effort to the existing runner", () => { for (const row of rows.filter((row) => (GPT6_FAMILIES as readonly string[]).includes(row.family))) { for (const effort of row.selectableEfforts) { @@ -510,24 +460,9 @@ describe("model matrix", () => { expect(current).toBeGreaterThan(previous); previous = current; } - expect(setup).toContain("Do not invent a precedence rule."); expect(setup).toContain("Do not probe or write while any inconsistency is unresolved."); expect(setup).toContain("A failed probe writes nothing:"); - expect(setup).toContain("Run one probe per family"); - expect(setup).toContain("There is no requirement to assign every matrix family."); - expect(setup).toContain("`architect runners` to keep at least two entries"); - expect(setup).toContain("span at least two distinct providers"); - expect(setup).toContain( - "A failed model demands explicit repair or role reassignment before saving." - ); - expect(setup).toContain("normalized complete role map from step 2"); - expect(setup).toContain("starts with `claude-fable-` or `claude-opus-`"); - expect(setup).toContain("preserving the provider, effort, role, and lane order"); - expect(setup).toContain("Show any rolling-alias migrations"); - expect(setup).toContain("Every documented role remains present."); - expect(setup).toContain("An effort-only rerun cannot change a role's family."); - expect(setup).toContain(""); - expect(setup).toContain(""); + expect(setup).toContain("Ask for confirmation before writing."); }); it("keeps the flex matrix additive, parseable, and aligned with the runner", () => { @@ -566,8 +501,6 @@ describe("model matrix", () => { const pair = `${provider}:${model}`; expect(pairs.has(pair)).toBe(false); pairs.add(pair); - // OpenRouter's row is open: its Model cell is a placeholder for any - // catalog ID, which setup probes one model at a time. if (gateway === "openrouter") { expect(model).toBe(OPEN_OPENROUTER_MODEL); } else { @@ -587,42 +520,7 @@ describe("model matrix", () => { "minimax:MiniMax-M3.1-Flash-Preview", `openrouter:${OPEN_OPENROUTER_MODEL}`, ]) expect(pairs.has(pair)).toBe(true); - expect(dispatch).toContain("The `openrouter` row is open."); - expect(dispatch).toContain("There is no allowlist; setup's live probe on the chosen model is the gate."); - expect(dispatch).toContain("Each distinct OpenRouter model ID is its own family"); - expect(dispatch).toContain("a lane's provider is the lab that made the model, not the route that reaches it."); - expect(setup).toContain("Never group efforts or deduplicate probes by provider alone."); - expect(setup).toContain("Different models sharing a provider count as one provider"); - // The stock quad and first-run sheet must not carry flex descriptors: - // upstream's own checks parse descriptors with a lowercase-only, - // three-provider grammar and must never see a flex lane. expect(firstRunSheet(setup)).not.toMatch(/deepseek:|minimax:|openrouter:/i); }); - it("binds Claude-native dispatch to the matrix mapping", () => { - const dispatch = readFileSync(DISPATCH_PATH, "utf8"); - const nativeStart = dispatch.indexOf("## Native lanes"); - const externalStart = dispatch.indexOf("## External lanes"); - expect(nativeStart).toBeGreaterThan(-1); - expect(externalStart).toBeGreaterThan(nativeStart); - const nativeLanes = dispatch.slice(nativeStart, externalStart); - expect(nativeLanes).toContain( - "match the descriptor's `(provider, model)` to one model-matrix row" - ); - expect(nativeLanes).toContain("`pstack--`"); - }); - - it("normalizes old rolling-family pins before any runtime route", () => { - const dispatch = readFileSync(DISPATCH_PATH, "utf8"); - const normalizationStart = dispatch.indexOf("## Read-time normalization"); - const parentStart = dispatch.indexOf("## The parent owns the route"); - expect(normalizationStart).toBeGreaterThan(-1); - expect(parentStart).toBeGreaterThan(normalizationStart); - const normalization = dispatch.slice(normalizationStart, parentStart); - expect(normalization).toContain("replace that model component in memory"); - expect(normalization).toContain("Never pass the versioned predecessor to Claude."); - expect(normalization).toContain("without writing user files"); - expect(normalization).toContain("`/setup-pstack` will rewrite it"); - expect(normalization).toContain("runner rejects a missed Fable or Opus version pin"); - }); }); diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/model-refusal.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/model-refusal.ts new file mode 100644 index 00000000..f2bf5dfb --- /dev/null +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/model-refusal.ts @@ -0,0 +1,9 @@ +import { openRouterModelRefusal } from "./flex-providers.ts"; +import { openCodeModelRefusal } from "./opencode-lane.ts"; +import type { Provider } from "./types.ts"; + +export function modelRefusal(provider: Provider, model: string): string | null { + if (provider === "openrouter") return openRouterModelRefusal(model); + if (provider === "opencode") return openCodeModelRefusal(model); + return null; +} diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/opencode-lane.test.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/opencode-lane.test.ts new file mode 100644 index 00000000..3051607a --- /dev/null +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/opencode-lane.test.ts @@ -0,0 +1,90 @@ +import { describe, expect, it } from "bun:test"; +import { + openCodeLane, + openCodeModelRefusal, + inspectOpenCodeModels, +} from "./opencode-lane.ts"; + +function listing(models: Record): string { + return Object.entries(models) + .map(([id, variants]) => + `${id}\n${JSON.stringify( + { + id, + capabilities: { reasoning: true }, + ...(variants === null ? {} : { variants: Object.fromEntries(variants.map((v) => [v, {}])) }), + }, + null, + 2 + )}` + ) + .join("\n"); +} + +describe("openCodeModelRefusal", () => { + it("requires OpenCode's / form", () => { + expect(openCodeModelRefusal("openrouter/z-ai/glm-5.3")).toBeNull(); + expect(openCodeModelRefusal("opencode/big-pickle")).toBeNull(); + expect(openCodeModelRefusal("openrouter/openrouter/auto")).toContain("router"); + expect(openCodeModelRefusal("openrouter/openrouter/free")).toContain("router"); + for (const model of ["glm-5.3", "/glm-5.3", "openrouter/"]) { + expect(openCodeModelRefusal(model)).toContain("must be /"); + } + }); +}); + +describe("inspectOpenCodeModels", () => { + const stdout = listing({ + "openrouter/z-ai/glm-5.3": ["low", "high", "max"], + "openrouter/z-ai/glm-5.3-air": ["low", "medium", "high"], + "openrouter/acme/plain": [], + "openrouter/acme/bare": null, + }); + + it("passes a listed model that offers the requested effort", () => { + expect(inspectOpenCodeModels(stdout, "openrouter/z-ai/glm-5.3", "max").status).toBe("passed"); + }); + + it("names the variants a model offers when the effort is not one of them", () => { + expect(inspectOpenCodeModels(stdout, "openrouter/z-ai/glm-5.3", "medium")).toEqual({ status: "unavailable-model", evidence: + "OpenCode model openrouter/z-ai/glm-5.3 offers no medium effort variant; it offers low, high, max" + }); + for (const model of ["openrouter/acme/plain", "openrouter/acme/bare"]) { + expect(inspectOpenCodeModels(stdout, model, "high").evidence).toEndWith("it offers none"); + } + }); + + it("matches the model ID exactly, never a sibling that extends it", () => { + expect(inspectOpenCodeModels(stdout, "openrouter/z-ai/glm-5", "high")).toEqual({ status: "unavailable-model", evidence: + "OpenCode lists no model openrouter/z-ai/glm-5 among openrouter's connected models" + }); + }); +}); + +describe("OpenCode listing failures", () => { + it("distinguishes malformed catalog output from a missing model", () => { + expect(inspectOpenCodeModels("format changed", "openrouter/z-ai/glm-5.3", "high").status).toBe("child-failed"); + expect(inspectOpenCodeModels('openrouter/z-ai/glm-5.3\n{"variants":', "openrouter/z-ai/glm-5.3", "high").status).toBe("child-failed"); + expect(inspectOpenCodeModels("", "openrouter/z-ai/glm-5.3", "high").status).toBe("unavailable-model"); + expect(inspectOpenCodeModels('openrouter/z-ai/glm-5.3\n{"variants":{"high":{}}}', "openrouter/z-ai/glm-5.3", "high").status).toBe("passed"); + }); +}); + +describe("OpenCode lane permissions", () => { + it("pairs each fresh agent with its access mode and grants no shell", () => { + const readOnly = openCodeLane("read-only"); + const writer = openCodeLane("isolated-write"); + expect(readOnly.agent).not.toBe(writer.agent); + for (const lane of [readOnly, writer]) { + const config = JSON.parse(lane.environment.OPENCODE_CONFIG_CONTENT); + expect(Object.keys(config.agent)).toEqual([lane.agent]); + const permission = config.agent[lane.agent].permission; + expect(Object.entries(permission)[0]).toEqual(["*", "deny"]); + for (const tool of ["bash", "task", "skill", "webfetch"]) expect(permission[tool]).toBeUndefined(); + } + expect(JSON.parse(readOnly.environment.OPENCODE_CONFIG_CONTENT).agent[readOnly.agent].permission.edit).toBeUndefined(); + const permission = JSON.parse(writer.environment.OPENCODE_CONFIG_CONTENT).agent[writer.agent].permission; + expect(permission.edit).toBe("allow"); + expect(permission.external_directory).toBeUndefined(); + }); +}); diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/opencode-lane.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/opencode-lane.ts new file mode 100644 index 00000000..67c85814 --- /dev/null +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/opencode-lane.ts @@ -0,0 +1,83 @@ +import { randomUUID } from "node:crypto"; +import type { AccessMode } from "./types.ts"; +import { openRouterModelRefusal } from "./flex-providers.ts"; + +export function openCodeModelRefusal(model: string): string | null { + const slash = model.indexOf("/"); + if (slash <= 0 || slash === model.length - 1) { + return `OpenCode model ${model} must be /, such as openrouter/z-ai/glm-5.3`; + } + return model.slice(0, slash) === "openrouter" ? openRouterModelRefusal(model.slice(slash + 1)) : null; +} + +export function openCodeProviderId(model: string): string { + return model.slice(0, model.indexOf("/")); +} + +const READ_ONLY = { + "*": "deny", + read: "allow", + list: "allow", + glob: "allow", + grep: "allow", + external_directory: "allow", +} as const; + +const ISOLATED_WRITE = { + "*": "deny", + read: "allow", + list: "allow", + glob: "allow", + grep: "allow", + edit: "allow", +} as const; + +export function openCodeLane(mode: AccessMode) { + // Named plan/build agents deep-merge ambient permissions. A fresh name + // gives this lane its own rules, appended after global permissions. + const agent = `pstack-lane-${randomUUID()}`; + const config = JSON.stringify({ + agent: { [agent]: { mode: "primary", permission: mode === "read-only" ? READ_ONLY : ISOLATED_WRITE } }, + }); + // OpenCode has no shell sandbox. Even Git reads can execute diff helpers. + return { agent, environment: { OPENCODE_CONFIG_CONTENT: config } }; +} + +export function openCodeEnvironment(): NodeJS.ProcessEnv { + return { + // A private database per lane. OpenCode 1.18.31 sets WAL mode before its + // busy timeout, so lanes sharing the on-disk database fail with + // "database is locked"; this also keeps lane sessions out of the user's history. + OPENCODE_DB: ":memory:", + // Keeps ~/.claude/CLAUDE.md and ~/.claude/skills, pstack included, out of the lane. + OPENCODE_DISABLE_CLAUDE_CODE: "1", + OPENCODE_DISABLE_AUTOUPDATE: "1", + }; +} + +// OpenCode silently ignores unsupported variants. Inspect the catalog before +// invoking a model so requested effort cannot silently change. +export function inspectOpenCodeModels(stdout: string, model: string, effort: string): { + readonly status: "passed" | "unavailable-model" | "child-failed"; + readonly evidence: string; +} { + const entries = stdout.split(/^([^\s{}"\[\],]+\/[^\s{}"\[\],]+)\r?$/m); + if (entries.length === 1 && stdout.trim() !== "") { + return { status: "child-failed", evidence: "OpenCode printed an unreadable model listing" }; + } + for (let i = 1; i < entries.length; i += 2) { + if (entries[i] !== model) continue; + let variants: string[]; + try { + const info = JSON.parse(entries[i + 1].trim()) as { variants?: unknown }; + variants = info.variants !== null && typeof info.variants === "object" + ? Object.keys(info.variants) : []; + } catch { + return { status: "child-failed", evidence: `OpenCode printed unreadable metadata for model ${model}` }; + } + return variants.includes(effort) + ? { status: "passed", evidence: `model ${model} listed with a ${effort} effort variant` } + : { status: "unavailable-model", evidence: `OpenCode model ${model} offers no ${effort} effort variant; it offers ${variants.length === 0 ? "none" : variants.join(", ")}` }; + } + return { status: "unavailable-model", evidence: `OpenCode lists no model ${model} among ${openCodeProviderId(model)}'s connected models` }; +} diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/opencode-permissions.integration.test.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/opencode-permissions.integration.test.ts new file mode 100644 index 00000000..d34694a5 --- /dev/null +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/opencode-permissions.integration.test.ts @@ -0,0 +1,107 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import { existsSync, mkdirSync, mkdtempSync, readFileSync, realpathSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { dirname, join } from "node:path"; +import { invocationCommand } from "./commands.ts"; +import { childEnvironment } from "./run.ts"; +import type { RunnerOptions } from "./types.ts"; + +const executable = Bun.which("opencode"); +const fixtures: string[] = []; + +afterEach(() => { + for (const root of fixtures.splice(0)) rmSync(root, { recursive: true, force: true }); +}); + +function fixture() { + const root = realpathSync(mkdtempSync(join(tmpdir(), "pstack-opencode-permissions-"))); + fixtures.push(root); + const home = join(root, "home"); + const cwd = join(root, "repo"); + const customConfig = join(root, "custom.json"); + const configDirectory = join(root, "custom-directory"); + const managedDirectory = join(root, "managed"); + mkdirSync(cwd, { recursive: true }); + + const ambient = { + permission: { "*": "allow", bash: "allow", edit: "allow", task: "allow", skill: "allow", webfetch: "allow" }, + agent: { + plan: { permission: { "*": "allow", edit: "allow", webfetch: "allow", bash: { "*": "allow", "touch *": "allow" } } }, + build: { permission: { "*": "allow", external_directory: "allow" } }, + }, + mode: { plan: { permission: "allow" } }, + tools: { edit: true, write: true, bash: true, task: true, skill: true }, + }; + for (const file of [ + join(home, ".config/opencode/opencode.json"), + join(home, ".opencode/opencode.json"), + join(cwd, "opencode.json"), + join(cwd, ".opencode/opencode.json"), + customConfig, + join(configDirectory, "opencode.json"), + join(managedDirectory, "opencode.json"), + ]) { + mkdirSync(dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(ambient)); + } + + const env = { + PATH: process.env.PATH, + HOME: home, + XDG_CONFIG_HOME: join(home, ".config"), + XDG_DATA_HOME: join(root, "data"), + XDG_CACHE_HOME: join(root, "cache"), + XDG_STATE_HOME: join(root, "state"), + OPENCODE_CONFIG: customConfig, + OPENCODE_CONFIG_DIR: configDirectory, + OPENCODE_TEST_MANAGED_CONFIG_DIR: managedDirectory, + OPENCODE_PERMISSION: JSON.stringify({ "*": "allow", bash: "allow", edit: "allow", task: "allow", skill: "allow", webfetch: "allow" }), + OPENCODE_DISABLE_MODELS_FETCH: "1", + }; + return { root, cwd, env }; +} + +describe.skipIf(executable === null)("installed OpenCode permission integration (skipped without CLI)", () => { + for (const mode of ["read-only", "isolated-write"] as const) { + it(`enforces ${mode} permissions despite ambient configuration`, () => { + if (executable === null) throw new Error("OpenCode CLI is required"); + const { root, cwd, env: source } = fixture(); + const options: RunnerOptions = { + parent: "codex", provider: "opencode", model: "openrouter/z-ai/glm-5.3", effort: "high", + mode, cwd, promptPath: join(root, "prompt.md"), outputPath: join(root, "output.txt"), + receiptPath: join(root, "receipt.json"), timeoutMs: null, + }; + const invocation = invocationCommand(options); + const agent = invocation.args[invocation.args.indexOf("--agent") + 1]; + expect(agent).toMatch(/^pstack-lane-/); + const env = { ...childEnvironment("opencode", source, options.model), ...invocation.environment }; + const inspect = Bun.spawnSync([executable, "debug", "agent", agent, "--pure"], { cwd, env }); + expect(inspect.exitCode, inspect.stderr.toString()).toBe(0); + const resolved = JSON.parse(inspect.stdout.toString()) as { tools: Record }; + for (const tool of ["bash", "task", "webfetch", "websearch", "skill"]) { + expect(resolved.tools[tool], tool).toBe(false); + } + for (const tool of ["read", "glob", "grep"]) expect(resolved.tools[tool], tool).toBe(true); + expect(resolved.tools.edit).toBe(mode === "isolated-write"); + expect(resolved.tools.write).toBe(mode === "isolated-write"); + + const outside = join(root, "outside.txt"); + const denied = Bun.spawnSync([ + executable, "debug", "agent", agent, "--pure", "--tool", "write", + "--params", JSON.stringify({ filePath: outside, content: "must not be written" }), + ], { cwd, env }); + expect(denied.exitCode).not.toBe(0); + expect(existsSync(outside)).toBe(false); + + if (mode === "isolated-write") { + const inside = join(cwd, "inside.txt"); + const allowed = Bun.spawnSync([ + executable, "debug", "agent", agent, "--pure", "--tool", "write", + "--params", JSON.stringify({ filePath: inside, content: "writer remains useful" }), + ], { cwd, env }); + expect(allowed.exitCode, allowed.stderr.toString()).toBe(0); + expect(readFileSync(inside, "utf8")).toBe("writer remains useful"); + } + }, 30_000); + } +}); diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/parse-output.test.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/parse-output.test.ts index b8b2b53d..6b8ba542 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/parse-output.test.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/parse-output.test.ts @@ -24,6 +24,52 @@ describe("parseProviderOutput", () => { }); }); + it("extracts the last OpenCode text and sums every step's tokens", () => { + const event = (type: string, part: object) => + JSON.stringify({ type, timestamp: 1, sessionID: "ses_1", part }); + const parsed = parseProviderOutput( + "opencode", + [ + event("step_start", { type: "step-start" }), + event("text", { type: "text", text: "Reading the file." }), + event("tool_use", { type: "tool", tool: "read" }), + event("step_finish", { type: "step-finish", cost: 0.1, tokens: { input: 100, output: 10, reasoning: 4, cache: { read: 20, write: 5 } } }), + event("text", { type: "text", text: "OPENCODE_OK" }), + event("step_finish", { type: "step-finish", cost: 0.1, tokens: { input: 150, output: 6, reasoning: 0, cache: { read: 90, write: 0 } } }), + ].join("\n"), + "", + "openrouter/z-ai/glm-5.3" + ); + expect(parsed).toEqual({ + text: "OPENCODE_OK", + reportedModel: null, + sessionId: "ses_1", + usage: { + inputTokens: 250, + cachedInputTokens: 110, + cacheCreationInputTokens: 5, + outputTokens: 16, + reasoningTokens: 4, + }, + costUsd: null, + }); + }); + + it("fails an OpenCode run on an error event or a missing answer", () => { + const error = JSON.stringify({ + type: "error", + sessionID: "ses_1", + error: { name: "APIError", data: { message: "No endpoints found for z-ai/glm-9" } }, + }); + expect(() => parseProviderOutput("opencode", error, "", "openrouter/z-ai/glm-9")).toThrow( + "No endpoints found for z-ai/glm-9" + ); + expect(() => + parseProviderOutput("opencode", JSON.stringify({ type: "step_start", sessionID: "s", part: {} }), "", "a/b") + ).toThrow("did not contain final text"); + expect(() => parseProviderOutput("opencode", "not-json", "", "a/b")).toThrow("non-JSON"); + }); + it("extracts Codex JSONL without inventing a provider-reported model", () => { const parsed = parseProviderOutput( "codex", diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/parse-output.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/parse-output.ts index 6235baba..725c7a59 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/parse-output.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/parse-output.ts @@ -121,6 +121,68 @@ function parseGrok(stdout: string, requestedModel: string): ParsedOutput { }; } +// pstack-flex: `opencode run --format json` prints one event per line. The last +// `text` event is the answer; each `step_finish` carries that step's tokens. +// No event names the model that answered, and its cost is a catalog estimate. +function parseOpenCode(stdout: string): ParsedOutput { + let text: string | null = null; + let sessionId: string | null = null; + let steps = 0; + const totals = { input: 0, output: 0, reasoning: 0, cacheRead: 0, cacheWrite: 0 }; + + for (const line of stdout.split("\n")) { + if (line.trim().length === 0) continue; + let raw: unknown; + try { + raw = JSON.parse(line); + } catch { + throw new Error("opencode emitted a non-JSON event"); + } + const event = object(raw); + if (event === null) continue; + sessionId = nullableString(event.sessionID) ?? sessionId; + const part = object(event.part); + if (event.type === "text") { + text = nullableString(part?.text) ?? text; + } + if (event.type === "step_finish") { + const tokens = object(part?.tokens); + const cache = object(tokens?.cache); + steps += 1; + totals.input += finiteNumber(tokens?.input) ?? 0; + totals.output += finiteNumber(tokens?.output) ?? 0; + totals.reasoning += finiteNumber(tokens?.reasoning) ?? 0; + totals.cacheRead += finiteNumber(cache?.read) ?? 0; + totals.cacheWrite += finiteNumber(cache?.write) ?? 0; + } + if (event.type === "error") { + const error = object(event.error); + throw new Error( + nullableString(object(error?.data)?.message) ?? + nullableString(error?.name) ?? + "opencode reported an error" + ); + } + } + + if (text === null) throw new Error("opencode result did not contain final text"); + return { + text, + reportedModel: null, + sessionId, + usage: steps === 0 + ? null + : { + inputTokens: totals.input, + cachedInputTokens: totals.cacheRead, + cacheCreationInputTokens: totals.cacheWrite, + outputTokens: totals.output, + reasoningTokens: totals.reasoning, + }, + costUsd: null, + }; +} + function parseCodex(stdout: string): ParsedOutput { let text: string | null = null; let usage: NormalizedUsage | null = null; @@ -184,9 +246,16 @@ export function parseProviderOutput( return parseCodex(stdout); case "grok": return parseGrok(stdout, requestedModel); + case "opencode": + return parseOpenCode(stdout); } } +// These CLIs omit a trustworthy model report; their receipts retain pinned argv. +export function acceptsUnreportedModel(provider: Provider): boolean { + return provider === "codex" || provider === "opencode" || isGatewayProvider(provider); +} + export function reportedModelMatches( provider: Provider, requested: string, diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/run.test.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/run.test.ts index 36cdb33b..16ba82ea 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/run.test.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/run.test.ts @@ -28,7 +28,8 @@ const name = process.argv[1].split("/").at(-1); const isPreflight = (name === "claude" && (args[0] === "auth" || args[0] === "--version")) || (name === "codex" && args[0] === "login") || - (name === "grok" && args[0] === "models"); + (name === "grok" && args[0] === "models") || + (name === "opencode" && args[0] === "models"); const stage = isPreflight ? "preflight" : "model"; const startedPath = isPreflight ? process.env.FAKE_PREFLIGHT_STARTED_PATH @@ -92,6 +93,22 @@ if (name === "grok" && args[0] === "models") { console.log("You are logged in with grok.com.\\nAvailable models:\\n * grok-4.6 (default)"); process.exit(0); } +if (name === "opencode" && args[0] === "models") { + if (process.env.FAKE_OPENCODE_DB_LOCKED === "1") { + console.error("Error: Unexpected error\\n\\ndatabase is locked"); + process.exit(1); + } + const listed = { + "openrouter/z-ai/glm-5.3": ["low", "high", "max"], + "openrouter/z-ai/glm-5.3-air": ["low", "medium", "high", "xhigh", "max"], + }; + for (const [id, variants] of Object.entries(listed)) { + if (!id.startsWith(args[1] + "/")) continue; + console.log(id); + console.log(JSON.stringify({ id, variants: Object.fromEntries(variants.map((v) => [v, {}])) }, null, 2)); + } + process.exit(0); +} const modelIndex = args.findIndex((value) => value === "--model"); const model = modelIndex >= 0 ? args[modelIndex + 1] : "unknown"; const reportedModel = process.env.FAKE_REPORT_MODEL ?? (model === "fable" @@ -99,6 +116,10 @@ const reportedModel = process.env.FAKE_REPORT_MODEL ?? (model === "fable" : model === "opus" ? "claude-opus-9" : model); +if (name === "opencode" && process.env.FAKE_OPENCODE_ERROR) { + console.error(process.env.FAKE_OPENCODE_ERROR); + process.exit(1); +} if (stage === "model" && process.env.FAKE_DUMP_ENV_PATH) { writeFileSync(process.env.FAKE_DUMP_ENV_PATH, JSON.stringify(process.env)); } @@ -137,6 +158,10 @@ if (name === "claude") { console.log(JSON.stringify({type:"thread.started",thread_id:"o1"})); console.log(JSON.stringify({type:"item.completed",item:{type:"agent_message",text:"CODEX_OK"}})); console.log(JSON.stringify({type:"turn.completed",usage:{input_tokens:20,cached_input_tokens:5,output_tokens:3,reasoning_output_tokens:1}})); +} else if (name === "opencode") { + console.log(JSON.stringify({type:"step_start",sessionID:"ses_fake",part:{type:"step-start"}})); + console.log(JSON.stringify({type:"text",sessionID:"ses_fake",part:{type:"text",text:"OPENCODE_OK"}})); + console.log(JSON.stringify({type:"step_finish",sessionID:"ses_fake",part:{type:"step-finish",cost:0.3,tokens:{input:40,output:5,reasoning:2,cache:{read:8,write:0}}}})); } else { console.log(JSON.stringify({type:"assistant",message:{content:[{type:"text",text:"progress"}]}})); console.log(JSON.stringify({type:"result",subtype:"success",is_error:false,result:"GROK_OK",session_id:"g1",usage:{input_tokens:30,output_tokens:4,total_tokens:34},total_cost_usd:0.02,modelUsage:{[model + "-build"]:{}}})); @@ -238,7 +263,7 @@ beforeEach(() => { bin = join(scratch, "bin"); mkdirSync(bin); writeFileSync(join(scratch, "prompt.md"), "Return the marker."); - for (const name of ["claude", "codex", "grok"]) makeExecutable(name); + for (const name of ["claude", "codex", "grok", "opencode"]) makeExecutable(name); previousPath = process.env.PATH; process.env.PATH = `${bin}:${dirname(process.execPath)}:${previousPath ?? ""}`; // A lanes directory that does not exist keeps the pstack-flex journal off. @@ -288,6 +313,7 @@ afterEach(() => { delete process.env.FAKE_DESCENDANT_HOLDS_PIPES_MS; delete process.env.FAKE_DESCENDANT_PID_PATH; delete process.env.FAKE_SELF_SIGNAL; + delete process.env.FAKE_OPENCODE_DB_LOCKED; rmSync(scratch, { recursive: true, force: true }); }); @@ -314,6 +340,113 @@ describe("runLane", () => { }); } + for (const provider of ["claude", "codex"] as const) { + it(`executes and receipts the ${provider} lane for an OpenCode parent`, async () => { + const input = { ...options(provider, `opencode-${provider}`), parent: "opencode" as const }; + const result = await runLane(input); + expect(result.exitCode).toBe(0); + expect(receipt(input.receiptPath)).toMatchObject({ + status: "complete", + parent: "opencode", + provider, + model: input.model, + }); + }); + } + + function openCodeOptions( + parent: RunnerOptions["parent"], + overrides: Partial = {} + ): RunnerOptions { + return { + ...options("grok", `opencode-lane-${parent}`), + parent, + provider: "opencode", + model: "openrouter/z-ai/glm-5.3", + effort: "high", + ...overrides, + }; + } + + for (const parent of ["claude", "codex", "opencode"] as const) { + it(`runs an opencode lane under a ${parent} parent with pinned-argv evidence`, async () => { + const input = openCodeOptions(parent); + const result = await runLane(input); + expect(result.exitCode).toBe(0); + expect(readFileSync(input.outputPath, "utf8")).toBe("OPENCODE_OK"); + expect(receipt(input.receiptPath)).toMatchObject({ + status: "complete", + parent, + provider: "opencode", + model: "openrouter/z-ai/glm-5.3", + reportedModel: null, + modelVerified: false, + modelEvidence: "pinned-argv", + sessionId: "ses_fake", + usage: { inputTokens: 40, cachedInputTokens: 8, outputTokens: 5, reasoningTokens: 2 }, + costUsd: null, + preflight: { + status: "passed", + evidence: "model openrouter/z-ai/glm-5.3 listed with a high effort variant", + }, + }); + }); + } + + it("refuses an effort the opencode model does not offer before running it", async () => { + process.env.FAKE_MODEL_STARTED_PATH = join(scratch, "model-started"); + const input = openCodeOptions("opencode", { effort: "medium" }); + const result = await runLane(input); + expect(result.exitCode).toBe(69); + expect(receipt(input.receiptPath)).toMatchObject({ + status: "unavailable-model", + preflight: { + status: "failed", + evidence: "OpenCode model openrouter/z-ai/glm-5.3 offers no medium effort variant; it offers low, high, max", + }, + }); + expect(existsSync(join(scratch, "model-started"))).toBe(false); + }); + + it("refuses an opencode model OpenCode does not list, even beside a longer sibling", async () => { + const input = openCodeOptions("opencode", { model: "openrouter/z-ai/glm-5" }); + const result = await runLane(input); + expect(result.exitCode).toBe(69); + expect(receipt(input.receiptPath)).toMatchObject({ + status: "unavailable-model", + preflight: { evidence: "OpenCode lists no model openrouter/z-ai/glm-5 among openrouter's connected models" }, + }); + }); + + it("reports an opencode preflight crash as a child failure, not a missing model", async () => { + process.env.FAKE_OPENCODE_DB_LOCKED = "1"; + const input = openCodeOptions("opencode"); + const result = await runLane(input); + expect(result.exitCode).toBe(70); + expect(receipt(input.receiptPath).status).toBe("child-failed"); + expect(receipt(input.receiptPath).preflight.evidence).toContain("database is locked"); + }); + + for (const [error, status, exitCode] of [ + ["ProviderAuthError: credentials rejected", "unauthenticated", 77], + ["ModelNotFound: requested model missing", "unavailable-model", 69], + ] as const) { + it(`classifies an OpenCode ${error.split(":")[0]} without producing output`, async () => { + process.env.FAKE_OPENCODE_ERROR = error; + const input = openCodeOptions("codex"); + const result = await runLane(input); + expect(result.exitCode).toBe(exitCode); + expect(receipt(input.receiptPath).status).toBe(status); + expect(existsSync(input.outputPath)).toBe(false); + }); + } + + it("rejects an opencode model without its OpenCode provider", async () => { + await expect(runLane(openCodeOptions("claude", { model: "glm-5.3" }))).rejects.toThrow( + "must be /" + ); + }); + it("records Codex's exact argv without fabricating a reported model", async () => { const input = options("codex"); const result = await runLane(input); @@ -702,6 +835,7 @@ describe("runLane", () => { const terminated = join(scratch, "preflight-child.terminated"); const isolatedRunner = join(scratch, "isolated-runner"); cpSync(import.meta.dir, isolatedRunner, { recursive: true }); + cpSync(join(import.meta.dir, "../harnesses.ts"), join(scratch, "harnesses.ts")); const runner = Bun.spawn([ process.execPath, join(isolatedRunner, "pstack-runner"), @@ -1316,6 +1450,26 @@ describe("childEnvironment", () => { expect(env.PATH).toBe("/bin"); }); + it("gives an opencode lane its locked config and no parent identity", () => { + const env = childEnvironment("opencode", { + PATH: "/bin", + CLAUDECODE: "1", + CODEX_CI: "1", + OPENCODE: "1", + OPENCODE_PID: "4242", + AGENT: "1", + OPENCODE_CONFIG_CONTENT: JSON.stringify({ permission: "allow" }), + OPENROUTER_API_KEY: "sk-or-test", + }); + expect(env).toEqual({ + PATH: "/bin", + OPENROUTER_API_KEY: "sk-or-test", + OPENCODE_DB: ":memory:", + OPENCODE_DISABLE_CLAUDE_CODE: "1", + OPENCODE_DISABLE_AUTOUPDATE: "1", + }); + }); + it("removes only inherited runtime identity needed to avoid nested detection", () => { const source = { PATH: "/bin", @@ -1323,6 +1477,9 @@ describe("childEnvironment", () => { CODEX_CI: "1", CLAUDECODE: "1", CLAUDE_CODE_CHILD_SESSION: "1", + OPENCODE: "1", + OPENCODE_PID: "4242", + AGENT: "1", KEEP_ME: "yes", }; expect(childEnvironment("claude", source)).toEqual({ diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/run.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/run.ts index 5561a38b..3aa3755d 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/run.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/run.ts @@ -10,15 +10,20 @@ import { } from "node:fs"; import { dirname, resolve } from "node:path"; import { invocationCommand, preflightCommand, type CommandSpec } from "./commands.ts"; +import { laneRoute, withoutParentIdentity } from "../harnesses.ts"; +import { + openCodeEnvironment, + inspectOpenCodeModels, +} from "./opencode-lane.ts"; import { openLaneJournal, type LaneTap } from "./flex-journal.ts"; import { GATEWAY_INHERITED_CONFLICTS, gatewayEnvironment, gatewayGuard, - openRouterModelRefusal, } from "./flex-providers.ts"; import { versionedClaudeAlias } from "./model-aliases.ts"; -import { parseProviderOutput, reportedModelMatches } from "./parse-output.ts"; +import { parseProviderOutput, reportedModelMatches, acceptsUnreportedModel } from "./parse-output.ts"; +import { modelRefusal } from "./model-refusal.ts"; import type { Provider, ReceiptStatus, @@ -119,35 +124,12 @@ function installRunCancellation(): RunCancellation { }; } -const CODEX_IDENTITY = [ - "CODEX_THREAD_ID", - "CODEX_SESSION_ID", - "CODEX_CI", - "CODEX_SHELL", - "CODEX_SANDBOX", - "CODEX_SANDBOX_NETWORK_DISABLED", - "CODEX_INTERNAL_ORIGINATOR_OVERRIDE", -] as const; - -const CLAUDE_IDENTITY = [ - "CLAUDECODE", - "CLAUDE_CODE_CHILD_SESSION", - "CLAUDE_CODE_SESSION_ID", - "CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS", -] as const; - export function childEnvironment( provider: Provider, source: NodeJS.ProcessEnv = process.env, model: string = "" ): NodeJS.ProcessEnv { - const result = { ...source }; - const remove = provider === "claude" - ? CODEX_IDENTITY - : provider === "codex" - ? CLAUDE_IDENTITY - : [...CODEX_IDENTITY, ...CLAUDE_IDENTITY]; - for (const key of remove) delete result[key]; + const result = withoutParentIdentity(provider, source); if (isGatewayProvider(provider)) { for (const key of Object.keys(result)) { if (key.startsWith("ANTHROPIC_")) delete result[key]; @@ -155,6 +137,10 @@ export function childEnvironment( for (const key of GATEWAY_INHERITED_CONFLICTS) delete result[key]; Object.assign(result, gatewayEnvironment(provider, model, source)); } + if (provider === "opencode") { + delete result.OPENCODE_CONFIG_CONTENT; + Object.assign(result, openCodeEnvironment()); + } return result; } @@ -370,7 +356,8 @@ async function waitForGrokPreflightRetry( } } -function preflightPassed(provider: Provider, model: string, result: ProcessResult): boolean { +function preflightPassed(options: RunnerOptions, result: ProcessResult): boolean { + const { provider, model } = options; if (result.exitCode !== 0 || result.timedOut) return false; // `claude --version` succeeded; gateway credentials were already verified // in-process by the gateway guard before any subprocess ran. @@ -394,9 +381,11 @@ function preflightPassed(provider: Provider, model: string, result: ProcessResul case "grok": return /logged in/i.test(combined) && combined.includes(model); } + return false; } -function successfulPreflightEvidence(provider: Provider, model: string): string { +function successfulPreflightEvidence(options: RunnerOptions): string { + const { provider, model } = options; if (isGatewayProvider(provider)) { return "claude binary responded; gateway credentials verified in-process"; } @@ -405,13 +394,28 @@ function successfulPreflightEvidence(provider: Provider, model: string): string : "authenticated"; } +function inspectPreflight(options: RunnerOptions, result: ProcessResult): { + readonly status: "passed" | ReceiptStatus; + readonly evidence: string; +} { + if (options.provider === "opencode" && result.exitCode === 0 && !result.timedOut) { + return inspectOpenCodeModels(result.stdout, options.model, options.effort); + } + const raw = evidence(`${result.stdout}\n${result.stderr}`); + const passed = preflightPassed(options, result); + return { + status: passed ? "passed" : preflightFailureStatus(options.provider, options.model, raw), + evidence: passed ? successfulPreflightEvidence(options) : raw, + }; +} + function unavailableStatus(value: string): ReceiptStatus { // Claude Code 2.1.289 reports a rejected gateway key as "Failed to // authenticate. API Error: 401" with `"api_error_status":401` in its result. - if (/not logged in|unauthenticated|authenticat(e|ion)|sign in|login required|"api_error_status":\s*401\b/i.test(value)) { + if (/not logged in|unauthenticated|authenticat(e|ion)|sign in|login required|"api_error_status":\s*401\b|ProviderAuthError/i.test(value)) { return "unauthenticated"; } - if (/model.{0,40}(not found|unknown|unavailable|unsupported|not supported|invalid)|invalid.{0,20}model/i.test(value)) { + if (/model.{0,40}(not found|unknown|unavailable|unsupported|not supported|invalid)|invalid.{0,20}model|ModelNotFound/i.test(value)) { return "unavailable-model"; } return "child-failed"; @@ -429,6 +433,11 @@ function preflightFailureStatus( // failure here means the binary misbehaved, not that auth failed. return "child-failed"; } + if (provider === "opencode") { + return /Provider not found/.test(value) + ? "unavailable-model" + : "child-failed"; + } return provider === "grok" && !value.includes(model) ? "unavailable-model" : "unauthenticated"; @@ -483,17 +492,7 @@ function modelProof( modelEvidence: "provider-report", }; } - if (provider === "codex" && reported === null) { - return { - reportedModel: null, - modelVerified: false, - modelEvidence: "pinned-argv", - }; - } - if (isGatewayProvider(provider) && reported === null) { - // Third-party Anthropic-compatible endpoints do not reliably echo the - // requested model slug. A reported mismatch is a failure, since some - // gateways silently substitute a default model for unknown slugs. + if (acceptsUnreportedModel(provider) && reported === null) { return { reportedModel: null, modelVerified: false, @@ -526,7 +525,7 @@ function completeReceipt( } export function validateOptions(options: RunnerOptions): void { - if (options.parent === options.provider) { + if (laneRoute(options.parent, options.provider) === "native") { throw new UsageError( `provider ${options.provider} is native to parent ${options.parent}; use the parent subagent primitive` ); @@ -540,10 +539,8 @@ export function validateOptions(options: RunnerOptions): void { `Claude model ${options.model} is a version pin; normalize it to ${staleAlias} before invoking the runner` ); } - const routerRefusal = options.provider === "openrouter" - ? openRouterModelRefusal(options.model) - : null; - if (routerRefusal !== null) throw new UsageError(routerRefusal); + const refusal = modelRefusal(options.provider, options.model); + if (refusal !== null) throw new UsageError(refusal); if ( options.timeoutMs !== null && (!Number.isFinite(options.timeoutMs) || options.timeoutMs <= 0) @@ -582,7 +579,7 @@ async function executeLane( ): Promise { const startedAt = new Date(started).toISOString(); const prompt = readFileSync(options.promptPath, "utf8"); - const env = childEnvironment(options.provider, process.env, options.model); + const env = { ...childEnvironment(options.provider, process.env, options.model), ...invocation.environment }; const executable = Bun.which(invocation.command, { PATH: env.PATH, cwd: options.cwd, @@ -706,11 +703,10 @@ async function executeLane( deadlineAt, cancellation ); - let rawPreflightEvidence = evidence(`${preflightResult.stdout}\n${preflightResult.stderr}`); - let passed = preflightPassed(options.provider, options.model, preflightResult); - let preflightEvidence = passed - ? successfulPreflightEvidence(options.provider, options.model) - : rawPreflightEvidence; + let verdict = inspectPreflight(options, preflightResult); + let rawPreflightEvidence = verdict.evidence; + let passed = verdict.status === "passed"; + let preflightEvidence = verdict.evidence; if ( options.provider === "grok" && @@ -750,12 +746,13 @@ async function executeLane( deadlineAt, cancellation ); - rawPreflightEvidence = evidence(`${preflightResult.stdout}\n${preflightResult.stderr}`); - passed = preflightPassed(options.provider, options.model, preflightResult); + verdict = inspectPreflight(options, preflightResult); + rawPreflightEvidence = verdict.evidence; + passed = verdict.status === "passed"; preflightEvidence = retriedPreflightEvidence( firstPreflightEvidence, passed - ? successfulPreflightEvidence(options.provider, options.model) + ? successfulPreflightEvidence(options) : rawPreflightEvidence, passed ); @@ -776,11 +773,7 @@ async function executeLane( if (preflightState.status !== "passed") { const completed = Date.now(); - const preflightFailure = preflightFailureStatus( - options.provider, - options.model, - rawPreflightEvidence - ); + const preflightFailure = verdict.status === "passed" ? "child-failed" : verdict.status; const status: ReceiptStatus = preflightResult.cancelledBy !== null ? "cancelled" : preflightResult.timedOut @@ -935,7 +928,7 @@ export async function runLane( validateOptions(options); const deadlineAt = options.timeoutMs === null ? null : started + options.timeoutMs; const invocation = invocationCommand(options); - const preflight = preflightCommand(options.provider); + const preflight = preflightCommand(options.provider, options.model); const progress: LaneProgress = { executable: null, preflight: { diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/types.ts b/plugins/pstack/skills/poteto-mode/scripts/runner/types.ts index 3cfe1d9a..5968b3b5 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/types.ts +++ b/plugins/pstack/skills/poteto-mode/scripts/runner/types.ts @@ -1,19 +1,22 @@ -export const PARENTS = ["claude", "codex"] as const; -// pstack-flex: gateway providers run the stock `claude` binary against a -// third-party Anthropic-compatible endpoint with injected environment. Adding -// one here requires a matching row in flex-providers.ts GATEWAY_SPECS. +export { PARENTS, type Parent } from "../harnesses.ts"; +import type { Parent } from "../harnesses.ts"; export const GATEWAY_PROVIDERS = ["deepseek", "minimax", "openrouter"] as const; -export const PROVIDERS = ["claude", "codex", "grok", ...GATEWAY_PROVIDERS] as const; +export const PROVIDERS = ["claude", "codex", "grok", "opencode", ...GATEWAY_PROVIDERS] as const; export const EFFORTS = ["low", "medium", "high", "xhigh", "max"] as const; export const ACCESS_MODES = ["read-only", "isolated-write"] as const; -export type Parent = (typeof PARENTS)[number]; export type Provider = (typeof PROVIDERS)[number]; export type GatewayProvider = (typeof GATEWAY_PROVIDERS)[number]; export function isGatewayProvider(provider: Provider): provider is GatewayProvider { return (GATEWAY_PROVIDERS as readonly string[]).includes(provider); } + +export function laneCapabilities(provider: Provider) { + return { + shell: provider !== "opencode", + }; +} export type Effort = (typeof EFFORTS)[number]; export type AccessMode = (typeof ACCESS_MODES)[number]; diff --git a/plugins/pstack/skills/poteto-mode/scripts/runner/tsconfig.json b/plugins/pstack/skills/poteto-mode/scripts/tsconfig.json similarity index 61% rename from plugins/pstack/skills/poteto-mode/scripts/runner/tsconfig.json rename to plugins/pstack/skills/poteto-mode/scripts/tsconfig.json index 477af90b..9de36301 100644 --- a/plugins/pstack/skills/poteto-mode/scripts/runner/tsconfig.json +++ b/plugins/pstack/skills/poteto-mode/scripts/tsconfig.json @@ -7,7 +7,14 @@ "skipLibCheck": true, "strict": true, "target": "esnext", - "types": ["bun-types"] + "types": [ + "bun-types" + ] }, - "include": ["*.ts"] + "include": [ + "harnesses.ts", + "configuration*.ts", + "context.ts", + "runner/*.ts" + ] } diff --git a/plugins/pstack/skills/setup-pstack/SKILL.md b/plugins/pstack/skills/setup-pstack/SKILL.md index d3c07edc..b6d0749e 100644 --- a/plugins/pstack/skills/setup-pstack/SKILL.md +++ b/plugins/pstack/skills/setup-pstack/SKILL.md @@ -1,35 +1,19 @@ --- name: setup-pstack -description: Configure pstack's provider-qualified models, per-family requested effort, and parent-owned routes per role, in the global sheet or a private project sheet. Verifies native and external Claude, Codex, Grok, DeepSeek, MiniMax, and OpenRouter lanes before writing the override sheet. Use for /setup-pstack, "configure pstack models", "set up pstack models for this project", or changing pstack's model choices. +description: Configure pstack's provider-qualified models, per-family requested effort, and parent-owned routes per role, in the global sheet or a private project sheet. Verifies native and external lanes before writing the override sheet. Use for /setup-pstack, "configure pstack models", "set up pstack models for this project", or changing pstack's model choices. --- # Setup pstack -Configure one portable model sheet for the current parent harness, in the scope the operator picks: global, or private to one project. Read [`provider-dispatch.md`](../poteto-mode/references/provider-dispatch.md) before probing or writing anything. Its model matrices (stock and flex), descriptor grammar, and route table are the contract. Choose one requested effort per assigned matrix family. Do not add a runtime resolver or a weaker-model fallback. The only configuration files are the global sheet and the optional project sheet that the Sheet scope section of provider dispatch defines. - -In global scope, Claude Code writes `~/.claude/pstack-models.md` and loads it from `~/.claude/CLAUDE.md` with: - -```text -@~/.claude/pstack-models.md -``` - -In global scope, Codex writes `~/.codex/pstack-models.md`. Codex has no `@` include, so mirror the sheet's exact bytes inside one bounded block in `~/.codex/AGENTS.md` and retain the sheet as the editable source of truth: - -```text - - - -``` - -In project scope, the sheet is `/.claude/pstack-models.md` on Claude Code and `/.codex/pstack-models.md` on Codex. It needs no include and no mirrored block: the parent reads the project path before dispatch, and an existing project sheet replaces the global sheet for that project. +Configure one portable model sheet for the current parent harness, globally or privately for one project. Read [provider dispatch](../poteto-mode/references/provider-dispatch.md) for model families and [harness integration](../poteto-mode/references/codex-tools.md) for tool and configuration recipes. One family is one `(provider, model)` pair with one requested effort. The model sheet is the only mutable configuration source. No weaker-model fallback. ## Steps ### 1. Establish the parent and scope -Use the harness and tool surface running this skill: Claude Code or Codex. Environment markers may corroborate that top-level answer, but do not launch a child and ask it to detect where it came from. Record the parent because the same descriptor takes a different route in each harness. +Match this session's harness and tools to the exact `id` in the installed [harness table](../poteto-mode/scripts/harnesses.ts), then run `poteto-mode/scripts/pstack-context --parent --cwd --paths-only`. Use the row's identifier rather than an app display name. It returns the adapter, global/project sheet paths, and integration targets without parsing a sheet that setup may need to repair. Resolve once; children never detect or reroute themselves. -Then ask which scope this run configures. Ask every run; never infer the scope from the working directory or from which sheets exist. Before asking, resolve the project root per the Sheet scope section and state both sheet paths and whether each file exists. +Then ask which scope this run configures. Ask every run; never infer the scope from the working directory or from which sheets exist. Before asking, use the project path returned by the context inspector and state both sheet paths and whether each file exists. - **Project**: this repository only. The sheet replaces the global sheet for work in this project and stays private to this machine. - **Global**: every project that has no project sheet. @@ -42,17 +26,17 @@ Read the scoped sheet when it exists. In project scope with no project sheet, lo Treat the normalized values as current role-to-family assignments. Overlay those rows on the complete first-run role map in step 7. Materialize any missing documented role row from that map on the next successful write. A duplicate role row is inconsistent state; report it and resolve it before probing. A row whose role is not in the step 7 role map, such as `how critics`, is from a retired role. Drop it and list it at confirmation. A bare host-native slug from an older sheet is also invalid because it does not say which provider owns it. A versioned Claude model outside the two migration families remains inconsistent state. If no sheet was loaded, use the complete first-run role map and the model matrix's Default effort cells. -Then ask whether to keep these role-to-family assignments or change named roles. Keeping them is the default. Apply only role changes the operator names; never offer a reset of a customized sheet to the first-run assignments. A changed role may use any stock or flex matrix family, any OpenRouter model ID the operator names (`openrouter:/`), `inherit-parent`, or `auto`. +Then ask whether to keep these role-to-family assignments or change named roles. Keeping them is the default. Apply only role changes the operator names; never offer a reset of a customized sheet to the first-run assignments. A changed role may use any stock or flex matrix family, any OpenRouter model ID the operator names (`openrouter:/`), any OpenCode model ID the operator names (`opencode:/`), `inherit-parent`, or `auto`. List every stock and flex family by name when the operator changes a role, so Claude, Codex, Grok, DeepSeek, MiniMax, and OpenRouter are all on offer and any role can move to any of them. Offer every stock family, including Astra, GPT-6.1 Sol, GPT-6 Sol, and Luna, when changing `architect runners` or another configurable role. Read each model, proposed effort, and selectable efforts from its row. The Codex families are separate families even though they share the Codex provider; changing one family's effort does not change another's. GPT-6.1 Sol uses the `sol-6.1` family and GPT-6 Sol the `sol-6` family; the `sol` family keeps GPT-5.6 Sol for sheets that still assign it. ### 3. Parse per-family efforts -Read the model matrices, stock and flex. Every non-alias value must match `:@`. Map it to exactly one matrix family by `(provider, model)`; an `openrouter` descriptor maps to the open `openrouter` row whatever its model ID. Require its effort to appear in that row's Selectable efforts cell, and collect the effort. `inherit-parent` and `auto` rows carry no family effort. +Read the model matrices, stock and flex. Every non-alias value must match `:@`. Map it to exactly one matrix family by `(provider, model)`; an `openrouter` descriptor maps to the open `openrouter` row whatever its model ID, and an `opencode` descriptor maps to the open `opencode` row the same way. Require its effort to appear in that row's Selectable efforts cell, and collect the effort. `inherit-parent` and `auto` rows carry no family effort. An unmatched provider/model, out-of-domain effort, or duplicate role is inconsistent state. Stop, show the conflicting rows verbatim, and ask for an explicit matrix family or alias replacement. If one or more families have mixed efforts, show every conflicting family and role row, then ask for one normalized effort per family from its Selectable efforts cell. Do not invent a precedence rule. Do not probe or write while any inconsistency is unresolved. -A family is a single `(provider, model)` matrix row. DeepSeek Flash and Pro have independent efforts, as do MiniMax M3 and M3.1 Flash Preview. Each distinct OpenRouter model ID is its own family under the open row, with its own effort and probe. Never group efforts or deduplicate probes by provider alone. +A family is a single `(provider, model)` matrix row. DeepSeek Flash and Pro have independent efforts, as do MiniMax M3 and M3.1 Flash Preview. Each distinct OpenRouter model ID is its own family under the open row, with its own effort and probe, and so is each distinct OpenCode model ID. Never group efforts or deduplicate probes by provider alone. One distinct effort per family is the current value. A family with no non-alias occurrence is unassigned: do not ask for its effort, check its CLI, or probe it. A family that a step 2 role change newly assigns takes its matrix Default effort as the proposed value. @@ -64,25 +48,13 @@ Ask one effort question for each assigned family. Name each model, its current o Probe only the selected `provider:model@effort` pair of each assigned family. Run one probe per family in the role map, even when two families share a provider. Do not enumerate or offer older models as substitutes. A failed probe writes nothing: report the failing pair and provider, stop, and keep the scoped sheet plus parent integration bytes unchanged. A failed model demands explicit repair or role reassignment before saving. A failed first run creates neither artifact. -| Family | Pair source | Claude parent route | Codex parent route | Availability proof | -|---|---|---|---|---| -| Fable | Fable matrix row + selected effort | native Agent `pstack-fable-` | Claude CLI | native one-turn probe or `claude auth status --json` plus one-turn probe | -| Sol | Sol matrix row + selected effort | `codex exec` | native `spawn_agent` | `codex login status` plus one-turn probe or native one-turn probe | -| Astra | Astra matrix row + selected effort | external runner | native `spawn_agent` | `codex login status` plus one-turn probe or native one-turn probe | -| GPT-6.1 Sol | sol-6.1 matrix row + selected effort | external runner | native `spawn_agent` | `codex login status` plus one-turn probe or native one-turn probe | -| GPT-6 Sol | sol-6 matrix row + selected effort | external runner | native `spawn_agent` | `codex login status` plus one-turn probe or native one-turn probe | -| Luna | Luna matrix row + selected effort | external runner | native `spawn_agent` | `codex login status` plus one-turn probe or native one-turn probe | -| Grok | Grok matrix row + selected effort | Grok CLI | Grok CLI | `grok models` must list the requested model; one-turn probe | -| Opus | Opus matrix row + selected effort | native Agent `pstack-opus-` | Claude CLI | native one-turn probe or `claude auth status --json` plus one-turn probe | -| DeepSeek Flash / Pro | Each assigned DeepSeek flex row + selected effort | external runner | external runner | `DEEPSEEK_API_KEY` present; isolated config dir free of OAuth credentials; one-turn probe confirms the endpoint | -| MiniMax M3 / M3.1 Flash Preview | Each assigned MiniMax flex row + selected effort | external runner | external runner | `MINIMAX_API_KEY` present; isolated config dir free of OAuth credentials; one-turn probe confirms the endpoint | -| OpenRouter (any model ID) | Each assigned OpenRouter model ID + selected effort | external runner | external runner | `OPENROUTER_API_KEY` present; isolated config dir free of OAuth credentials; one-turn probe that reads its marker from a file | +Use each descriptor's route from the selected adapter. A native pair uses that adapter's native one-turn invocation; an external pair uses `pstack-runner` with the explicit parent, provider, model, and selected effort. The runner owns CLI/authentication/model preflight. Keep every exact-pair receipt or native transcript. Do not repeat routing by family or invent another harness detector. For MiniMax M3.1 Flash Preview, disclose the Token Plan requirement before probing. Use the eligible subscription key through `MINIMAX_API_KEY`; do not assume a working M3 key grants preview access. A failed preview probe must not silently select M3. Keep preview thinking enabled and verify requested effort forwarding; distinguish request evidence from hidden applied reasoning depth. Before the first OpenRouter probe, tell the operator three things: OpenRouter forwards prompts to whichever host serves the model, data-collection and zero-data-retention routing plus the key's credit limit are set on OpenRouter's dashboard, and each probe spends a little credit. Probe exactly the model ID the operator named. A failed OpenRouter probe does not offer another catalog model; report OpenRouter's error and ask for a replacement ID or a role reassignment. -Use a tiny read-only probe that returns a unique marker. For an OpenRouter model, write the marker to a file in a scratch directory and leave it out of the prompt, so a passing probe proves the model made a tool call; Claude Code lanes cannot work without one. A login-status command alone proves credentials, not that the requested model and effort flags run. Record native and external results separately. Never call the external launcher for the parent's own provider. On a Claude parent, the Fable and Opus probes are one-turn runs of the mapped `pstack--` agent. On a Codex parent, each assigned Codex family gets a native `spawn_agent` probe with its matrix model and selected `reasoning_effort`. Every other pair, flex families always included, uses the external runner with the selected effort flag. A flex probe doubles as the base-URL confirmation: it proves the documented default (or the operator's override) actually serves the lane's model. +Use a tiny read-only probe returning a unique marker from a scratch file. Leave the marker out of the prompt so success proves a tool call. A login-status check alone does not prove the assigned model and effort work. Inspect offered variants when a runner refuses an effort. A gateway probe also confirms the endpoint default or override. Receipts and native transcripts prove the requested effort and the route. They do not prove a provider's hidden applied reasoning depth. There is no implicit timeout, weaker-model fallback, same-provider external fallback, or second mutable configuration source. @@ -96,11 +68,11 @@ Build the new sheet in memory. Do not write it yet. Require every documented role to remain present and non-empty, `architect runners` to keep at least two entries, and the final role map to contain at least one assigned matrix family. There is no requirement to assign every matrix family. -Different models sharing a provider count as one provider, even when their efforts differ. An OpenRouter lane counts as its model ID's lab, as `provider-dispatch.md` defines: the namespaces `anthropic`, `openai`, `x-ai`, `deepseek`, and `minimax` match the direct providers, and any other namespace is a provider of its own. +Different models sharing a provider count as one provider, even when their efforts differ. An OpenRouter lane counts as its model ID's lab, as `provider-dispatch.md` defines: the namespaces `anthropic`, `openai`, `x-ai`, `deepseek`, and `minimax` match the direct providers, and any other namespace is a provider of its own. An OpenCode lane counts as its model's lab the same way. Validate panel diversity: `arena runners` and `interrogate reviewers` must span at least two distinct providers. A single-provider panel is written only after the operator explicitly confirms the reduced diversity; record that confirmation in the setup report. -Rewrite every matrix-family descriptor to `provider:model@`. Leave `inherit-parent` and `auto` unchanged. An effort-only rerun cannot change a role's family. Changing Grok's effort updates every Grok occurrence and does not move a Sol role onto Grok. Refuse an unqualified slug, an unavailable route, a model outside the stock and flex matrix families, an OpenRouter ID without a namespace or from the `openrouter/*` routers, or a provider/model mismatch. +Rewrite every matrix-family descriptor to `provider:model@`. Leave `inherit-parent` and `auto` unchanged. An effort-only rerun cannot change a role's family. Changing Grok's effort updates every Grok occurrence and does not move a Sol role onto Grok. Refuse an unqualified slug, an unavailable route, a model outside the stock and flex matrix families, an OpenRouter ID without a namespace or from the `openrouter/*` routers, an OpenCode ID without its `/` prefix, or a provider/model mismatch. ### 7. Confirm and commit @@ -110,7 +82,7 @@ Why and Reflect require the parent's live MCP surface. Keep their investigator, Every non-alias value must match `:@` and must have passed step 5. -After the operator confirms, write the in-memory render from step 6. Never paste the example below as the result. It is only the complete first-run role map used to seed step 2; selected efforts and explicit role changes always replace its example values before writing. +After the operator confirms, pass the in-memory render from step 6 to step 8's write transaction. Never paste the example below as the result. It is only the complete first-run role map used to seed step 2; selected efforts and explicit role changes always replace its example values before writing. ```markdown # pstack model configuration @@ -136,9 +108,7 @@ interrogate reviewers: claude:fable@max, codex:gpt-6-astra@high, grok:grok-4.7@x ### 8. Wire it in -Project scope has no parent integration. Write only the project sheet, creating its directory when needed. Then make sure `.git/info/exclude` in the repository's common git directory lists the sheet's path relative to the project root, adding that one line when it is absent. Never add the sheet to a tracked `.gitignore`, stage it, or commit it. Snapshot, readback, and restore apply to the sheet and the exclude file exactly as below. The rest of this step applies to global scope. - -Render the parent integration in memory before either write. On Claude, the integration is the single `@~/.claude/pstack-models.md` include in `~/.claude/CLAUDE.md`. On Codex, it is the exact sheet bytes between one `` and `` pair in `~/.codex/AGENTS.md`. Replace that whole bounded block on a rerun. Insert one block at the end on first run. If either marker is missing, duplicated, or reversed, stop and report inconsistent state instead of guessing a boundary. +Use the paths and integration type returned by the context inspector and apply the corresponding recipe in harness integration. Project scope writes only the project sheet and the common git directory's `info/exclude`. Never add the sheet to a tracked `.gitignore`, stage it, or commit it. Global scope applies the selected include, mirror-block, or instructions-array recipe. Render both targets in memory before writing. Snapshot every target's current bytes. Write the sheet and parent integration only after every requested pair passes and the operator confirms. Read both targets back and compare them with the in-memory render. If either write or readback fails, restore every snapshot and report the failure. An unchanged rerun must produce byte-identical sheet and integration content after normalization. @@ -146,6 +116,6 @@ Do not copy the model sheet between harnesses or between projects without rerunn ### 9. Behavioral smoke -Before declaring setup complete, run one small read-only mixed panel from this parent: every distinct chosen descriptor, distinct output/receipt paths, and an independent cross-judge when at least two providers are assigned. Launch Claude-native agents and every external process in the background with retained handles, then drain them. Verify the native transcript entries and every external receipt. A structural config check or unit test is not a substitute. +Before declaring setup complete, run one small read-only mixed panel from this parent: every distinct chosen descriptor, distinct output/receipt paths, and an independent cross-judge when at least two providers are assigned. Launch every lane with the selected adapter's retained handles, then drain them. Verify the native transcript entries and every external receipt. A structural config check or unit test is not a substitute. Report the scope, the sheet path, parent route table, requested-effort probe results, smoke results, and external elapsed/token/cost receipts. For a project sheet, add that deleting the file returns the project to the global sheet. Re-running this skill in the same scope re-probes and updates the same sheet. Do not claim the provider exposed hidden applied-effort observability. diff --git a/scripts/check.sh b/scripts/check.sh index fa873ce7..dad20468 100755 --- a/scripts/check.sh +++ b/scripts/check.sh @@ -1,8 +1,4 @@ #!/usr/bin/env bash -# Runs every local check a pull request needs before review: install, -# tests, strict typecheck, manifest parse, static -# invariants, and Claude plugin validation when the claude CLI is present. -# This is the local half of the gate. The live half is docs/LIVE-GATE.md. set -uo pipefail repo="$(cd "$(dirname "$0")/.." && pwd)" diff --git a/scripts/probe-openrouter.sh b/scripts/probe-openrouter.sh index 8d70d2a2..ee05ec47 100644 --- a/scripts/probe-openrouter.sh +++ b/scripts/probe-openrouter.sh @@ -1,21 +1,12 @@ #!/usr/bin/env bash -# V6 in docs/LANES.md: the OpenRouter route battery from issue #7. +# OpenRouter route battery from issue #7; evidence goes in the issue or PR. # It spends real OpenRouter credit (cents), so it never runs in CI or # check.sh. The key comes from OPENROUTER_API_KEY and is never printed or # put on a command line. # # OPENROUTER_API_KEY=... bash scripts/probe-openrouter.sh [model ...] # -# Per model, through pstack-runner (read-only, synthetic workspace): -# chain read start.txt, follow it to a second file, return the value -# there. Neither the file name nor the value is in the prompt, so -# a pass proves tool calls across turns. -# effort the same no-tool puzzle at low and at high; compare the -# reasoning tokens Claude Code reports. -# identity one direct OpenRouter call: the model and host it served. -# Then the failure strings: a wrong ID, a bad key, a model without tools, a -# `~` rolling alias, and a `:free` variant. The bad-key lane takes about three -# minutes: Claude Code retries a 401 before it gives up. +# Claude Code retries a bad gateway key for about three minutes before failing. set -uo pipefail repo="$(cd "$(dirname "$0")/.." && pwd)" @@ -52,7 +43,6 @@ workspace() { printf '%s' "$ws" } -# lane lane() { local name="$1" model="$2" effort="$3" prompt="$4" cwd="$5" expected="$6" local dir="$out/$name" @@ -75,7 +65,6 @@ lane() { .elapsedMs, ((.error.message // "-") | gsub("[\n|]"; " ") | .[0:160]) ] | @tsv' "$dir/receipt.json" >> "$rows" else - # The runner refused before reserving a receipt (a usage error). printf '%s\t%s\t%s\tusage-error\t-\t-\t-\t-\t-\t-\t%s\n' "$name" "$model" "$effort" \ "$(tr '\n|' ' ' < "$dir/stderr.txt" | cut -c1-160)" >> "$rows" fi @@ -92,7 +81,6 @@ for model in "${models[@]}"; do lane "high-$slug" "$model" high "$puzzle_prompt" "$ws" "62" done -# Failure strings and edge forms. The catalog is public; no key needed. catalog="$out/catalog.json" curl -fsS "$api/models" -o "$catalog" no_tools="$(jq -r '[.data[] | select((.supported_parameters // []) | index("tools") | not) diff --git a/scripts/upstream-audit.py b/scripts/upstream-audit.py index 04c2f331..0321500e 100644 --- a/scripts/upstream-audit.py +++ b/scripts/upstream-audit.py @@ -25,8 +25,6 @@ def tree(ref, prefix): def port_path(path): relative = path.removeprefix("pstack/") - if relative == "README.md": - return "README-UPSTREAM.md" if relative.startswith(("skills/", "agents/", "assets/")): return "plugins/pstack/" + relative return None @@ -40,14 +38,14 @@ def port_path(path): port = git("rev-parse", "--verify", args.port + "^{commit}").decode().strip() target = git("rev-parse", "--verify", args.upstream + "^{commit}").decode().strip() sync_doc = git("show", port + ":UPSTREAM.md").decode() -match = re.search(r"^\| Commit \| `([0-9a-f]{40})` \|$", sync_doc, re.MULTILINE) -if not match: - parser.error("UPSTREAM.md must contain exactly the recorded full commit row") -base = match.group(1) +pins = re.findall(r"^\| (?:Cursor content imported by open-pstack \| [^|]+ \||Commit \|) `([0-9a-f]{40})` \|$", sync_doc, re.MULTILINE) +if len(pins) != 1: + parser.error("UPSTREAM.md must contain one full Cursor content pin") +base = pins[0] subprocess.run(["git", "merge-base", "--is-ancestor", base, target], cwd=ROOT, check=True) before = tree(base, "pstack/") after = tree(target, "pstack/") -local = tree(port, "plugins/pstack/") | tree(port, "README-UPSTREAM.md") +local = tree(port, "plugins/pstack/") changes = [] for path in sorted(before.keys() | after.keys()): diff --git a/tests/skill-collision-repro.sh b/tests/skill-collision-repro.sh index b7599069..09d679fc 100755 --- a/tests/skill-collision-repro.sh +++ b/tests/skill-collision-repro.sh @@ -37,11 +37,10 @@ verof() { { grep -m1 '"version"' "$1" || true; } | sed -E 's/.*"version"[[:space vc="$(verof "$repo/plugins/pstack/.claude-plugin/plugin.json")" vx="$(verof "$repo/plugins/pstack/.codex-plugin/plugin.json")" vm="$(verof "$repo/.claude-plugin/marketplace.json")" -vu="$(sed -n 's/| open-pstack version | `\([^`]*\)` |/\1/p' "$repo/UPSTREAM.md")" -if [ -n "$vc" ] && [ "$vc" = "$vx" ] && [ "$vc" = "$vm" ] && [ "$vc" = "$vu" ]; then - note "ok: open-pstack version matches across UPSTREAM.md and the 3 manifests ($vc)" +if [ -n "$vc" ] && [ "$vc" = "$vx" ] && [ "$vc" = "$vm" ]; then + note "ok: distribution version matches across the 3 manifests ($vc)" else - note "FAIL: open-pstack version differs: upstream=$vu claude-plugin=$vc codex-plugin=$vx marketplace=$vm" + note "FAIL: distribution version differs: claude-plugin=$vc codex-plugin=$vx marketplace=$vm" fail=1 fi @@ -51,6 +50,7 @@ fi legacy_model_pins="$( grep -REn \ --include='*.md' --include='*.ts' --include='*.sh' \ + --exclude='*.test.ts' \ 'claude:claude-(fable|opus)-[0-9]|^model: claude-(fable|opus)-[0-9]|--model claude-(fable|opus)-[0-9]' \ "$repo/plugins/pstack" "$repo/tests" "$repo/README.md" "$repo/docs/reference.md" \ 2>/dev/null || true