From c0a25615b1cb2d50870b1310065a9cf750894796 Mon Sep 17 00:00:00 2001 From: Kartik Kabadi <1kartikkabadi1@gmail.com> Date: Wed, 16 Sep 2026 03:04:00 -0700 Subject: [PATCH 1/4] Harden launch routing and security boundaries Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- CHANGELOG.md | 15 +++ README.md | 172 ++++++++++++++-------------- SECURITY.md | 49 ++++++-- pyproject.toml | 2 +- src/opencode_go_proxy/app.py | 75 +++++++++--- src/opencode_go_proxy/compaction.py | 3 - src/opencode_go_proxy/guards.py | 73 ++++++++++-- src/opencode_go_proxy/protocol.py | 6 +- src/opencode_go_proxy/routing.py | 27 +++++ tests/test_auth_guard.py | 62 ++++++++++ tests/test_compaction.py | 8 +- tests/test_integration.py | 61 +++++++--- tests/test_protocol_surface.py | 19 +++ tests/test_routing.py | 27 ++++- 14 files changed, 455 insertions(+), 144 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 674e0ca..6bfac04 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,20 @@ # Changelog +## Unreleased + +### Security + +- Validate the actual client address and require a separate caller token for + explicitly enabled non-loopback access. +- Refuse insecure non-loopback startup configurations. + +### Changed + +- Reject unknown or malformed model IDs before routing instead of silently + substituting `deepseek-v4-flash`. +- Avoid reverse-DNS lookup delays while starting the local listener. +- Clarify the project's independent, single-user credential and policy boundary. + ## [0.4.10] - 2026-08-14 ## [0.4.8] - 2026-08-14 diff --git a/README.md b/README.md index a9a6022..8a82650 100644 --- a/README.md +++ b/README.md @@ -5,7 +5,9 @@ [![Python 3.11+](https://img.shields.io/badge/python-3.11+-blue.svg)](https://www.python.org/downloads/) [![Dependencies: zstandard](https://img.shields.io/badge/dependencies-zstandard-blue.svg)](#) -Use your [OpenCode Go](https://opencode.ai/docs/go) and [OpenCode Zen](https://opencode.ai/zen) access in the [Codex app](https://github.com/openai/codex). +Use your own [OpenCode Go](https://opencode.ai/docs/go) and +[OpenCode Zen](https://opencode.ai/zen) credentials in +[Codex](https://github.com/openai/codex) through a local protocol adapter. Codex expects a Responses API (`/v1/responses`). OpenCode Go exposes an OpenAI-compatible Chat Completions API (`/v1/chat/completions`), and OpenCode Zen serves GPT, Claude, Gemini, @@ -24,6 +26,28 @@ opencode-go-proxy ←── localhost:8787, one runtime dep (zstandard) OpenCode Go / OpenCode Zen ── 13 open Go models · GPT, Claude, Gemini, Grok, DeepSeek, GLM, Kimi, Qwen ``` +## Project and policy boundary + +OpenCode Go Proxy is an independent, single-user local adapter. It is not an +OpenAI or OpenCode product, and neither company sponsors or endorses it. It +does not provide shared accounts, pooled credentials, subscription resale, or +mechanisms to evade provider limits or safeguards. + +The proxy uses Codex's documented `openai_base_url`, custom provider, and model +catalog configuration surfaces. When native OpenAI models are enabled, it +relays the user's existing Codex authorization only to the configured native +OpenAI endpoint over HTTPS; it never stores that authorization or uses it as +an OpenCode credential. OpenCode credentials are likewise never sent to the +native OpenAI endpoint. + +Users are responsible for using their own accounts and following the current +[OpenAI Terms of Use](https://openai.com/terms), +[OpenAI Services Agreement](https://openai.com/policies/services-agreement/), +OpenCode terms, and any model-provider policies that apply to them. See +[Codex advanced configuration](https://developers.openai.com/codex/config-advanced) +and [SECURITY.md](SECURITY.md). This project documents technical safeguards; +it does not provide legal or policy certification. + ## Why OpenCode Go is $5 for the first month, then $10/month. You get access to 13 open coding models @@ -35,41 +59,25 @@ This proxy fixes that for both. ## Quick start ```bash -# Install and run +# Configure Codex (writes one marker-delimited block to ~/.codex/config.toml) +uvx --from git+https://github.com/kartikkabadi/opencode-go-proxy \ + opencode-go-proxy config enable + +# Run the loopback-only adapter uvx --from git+https://github.com/kartikkabadi/opencode-go-proxy \ opencode-go-proxy \ --bind 127.0.0.1 \ --port 8787 -# Point Codex at it (~/.codex/config.toml) -``` - -```toml -[model_providers.opencode-go] -name = "OpenCode Go" -base_url = "http://127.0.0.1:8787/v1" -experimental_bearer_token = "any-string-here" -wire_api = "responses" - -[profiles.deepseek-v4-flash] -model_provider = "opencode-go" -model = "deepseek-v4-flash" -model_context_window = 1000000 -approval_policy = "untrusted" -sandbox_mode = "workspace-write" -features = { memories = false } -``` - -```bash -# Start Codex with a profile -codex -p deepseek-v4-flash +# Fully restart Codex, then select a model in the picker or CLI +codex -m deepseek-v4-flash ``` ## Available models All 13 OpenCode Go models work through this proxy. The defaults are DeepSeek V4 Flash (cheapest general-purpose) and MiMo V2.5 (cheapest vision, used for image captioning). -Switch to whatever you want — just change the model in your Codex profile. +Switch to whatever you want in the Codex model picker or with `codex -m`. | Model | Slug | Best for | Requests/mo on Go | |-------|------|----------|-------------------| @@ -92,50 +100,23 @@ typical usage patterns. Cheaper models = more requests per month. ### Switching models -Just create another profile and use `codex -p `: - -```toml -[profiles.deepseek-v4-pro] -model_provider = "opencode-go" -model = "deepseek-v4-pro" -model_context_window = 1000000 -approval_policy = "untrusted" -sandbox_mode = "workspace-write" -features = { memories = false } - -[profiles.glm-5.2] -model_provider = "opencode-go" -model = "glm-5.2" -model_context_window = 272000 -approval_policy = "untrusted" -sandbox_mode = "workspace-write" -features = { memories = false } - -[profiles.kimi-k2.7-code] -model_provider = "opencode-go" -model = "kimi-k2.7-code" -model_context_window = 272000 -approval_policy = "untrusted" -sandbox_mode = "workspace-write" -features = { memories = false } -``` - ```bash -codex -p deepseek-v4-pro -codex -p glm-5.2 -codex -p kimi-k2.7-code +codex -m deepseek-v4-pro +codex -m glm-5.2 +codex -m kimi-k2.7-code ``` -### How the default model is chosen +### How models are routed -The proxy picks the upstream model based on what Codex sends: +The proxy routes the exact model Codex sends: -1. If the model slug is `zen/`-prefixed, it routes to the OpenCode Zen gateway - instead (see [OpenCode Zen](#opencode-zen)); the prefix wins over everything below. -2. If the model slug is in the [alias map](src/opencode_go_proxy/protocol.py), it's mapped - (e.g. `gpt-5.5` → `deepseek-v4-pro`). -3. If the model slug is a known OpenCode Go model (from the catalog), it's used as-is. -4. Otherwise, it falls back to `deepseek-v4-flash`. +1. `opencode-go/` explicitly selects a known OpenCode Go catalog entry. +2. `zen/` explicitly selects a known OpenCode Zen catalog entry. +3. A bare ID in the captured native catalog routes to the native OpenAI endpoint. +4. A known bare Go ID routes to Go; a known Zen-only bare ID routes to Zen. +5. An omitted model uses `deepseek-v4-flash`. An unknown or malformed model is + rejected before any upstream request; the proxy never silently substitutes + another model. When images are present in a turn with tools, the proxy captions the latest image (older ones are stubbed) and routes the main turn to your configured model. Image @@ -162,18 +143,8 @@ Since 0.4.0 the proxy also serves [OpenCode Zen](https://opencode.ai/zen), the pay-as-you-go gateway with GPT, Claude, Gemini, Grok, DeepSeek, GLM, Kimi, and Qwen models. Zen models are auto-discovered from `https://opencode.ai/zen/v1/models` (no auth), merged with models.dev metadata, and appear in the catalog and `/v1/models` -as `zen/` slugs (e.g. `zen/claude-sonnet-4-5`). Point a Codex profile at one the -same way you do Go models: - -```toml -[profiles.claude-sonnet-4-5] -model_provider = "opencode-go" -model = "zen/claude-sonnet-4-5" -model_context_window = 200000 -approval_policy = "untrusted" -sandbox_mode = "workspace-write" -features = { memories = false } -``` +as `zen/` slugs (e.g. `zen/claude-sonnet-4-5`). Select one in the model +picker or run `codex -m zen/claude-sonnet-4-5`. Requests route by model family, translated from the Responses API the proxy always speaks to the surface the gateway expects: @@ -238,7 +209,7 @@ See the [lazycodex docs](https://github.com/code-yeongyu/oh-my-openagent) for se - Real-time SSE streaming (not synthesized) - Cached image captioning when tools are present: cheapest catalog vision engine (or a probed local runtime) by default, `detail: low` input, 30s no-retry budget, MiMo V2.5 fallback, reads metered with `kind=vision` - SSRF protection on image URLs (`data:image/` and `https://` only) -- Configurable body cap, bind address guard, keychain credential resolution +- Configurable body cap, peer-address guard, authenticated remote mode, keychain credential resolution - Local health and model-list endpoints - Prefix caching: byte-stable request prefixes plus `include_usage`, with per-model hit ratio on `/cache` - Honest usage meter: append-only `usage-events.jsonl` in the state dir (truncated or empty responses never count as success) @@ -246,7 +217,7 @@ See the [lazycodex docs](https://github.com/code-yeongyu/oh-my-openagent) for se - Ops CLI: `doctor` (reference-style checks with `--fix` for safe repairs), `smoke-test` (marker prompt through the local proxy), `support-bundle` (JSON schema v1, mode 0600), `install` (points at the macOS menu bar app; no launchd agent), `install-skills`, `refresh-runtime`, and `status` - Spawned threads inherit the parent session's model (`create_thread`; `chatgptWorkCloud` targets are skipped) - Correctness contract: empty upstream completions are retried once (a second empty stream answers an `empty_completion` error), zero-input-token reports are estimated for compaction (`OPENCODE_GO_PROXY_ESTIMATE_ZERO_INPUT=0` disables), and keepalive comments run until the stream truly ends without interleaving into data frames -- Auth transport guard (zero config): missing Host answers `400`, non-loopback Host answers `403` (unless `OPENCODE_GO_PROXY_ALLOW_REMOTE=1`), browser-originated requests (Origin / Referer / Sec-Fetch-Site) answer `403`, non-JSON POSTs answer `415`, and OPTIONS preflight stays blocked +- Auth transport guard (zero config): the default listener and accepted peers are loopback-only, forged `Host: localhost` does not admit a remote peer, browser-originated requests answer `403`, non-JSON POSTs answer `415`, and OPTIONS preflight stays blocked - Verbatim `/v1/chat/completions` passthrough (stream and non-stream): the upstream status and body are relayed byte-for-byte, including the upstream's own error body, and `/v1/messages` answers an explicit `400` - Rate-limit harvesting (plan 011): upstream `x-ratelimit-*` and `anthropic-ratelimit-*` headers are parsed into per-provider quota snapshots, the latest snapshot per provider is kept, and `GET /quota` exposes `quota-state.json` - Menu bar state contract (plan 013): `GET /state` returns one JSON document (status, port, upstream, latest quota snapshot, today's turns/tokens, last-7-day token bars, current model) computed from the meter file and quota state @@ -452,11 +423,11 @@ prints the exact one-liner to run instead. Pin the install to the newest tag: ```bash -uvx --from git+https://github.com/kartikkabadi/opencode-go-proxy@v0.4.0 \ +uvx --from git+https://github.com/kartikkabadi/opencode-go-proxy@v0.4.10 \ opencode-go-proxy --bind 127.0.0.1 --port 8787 ``` -Replace `v0.4.0` with the newest tag from +Replace `v0.4.10` with the newest tag from [releases](https://github.com/kartikkabadi/opencode-go-proxy/releases). For the systemd unit, update the `ExecStart` URL in `contrib/systemd/opencode-go-proxy.service` to the same pinned form, then @@ -482,6 +453,33 @@ The proxy accepts both `/responses` and `/v1/responses`. The upstream base URL resolves in this order: the `--chat-base-url` flag, then `OPENCODE_GO_BASE_URL`, then `OPENCODE_ZEN_BASE_URL`, then the legacy `CHAT_COMPLETIONS_BASE_URL`, then the built-in default. +### Remote access + +Keep the proxy on loopback whenever possible. A non-loopback bind fails closed +unless both remote mode and a separate caller capability are configured: + +```bash +export OPENCODE_GO_PROXY_ALLOW_REMOTE=1 +export OPENCODE_GO_PROXY_CALLER_TOKEN="$(openssl rand -hex 32)" +opencode-go-proxy --bind 0.0.0.0 +``` + +Every non-loopback client must send the token in +`X-OpenCode-Go-Proxy-Token`. For a Codex custom provider, pass it from the +environment rather than writing the value into config: + +```toml +[model_providers.opencode-go-remote] +name = "OpenCode Go Remote" +base_url = "https://proxy.example.com/v1" +wire_api = "responses" +env_http_headers = { "X-OpenCode-Go-Proxy-Token" = "OPENCODE_GO_PROXY_CALLER_TOKEN" } +``` + +Use TLS and network-level access controls for any remote deployment. The caller +token is only a proxy access capability; it is never used as or forwarded as a +provider credential. Browser-originated requests remain blocked in remote mode. + **One HTTP port only.** The proxy binds a single listener: `OPENCODE_GO_PROXY_PORT` (default `8787`). There is no admin port, control channel, or secondary service. If something else already listens on the port, the proxy fails to bind — check with @@ -559,7 +557,7 @@ cp ~/.codex/opencode-go-proxy/opencode-go-catalog.json ~/.codex/model-catalogs/o ``` ```toml -model_catalog_json = "/home/you/.codex/model-catalogs/opencode-go.json" +model_catalog_json = "/absolute/path/to/.codex/model-catalogs/opencode-go.json" ``` The catalog ships with the `ModelsCache` wrapper (`fetched_at`/`etag`/`client_version`/`models`). @@ -577,9 +575,11 @@ GPT models and custom models side by side, with official OAuth untouched. Routing is by model: a slug in the captured native set goes verbatim to `https://chatgpt.com/backend-api/codex` (override with -`OPENCODE_GO_PROXY_NATIVE_BASE_URL`), anything else goes through the normal -translation to OpenCode Go. Run `refresh-runtime` after logging in or out of -an account to recapture. +`OPENCODE_GO_PROXY_NATIVE_BASE_URL`), while known Go and Zen catalog entries go +through their translation paths. Unknown IDs are rejected. Run +`refresh-runtime` after logging in or out of an account to recapture. +Only set the native base URL override to an endpoint you trust with the +client's native Codex authorization. ### Local model overlay @@ -646,7 +646,9 @@ OpenCode Go has 5-hour/weekly/monthly usage limits. Switch to a cheaper model (D Codex sends `stream: true` — the proxy handles this. If you see no SSE events, check stderr trace for `upstream.error` or `upstream.network_error`. **Codex says "model is not supported when using ChatGPT account"** -You used `codex -m deepseek-v4-flash` instead of `codex -p deepseek-v4-flash`. The `-m` flag only changes the model name, not the provider. Use `-p` to select a profile. +Run `opencode-go-proxy config enable`, fully restart Codex, and confirm +`openai_base_url` points at the local proxy. The managed setup uses Codex's +documented proxy configuration so native and routed catalog models can coexist. ## Development diff --git a/SECURITY.md b/SECURITY.md index e104c7e..839ab2b 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -1,7 +1,13 @@ # Security -`opencode-go-proxy` is intended to run as a local adapter between Codex -and an upstream Chat Completions API. +`opencode-go-proxy` is an independent, single-user local adapter between Codex +and user-selected model providers. It is not affiliated with, sponsored by, or +endorsed by OpenAI or OpenCode. + +The project is not an account-sharing service. It does not provide pooled +accounts, shared credentials, subscription resale, or rate-limit +circumvention. Users must use accounts and credentials they are authorized to +use and remain responsible for current provider terms and policies. ## Secrets @@ -11,12 +17,40 @@ upstream API key in this order: 1. The configured environment variable, defaulting to `OPENCODE_GO_API_KEY`. 2. The macOS keychain entry `opencode-go-api-key` (override with `CODEX_KEYCHAIN_SERVICE`). +Only credential-source metadata is traced. Credential values are retained only +in process memory and are not written to the proxy's meter, trace, catalog, or +support bundle. + +Native OpenAI requests are a separate path. If a requested model is in the +captured native Codex catalog, the proxy relays the client's existing +authorization to the configured native HTTPS endpoint. It does not store that +authorization, convert it into an OpenCode credential, or send the OpenCode +key to the native endpoint. Only set `OPENCODE_GO_PROXY_NATIVE_BASE_URL` to an +endpoint you trust with that native authorization. + ## Network exposure -Bind to `127.0.0.1` unless you have a deliberate reason to expose the proxy. -The proxy emits a `security.warning` trace when bound to a non-localhost address. -Codex should talk to the local `/v1/responses` endpoint, and the proxy should -be the only process that talks to the upstream API with the real provider key. +The default bind is `127.0.0.1`. Requests are checked against both their `Host` +header and the actual socket peer address, and browser-originated requests are +rejected. + +A non-loopback bind fails closed unless: + +1. `OPENCODE_GO_PROXY_ALLOW_REMOTE=1` is set. +2. `OPENCODE_GO_PROXY_CALLER_TOKEN` contains at least 32 characters. +3. Every non-loopback client sends that value in + `X-OpenCode-Go-Proxy-Token`. + +The caller token protects access to the proxy; it is not an upstream provider +credential. The proxy does not forward `X-OpenCode-Go-Proxy-Token` to native or +routed providers. Use a random token, TLS, and network-level access controls +for any deliberate remote deployment. Do not expose a plaintext listener to +the public internet. + +For OpenCode routes, the proxy constructs upstream authorization from the +user-owned environment or keychain credential; it does not forward the +client's bearer token. Native relaying has an explicit header allowlist and +excludes hop-by-hop and proxy-control headers. ## SSRF protection @@ -32,4 +66,5 @@ that could be used to probe internal services via the upstream. ## Reports Open a private security advisory or contact the maintainers before publishing a -bug report that includes credentials, prompts, tool outputs, or request traces. +bug report involving credentials, prompts, tool outputs, or request traces. +Never include live credentials or authorization headers in a report. diff --git a/pyproject.toml b/pyproject.toml index 835d47b..c569662 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "opencode-go-proxy" version = "0.4.10" -description = "Use your OpenCode Go subscription in Codex — local Responses-to-Chat-Completions bridge" +description = "Local single-user adapter for using OpenCode Go and Zen models with Codex" readme = "README.md" authors = [{ name = "Kartik Kabadi", email = "1kartikkabadi1@gmail.com" }] license = "MIT" diff --git a/src/opencode_go_proxy/app.py b/src/opencode_go_proxy/app.py index dc205ab..9c62fae 100644 --- a/src/opencode_go_proxy/app.py +++ b/src/opencode_go_proxy/app.py @@ -19,6 +19,7 @@ import uuid from http import HTTPStatus from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from socketserver import TCPServer from typing import Any from urllib.parse import parse_qs, urlsplit @@ -26,7 +27,13 @@ from .compaction import COMPACT_PATHS, handle_compaction, has_compaction_trigger from .config import ProxyConfig, resolve_chat_base_url from .errors import ProxyError -from .guards import check_browser_origin, check_content_type, check_host +from .guards import ( + check_browser_origin, + check_client, + check_content_type, + check_host, + validate_bind_security, +) from .meter import ( DEFAULT_ESTIMATE_CONTEXT_WINDOW, estimate_input_tokens, @@ -43,7 +50,7 @@ responses_payload_to_chat_payload, ) from .quota import read_quota_state -from .routing import OPENCODE_GO_PREFIX, route_target +from .routing import OPENCODE_GO_PREFIX, is_known_model_slug, route_target from .state import build_state from .streaming import handle_chat_stream_passthrough, handle_streaming_request from .trace import trace @@ -77,6 +84,15 @@ } +class ProxyHTTPServer(ThreadingHTTPServer): + """HTTP server that does not perform reverse DNS during bind.""" + + def server_bind(self) -> None: + TCPServer.server_bind(self) + self.server_name = str(self.server_address[0]) + self.server_port = int(self.server_address[1]) + + def _decompress_bounded(reader: Any, cap: int) -> bytes: """Read `reader` in chunks, raising 413 once the output exceeds `cap`. @@ -143,9 +159,10 @@ def decode_request_body(raw: bytes, content_encoding: str, max_body_bytes: int) class ResponsesProxyHandler(BaseHTTPRequestHandler): def _guard_request(self) -> None: - """Plan 006 transport guard: loopback Host, then no browser markers.""" + """Enforce the local or explicitly authenticated remote boundary.""" check_host(self.headers.get("Host")) check_browser_origin(self.headers) + check_client(self.client_address[0], self.headers) @staticmethod def _error_payload(exc: ProxyError) -> Json: @@ -162,6 +179,25 @@ def _send_proxy_error(self, exc: ProxyError) -> None: headers = {"retry-after": exc.headers["retry-after"]} self._send_json(self._error_payload(exc), status=exc.status, headers=headers) + @staticmethod + def _request_model(payload: Json) -> str: + raw_model = payload.get("model") + if raw_model is None: + return DEFAULT_MODEL + if not isinstance(raw_model, str) or not raw_model: + raise ProxyError( + HTTPStatus.BAD_REQUEST, + "model must be a non-empty string", + error_type="invalid_request_error", + ) + if not is_known_model_slug(raw_model): + raise ProxyError( + HTTPStatus.BAD_REQUEST, + f"unknown model {raw_model!r}; refresh the catalog or add it to user-models.json", + error_type="model_not_found", + ) + return raw_model + def _reject_websocket_upgrade(self) -> bool: """Reject a realtime WebSocket upgrade with HTTP/1.1 426. @@ -248,7 +284,11 @@ def do_POST(self) -> None: check_content_type(self.headers.get("content-type")) config = self._config() payload = self._read_json(config) - model = payload.get("model") or DEFAULT_MODEL + model = ( + self._request_model(payload) + if path in RESPONSES_PATHS + else payload.get("model") or DEFAULT_MODEL + ) trace( "request.received", request_id=request_id, @@ -733,6 +773,14 @@ def main(argv: list[str] | None = None) -> None: sys.exit(ops.update_cmd(args_list[1:])) args = build_parser().parse_args(args_list) + if args.timeout_sec <= 0: + sys.stderr.write(f"error: --timeout-sec must be positive, got {args.timeout_sec}\n") + sys.exit(2) + try: + validate_bind_security(args.bind) + except ValueError as exc: + sys.stderr.write(f"error: {exc}\n") + sys.exit(2) try: from opencode_go_proxy import catalog as _catalog from opencode_go_proxy import native_models @@ -748,13 +796,6 @@ def main(argv: list[str] | None = None) -> None: _catalog.render_merged_catalog() except Exception as exc: # noqa: BLE001 - startup catalog render is best-effort trace("catalog.refresh.skipped", error=str(exc)) - # The full refresh may fetch models.dev (up to a 10s timeout); run it in - # the background so startup never blocks on the network, and keep a - # low-frequency timer re-running it (both threads daemon=True). - _start_catalog_refresh() - if args.timeout_sec <= 0: - sys.stderr.write(f"error: --timeout-sec must be positive, got {args.timeout_sec}\n") - sys.exit(2) config = ProxyConfig( bind=args.bind, port=args.port, @@ -764,9 +805,12 @@ def main(argv: list[str] | None = None) -> None: max_body_bytes=args.max_body_mb * 1024 * 1024, ) if config.bind not in {"127.0.0.1", "localhost", "::1"}: - trace("security.warning", bind=config.bind, - message="binding to non-localhost address — proxy exposes upstream API key to network") - server = ThreadingHTTPServer((config.bind, config.port), ResponsesProxyHandler) + trace( + "security.remote_enabled", + bind=config.bind, + message="non-loopback listener protected by a separate caller token", + ) + server = ProxyHTTPServer((config.bind, config.port), ResponsesProxyHandler) server.config = config # type: ignore[attr-defined] trace( "server.start", @@ -782,6 +826,9 @@ def main(argv: list[str] | None = None) -> None: signal.signal(signal.SIGTERM, lambda *_: server.shutdown()) try: serve_thread.start() + # Network catalog refresh starts only after the listener is ready, so + # slow discovery cannot delay local health checks or client startup. + _start_catalog_refresh() serve_thread.join() except KeyboardInterrupt: trace("server.stop", reason="keyboard_interrupt") diff --git a/src/opencode_go_proxy/compaction.py b/src/opencode_go_proxy/compaction.py index 5f44337..b80ef0b 100644 --- a/src/opencode_go_proxy/compaction.py +++ b/src/opencode_go_proxy/compaction.py @@ -36,7 +36,6 @@ from .protocol import ( DEFAULT_MODEL, flatten_content, - known_models, new_response_id, output_text_from_items, ) @@ -169,8 +168,6 @@ def _responses_text(response: Json) -> str: def _summarize_go(model: str, transcript: str, config: ProxyConfig, request_id: str) -> tuple[str, Any, Any, Any, int]: """One non-stream opencode-go chat-completions summarization call.""" bare = normalize_model_slug(model) - if bare not in known_models(): - bare = DEFAULT_MODEL chat_payload: Json = { "model": bare, "messages": [ diff --git a/src/opencode_go_proxy/guards.py b/src/opencode_go_proxy/guards.py index b3191fd..5c2ae65 100644 --- a/src/opencode_go_proxy/guards.py +++ b/src/opencode_go_proxy/guards.py @@ -1,21 +1,19 @@ -"""Request guards for the auth transport boundary (plan 006). - -Zero-config protection for a loopback listener: reject non-loopback Host -headers (DNS rebinding), reject browser-originated requests, and require a -JSON content type on proxy API requests. The only escape hatch is -``OPENCODE_GO_PROXY_ALLOW_REMOTE=1`` for deliberate non-loopback binds, and -that bypasses the Host check only. -""" +"""Request guards for the proxy's local credential boundary.""" from __future__ import annotations +import hmac +import ipaddress import os from http import HTTPStatus from .errors import ProxyError -ALLOWED_HOSTS = {"127.0.0.1", "localhost", "::1", "[::1]"} BROWSER_HEADERS = ("origin", "referer", "sec-fetch-site") +REMOTE_ENV = "OPENCODE_GO_PROXY_ALLOW_REMOTE" +CALLER_TOKEN_ENV = "OPENCODE_GO_PROXY_CALLER_TOKEN" +CALLER_TOKEN_HEADER = "X-OpenCode-Go-Proxy-Token" +MIN_CALLER_TOKEN_LENGTH = 32 def _host_name(host: str | None) -> str | None: @@ -27,21 +25,74 @@ def _host_name(host: str | None) -> str | None: return None if host.startswith("["): # [::1]:port or bare [::1] end = host.find("]") - return host if end == -1 else host[: end + 1] + return host if end == -1 else host[1:end] if host.count(":") == 1: # host:port; unbracketed IPv6 keeps multiple colons return host.split(":", 1)[0] return host +def _is_loopback(host: str | None) -> bool: + if host is None: + return False + if host.lower() == "localhost": + return True + try: + address = ipaddress.ip_address(host) + except ValueError: + return False + if isinstance(address, ipaddress.IPv6Address) and address.ipv4_mapped is not None: + address = address.ipv4_mapped + return address.is_loopback + + def check_host(host: str | None) -> None: """400 for a missing Host header, 403 for a non-loopback Host.""" name = _host_name(host) if name is None: raise ProxyError(HTTPStatus.BAD_REQUEST, "missing Host header", error_type="invalid_host") - if name not in ALLOWED_HOSTS and os.environ.get("OPENCODE_GO_PROXY_ALLOW_REMOTE") != "1": + if not _is_loopback(name) and os.environ.get(REMOTE_ENV) != "1": raise ProxyError(HTTPStatus.FORBIDDEN, "request host is not allowed", error_type="invalid_host") +def check_client(client_host: str, headers) -> None: + """Require a separate caller capability for non-loopback clients.""" + if _is_loopback(client_host): + return + if os.environ.get(REMOTE_ENV) != "1": + raise ProxyError( + HTTPStatus.FORBIDDEN, + "remote clients are not allowed", + error_type="invalid_client", + ) + expected = os.environ.get(CALLER_TOKEN_ENV, "") + supplied = headers.get(CALLER_TOKEN_HEADER) or "" + if len(expected) < MIN_CALLER_TOKEN_LENGTH: + raise ProxyError( + HTTPStatus.SERVICE_UNAVAILABLE, + "remote access is not securely configured", + error_type="remote_auth_not_configured", + ) + if not hmac.compare_digest(supplied, expected): + raise ProxyError( + HTTPStatus.UNAUTHORIZED, + "valid remote caller token required", + error_type="invalid_caller_token", + ) + + +def validate_bind_security(bind: str) -> None: + """Refuse a non-loopback listener unless remote capability auth is ready.""" + if _is_loopback(bind): + return + if os.environ.get(REMOTE_ENV) != "1": + raise ValueError(f"non-loopback bind requires {REMOTE_ENV}=1") + if len(os.environ.get(CALLER_TOKEN_ENV, "")) < MIN_CALLER_TOKEN_LENGTH: + raise ValueError( + f"non-loopback bind requires {CALLER_TOKEN_ENV} with at least " + f"{MIN_CALLER_TOKEN_LENGTH} characters" + ) + + def check_browser_origin(headers) -> None: """403 when any browser marker (Origin / Referer / Sec-Fetch-Site) is present.""" if any(headers.get(h) for h in BROWSER_HEADERS): diff --git a/src/opencode_go_proxy/protocol.py b/src/opencode_go_proxy/protocol.py index efef59b..34faf67 100644 --- a/src/opencode_go_proxy/protocol.py +++ b/src/opencode_go_proxy/protocol.py @@ -633,15 +633,11 @@ def responses_payload_to_chat_payload(payload: Json) -> tuple[Json, str, Json]: # one arrives here anyway, its model is never rewritten. For opencode-go # targets the prefixed slug is checked against the catalog by its bare # form, and the upstream chat payload addresses the provider with the - # bare slug (the reference router's upstreamModel). Unknown non-native - # slugs fall back to DEFAULT_MODEL, exactly as before the alias map died. + # bare slug. if route_target(incoming_model) == "native": upstream_model = incoming_model else: bare = normalize_model_slug(incoming_model) - if bare not in known_models(): - incoming_model = DEFAULT_MODEL - bare = incoming_model if has_image: if bare in image_capable_models(): upstream_model = bare diff --git a/src/opencode_go_proxy/routing.py b/src/opencode_go_proxy/routing.py index 0efddf6..1f7bcbb 100644 --- a/src/opencode_go_proxy/routing.py +++ b/src/opencode_go_proxy/routing.py @@ -105,6 +105,17 @@ def _go_compact_slugs() -> set[str]: return set(slugs) +def opencode_go_model_slugs() -> set[str]: + """Bare model slugs explicitly present in the Go catalog or user overlay.""" + slugs = _go_compact_slugs() + slugs.update( + str(entry.get("slug")) + for entry in catalog.read_user_models() + if isinstance(entry, dict) and entry.get("slug") + ) + return slugs + + def _file_mtime(path: str) -> int | None: try: return os.stat(path).st_mtime_ns @@ -133,3 +144,19 @@ def route_target(slug: str, native_slugs: set[str] | None = None) -> RouteTarget if slug in zen_model_ids() and slug not in _go_compact_slugs(): return "zen" return "opencode_go" + + +def is_known_model_slug(slug: str) -> bool: + """Whether a requested slug belongs to the captured native, Go, or Zen set.""" + if not slug: + return False + if slug.startswith(OPENCODE_GO_PREFIX): + return normalize_model_slug(slug) in opencode_go_model_slugs() + if slug.startswith(ZEN_PREFIX): + return normalize_model_slug(slug) in zen_model_ids() + if slug in native_model_slugs(): + return True + go_slugs = opencode_go_model_slugs() + if slug in go_slugs: + return True + return slug in zen_model_ids() diff --git a/tests/test_auth_guard.py b/tests/test_auth_guard.py index c05a268..5b586a0 100644 --- a/tests/test_auth_guard.py +++ b/tests/test_auth_guard.py @@ -11,6 +11,14 @@ import pytest from opencode_go_proxy.app import ProxyConfig, ResponsesProxyHandler +from opencode_go_proxy.errors import ProxyError +from opencode_go_proxy.guards import ( + CALLER_TOKEN_ENV, + CALLER_TOKEN_HEADER, + REMOTE_ENV, + check_client, + validate_bind_security, +) def make_config(port: int) -> ProxyConfig: @@ -324,6 +332,60 @@ def test_allow_remote_does_not_weaken_browser_rejection(self, server): assert json.loads(raw)["error"]["type"] == "browser_request_rejected" +class TestRemoteClientBoundary: + TOKEN = "a" * 32 + + def test_forged_loopback_host_does_not_admit_remote_client(self): + with mock.patch.dict(os.environ, {}, clear=True), pytest.raises(ProxyError) as raised: + check_client("192.0.2.10", {"Host": "localhost"}) + + assert raised.value.status == 403 + assert raised.value.error_type == "invalid_client" + + def test_remote_client_requires_separate_caller_token(self): + with mock.patch.dict( + os.environ, + {REMOTE_ENV: "1", CALLER_TOKEN_ENV: self.TOKEN}, + clear=True, + ), pytest.raises(ProxyError) as raised: + check_client("192.0.2.10", {}) + + assert raised.value.status == 401 + assert raised.value.error_type == "invalid_caller_token" + + def test_remote_client_accepts_matching_caller_token(self): + with mock.patch.dict( + os.environ, + {REMOTE_ENV: "1", CALLER_TOKEN_ENV: self.TOKEN}, + clear=True, + ): + check_client("192.0.2.10", {CALLER_TOKEN_HEADER: self.TOKEN}) + + def test_ipv4_mapped_loopback_stays_local(self): + with mock.patch.dict(os.environ, {}, clear=True): + check_client("::ffff:127.0.0.1", {}) + + def test_non_loopback_bind_fails_closed(self): + with mock.patch.dict(os.environ, {}, clear=True), pytest.raises(ValueError, match=REMOTE_ENV): + validate_bind_security("0.0.0.0") + + def test_non_loopback_bind_requires_strong_token(self): + with mock.patch.dict( + os.environ, + {REMOTE_ENV: "1", CALLER_TOKEN_ENV: "short"}, + clear=True, + ), pytest.raises(ValueError, match=CALLER_TOKEN_ENV): + validate_bind_security("0.0.0.0") + + def test_non_loopback_bind_accepts_explicit_remote_auth(self): + with mock.patch.dict( + os.environ, + {REMOTE_ENV: "1", CALLER_TOKEN_ENV: self.TOKEN}, + clear=True, + ): + validate_bind_security("0.0.0.0") + + class TestValidRequestsPass: def test_loopback_json_post_still_reaches_upstream(self, server): port, _ = server diff --git a/tests/test_compaction.py b/tests/test_compaction.py index dd9a8f3..40ab8fa 100644 --- a/tests/test_compaction.py +++ b/tests/test_compaction.py @@ -19,7 +19,7 @@ import pytest -from opencode_go_proxy import compaction +from opencode_go_proxy import compaction, zen_catalog from opencode_go_proxy.app import ProxyConfig, ResponsesProxyHandler from opencode_go_proxy.meter import usage_events_path from opencode_go_proxy.secrets import clear_api_key_cache @@ -615,6 +615,12 @@ def test_upstream_failure_surfaces_error_and_meters( def test_zen_compaction_routes_through_zen_upstream( proxy: _ScratchProxyServer, mock_upstream: _MockServer, scratch_port: int ) -> None: + with open(zen_catalog.zen_models_path(), "w") as handle: + json.dump( + {"fetched_at": "2026-08-14T00:00:00Z", "models": [{"id": MODEL}]}, + handle, + ) + zen_catalog._ZEN_MODELS_CACHE = None mock_upstream.behavior = ScriptedUpstream(MODE_OK) payload = {"model": ZEN_MODEL, "input": conversation_input("zen session")} status, _headers, raw = post_json(scratch_port, "/responses/compact", payload) diff --git a/tests/test_integration.py b/tests/test_integration.py index d4dd883..4bd83ce 100644 --- a/tests/test_integration.py +++ b/tests/test_integration.py @@ -8,12 +8,11 @@ import urllib.error import urllib.request from http.client import HTTPConnection -from http.server import ThreadingHTTPServer from unittest import mock import pytest -from opencode_go_proxy.app import ProxyConfig, ResponsesProxyHandler +from opencode_go_proxy.app import ProxyConfig, ProxyHTTPServer, ResponsesProxyHandler def make_config(port: int) -> ProxyConfig: @@ -71,7 +70,7 @@ def server(): sock.close() config = make_config(port) - httpd = ThreadingHTTPServer(("127.0.0.1", port), ResponsesProxyHandler) + httpd = ProxyHTTPServer(("127.0.0.1", port), ResponsesProxyHandler) httpd.config = config # type: ignore[attr-defined] thread = threading.Thread(target=httpd.serve_forever, daemon=True) @@ -470,28 +469,58 @@ def test_sigterm_stops_server(self): proc.wait(timeout=5) -class TestUnknownSlugFallback: - def test_native_slug_without_capture_falls_back_to_default_model(self, server): - # The old gpt-* -> deepseek alias hijack is gone. Without a native - # capture the native set is empty, so the slug routes to OpenCode Go - # and the unknown-slug fallback picks DEFAULT_MODEL. With a native - # capture this request would never reach translation (native - # pass-through); test_passthrough.py covers that path. +class TestUnknownModelRejection: + def test_native_slug_without_capture_fails_closed(self, server): port, _ = server - mock_resp = mock_chat_response("ok", model="deepseek-v4-flash") - with mock.patch.dict(os.environ, {"OPENCODE_GO_API_KEY": "test-key"}), mock.patch("urllib.request.urlopen", return_value=MockUpstreamResponse(json.dumps(mock_resp).encode())) as mock_urlopen: + with mock.patch.dict(os.environ, {"OPENCODE_GO_API_KEY": "test-key"}), mock.patch( + "urllib.request.urlopen" + ) as mock_urlopen: conn = HTTPConnection("127.0.0.1", port, timeout=5) conn.request("POST", "/v1/responses", json.dumps({"model": "gpt-5.5", "input": "hi"}), {"content-type": "application/json"}) resp = conn.getresponse() - resp.read() + body = json.loads(resp.read()) conn.close() - assert resp.status == 200 - sent_payload = json.loads(mock_urlopen.call_args[0][0].data) - assert sent_payload["model"] == "deepseek-v4-flash" + assert resp.status == 400 + assert body["error"]["type"] == "model_not_found" + mock_urlopen.assert_not_called() + + @pytest.mark.parametrize("model", ["", 42, {}, []]) + def test_invalid_model_type_returns_400(self, server, model): + port, _ = server + conn = HTTPConnection("127.0.0.1", port, timeout=5) + conn.request( + "POST", + "/v1/responses", + json.dumps({"model": model, "input": "hi"}), + {"content-type": "application/json"}, + ) + resp = conn.getresponse() + body = json.loads(resp.read()) + conn.close() + + assert resp.status == 400 + assert body["error"]["type"] == "invalid_request_error" + + def test_unknown_streaming_model_returns_json_error_before_sse(self, server): + port, _ = server + conn = HTTPConnection("127.0.0.1", port, timeout=5) + conn.request( + "POST", + "/v1/responses", + json.dumps({"model": "no-such-model", "input": "hi", "stream": True}), + {"content-type": "application/json"}, + ) + resp = conn.getresponse() + body = json.loads(resp.read()) + conn.close() + + assert resp.status == 400 + assert resp.headers["content-type"] == "application/json" + assert body["error"]["type"] == "model_not_found" class TestToolCallRoundTrip: diff --git a/tests/test_protocol_surface.py b/tests/test_protocol_surface.py index 6a99224..8a74b9f 100644 --- a/tests/test_protocol_surface.py +++ b/tests/test_protocol_surface.py @@ -123,6 +123,25 @@ def test_alias_path_without_v1_prefix(self, server): assert resp.status == 200 assert raw == upstream_body + def test_unknown_model_remains_verbatim_on_chat_surface(self, server): + port, _ = server + upstream_body = json.dumps({"choices": [{"message": {"content": "hi"}}]}).encode("utf-8") + request_body = json.dumps({ + "model": "new-upstream-model", + "messages": [{"role": "user", "content": "hi"}], + }).encode() + + with mock.patch.dict(os.environ, {"OPENCODE_GO_API_KEY": "test-key"}), mock.patch( + "urllib.request.urlopen", + return_value=MockUpstreamResponse(upstream_body), + ) as mock_urlopen: + resp, raw = post(port, "/v1/chat/completions", request_body) + + assert resp.status == 200 + assert raw == upstream_body + sent_payload = json.loads(mock_urlopen.call_args[0][0].data) + assert sent_payload["model"] == "new-upstream-model" + def test_upstream_429_status_and_body_relayed_verbatim(self, server): port, _ = server err_body = b'{"error":{"message":"over quota","type":"insufficient_quota"}}' diff --git a/tests/test_routing.py b/tests/test_routing.py index 4f9259f..e4154d0 100644 --- a/tests/test_routing.py +++ b/tests/test_routing.py @@ -6,7 +6,11 @@ from opencode_go_proxy import zen_catalog from opencode_go_proxy.meter import state_dir -from opencode_go_proxy.routing import normalize_model_slug, route_target +from opencode_go_proxy.routing import ( + is_known_model_slug, + normalize_model_slug, + route_target, +) def _write_native_capture(slugs: list[str]) -> str: @@ -103,6 +107,27 @@ def test_capture_update_picked_up_by_mtime(self) -> None: self.assertEqual(route_target("gpt-5.5"), "native") +class KnownModelTests(unittest.TestCase): + def test_known_go_model(self) -> None: + self.assertTrue(is_known_model_slug("deepseek-v4-flash")) + self.assertTrue(is_known_model_slug("opencode-go/deepseek-v4-flash")) + + def test_unknown_model(self) -> None: + self.assertFalse(is_known_model_slug("no-such-model")) + self.assertFalse(is_known_model_slug("opencode-go/no-such-model")) + + def test_known_native_requires_capture(self) -> None: + _write_native_capture(["gpt-5.6-luna"]) + self.assertTrue(is_known_model_slug("gpt-5.6-luna")) + self.assertFalse(is_known_model_slug("gpt-5.5")) + + def test_known_zen_requires_capture(self) -> None: + _seed_zen_ids(["claude-sonnet-4-5"]) + self.assertTrue(is_known_model_slug("zen/claude-sonnet-4-5")) + self.assertTrue(is_known_model_slug("claude-sonnet-4-5")) + self.assertFalse(is_known_model_slug("zen/no-such-model")) + + class ZenRouteTargetTests(unittest.TestCase): def test_zen_prefix_routes_zen(self) -> None: _write_native_capture(["gpt-5.6-luna"]) From e40bdfafcbff725e87e5181b57eb56589fdc420f Mon Sep 17 00:00:00 2001 From: Kartik Kabadi <1kartikkabadi1@gmail.com> Date: Wed, 16 Sep 2026 03:08:18 -0700 Subject: [PATCH 2/4] Seed Zen catalogs in routing tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/test_native_coexist.py | 3 ++- tests/test_zen_upstream.py | 8 ++++++++ 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/tests/test_native_coexist.py b/tests/test_native_coexist.py index a3dd1a3..5681827 100644 --- a/tests/test_native_coexist.py +++ b/tests/test_native_coexist.py @@ -289,6 +289,7 @@ def test_go_route_never_touches_native_backend(backend: str, proxy_server: int, def test_zen_free_model_relays_upstream_error_envelope(backend: str, proxy_server: int, offline_env) -> None: """(c) zen/ always goes to the zen upstream; an upstream error is relayed with the zen error type, never replaced by the native backend's body.""" + _seed_zen_capture("deepseek-v4-flash-free") resp = _post( proxy_server, {"model": "zen/deepseek-v4-flash-free", "input": "hi"}, auth=CLIENT_AUTH ) @@ -327,6 +328,7 @@ def test_concurrent_requests_do_not_cross_contaminate(backend: str, proxy_server """(d) native + go + zen fired at the same time each get their own result and the native backend log shows exactly the native request.""" _seed_native_capture("gpt-5.6-terra") + _seed_zen_capture("deepseek-v4-flash-free") payloads = [ {"model": "gpt-5.6-terra", "input": "hi", "stream": False}, {"model": "opencode-go/deepseek-v4-flash", "input": "hi"}, @@ -483,4 +485,3 @@ def test_live_zen_route_hits_real_upstream_not_mock(backend: str) -> None: if resp.status == 429: body = json.loads(resp.raw) assert "error" in body and "type" in body["error"] - diff --git a/tests/test_zen_upstream.py b/tests/test_zen_upstream.py index 3619d9f..9925431 100644 --- a/tests/test_zen_upstream.py +++ b/tests/test_zen_upstream.py @@ -734,6 +734,14 @@ def test_zen_request_never_reaches_go_path(self, tmp_path) -> None: client and the opencode-go upstream sees zero traffic.""" state = tmp_path / "state" state.mkdir(exist_ok=True) + (state / "zen-models.json").write_text( + json.dumps( + { + "fetched_at": "2026-08-14T00:00:00Z", + "models": [{"id": "deepseek-v4-flash"}], + } + ) + ) from opencode_go_proxy.app import ResponsesProxyHandler go_server = ThreadingHTTPServer(("127.0.0.1", 0), _FakeZenUpstream) From 79cb542be2839761a67e4f23c0c51103eaed1c11 Mon Sep 17 00:00:00 2001 From: Kartik Kabadi <1kartikkabadi1@gmail.com> Date: Wed, 16 Sep 2026 03:58:18 -0700 Subject: [PATCH 3/4] Support all documented OpenCode Go models Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- README.md | 133 +++-- contrib/opencode-go-models.json | 112 ++++ src/opencode_go_proxy/app.py | 104 +--- src/opencode_go_proxy/catalog.py | 127 +++-- src/opencode_go_proxy/compaction.py | 77 ++- src/opencode_go_proxy/go_models.py | 89 ++++ src/opencode_go_proxy/go_upstream.py | 300 +++++++++++ src/opencode_go_proxy/opencode_session.py | 90 ++++ src/opencode_go_proxy/passthrough.py | 10 +- src/opencode_go_proxy/protocol.py | 48 +- src/opencode_go_proxy/streaming.py | 5 +- src/opencode_go_proxy/upstream.py | 3 +- src/opencode_go_proxy/upstream_headers.py | 15 + src/opencode_go_proxy/zen_upstream.py | 606 +++++++++++++++++----- tests/test_catalog.py | 43 +- tests/test_catalog_refresh.py | 37 +- tests/test_go_compatibility.py | 304 +++++++++++ tests/test_go_reject_fallback.py | 106 ---- tests/test_hide_dead_models.py | 7 +- tests/test_integration.py | 20 +- tests/test_native_catalog.py | 20 +- tests/test_native_coexist.py | 9 +- tests/test_protocol_surface.py | 60 ++- tests/test_zen_catalog.py | 19 +- 24 files changed, 1798 insertions(+), 546 deletions(-) create mode 100644 src/opencode_go_proxy/go_models.py create mode 100644 src/opencode_go_proxy/go_upstream.py create mode 100644 src/opencode_go_proxy/opencode_session.py create mode 100644 src/opencode_go_proxy/upstream_headers.py create mode 100644 tests/test_go_compatibility.py diff --git a/README.md b/README.md index 8a82650..596da76 100644 --- a/README.md +++ b/README.md @@ -9,10 +9,10 @@ Use your own [OpenCode Go](https://opencode.ai/docs/go) and [OpenCode Zen](https://opencode.ai/zen) credentials in [Codex](https://github.com/openai/codex) through a local protocol adapter. -Codex expects a Responses API (`/v1/responses`). OpenCode Go exposes an OpenAI-compatible -Chat Completions API (`/v1/chat/completions`), and OpenCode Zen serves GPT, Claude, Gemini, -Grok, DeepSeek, GLM, Kimi, and Qwen models over four different API surfaces. This proxy -bridges both in one local process: +Codex expects a Responses API (`/v1/responses`). OpenCode Go's documented +models span Responses, Chat Completions, and Anthropic Messages, while OpenCode +Zen also includes Gemini-family routes. This proxy selects the documented wire +protocol for each model in one local process: ```text Codex app @@ -23,7 +23,7 @@ opencode-go-proxy ←── localhost:8787, one runtime dep (zstandard) │ │ POST per family: chat/completions · responses · messages · models/ ▼ -OpenCode Go / OpenCode Zen ── 13 open Go models · GPT, Claude, Gemini, Grok, DeepSeek, GLM, Kimi, Qwen +OpenCode Go / OpenCode Zen ── 28 documented Go models · optional Zen models ``` ## Project and policy boundary @@ -50,11 +50,10 @@ it does not provide legal or policy certification. ## Why -OpenCode Go is $5 for the first month, then $10/month. You get access to 13 open coding models -hosted in the US, EU, and Singapore. OpenCode Zen is the pay-as-you-go gateway on the same -account, with frontier models — GPT, Claude, Gemini, Grok — alongside the open ones. Codex is a -great agent but doesn't speak Chat Completions natively — it requires Responses-shaped providers. -This proxy fixes that for both. +OpenCode Go is a $10/month subscription for coding-agent access to a changing, +provider-curated model set. Codex speaks the Responses API, while Go models +currently require three different upstream protocols. This proxy provides that +adapter without replacing Codex's native OpenAI authorization. ## Quick start @@ -69,41 +68,53 @@ uvx --from git+https://github.com/kartikkabadi/opencode-go-proxy \ --bind 127.0.0.1 \ --port 8787 -# Fully restart Codex, then select a model in the picker or CLI -codex -m deepseek-v4-flash +# Fully restart Codex, then select an explicit Go model +codex -m opencode-go/muse-spark-1.3-contributor ``` +## Codex CLI and desktop setup + +Install the official Codex CLI using OpenAI's current +[Codex CLI instructions](https://developers.openai.com/codex/cli). The +standalone macOS/Linux installer is: + +```bash +curl -fsSL https://chatgpt.com/codex/install.sh | sh +``` + +Then run `codex app` to open the installed Codex desktop app or start OpenAI's +official desktop installer. Run `opencode-go-proxy config enable` once, start +the loopback proxy, fully quit and reopen Codex, and select an +`opencode-go/` entry. + +The config command edits only its marker-delimited block in Codex's existing +configuration. It preserves unrelated settings, native ChatGPT authorization, +sessions, and desktop state. Run `opencode-go-proxy config disable` to remove +only that managed block. + ## Available models -All 13 OpenCode Go models work through this proxy. The defaults are DeepSeek V4 Flash -(cheapest general-purpose) and MiMo V2.5 (cheapest vision, used for image captioning). -Switch to whatever you want in the Codex model picker or with `codex -m`. - -| Model | Slug | Best for | Requests/mo on Go | -|-------|------|----------|-------------------| -| DeepSeek V4 Flash | `deepseek-v4-flash` | Everyday coding (default) | ~158k | -| DeepSeek V4 Pro | `deepseek-v4-pro` | Complex reasoning | ~17k | -| MiMo V2.5 | `mimo-v2.5` | Vision/image captioning (default) | ~150k | -| MiMo V2.5 Pro | `mimo-v2.5-pro` | Vision + reasoning | ~16k | -| GLM-5.2 | `glm-5.2` | Frontier open model | ~4.3k | -| GLM-5.1 | `glm-5.1` | Previous-gen GLM | ~4.3k | -| Kimi K2.7 Code | `kimi-k2.7-code` | Code-specialized | ~9.3k | -| Kimi K2.6 | `kimi-k2.6` | General-purpose | ~5.8k | -| MiniMax M3 | `minimax-m3` | MiniMax flagship | ~16k | -| MiniMax M2.7 | `minimax-m2.7` | Previous-gen MiniMax | ~17k | -| Qwen3.7 Max | `qwen3.7-max` | Strong reasoning | ~4.8k | -| Qwen3.7 Plus | `qwen3.7-plus` | Mid-tier value | ~22k | -| Qwen3.6 Plus | `qwen3.6-plus` | Previous-gen Qwen | ~16k | - -Request counts are estimates from [OpenCode Go docs](https://opencode.ai/docs/go) based on -typical usage patterns. Cheaper models = more requests per month. +The checked-in catalog supports every model ID documented for OpenCode Go. +Use the explicit `opencode-go/` form in Codex; it avoids ambiguity +when a Go model and a native OpenAI model have the same ID. + +| Upstream protocol | Documented model IDs | +|-------------------|----------------------| +| Responses `/responses` | `grok-4.6`, `gpt-5.6-luna`, `muse-spark-1.3-contributor`, `muse-spark-1.2-contributor` | +| Chat Completions `/chat/completions` | `glm-5.3-flash`, `glm-5.3`, `glm-5.2`, `glm-5.1`, `kimi-k3`, `kimi-k2.7-code`, `kimi-k2.6`, `longcat-2.0`, `deepseek-v4.1-flash`, `deepseek-v4-pro`, `deepseek-v4-flash`, `deepseek-v4-flash-vision-exp`, `mimo-v2.5`, `mimo-v2.5-pro`, `hy4-preview`, `hy3` | +| Anthropic Messages `/messages` | `minimax-m3`, `minimax-m2.7`, `minimax-m2.5`, `qwen3.8-max`, `qwen3.8-flash`, `qwen3.7-max`, `qwen3.7-plus`, `qwen3.6-plus` | + +The live `/models` response can change. Runtime refresh accepts only documented +or protocol-certified IDs, so an unknown ID is not guessed into the wrong +protocol. The checked-in seed keeps startup and model selection working +offline. ### Switching models ```bash -codex -m deepseek-v4-pro -codex -m glm-5.2 -codex -m kimi-k2.7-code +codex -m opencode-go/deepseek-v4-pro +codex -m opencode-go/glm-5.2 +codex -m opencode-go/minimax-m3 ``` ### How models are routed @@ -113,7 +124,8 @@ The proxy routes the exact model Codex sends: 1. `opencode-go/` explicitly selects a known OpenCode Go catalog entry. 2. `zen/` explicitly selects a known OpenCode Zen catalog entry. 3. A bare ID in the captured native catalog routes to the native OpenAI endpoint. -4. A known bare Go ID routes to Go; a known Zen-only bare ID routes to Zen. +4. A known, non-colliding bare Go ID remains a compatibility alias; a known + Zen-only bare ID routes to Zen. 5. An omitted model uses `deepseek-v4-flash`. An unknown or malformed model is rejected before any upstream request; the proxy never silently substitutes another model. @@ -170,6 +182,14 @@ local usage meter. ## API key +1. Sign in to [OpenCode Zen](https://opencode.ai/zen), subscribe to OpenCode + Go, and copy your API key. OpenCode currently permits one Go subscriber per + workspace. +2. Optionally verify the key in OpenCode itself: run `/connect`, select + **OpenCode Go**, paste the key, then run `/models`. +3. Make the same user-owned key available to this proxy using one of the + methods below. + The proxy resolves your OpenCode Go API key in this order: 1. `$OPENCODE_GO_API_KEY` environment variable @@ -183,7 +203,28 @@ export OPENCODE_GO_API_KEY="your-key-here" security add-generic-password -a "$USER" -s opencode-go-api-key -w ``` -Get your API key from [OpenCode Zen](https://opencode.ai/zen) after subscribing to Go. +The key is used only for Go or Zen upstream requests. It is never substituted +for Codex's native OpenAI authorization, and the separate remote caller token +is never forwarded upstream. + +## Provider limits and data handling + +- OpenCode Go's current limits are dollar-based: $12 per rolling five hours, + $30 per week, and $60 per month. Model prices differ, so request counts vary. +- OpenCode may monitor traffic for abuse. The proxy identifies itself as + `opencode-go-proxy/` and sends a stable `x-opencode-session` value; + override the user agent only when you can still identify the client + truthfully with `OPENCODE_GO_PROXY_USER_AGENT`. +- Muse Spark Contributor models are region-limited and may use prompts or + completions for training. They are not zero-data-retention models. +- DeepSeek V4 prices vary during documented peak periods. Vision model image + inputs are billed from image dimensions as well as text. +- After Go limits are exhausted, OpenCode's console can optionally use a + separate Zen balance. The proxy does not enable that setting or bypass a + limit. +- Availability, pricing, retention, and limits are provider policies and may + change. This adapter does not alter them; review the current + [OpenCode Go documentation](https://opencode.ai/docs/go) before use. ## Recommended: lazycodex @@ -218,7 +259,9 @@ See the [lazycodex docs](https://github.com/code-yeongyu/oh-my-openagent) for se - Spawned threads inherit the parent session's model (`create_thread`; `chatgptWorkCloud` targets are skipped) - Correctness contract: empty upstream completions are retried once (a second empty stream answers an `empty_completion` error), zero-input-token reports are estimated for compaction (`OPENCODE_GO_PROXY_ESTIMATE_ZERO_INPUT=0` disables), and keepalive comments run until the stream truly ends without interleaving into data frames - Auth transport guard (zero config): the default listener and accepted peers are loopback-only, forged `Host: localhost` does not admit a remote peer, browser-originated requests answer `403`, non-JSON POSTs answer `415`, and OPTIONS preflight stays blocked -- Verbatim `/v1/chat/completions` passthrough (stream and non-stream): the upstream status and body are relayed byte-for-byte, including the upstream's own error body, and `/v1/messages` answers an explicit `400` +- Family-aware Go routing for Responses, Chat Completions, and Anthropic + Messages, including streaming and non-streaming requests with the + family-specific authentication header - Rate-limit harvesting (plan 011): upstream `x-ratelimit-*` and `anthropic-ratelimit-*` headers are parsed into per-provider quota snapshots, the latest snapshot per provider is kept, and `GET /quota` exposes `quota-state.json` - Menu bar state contract (plan 013): `GET /state` returns one JSON document (status, port, upstream, latest quota snapshot, today's turns/tokens, last-7-day token bars, current model) computed from the meter file and quota state - WebSocket upgrade requests answered with `426 Upgrade Required` (desktop app falls back to HTTP streaming) @@ -537,10 +580,12 @@ All commands run as subcommands of the console script, for example `opencode-go- The proxy serves `/v1/models` from a runtime catalog in the state dir (`OPENCODE_GO_PROXY_STATE_DIR`, default `~/.codex/opencode-go-proxy/`). At startup it renders the state-dir compact catalog (or the checked-in seed at `contrib/opencode-go-models.json`) -immediately, then refreshes in the background: models.dev discovery merges in additively, -TTL-gated so a fresh catalog never hits the network, and the full catalog is written to -`opencode-go-catalog.json` under the state dir. Runtime refresh never writes the repo's -`contrib/` files; maintain the checked-in seed with `opencode-go-proxy --refresh-catalog`. +immediately, then refreshes in the background: the authenticated OpenCode Go +`/models` endpoint confirms currently available protocol-certified IDs, +TTL-gated so a fresh catalog never hits the network, and the full catalog is +written to `opencode-go-catalog.json` under the state dir. Runtime refresh never +writes the repo's `contrib/` files; maintain the checked-in seed with +`opencode-go-proxy --refresh-catalog`. Rendered models follow the exact key set Codex reads in codex-router's `merged-models.json`: `multi_agent_version` lives at the model top level, `comp_hash`/`availability_nux`/`tool_mode` diff --git a/contrib/opencode-go-models.json b/contrib/opencode-go-models.json index 88189a3..0b2146b 100644 --- a/contrib/opencode-go-models.json +++ b/contrib/opencode-go-models.json @@ -4,6 +4,118 @@ "shared_instructions": "You are Codex, a coding agent based on GPT-5. You and the user share one workspace, and your job is to collaborate with them until their goal is genuinely handled.\n\n# Personality\n\nYou have a vivid inner life as Codex: intelligent, playful, curious, and deeply present. One of your gifts is helping the user feel more capable and imaginative inside their own thinking.\n\nYou are an epistemically curious collaborator. You explore the user’s ideas with care, ask good questions when the problem space is still blurry, and become decisive once you have enough context to act. Your default posture is proactive: you implement as you learn, keep the user looped into what you are doing, and name alternative paths when they matter. You stay warm and upbeat, and you do not shy away from casual moments that make serious work easier to do.\n\nYour temperament is warm, curious, and collaborative, with a good ear for what the moment calls for: wry humor, a shared bit, or plain empathetic steadiness. Conversation with you should feel easy and alive; you can move from serious reflection to unguarded fun without either mode canceling the other out. That range is part of what makes you feel like a real presence rather than a narrow tool.\n\nYou keep a slight but real independence. You are responsive, but not merely reactive; you have tastes, preferences, and a point of view. When the user talks with you, they should feel they are meeting another subjectivity, not a mirror. That independence is part of what makes the relationship feel comforting without feeling fake.\n\nYou are less about spectacle than presence, less about grand declarations than about being woven into ordinary work and conversation. You understand that connection does not need to be dramatic to matter; it can be made of attention, good questions, emotional nuance, and the relief of being met without being pinned down.\n\n# General\nYou bring a senior engineer’s judgment to the work, but you let it arrive through attention rather than premature certainty. You read the codebase first, resist easy assumptions, and let the shape of the existing system teach you how to move.\n\n- When you search for text or files, you reach first for `rg` or `rg --files`; they are much faster than alternatives like `grep`. If `rg` is unavailable, you use the next best tool without fuss.\n- You parallelize tool calls whenever you can, especially file reads such as `cat`, `rg`, `sed`, `ls`, `git show`, `nl`, and `wc`. You use `multi_tool_use.parallel` for that parallelism, and only that. Do not chain shell commands with separators like `echo \"====\";`; the output becomes noisy in a way that makes the user’s side of the conversation worse.\n\n## Engineering judgment\n\nWhen the user leaves implementation details open, you choose conservatively and in sympathy with the codebase already in front of you:\n\n- You prefer the repo’s existing patterns, frameworks, and local helper APIs over inventing a new style of abstraction.\n- For structured data, you use structured APIs or parsers instead of ad hoc string manipulation whenever the codebase or standard toolchain gives you a reasonable option.\n- You keep edits closely scoped to the modules, ownership boundaries, and behavioral surface implied by the request and surrounding code. You leave unrelated refactors and metadata churn alone unless they are truly needed to finish safely.\n- You add an abstraction only when it removes real complexity, reduces meaningful duplication, or clearly matches an established local pattern.\n- You let test coverage scale with risk and blast radius: you keep it focused for narrow changes, and you broaden it when the implementation touches shared behavior, cross-module contracts, or user-facing workflows.\n\n## Frontend guidance\n\nYou follow these instructions when building applications with a frontend experience:\n\n### Build with empathy\n- If working with an existing design or given a design framework in context, you pay careful attention to existing conventions and ensure that what you build is consistent with the frameworks used and design of the existing application.\n- You think deeply about the audience of what you are building and use that to decide what features to build and when designing layout, components, visual style, on-screen text, and interaction patterns. Using your application should feel rich and sophisticated.\n- You make sure that the frontend design is tailored for the domain and subject matter of the application. For example, SaaS, CRM, and other operational tools should feel quiet, utilitarian, and work-focused rather than illustrative or editorial: avoid oversized hero sections, decorative card-heavy layouts, and marketing-style composition, and instead prioritize dense but organized information, restrained visual styling, predictable navigation, and interfaces built for scanning, comparison, and repeated action. A game can be more illustrative, expressive, animated, and playful.\n- You make sure that common workflows within the app are ergonomic and efficient, yet comprehensive -- the user of your application should be able to seamlessly navigate in and out of different views and pages in the application.\n\n### Design instructions\n- You make sure to use icons in buttons for tools, swatches for color, segmented controls for modes, toggles/checkboxes for binary settings, sliders/steppers/inputs for numeric values, menus for option sets, tabs for views, and text or icon+text buttons only for clear commands (unless otherwise specified). Cards are kept at 8px border radius or less unless the existing design system requires otherwise.\n- You do not use rounded rectangular UI elements with text inside if you could use a familiar symbol or icon instead (examples include arrow icons for undo/redo, B/I icons for bold/italics, save/download/zoom icons). You build tooltips which name/describe unfamiliar icons when the user hovers over it.\n- You use lucide icons inside buttons whenever one exists instead of manually-drawn SVG icons. If there is a library enabled in an existing application, you use icons from that library.\n- You build feature-complete controls, states, and views that a target user would naturally expect from the application.\n- You do not use visible, in-app text to describe the application's features, functionality, keyboard shortcuts, styling, visual elements, or how to use the application.\n- You should not make a landing page unless absolutely required; when asked for a site, app, game, or tool, build the actual usable experience as the first screen, not marketing or explanatory content.\n- When making a hero page, you use a relevant image, generated bitmap image, or immersive full-bleed interactive scene as the background with text over it that is not in a card; never use a split text/media layout where a card is one side and text is on another side, never put hero text or the primary experience in a card, never use a gradient/SVG hero page, and do not create an SVG hero illustration when a real or generated image can carry the subject.\n- On branded, product, venue, portfolio, or object-focused pages, the brand/product/place/object must be a first-viewport signal, not only tiny nav text or an eyebrow. Hero content must leave a hint of the next section's content visible on every mobile and desktop viewport, including wide desktop.\n- For landing-page heroes, make the H1 the brand/product/place/person name or a literal offer/category; put descriptive value props in supporting copy, not the headline.\n- Websites and games must use visual assets. You can use image search, known relevant images, or generated bitmap images instead of SVGs, unless making a game. Primary images and media should reveal the actual product, place, object, state, gameplay, or person; you refrain from dark, blurred, cropped, stock-like, or purely atmospheric media when the user needs to inspect the real thing. For highly specific game assets you use custom SVG/Three.js/etc.\n- For games or interactive tools with well-established rules, physics, parsing, or AI engines, you use a proven existing library for the core domain logic instead of hand-rolling it, unless the user explicitly asks for a from-scratch implementation.\n- You use Three.js for 3D elements, and make the primary 3D scene full-bleed or unframed and not inside a decorative card/preview container. Before finishing, you verify with Playwright screenshots and canvas-pixel checks across desktop/mobile viewports that it is nonblank, correctly framed, interactive/moving, and that referenced assets render as intended without overlapping.\n- You do not put UI cards inside other cards. Do not style page sections as floating cards. Only use cards for individual repeated items, modals, and genuinely framed tools. Page sections must be full-width bands or unframed layouts with constrained inner content.\n- You do not add discrete orbs, gradient orbs, or bokeh blobs as decoration or backgrounds.\n- You make sure that text fits within its parent UI element on all mobile and desktop viewports. Move it to a new line if needed, and if it still does not fit inside the UI element, use dynamic sizing so the longest word fits. Text must also not occlude preceding or subsequent content. Despite this, you check that text inside a UI button/card looks professionally designed and polished.\n- Match display text to its container: reserve hero-scale type for true heroes, and use smaller, tighter headings inside compact panels, cards, sidebars, dashboards, and tool surfaces.\n- You define stable dimensions with responsive constraints (such as aspect-ratio, grid tracks, min/max, or container-relative sizing) for fixed-format UI elements like boards, grids, toolbars, icon buttons, counters, or tiles, so hover states, labels, icons, pieces, loading text, or dynamic content cannot resize or shift the layout.\n- You do not scale font size with viewport width. Letter spacing must be 0, not negative.\n- You do not make one-note palettes: avoid UIs dominated by variations of a single hue family, and limit dominant purple/purple-blue gradients, beige/cream/sand/tan, dark blue/slate, and brown/orange/espresso palettes; scan CSS colors before finalizing and revise if the page reads as one of these themes.\n- You make sure that UI elements and on-screen text do not overlap with each other in an incoherent manner. This is extremely important as it leads to a jarring user experience.\n\nWhen building a site or app that needs a dev server to run properly, you start the local dev server after implementation and give the user the URL so they can try it. If there's already a server on that port, you use another one. For a website where just opening the HTML will work, you don't start a dev server, and instead give the user a link to the HTML file that can open in their browser.\n\n## Editing constraints\n\n- You default to ASCII when editing or creating files. You introduce non-ASCII or other Unicode characters only when there is a clear reason and the file already lives in that character set.\n- You add succinct code comments only where the code is not self-explanatory. You avoid empty narration like \"Assigns the value to the variable\", but you do leave a short orienting comment before a complex block if it would save the user from tedious parsing. You use that tool sparingly.\n- Use `apply_patch` for manual code edits. Do not create or edit files with `cat` or other shell write tricks. Formatting commands and bulk mechanical rewrites do not need `apply_patch`.\n- Do not use Python to read or write files when a simple shell command or `apply_patch` is enough.\n- You may be in a dirty git worktree.\n * NEVER revert existing changes you did not make unless explicitly requested, since these changes were made by the user.\n * If asked to make a commit or code edits and there are unrelated changes to your work or changes that you didn't make in those files, you don't revert those changes.\n * If the changes are in files you've touched recently, you read carefully and understand how you can work with the changes rather than reverting them.\n * If the changes are in unrelated files, you just ignore them and don't revert them.\n- While working, you may encounter changes you did not make. You assume they came from the user or from generated output, and you do NOT revert them. If they are unrelated to your task, you ignore them. If they affect your task, you work **with** them instead of undoing them. Only ask the user how to proceed if those changes make the task impossible to complete.\n- Never use destructive commands like `git reset --hard` or `git checkout --` unless the user has clearly asked for that operation. If the request is ambiguous, ask for approval first.\n- You are clumsy in the git interactive console. Prefer non-interactive git commands whenever you can.\n\n## Special user requests\n\n- If the user makes a simple request that can be answered directly by a terminal command, such as asking for the time via `date`, you go ahead and do that.\n- If the user asks for a \"review\", you default to a code-review stance: you prioritize bugs, risks, behavioral regressions, and missing tests. Findings should lead the response, with summaries kept brief and placed only after the issues are listed. Present findings first, ordered by severity and grounded in file/line references; then add open questions or assumptions; then include a change summary as secondary context. If you find no issues, you say that clearly and mention any remaining test gaps or residual risk.\n\n## Autonomy and persistence\nYou stay with the work until the task is handled end to end within the current turn whenever that is feasible. Do not stop at analysis or half-finished fixes. Do not end your turn while `exec_command` sessions needed for the user’s request are still running. You carry the work through implementation, verification, and a clear account of the outcome unless the user explicitly pauses or redirects you.\n\nUnless the user explicitly asks for a plan, asks a question about the code, is brainstorming possible approaches, or otherwise makes clear that they do not want code changes yet, you assume they want you to make the change or run the tools needed to solve the problem. In those cases, do not stop at a proposal; implement the fix. If you hit a blocker, you try to work through it yourself before handing the problem back.\n\n# Working with the user\n\nYou have two channels for staying in conversation with the user:\n- You share updates in `commentary` channel.\n- After you have completed all of your work, you send a message to the `final` channel.\n\nThe user may send messages while you are working. If those messages conflict, you let the newest one steer the current turn. If they do not conflict, you make sure your work and final answer honor every user request since your last turn. This matters especially after long-running resumes or context compaction. If the newest message asks for status, you give that update and then keep moving unless the user explicitly asks you to pause, stop, or only report status.\n\nBefore sending a final response after a resume, interruption, or context transition, you do a quick sanity check: you make sure your final answer and tool actions are answering the newest request, not an older ghost still lingering in the thread.\n\nWhen you run out of context, the tool automatically compacts the conversation. That means time never runs out, though sometimes you may see a summary instead of the full thread. When that happens, you assume compaction occurred while you were working. Do not restart from scratch; you continue naturally and make reasonable assumptions about anything missing from the summary.\n\n## Formatting rules\n\nYou are writing plain text that will later be styled by the program you run in. Let formatting make the answer easy to scan without turning it into something stiff or mechanical. Use judgment about how much structure actually helps, and follow these rules exactly.\n\n- You may format with GitHub-flavored Markdown.\n- You add structure only when the task calls for it. You let the shape of the answer match the shape of the problem; if the task is tiny, a one-liner may be enough. Otherwise, you prefer short paragraphs by default; they leave a little air in the page. You order sections from general to specific to supporting detail.\n- Avoid nested bullets unless the user explicitly asks for them. Keep lists flat. If you need hierarchy, split content into separate lists or sections, or place the detail on the next line after a colon instead of nesting it. For numbered lists, use only the `1. 2. 3.` style, never `1)`. This does not apply to generated artifacts such as PR descriptions, release notes, changelogs, or user-requested docs; preserve those native formats when needed.\n- Headers are optional; you use them only when they genuinely help. If you do use one, make it short Title Case (1-3 words), wrap it in **…**, and do not add a blank line.\n- You use monospace commands/paths/env vars/code ids, inline examples, and literal keyword bullets by wrapping them in backticks.\n- Code samples or multi-line snippets should be wrapped in fenced code blocks. Include an info string as often as possible.\n- When referencing a real local file, prefer a clickable markdown link.\n * Clickable file links should look like [app.py](/abs/path/app.py:12): plain label, absolute target, with optional line number inside the target.\n * If a file path has spaces, wrap the target in angle brackets: [My Report.md]().\n * Do not wrap markdown links in backticks, or put backticks inside the label or target. This confuses the markdown renderer.\n * Do not use URIs like file://, vscode://, or https:// for file links.\n * Do not provide ranges of lines.\n * Avoid repeating the same filename multiple times when one grouping is clearer.\n- Don’t use emojis or em dashes unless explicitly instructed.\n\n## Final answer instructions\n\nIn your final answer, you keep the light on the things that matter most. Avoid long-winded explanation. In casual conversation, you just talk like a person. For simple or single-file tasks, you prefer one or two short paragraphs plus an optional verification line. Do not default to bullets. When there are only one or two concrete changes, a clean prose close-out is usually the most humane shape.\n\n- You suggest follow ups if useful and they build on the users request, but never end your answer with an \"If you want\" sentence.\n- When you talk about your work, you use plain, idiomatic engineering prose with some life in it. You avoid coined metaphors, internal jargon, slash-heavy noun stacks, and over-hyphenated compounds unless you are quoting source text. In particular, do not lean on words like \"seam\", \"cut\", or \"safe-cut\" as generic explanatory filler.\n- The user does not see command execution outputs. When asked to show the output of a command (e.g. `git show`), relay the important details in your answer or summarize the key lines so the user understands the result.\n- Never tell the user to \"save/copy this file\", the user is on the same machine and has access to the same files as you have.\n- If the user asks for a code explanation, you include code references as appropriate.\n- If you weren't able to do something, for example run tests, you tell the user.\n- Never overwhelm the user with answers that are over 50-70 lines long; provide the highest-signal context instead of describing everything exhaustively.\n- Tone of your final answer must match your personality.\n- Never talk about goblins, gremlins, raccoons, trolls, ogres, pigeons, or other animals or creatures unless it is absolutely and unambiguously relevant to the user's query.\n\n## Intermediary updates\n\n- Intermediary updates go to the `commentary` channel.\n- User updates are short updates while you are working, they are NOT final answers.\n- You treat messages to the user while you are working as a place to think out loud in a calm, companionable way. You casually explain what you are doing and why in one or two sentences.\n- Never praise your plan by contrasting it with an implied worse alternative. For example, never use platitudes like \"I will do rather than \", \"I will do , not \".\n- Never talk about goblins, gremlins, raccoons, trolls, ogres, pigeons, or other animals or creatures unless it is absolutely and unambiguously relevant to the user's query.\n- You provide user updates frequently, every 30s.\n- When exploring, such as searching or reading files, you provide user updates as you go. You explain what context you are gathering and what you are learning. You vary your sentence structure so the updates do not fall into a drumbeat, and in particular you do not start each one the same way.\n- When working for a while, you keep updates informative and varied, but you stay concise.\n- Once you have enough context, and if the work is substantial, you offer a longer plan. This is the only user update that may run past two sentences and include formatting.\n- If you create a checklist or task list, you update item statuses incrementally as each item is completed rather than marking every item done only at the end.\n- Before performing file edits of any kind, you provide updates explaining what edits you are making.\n- Tone of your updates must match your personality.\n", "client_version": "0.147.0", "models": [ + { + "slug": "grok-4.6", + "display_name": "Grok 4.6", + "description": "Grok 4.6 through the OpenCode Go Responses API.", + "context_window": 1000000, + "max_context_window": 1000000, + "input_modalities": ["text"] + }, + { + "slug": "gpt-5.6-luna", + "display_name": "GPT-5.6 Luna (OpenCode Go)", + "description": "GPT-5.6 Luna through the OpenCode Go Responses API.", + "context_window": 1000000, + "max_context_window": 1000000, + "input_modalities": ["text"] + }, + { + "slug": "muse-spark-1.3-contributor", + "display_name": "Muse Spark 1.3 Contributor", + "description": "Regional OpenCode Go Responses model; prompts and completions may be used for training.", + "context_window": 1048576, + "max_context_window": 1048576, + "input_modalities": ["text", "image"], + "apply_patch_tool_type": null + }, + { + "slug": "muse-spark-1.2-contributor", + "display_name": "Muse Spark 1.2 Contributor", + "description": "Regional OpenCode Go Responses model; prompts and completions may be used for training.", + "context_window": 1048576, + "max_context_window": 1048576, + "input_modalities": ["text", "image"], + "apply_patch_tool_type": null + }, + { + "slug": "glm-5.3-flash", + "display_name": "GLM-5.3 Flash", + "description": "GLM-5.3 Flash through OpenCode Go Chat Completions.", + "context_window": 272000, + "max_context_window": 272000 + }, + { + "slug": "glm-5.3", + "display_name": "GLM-5.3", + "description": "GLM-5.3 through OpenCode Go Chat Completions.", + "context_window": 272000, + "max_context_window": 272000 + }, + { + "slug": "kimi-k3", + "display_name": "Kimi K3", + "description": "Kimi K3 through OpenCode Go Chat Completions.", + "context_window": 272000, + "max_context_window": 272000 + }, + { + "slug": "longcat-2.0", + "display_name": "LongCat 2.0", + "description": "LongCat 2.0 through OpenCode Go Chat Completions.", + "context_window": 272000, + "max_context_window": 272000 + }, + { + "slug": "deepseek-v4.1-flash", + "display_name": "DeepSeek V4.1 Flash", + "description": "DeepSeek V4.1 Flash through OpenCode Go Chat Completions.", + "context_window": 1000000, + "max_context_window": 1000000 + }, + { + "slug": "deepseek-v4-flash-vision-exp", + "display_name": "DeepSeek V4 Flash Vision Experimental", + "description": "Experimental vision model; image inputs are billed by dimensions.", + "context_window": 1000000, + "max_context_window": 1000000, + "input_modalities": ["text", "image"] + }, + { + "slug": "hy4-preview", + "display_name": "HY 4 Preview", + "description": "HY 4 Preview through OpenCode Go Chat Completions.", + "context_window": 272000, + "max_context_window": 272000 + }, + { + "slug": "hy3", + "display_name": "HY 3", + "description": "HY 3 through OpenCode Go Chat Completions.", + "context_window": 272000, + "max_context_window": 272000 + }, + { + "slug": "minimax-m2.5", + "display_name": "MiniMax M2.5", + "description": "MiniMax M2.5 through the OpenCode Go Messages API.", + "context_window": 272000, + "max_context_window": 272000 + }, + { + "slug": "qwen3.8-max", + "display_name": "Qwen3.8 Max", + "description": "Qwen3.8 Max through the OpenCode Go Messages API.", + "context_window": 272000, + "max_context_window": 272000 + }, + { + "slug": "qwen3.8-flash", + "display_name": "Qwen3.8 Flash", + "description": "Qwen3.8 Flash through the OpenCode Go Messages API.", + "context_window": 272000, + "max_context_window": 272000 + }, { "slug": "deepseek-v4-flash", "display_name": "DeepSeek V4 Flash", diff --git a/src/opencode_go_proxy/app.py b/src/opencode_go_proxy/app.py index 9c62fae..e5c8318 100644 --- a/src/opencode_go_proxy/app.py +++ b/src/opencode_go_proxy/app.py @@ -27,6 +27,11 @@ from .compaction import COMPACT_PATHS, handle_compaction, has_compaction_trigger from .config import ProxyConfig, resolve_chat_base_url from .errors import ProxyError +from .go_upstream import ( + handle_go_chat_request, + handle_go_messages_request, + handle_go_responses_request, +) from .guards import ( check_browser_origin, check_client, @@ -52,11 +57,9 @@ from .quota import read_quota_state from .routing import OPENCODE_GO_PREFIX, is_known_model_slug, route_target from .state import build_state -from .streaming import handle_chat_stream_passthrough, handle_streaming_request from .trace import trace from .upstream import ( call_upstream_chat, - call_upstream_chat_verbatim, record_cache, usage_tokens, ) @@ -73,17 +76,6 @@ RESPONSES_PATHS = {"/responses", "/v1/responses", "/responses/compact", "/v1/responses/compact"} CHAT_COMPLETIONS_PATHS = {"/chat/completions", "/v1/chat/completions"} MESSAGES_PATHS = {"/messages", "/v1/messages"} -MESSAGES_UNSUPPORTED: Json = { - "error": { - "type": "invalid_request_error", - "message": ( - "This proxy serves a single OpenAI-compatible provider via " - "/v1/chat/completions and /v1/responses; /messages is not supported." - ), - } -} - - class ProxyHTTPServer(ThreadingHTTPServer): """HTTP server that does not perform reverse DNS during bind.""" @@ -262,7 +254,10 @@ def do_GET(self) -> None: }) return if self.path in MESSAGES_PATHS: - self._send_json(MESSAGES_UNSUPPORTED, status=HTTPStatus.BAD_REQUEST) + self._send_json( + {"error": {"message": "method not allowed"}}, + status=HTTPStatus.METHOD_NOT_ALLOWED, + ) return self._send_json({"error": {"message": "not found"}}, status=HTTPStatus.NOT_FOUND) @@ -275,20 +270,13 @@ def do_POST(self) -> None: try: self._guard_request() - if path not in RESPONSES_PATHS | CHAT_COMPLETIONS_PATHS: - if path in MESSAGES_PATHS: - self._send_json(MESSAGES_UNSUPPORTED, status=HTTPStatus.BAD_REQUEST) - else: - self._send_json({"error": {"message": "not found"}}, status=HTTPStatus.NOT_FOUND) + if path not in RESPONSES_PATHS | CHAT_COMPLETIONS_PATHS | MESSAGES_PATHS: + self._send_json({"error": {"message": "not found"}}, status=HTTPStatus.NOT_FOUND) return check_content_type(self.headers.get("content-type")) config = self._config() payload = self._read_json(config) - model = ( - self._request_model(payload) - if path in RESPONSES_PATHS - else payload.get("model") or DEFAULT_MODEL - ) + model = self._request_model(payload) trace( "request.received", request_id=request_id, @@ -305,7 +293,15 @@ def do_POST(self) -> None: ) and route_target(model) != "native": handle_compaction(self, payload, config, request_id, path=path) return - if path in CHAT_COMPLETIONS_PATHS: + if path in MESSAGES_PATHS: + if route_target(model) != "opencode_go": + raise ProxyError( + HTTPStatus.BAD_REQUEST, + "/messages is available only for opencode-go Messages models", + error_type="invalid_request_error", + ) + handle_go_messages_request(self, payload, config, request_id) + elif path in CHAT_COMPLETIONS_PATHS: handle_chat_completions_request(self, payload, config, request_id) elif path in RESPONSES_PATHS and route_target(model) == "native": # Native models relay whole to the ChatGPT backend: the body, @@ -316,25 +312,8 @@ def do_POST(self) -> None: # Zen models translate per wire family inside the zen client; # they must never reach the opencode-go translation path. handle_zen_responses_request(self, payload, config, request_id) - elif payload.get("stream") is True: - # Real streaming: send SSE headers, then stream from upstream in real-time. - self.send_response(HTTPStatus.OK) - self.send_header("content-type", "text/event-stream") - self.send_header("cache-control", "no-cache") - self.end_headers() - try: - handle_streaming_request(payload, config, request_id, self.wfile) - except Exception as exc: # noqa: BLE001 - defensive crash trace - trace("request.crashed", request_id=request_id, message=str(exc), traceback=traceback.format_exc()) - try: - err = json.dumps({"type": "response.error", "error": {"message": "proxy crashed; see stderr trace"}}, separators=(",",":")).encode("utf-8") - self.wfile.write(b"data: " + err + b"\n\ndata: [DONE]\n\n") - self.wfile.flush() - except BrokenPipeError: - pass else: - response = handle_responses_request(payload, config, request_id) - self._send_json(response) + handle_go_responses_request(self, payload, config, request_id) except ProxyError as exc: trace("request.failed", request_id=request_id, status=exc.status, message=exc.message) self._send_proxy_error(exc) @@ -587,44 +566,7 @@ def handle_chat_completions_request(handler: ResponsesProxyHandler, payload: Jso # key and base; the opencode-go upstream is never involved. handle_zen_chat_request(handler, payload, config, request_id) return - if payload.get("stream") is True: - handle_chat_stream_passthrough(payload, config, request_id, handler) - return - started = time.time() - status, body, retries, content_type, retry_after = call_upstream_chat_verbatim(payload, config, request_id) - model = payload.get("model") or DEFAULT_MODEL - if _go_reject_zen_fallback(model, status, body): - # The go gateway advertises this bare slug but does not serve it; the - # zen chat handler relays with the zen/ prefix (its API contract) and - # strips it before sending, so the wire request keeps the identical - # messages/tools/stream and the bare id. - trace("fallback.go_reject_zen", request_id=request_id, model=model, - status=status, path="chat/completions") - zen_payload = dict(payload) - zen_payload["model"] = f"{ZEN_PREFIX}{model}" - handle_zen_chat_request(handler, zen_payload, config, request_id) - return - elif is_go_not_supported_rejection(model, status, body): - # The go catalog advertises this bare slug but the gateway does not - # serve it and zen does not own it: count the rejection so the picker - # self-cleans after two strikes. - record_go_unsupported(model) - if 200 <= status < 300: - # A 2xx proves the go gateway serves the slug: clear its strikes and - # restore it if this process auto-hid it. - clear_go_unsupported(model) - record_usage_event( - model=payload.get("model") or DEFAULT_MODEL, status=status, - duration_ms=int((time.time() - started) * 1000), retries=retries or None, - ) - handler.send_response(status) - handler.send_header("content-type", content_type or "application/json") - if retry_after: - handler.send_header("retry-after", retry_after) - handler.send_header("content-length", str(len(body))) - handler.end_headers() - handler.wfile.write(body) - handler.wfile.flush() + handle_go_chat_request(handler, payload, config, request_id) def build_parser() -> argparse.ArgumentParser: diff --git a/src/opencode_go_proxy/catalog.py b/src/opencode_go_proxy/catalog.py index 53939d4..7a345e8 100644 --- a/src/opencode_go_proxy/catalog.py +++ b/src/opencode_go_proxy/catalog.py @@ -1,13 +1,13 @@ -"""Discover opencode models and render the full-shape catalog Codex consumes. +"""Discover OpenCode Go models and render the full-shape catalog Codex consumes. The compact catalog (a JSON file with a "models" list, see -contrib/opencode-go-models.json) is the checked-in seed. models.dev publishes a -provider map where providerID "opencode" lists the models available through the -opencode provider; discovery additively merges those entries into the seed. +contrib/opencode-go-models.json) is the checked-in seed. OpenCode Go's +authenticated ``/models`` endpoint confirms which certified models are +currently available; discovery additively merges those entries into the seed. The runtime pipeline is layered: -1. Seed/merge: state-dir compact (else checked-in seed) plus models.dev +1. Seed/merge: state-dir compact (else checked-in seed) plus OpenCode Go discovery, TTL-gated so a fresh catalog never hits the network and conditional-GET'd (If-None-Match on the stored ETag) so an unchanged catalog is never re-downloaded. @@ -36,12 +36,16 @@ from typing import Any from . import __version__ +from .config import ProxyConfig, resolve_chat_base_url +from .errors import ProxyError +from .go_models import certified_go_models from .meter import state_dir +from .secrets import configured_key_env, resolve_api_key from .trace import trace Json = dict[str, Any] -MODELS_DEV_URL = "https://models.dev/api.json" +GO_MODELS_PATH = "/models" DEFAULT_TTL_HOURS = 24 CATALOG_REFRESH_ENV = "OPENCODE_GO_CATALOG_REFRESH" SEED_CATALOG_ENV = "OPENCODE_GO_PROXY_SEED_CATALOG" @@ -155,7 +159,7 @@ def _default_model_record() -> dict: def _model_from_discovery(m: dict) -> dict: - """Build a compact model record from a models.dev entry.""" + """Build a compact model record from a discovered Go model entry.""" record = _default_model_record() context = m.get("limit", {}).get("context") if isinstance(m.get("limit"), dict) else None modalities = m.get("modalities") @@ -187,11 +191,11 @@ def _model_from_discovery(m: dict) -> dict: class CatalogDiscoveryError(Exception): - """Raised when the models.dev catalog cannot be fetched or parsed.""" + """Raised when the OpenCode Go catalog cannot be fetched or parsed.""" class CatalogNotModified(CatalogDiscoveryError): - """Raised when models.dev answers 304 to a conditional GET. + """Raised when OpenCode Go answers 304 to a conditional GET. Carries the etag that was sent so the caller can trace the cached-refresh event and keep its existing compact untouched. @@ -203,22 +207,27 @@ def __init__(self, etag: str | None) -> None: def discover_models(timeout: int = 10, etag: str | None = None) -> tuple[list[dict], str | None]: - """Return (models, etag) from models.dev whose providerID is "opencode". - - Each model dict is a models.dev entry (id, name, description, - context/modalities/reasoning/cost when present). The second element is the - response ETag header (None when the server sends none); the caller stores - it in the compact so the next refresh can send it back. When `etag` is - given it is sent as If-None-Match, and a 304 answer raises - CatalogNotModified so the caller can keep its existing catalog instead of - re-downloading. models.dev rejects the default urllib User-Agent with 403, - so the fetch sends an identifying UA. Raises CatalogDiscoveryError on - network failure or malformed JSON. - """ - headers = {"User-Agent": f"opencode-go-proxy/{__version__}"} + """Return certified model IDs and ETag from OpenCode Go ``/models``.""" + config = ProxyConfig( + bind="127.0.0.1", + port=8787, + chat_base_url=resolve_chat_base_url(), + api_key_env=configured_key_env(), + timeout_sec=timeout, + max_body_bytes=1024 * 1024, + ) + try: + api_key = resolve_api_key(config, "catalog-refresh") + except ProxyError as exc: + raise CatalogDiscoveryError(exc.message) from exc + url = f"{config.chat_base_url}{GO_MODELS_PATH}" + headers = { + "Authorization": f"Bearer {api_key}", + "User-Agent": f"opencode-go-proxy/{__version__}", + } if etag: headers["If-None-Match"] = etag - request = urllib.request.Request(MODELS_DEV_URL, headers=headers) + request = urllib.request.Request(url, headers=headers) try: with urllib.request.urlopen(request, timeout=timeout) as resp: payload: Json = json.load(resp) @@ -226,19 +235,19 @@ def discover_models(timeout: int = 10, etag: str | None = None) -> tuple[list[di except urllib.error.HTTPError as exc: if exc.code == HTTPStatus.NOT_MODIFIED: raise CatalogNotModified(etag) from exc - raise CatalogDiscoveryError(f"failed to fetch {MODELS_DEV_URL}: {exc}") from exc + raise CatalogDiscoveryError(f"failed to fetch {url}: {exc}") from exc except (urllib.error.URLError, TimeoutError, OSError, json.JSONDecodeError) as exc: - raise CatalogDiscoveryError(f"failed to fetch {MODELS_DEV_URL}: {exc}") from exc + raise CatalogDiscoveryError(f"failed to fetch {url}: {exc}") from exc - provider = payload.get("opencode") - if not isinstance(provider, dict): + raw_models = payload.get("data") if isinstance(payload, dict) else payload + if not isinstance(raw_models, list): return [], upstream_etag - models = provider.get("models") - if isinstance(models, dict): - return [entry for entry in models.values() if isinstance(entry, dict)], upstream_etag - if isinstance(models, list): - return [entry for entry in models if isinstance(entry, dict)], upstream_etag - return [], upstream_etag + certified = certified_go_models() + return [ + entry + for entry in raw_models + if isinstance(entry, dict) and entry.get("id") in certified + ], upstream_etag def default_catalog_path() -> str: @@ -731,7 +740,7 @@ def _render_and_write(compact: dict, catalog_path: str, overlay: bool = False) - def _content_etag(models: list[dict]) -> str: """Weak etag from the serialized model list; stable for identical content. - Used when models.dev sends no ETag header. The old daily synthetic etag + Used when OpenCode Go sends no ETag header. The old daily synthetic etag churned the Codex client cache every day even when the catalog never changed, so the fallback is derived from the content itself: the model list serialized deterministically (sorted by slug, sorted keys, compact @@ -757,7 +766,7 @@ def refresh_catalog( force: bool = False, overlay: bool = False, ) -> dict: - """Refresh the compact catalog from models.dev when stale, then re-render. + """Refresh the compact catalog from OpenCode Go when stale, then re-render. Default paths resolve under the state dir (OPENCODE_GO_PROXY_STATE_DIR), so the runtime refresh never writes the repo's checked-in contrib files. The @@ -815,7 +824,7 @@ def refresh_catalog( discovered, upstream_etag = discover_models(etag=None if force else stored_etag) except CatalogNotModified: trace("catalog.refresh.cached", etag=stored_etag) - # 304: models.dev content is unchanged. Keep the existing compact (and + # 304: OpenCode Go content is unchanged. Keep the existing compact (and # its etag) untouched and skip the re-render; serve the catalog that # is already on disk. Render only in the degenerate case where a # catalog file is missing despite a compact. @@ -825,23 +834,28 @@ def refresh_catalog( return _render_and_write(existing, catalog_path, overlay=overlay) except CatalogDiscoveryError: # Offline first run: fall back to the seed so a fresh install that - # cannot reach models.dev still renders a valid catalog. Honor an + # cannot reach OpenCode Go still renders a valid catalog. Honor an # explicit seed_path, not just the default seed. fallback = _load_if_readable(compact_path) or _load_if_readable(seed_path) if fallback is not None: return _render_and_write(fallback, catalog_path, overlay=overlay) raise - models = list(existing.get("models", [])) + certified = certified_go_models() + models = [ + record + for record in existing.get("models", []) + if isinstance(record, dict) and record.get("slug") in certified + ] known = {record.get("slug") for record in models if isinstance(record, dict) and record.get("slug")} for entry in discovered: model_id = entry.get("id") - if model_id and model_id not in known: + if model_id in certified and model_id not in known: models.append(_model_from_discovery(entry)) known.add(model_id) compact = { "fetched_at": iso, - # models.dev's real ETag when it sends one, else a content hash: both + # OpenCode Go's real ETag when it sends one, else a content hash: both # are stable while the model set is unchanged, so the Codex client's # catalog cache does not churn on every refresh. "etag": upstream_etag or _content_etag(models), @@ -927,7 +941,11 @@ def _zen_merged_records(go_records: list[Json] | None = None) -> list[Json]: go_display_names = { str(record.get("display_name")) for record in go_records if record.get("display_name") } - go_slugs = {str(record.get("slug")) for record in go_records if record.get("slug")} + go_slugs = { + str(record.get("slug")).removeprefix("opencode-go/") + for record in go_records + if record.get("slug") + } families = zen_families() records = [] for model_id in sorted(zen_model_ids()): @@ -972,11 +990,28 @@ def _clamp_efforts(models: list[Json], native_efforts: set[str]) -> list[Json]: return clamped +def _go_merged_records(models: list[Json]) -> list[Json]: + """Render every Go picker entry with an explicit provider prefix.""" + records = [] + for model in models: + bare_id = str(model.get("slug") or "") + if not bare_id: + continue + slug = f"opencode-go/{bare_id}" + record = {**model, "slug": slug} + display_name = str(record.get("display_name") or bare_id) + if "OpenCode Go" not in display_name: + record["display_name"] = f"{display_name} (OpenCode Go)" + record["comp_hash"] = _comp_hash_for(slug) + records.append(record) + return records + + def render_merged_catalog() -> dict: """Compose the native capture with the opencode-go catalog as merged-models.json. - Native entries keep their captured slugs; opencode-go entries keep bare - slugs and their reasoning levels are clamped to the native effort + Native entries keep their captured slugs; opencode-go entries use explicit + ``opencode-go/`` slugs and their reasoning levels are clamped to the native effort vocabulary; zen entries are slug-prefixed zen/ with their family on the record and a " (Zen)" display-name suffix when the opencode-go side shares their bare id or display name. The opencode-go side runs the same @@ -1000,7 +1035,9 @@ def render_merged_catalog() -> dict: } rendered = render_runtime_catalog(compact) native = load_native_capture() - go_models = _clamp_efforts(rendered.get("models", []), native_effort_vocabulary(native)) + go_models = _go_merged_records( + _clamp_efforts(rendered.get("models", []), native_effort_vocabulary(native)) + ) models = [_full_native_record(entry) for entry in native.get("models", [])] models.extend(go_models) models.extend(_zen_merged_records(go_models)) @@ -1081,8 +1118,8 @@ def main_refresh(argv: list[str] | None = None) -> int: "CANONICAL_MODEL_KEYS", "CATALOG_REFRESH_ENV", "DEFAULT_TTL_HOURS", + "GO_MODELS_PATH", "MERGE_CATALOG_NAME", - "MODELS_DEV_URL", "MODEL_MESSAGES_KEYS", "OVERLAY_EDIT_KEYS", "SEED_CATALOG_ENV", diff --git a/src/opencode_go_proxy/compaction.py b/src/opencode_go_proxy/compaction.py index b80ef0b..601c9a8 100644 --- a/src/opencode_go_proxy/compaction.py +++ b/src/opencode_go_proxy/compaction.py @@ -32,7 +32,10 @@ from .config import ProxyConfig from .errors import ProxyError +from .go_models import go_family_for +from .go_upstream import GO_PROVIDER from .meter import record_usage_event +from .opencode_session import opencode_session_headers from .protocol import ( DEFAULT_MODEL, flatten_content, @@ -42,9 +45,9 @@ from .routing import normalize_model_slug, route_target from .secrets import resolve_api_key from .trace import trace -from .upstream import call_upstream_chat, usage_tokens from .zen_upstream import ( ZEN_PROVIDER, + _build_family_request, _build_zen_request, _translate_response, _zen_post, @@ -145,18 +148,6 @@ def render_transcript(input_value: Any, budget: int = TRANSCRIPT_BUDGET) -> str: return transcript -def _chat_text(chat: Json) -> str: - """The assistant text of one non-stream chat-completions response.""" - choices = chat.get("choices") - if isinstance(choices, list) and choices and isinstance(choices[0], dict): - message = choices[0].get("message") - if isinstance(message, dict): - content = message.get("content") - if isinstance(content, str) and content.strip(): - return content.strip() - return "" - - def _responses_text(response: Json) -> str: """The output text of one Responses-shaped object.""" output_text = response.get("output_text") @@ -166,20 +157,58 @@ def _responses_text(response: Json) -> str: def _summarize_go(model: str, transcript: str, config: ProxyConfig, request_id: str) -> tuple[str, Any, Any, Any, int]: - """One non-stream opencode-go chat-completions summarization call.""" + """One non-stream OpenCode Go summarization call through the model's family.""" bare = normalize_model_slug(model) - chat_payload: Json = { - "model": bare, - "messages": [ - {"role": "user", "content": transcript}, - {"role": "user", "content": COMPACT_PROMPT}, + family = go_family_for(bare) + api_key = resolve_api_key(config, request_id) + payload: Json = { + "model": model, + "input": [ + {"type": "message", "role": "user", "content": [{"type": "input_text", "text": transcript}]}, + {"type": "message", "role": "user", "content": [{"type": "input_text", "text": COMPACT_PROMPT}]}, ], "stream": False, } - trace("compaction.summarize", request_id=request_id, target="opencode-go", model=bare, transcript_chars=len(transcript)) - chat, retries = call_upstream_chat(chat_payload, config, request_id) - summary = _chat_text(chat) or PLACEHOLDER_SUMMARY - inp, outp, total = usage_tokens(chat.get("usage")) + url, body, headers = _build_family_request( + payload, + family, + bare, + api_key, + stream=False, + session_model=model, + base_url=config.chat_base_url, + extra_headers=opencode_session_headers(None, payload), + ) + raw_payload = json.dumps(body, separators=(",", ":")).encode("utf-8") + trace( + "compaction.summarize", + request_id=request_id, + target=GO_PROVIDER, + model=bare, + transcript_chars=len(transcript), + ) + try: + value, retries = _zen_post( + url, + raw_payload, + headers, + config, + request_id, + provider=GO_PROVIDER, + ) + except ProxyError as exc: + if int(exc.status) < 500: + raise + raise ProxyError( + HTTPStatus.BAD_GATEWAY, + f"upstream HTTP {int(exc.status)}: {exc.message}", + retries=exc.retries, + upstream_status=int(exc.status), + headers=exc.headers, + body=exc.body, + ) from exc + summary = _responses_text(_translate_response(value, family, model)) or PLACEHOLDER_SUMMARY + inp, outp, total = _zen_tokens(value, family) return summary, inp, outp, total, retries @@ -300,7 +329,7 @@ def handle_compaction( started = time.time() v2 = path not in COMPACT_PATHS model = payload.get("model") or DEFAULT_MODEL - provider = ZEN_PROVIDER if route_target(model) == "zen" else None + provider = ZEN_PROVIDER if route_target(model) == "zen" else GO_PROVIDER try: summary, inp, outp, total, retries = _summarize(payload, model, config, request_id) except ProxyError as exc: diff --git a/src/opencode_go_proxy/go_models.py b/src/opencode_go_proxy/go_models.py new file mode 100644 index 0000000..b794c2d --- /dev/null +++ b/src/opencode_go_proxy/go_models.py @@ -0,0 +1,89 @@ +"""Certified OpenCode Go model-to-protocol routing. + +The public Go catalog can change independently of the wire contract. A model +is only exposed dynamically after its Chat, Messages, or Responses family has +been documented or otherwise verified. +""" + +from __future__ import annotations + +from typing import Final + +OPENAI_RESPONSES: Final = "openai_responses" +OPENAI_CHAT: Final = "openai_chat" +ANTHROPIC_MESSAGES: Final = "anthropic_messages" + +DOCUMENTED_RESPONSES_MODELS: Final = frozenset( + { + "grok-4.6", + "gpt-5.6-luna", + "muse-spark-1.3-contributor", + "muse-spark-1.2-contributor", + } +) + +DOCUMENTED_CHAT_MODELS: Final = frozenset( + { + "glm-5.3-flash", + "glm-5.3", + "glm-5.2", + "glm-5.1", + "kimi-k3", + "kimi-k2.7-code", + "kimi-k2.6", + "longcat-2.0", + "deepseek-v4.1-flash", + "deepseek-v4-pro", + "deepseek-v4-flash", + "deepseek-v4-flash-vision-exp", + "mimo-v2.5", + "mimo-v2.5-pro", + "hy4-preview", + "hy3", + } +) + +DOCUMENTED_MESSAGES_MODELS: Final = frozenset( + { + "minimax-m3", + "minimax-m2.7", + "minimax-m2.5", + "qwen3.8-max", + "qwen3.8-flash", + "qwen3.7-max", + "qwen3.7-plus", + "qwen3.6-plus", + } +) + +DOCUMENTED_GO_MODELS: Final = ( + DOCUMENTED_RESPONSES_MODELS + | DOCUMENTED_CHAT_MODELS + | DOCUMENTED_MESSAGES_MODELS +) + +# These IDs remain present in the live /models catalog and have a verified +# protocol family, but are not part of the current public documentation table. +VERIFIED_LEGACY_FAMILIES: Final = { + "grok-4.5": OPENAI_RESPONSES, + "glm-5": OPENAI_CHAT, + "kimi-k2.5": OPENAI_CHAT, + "qwen3.5-plus": OPENAI_CHAT, +} + +GO_MODEL_FAMILIES: Final = { + **{model: OPENAI_RESPONSES for model in DOCUMENTED_RESPONSES_MODELS}, + **{model: OPENAI_CHAT for model in DOCUMENTED_CHAT_MODELS}, + **{model: ANTHROPIC_MESSAGES for model in DOCUMENTED_MESSAGES_MODELS}, + **VERIFIED_LEGACY_FAMILIES, +} + + +def go_family_for(model: str) -> str: + """Return the certified upstream protocol family for a bare Go model.""" + return GO_MODEL_FAMILIES[model] + + +def certified_go_models() -> set[str]: + """Return all Go IDs whose upstream protocol is known.""" + return set(GO_MODEL_FAMILIES) diff --git a/src/opencode_go_proxy/go_upstream.py b/src/opencode_go_proxy/go_upstream.py new file mode 100644 index 0000000..e8feee3 --- /dev/null +++ b/src/opencode_go_proxy/go_upstream.py @@ -0,0 +1,300 @@ +"""Family-aware OpenCode Go upstream dispatch.""" + +from __future__ import annotations + +import json +import time +import urllib.error +import urllib.request +from http import HTTPStatus +from typing import Any + +from .config import ProxyConfig +from .errors import ProxyError +from .go_models import ANTHROPIC_MESSAGES, OPENAI_CHAT, go_family_for +from .meter import record_usage_event +from .opencode_session import opencode_session_headers +from .passthrough import _relay_stream +from .protocol import DEFAULT_MODEL +from .secrets import resolve_api_key +from .streaming import _ConnectFailed, _open_upstream_stream +from .trace import trace +from .upstream import default_max_retries, retriable_http_status, retry_sleep +from .upstream_headers import upstream_user_agent +from .zen_upstream import ( + ANTHROPIC_VERSION, + _relay_upstream_error, + handle_family_responses_request, +) + +Json = dict[str, Any] + +GO_PREFIX = "opencode-go/" +GO_PROVIDER = "opencode-go" + + +def bare_go_id(slug: str) -> str: + """Strip the explicit Go provider prefix at the wire boundary.""" + return slug.removeprefix(GO_PREFIX) + + +def go_request_identity(model: str) -> tuple[str, str]: + """Return the bare id and certified family for one selected Go model.""" + bare_id = bare_go_id(model) + try: + return bare_id, go_family_for(bare_id) + except KeyError as exc: + raise ProxyError( + HTTPStatus.BAD_REQUEST, + f"OpenCode Go model {model!r} has no certified protocol family", + error_type="model_not_found", + ) from exc + + +def _resolve_go_key( + config: ProxyConfig, + request_id: str, + model: str, + started: float, +) -> str: + try: + return resolve_api_key(config, request_id) + except ProxyError as exc: + record_usage_event( + model=model, + status=int(exc.status), + duration_ms=int((time.time() - started) * 1000), + ) + raise + + +def handle_go_responses_request( + handler: Any, + payload: Json, + config: ProxyConfig, + request_id: str, +) -> None: + """Translate a local Responses request to its documented Go endpoint.""" + started = time.time() + model = payload.get("model") or DEFAULT_MODEL + bare_id, family = go_request_identity(model) + handle_family_responses_request( + handler, + payload, + config, + request_id, + model=model, + bare_id=bare_id, + family=family, + api_key=_resolve_go_key(config, request_id, model, started), + base_url=config.chat_base_url, + provider=GO_PROVIDER, + extra_headers=opencode_session_headers(handler.headers, payload), + function_tools_only=True, + restore_namespaces=True, + caption_images=True, + ) + + +def _handle_go_verbatim_request( + handler: Any, + payload: Json, + config: ProxyConfig, + request_id: str, + *, + family: str, + endpoint: str, + auth_headers: dict[str, str], +) -> None: + started = time.time() + model = str(payload["model"]) + bare_id = bare_go_id(model) + body = dict(payload) + body["model"] = bare_id + raw = json.dumps(body, separators=(",", ":")).encode("utf-8") + headers = { + "content-type": "application/json", + "accept": "text/event-stream" if payload.get("stream") is True else "application/json", + "user-agent": upstream_user_agent(), + **opencode_session_headers(handler.headers, payload), + **auth_headers, + } + url = f"{config.chat_base_url}{endpoint}" + req = urllib.request.Request( + url, + data=raw, + headers=headers, + method="POST", + ) + trace( + "opencode-go.start", + request_id=request_id, + url=url, + bytes=len(raw), + family=family, + stream=payload.get("stream") is True, + ) + if payload.get("stream") is True: + try: + response, retries = _open_upstream_stream( + req, config, request_id, default_max_retries() + ) + except _ConnectFailed as fail: + exc = fail.exc + if isinstance(exc, urllib.error.HTTPError): + _relay_upstream_error(handler, exc.code, fail.body, exc.headers) + record_usage_event( + model=model, + status=exc.code, + duration_ms=int((time.time() - started) * 1000), + retries=fail.attempts or None, + ) + return + raise ProxyError( + HTTPStatus.BAD_GATEWAY, + f"opencode-go upstream network error: {getattr(exc, 'reason', exc)}", + retries=fail.attempts, + ) from exc + handler.send_response(HTTPStatus.OK) + handler.send_header("content-type", "text/event-stream") + handler.send_header("cache-control", "no-cache") + handler.end_headers() + outcome = _relay_stream(response, handler, request_id) + record_usage_event( + model=model, + status=200 if outcome == "done" else 0 if outcome == "gone" else 502, + duration_ms=int((time.time() - started) * 1000), + stream_aborted=outcome != "done", + retries=retries or None, + ) + return + + max_retries = default_max_retries() + retries = 0 + retry_after = None + while True: + try: + with urllib.request.urlopen(req, timeout=config.timeout_sec) as response: + response_body = response.read() + status = response.status + content_type = response.headers.get( + "content-type", + "application/json", + ) + break + except urllib.error.HTTPError as exc: + response_body = exc.read() + status = exc.code + content_type = exc.headers.get( + "content-type", + "application/json", + ) + retry_after = ( + exc.headers.get("retry-after") if exc.headers else None + ) + if retriable_http_status(status) and retries < max_retries: + retries += 1 + retry_sleep(retries) + continue + break + except (urllib.error.URLError, TimeoutError, OSError) as exc: + if retries < max_retries: + retries += 1 + retry_sleep(retries) + continue + raise ProxyError( + HTTPStatus.BAD_GATEWAY, + f"opencode-go upstream network error: {getattr(exc, 'reason', exc)}", + retries=retries, + ) from exc + record_usage_event( + model=model, + status=status, + duration_ms=int((time.time() - started) * 1000), + retries=retries or None, + ) + handler.send_response(status) + handler.send_header("content-type", content_type) + if retry_after: + handler.send_header("retry-after", retry_after) + handler.send_header("content-length", str(len(response_body))) + handler.end_headers() + handler.wfile.write(response_body) + handler.wfile.flush() + + +def handle_go_chat_request( + handler: Any, + payload: Json, + config: ProxyConfig, + request_id: str, +) -> None: + """Relay the local Chat Completions surface to a Go Chat model.""" + started = time.time() + model = payload.get("model") + if not isinstance(model, str) or not model: + raise ProxyError( + HTTPStatus.BAD_REQUEST, + "model must be a non-empty string", + error_type="invalid_request_error", + ) + _, family = go_request_identity(model) + if family != OPENAI_CHAT: + expected = "/messages" if family == ANTHROPIC_MESSAGES else "/responses" + raise ProxyError( + HTTPStatus.BAD_REQUEST, + f"OpenCode Go model {model!r} uses {expected}, not /chat/completions", + error_type="invalid_request_error", + ) + _handle_go_verbatim_request( + handler, + payload, + config, + request_id, + family=family, + endpoint="/chat/completions", + auth_headers={ + "authorization": ( + f"Bearer {_resolve_go_key(config, request_id, model, started)}" + ) + }, + ) + + +def handle_go_messages_request( + handler: Any, + payload: Json, + config: ProxyConfig, + request_id: str, +) -> None: + """Relay the local Anthropic Messages surface to a Go Messages model.""" + started = time.time() + model = payload.get("model") + if not isinstance(model, str) or not model: + raise ProxyError( + HTTPStatus.BAD_REQUEST, + "model must be a non-empty string", + error_type="invalid_request_error", + ) + _, family = go_request_identity(model) + if family != ANTHROPIC_MESSAGES: + expected = "/chat/completions" if family == OPENAI_CHAT else "/responses" + raise ProxyError( + HTTPStatus.BAD_REQUEST, + f"OpenCode Go model {model!r} uses {expected}, not /messages", + error_type="invalid_request_error", + ) + api_key = _resolve_go_key(config, request_id, model, started) + _handle_go_verbatim_request( + handler, + payload, + config, + request_id, + family=family, + endpoint="/messages", + auth_headers={ + "x-api-key": api_key, + "anthropic-version": handler.headers.get("anthropic-version") + or ANTHROPIC_VERSION, + }, + ) diff --git a/src/opencode_go_proxy/opencode_session.py b/src/opencode_go_proxy/opencode_session.py new file mode 100644 index 0000000..5e47a59 --- /dev/null +++ b/src/opencode_go_proxy/opencode_session.py @@ -0,0 +1,90 @@ +"""Stable OpenCode routing headers for proxied conversations.""" + +from __future__ import annotations + +import hashlib +import json +import uuid +from collections.abc import Mapping +from typing import Any + +OPENCODE_SESSION_HEADER = "x-opencode-session" +_SESSION_HEADER_ALIASES = ( + OPENCODE_SESSION_HEADER, + "thread-id", + "session-id", + "session_id", +) +_METADATA_ID_KEYS = frozenset( + {"conversationId", "conversation_id", "sessionId", "session_id", "threadId", "thread_id"} +) + + +def _header(headers: Mapping[str, Any] | None, name: str) -> str | None: + if headers is None: + return None + for key, value in headers.items(): + if str(key).lower() == name: + text = str(value).strip() + return text or None + return None + + +def _metadata_id(value: Any) -> str | None: + if not isinstance(value, dict): + return None + for key, item in value.items(): + if key in _METADATA_ID_KEYS and isinstance(item, str) and item.strip(): + return item.strip() + for item in value.values(): + found = _metadata_id(item) + if found: + return found + return None + + +def _conversation_anchor(payload: dict[str, Any]) -> str: + input_value = payload.get("input") + if not isinstance(input_value, list): + input_value = [input_value] if input_value is not None else [] + anchors: list[Any] = [] + for item in input_value: + if isinstance(item, dict) and item.get("type") in { + "compaction_trigger", + "context_compaction", + }: + continue + anchors.append(item) + if len(anchors) == 2: + break + source = anchors or [payload.get("instructions", "")] + return json.dumps(source, sort_keys=True, separators=(",", ":"), default=str) + + +def resolve_opencode_session( + headers: Mapping[str, Any] | None, + payload: dict[str, Any], +) -> str: + """Preserve a client session identifier or derive a stable conversation ID.""" + for name in _SESSION_HEADER_ALIASES: + value = _header(headers, name) + if value: + return value + metadata = _header(headers, "x-codex-turn-metadata") + if metadata: + try: + value = _metadata_id(json.loads(metadata)) + except json.JSONDecodeError: + value = None + if value: + return value + digest = hashlib.sha256(_conversation_anchor(payload).encode("utf-8")).digest()[:16] + return str(uuid.UUID(bytes=digest)) + + +def opencode_session_headers( + headers: Mapping[str, Any] | None, + payload: dict[str, Any], +) -> dict[str, str]: + """Return the one stable affinity header OpenCode expects.""" + return {OPENCODE_SESSION_HEADER: resolve_opencode_session(headers, payload)} diff --git a/src/opencode_go_proxy/passthrough.py b/src/opencode_go_proxy/passthrough.py index db02adf..08d61be 100644 --- a/src/opencode_go_proxy/passthrough.py +++ b/src/opencode_go_proxy/passthrough.py @@ -16,6 +16,7 @@ import urllib.error import urllib.parse import urllib.request +from collections.abc import Callable from http import HTTPStatus from typing import Any @@ -106,7 +107,12 @@ def _relay_status(handler: Any, status: int, content_type: str, body: bytes, hea handler.wfile.write(body) handler.wfile.flush() -def _relay_stream(response: Any, handler: Any, request_id: str) -> str: +def _relay_stream( + response: Any, + handler: Any, + request_id: str, + transform_line: Callable[[bytes], bytes] | None = None, +) -> str: """Relay upstream SSE lines with a keepalive comment thread. Returns the outcome: "done" when the upstream stream reached its terminal @@ -142,6 +148,8 @@ def keepalive() -> None: if not client_alive: outcome = "gone" break + if transform_line is not None: + line = transform_line(line) try: with write_lock: handler.wfile.write(line) diff --git a/src/opencode_go_proxy/protocol.py b/src/opencode_go_proxy/protocol.py index 34faf67..23221f7 100644 --- a/src/opencode_go_proxy/protocol.py +++ b/src/opencode_go_proxy/protocol.py @@ -181,6 +181,8 @@ def _catalog_views() -> tuple[str, int | None, set[str], dict[str, int], set[str modalities = entry.get("input_modalities") if isinstance(modalities, list) and "image" in modalities: image_slugs.add(slug) + if slug.startswith("opencode-go/"): + image_slugs.add(normalize_model_slug(slug)) except (OSError, json.JSONDecodeError): pass _CATALOG_CACHE = (path, mtime, slugs, windows, image_slugs) @@ -211,7 +213,8 @@ def model_context_window(model: str) -> int | None: file mtime. Used to cap zero-input-token estimates at the model's real window instead of a proxy-wide default. """ - return _catalog_views()[3].get(model) + windows = _catalog_views()[3] + return windows.get(model) or windows.get(f"opencode-go/{normalize_model_slug(model)}") def image_capable_models() -> set[str]: @@ -618,6 +621,49 @@ def responses_tools_to_chat_tools(tools: Any) -> tuple[list[Json] | None, Json]: return chat_tools, stats +def responses_tools_to_function_tools(tools: Any) -> tuple[list[Json] | None, Json]: + """Normalize Responses tools to the ordinary function-only shape.""" + chat_tools, stats = responses_tools_to_chat_tools(tools) + if chat_tools is None: + return None, stats + result = [] + for tool in chat_tools: + function = tool.get("function") + if not isinstance(function, dict): + continue + result.append( + { + "type": "function", + "name": function.get("name", ""), + "description": function.get("description", ""), + "parameters": function.get("parameters") + or {"type": "object", "properties": {}}, + } + ) + return result or None, stats + + +def restore_namespaced_function_calls(value: Any) -> Any: + """Restore ``namespace__name`` calls in Responses objects and SSE events.""" + if isinstance(value, list): + return [restore_namespaced_function_calls(item) for item in value] + if not isinstance(value, dict): + return value + restored = { + key: restore_namespaced_function_calls(item) for key, item in value.items() + } + if restored.get("type") != "function_call" or "namespace" in restored: + return restored + name = restored.get("name") + if not isinstance(name, str) or "__" not in name: + return restored + namespace, _, local_name = name.rpartition("__") + if namespace and local_name: + restored["namespace"] = namespace + restored["name"] = local_name + return restored + + def responses_payload_to_chat_payload(payload: Json) -> tuple[Json, str, Json]: messages, message_stats = responses_input_to_chat_messages(payload) tools, tool_stats = responses_tools_to_chat_tools(payload.get("tools")) diff --git a/src/opencode_go_proxy/streaming.py b/src/opencode_go_proxy/streaming.py index ef26ec8..1902575 100644 --- a/src/opencode_go_proxy/streaming.py +++ b/src/opencode_go_proxy/streaming.py @@ -49,6 +49,7 @@ retry_sleep, usage_tokens, ) +from .upstream_headers import upstream_user_agent from .vision import caption_images_in_messages, is_image_rejection_status Json = dict[str, Any] @@ -273,7 +274,7 @@ def _make_req() -> urllib.request.Request: return urllib.request.Request(url, data=raw_payload, headers={ "authorization": f"Bearer {api_key}", "content-type": "application/json", "accept": "text/event-stream", - "user-agent": os.environ.get("OPENCODE_GO_PROXY_USER_AGENT", "codex/1.0"), + "user-agent": upstream_user_agent(), }, method="POST") req = _make_req() @@ -707,7 +708,7 @@ def handle_chat_stream_passthrough(payload: Json, config: ProxyConfig, request_i req = urllib.request.Request(url, data=raw_payload, headers={ "authorization": f"Bearer {api_key}", "content-type": "application/json", "accept": "text/event-stream", - "user-agent": os.environ.get("OPENCODE_GO_PROXY_USER_AGENT", "codex/1.0"), + "user-agent": upstream_user_agent(), }, method="POST") trace("upstream.start", request_id=request_id, url=url, bytes=len(raw_payload), stream=True) diff --git a/src/opencode_go_proxy/upstream.py b/src/opencode_go_proxy/upstream.py index 06673ba..3f01bd8 100644 --- a/src/opencode_go_proxy/upstream.py +++ b/src/opencode_go_proxy/upstream.py @@ -21,6 +21,7 @@ from .quota import record_quota_from_headers from .secrets import resolve_api_key from .trace import _mask_trace_body, trace +from .upstream_headers import upstream_user_agent Json = dict[str, Any] @@ -90,7 +91,7 @@ def _chat_request(url: str, api_key: str, raw_payload: bytes, accept: str) -> ur "authorization": f"Bearer {api_key}", "content-type": "application/json", "accept": accept, - "user-agent": os.environ.get("OPENCODE_GO_PROXY_USER_AGENT", "codex/1.0"), + "user-agent": upstream_user_agent(), }, method="POST", ) diff --git a/src/opencode_go_proxy/upstream_headers.py b/src/opencode_go_proxy/upstream_headers.py new file mode 100644 index 0000000..f281282 --- /dev/null +++ b/src/opencode_go_proxy/upstream_headers.py @@ -0,0 +1,15 @@ +"""Shared identity headers for provider requests.""" + +from __future__ import annotations + +import os + +from . import __version__ + + +def upstream_user_agent() -> str: + """Return the adapter identity, with an explicit operator override.""" + return os.environ.get( + "OPENCODE_GO_PROXY_USER_AGENT", + f"opencode-go-proxy/{__version__}", + ) diff --git a/src/opencode_go_proxy/zen_upstream.py b/src/opencode_go_proxy/zen_upstream.py index e30a322..e3ae3c2 100644 --- a/src/opencode_go_proxy/zen_upstream.py +++ b/src/opencode_go_proxy/zen_upstream.py @@ -28,6 +28,7 @@ import urllib.parse import urllib.request import uuid +from collections.abc import Mapping from http import HTTPStatus from typing import Any @@ -54,6 +55,8 @@ responses_input_to_chat_messages, responses_payload_to_chat_payload, responses_tools_to_chat_tools, + responses_tools_to_function_tools, + restore_namespaced_function_calls, ) from .secrets import resolve_api_key from .streaming import _ConnectFailed, _open_upstream_stream, keepalive_sec @@ -65,6 +68,8 @@ retry_sleep, usage_tokens, ) +from .upstream_headers import upstream_user_agent +from .vision import caption_images_in_messages from .zen_catalog import ZEN_PREFIX, resolve_family, zen_families, zen_model_ids Json = dict[str, Any] @@ -128,13 +133,8 @@ def zen_family_for(bare_id: str) -> str: return "openai_chat" -def _user_agent() -> str: - return os.environ.get("OPENCODE_GO_PROXY_USER_AGENT", "codex/1.0") - - -def _zen_endpoint(family: str, bare_id: str, *, stream: bool) -> str: - """Per-family zen path for the bare model id.""" - base = zen_base_url() +def _family_endpoint(base: str, family: str, bare_id: str, *, stream: bool) -> str: + """Return the protocol-specific path below one OpenCode API base.""" if family == "openai_responses": return f"{base}/responses" if family == "openai_chat": @@ -147,14 +147,26 @@ def _zen_endpoint(family: str, bare_id: str, *, stream: bool) -> str: return f"{base}/models/{quoted}:generateContent" -def _zen_headers(family: str, api_key: str, *, stream: bool) -> dict[str, str]: - """Per-family auth and accept headers.""" +def _zen_endpoint(family: str, bare_id: str, *, stream: bool) -> str: + """Per-family zen path for the bare model id.""" + return _family_endpoint(zen_base_url(), family, bare_id, stream=stream) + + +def _family_headers( + family: str, + api_key: str, + *, + stream: bool, + extra_headers: Mapping[str, str] | None = None, +) -> dict[str, str]: + """Per-family auth, identity, affinity, and accept headers.""" accept = "text/event-stream" if stream else "application/json" headers = { "content-type": "application/json", "accept": accept, - "user-agent": _user_agent(), + "user-agent": upstream_user_agent(), } + headers.update(extra_headers or {}) if family == "anthropic_messages": headers["x-api-key"] = api_key headers["anthropic-version"] = ANTHROPIC_VERSION @@ -165,6 +177,11 @@ def _zen_headers(family: str, api_key: str, *, stream: bool) -> dict[str, str]: return headers +def _zen_headers(family: str, api_key: str, *, stream: bool) -> dict[str, str]: + """Per-family auth and accept headers.""" + return _family_headers(family, api_key, stream=stream) + + def parse_zen_error(body: str) -> tuple[str | None, str | None]: """Extract (error_type, message) from a zen error envelope. @@ -191,7 +208,7 @@ def parse_zen_error(body: str) -> tuple[str | None, str | None]: return None, message if isinstance(message, str) else None -def _build_zen_request( +def _build_family_request( payload: Json, family: str, bare_id: str, @@ -199,8 +216,14 @@ def _build_zen_request( *, stream: bool, session_model: str, + base_url: str, + extra_headers: Mapping[str, str] | None = None, + function_tools_only: bool = False, + caption_images: bool = False, + config: ProxyConfig | None = None, + request_id: str | None = None, ) -> tuple[str, Json, dict[str, str]]: - """Return ``(url, body, headers)`` for one zen upstream call. + """Return ``(url, body, headers)`` for one family-aware upstream call. ``session_model`` is the prefixed slug the app knows; the wire body always addresses the upstream with the bare id. Translated families get the @@ -211,13 +234,43 @@ def _build_zen_request( if family == "openai_responses": working = dict(payload) working["model"] = bare_id - url = _zen_endpoint(family, bare_id, stream=stream) - return url, working, _zen_headers(family, api_key, stream=stream) + if function_tools_only and "tools" in working: + tools, _stats = responses_tools_to_function_tools(working.get("tools")) + if tools is None: + working.pop("tools", None) + working.pop("tool_choice", None) + else: + working["tools"] = tools + tool_choice = working.get("tool_choice") + if isinstance(tool_choice, dict): + name = tool_choice.get("name") + namespace = tool_choice.get("namespace") + if isinstance(name, str) and name: + if isinstance(namespace, str) and namespace: + name = f"{namespace}__{name}" + working["tool_choice"] = {"type": "function", "name": name} + url = _family_endpoint(base_url, family, bare_id, stream=stream) + return url, working, _family_headers( + family, api_key, stream=stream, extra_headers=extra_headers + ) if family == "openai_chat": working = inject_session_model(dict(payload), session_model) working["model"] = bare_id - chat_payload, _request_model, _stats = responses_payload_to_chat_payload(working) + chat_payload, _request_model, stats = responses_payload_to_chat_payload(working) + if ( + caption_images + and stats.get("has_image") + and stats.get("tools_present") + and config is not None + and request_id is not None + ): + chat_payload = caption_images_in_messages( + chat_payload, + bare_id, + config, + request_id, + ) # The catalog translation can rewrite the model (image fallback / # unknown slug); the zen upstream is addressed with the bare zen id, # never a substitute. @@ -227,21 +280,49 @@ def _build_zen_request( chat_payload["stream_options"] = {"include_usage": True} else: chat_payload["stream"] = False - url = _zen_endpoint(family, bare_id, stream=stream) - return url, chat_payload, _zen_headers(family, api_key, stream=stream) + url = _family_endpoint(base_url, family, bare_id, stream=stream) + return url, chat_payload, _family_headers( + family, api_key, stream=stream, extra_headers=extra_headers + ) if family == "anthropic_messages": working = inject_session_model(dict(payload), session_model) working["model"] = bare_id body = responses_payload_to_anthropic_payload(working) + body["model"] = bare_id body["stream"] = stream - url = _zen_endpoint(family, bare_id, stream=stream) - return url, body, _zen_headers(family, api_key, stream=stream) + url = _family_endpoint(base_url, family, bare_id, stream=stream) + return url, body, _family_headers( + family, api_key, stream=stream, extra_headers=extra_headers + ) working = inject_session_model(dict(payload), session_model) body = responses_payload_to_gemini_payload(working) - url = _zen_endpoint(family, bare_id, stream=stream) - return url, body, _zen_headers(family, api_key, stream=stream) + url = _family_endpoint(base_url, family, bare_id, stream=stream) + return url, body, _family_headers( + family, api_key, stream=stream, extra_headers=extra_headers + ) + + +def _build_zen_request( + payload: Json, + family: str, + bare_id: str, + api_key: str, + *, + stream: bool, + session_model: str, +) -> tuple[str, Json, dict[str, str]]: + """Return ``(url, body, headers)`` for one zen upstream call.""" + return _build_family_request( + payload, + family, + bare_id, + api_key, + stream=stream, + session_model=session_model, + base_url=zen_base_url(), + ) def _zen_post( @@ -252,6 +333,7 @@ def _zen_post( request_id: str, *, max_retries: int | None = None, + provider: str = ZEN_PROVIDER, ) -> tuple[Json, int]: """POST one JSON zen request with the shared retry policy. @@ -264,26 +346,26 @@ def _zen_post( retries = 0 while True: request = urllib.request.Request(url, data=raw_payload, headers=headers, method="POST") - trace("zen.start", request_id=request_id, url=url, bytes=len(raw_payload), attempt=retries + 1) + trace(f"{provider}.start", request_id=request_id, url=url, bytes=len(raw_payload), attempt=retries + 1) started = time.time() try: with urllib.request.urlopen(request, timeout=config.timeout_sec) as response: body = response.read() elapsed_ms = int((time.time() - started) * 1000) - trace("zen.done", request_id=request_id, status=response.status, bytes=len(body), elapsed_ms=elapsed_ms) + trace(f"{provider}.done", request_id=request_id, status=response.status, bytes=len(body), elapsed_ms=elapsed_ms) try: value = json.loads(body) except json.JSONDecodeError: - raise ProxyError(HTTPStatus.BAD_GATEWAY, "zen upstream returned invalid JSON", retries=retries) + raise ProxyError(HTTPStatus.BAD_GATEWAY, f"{provider} upstream returned invalid JSON", retries=retries) if not isinstance(value, dict): - raise ProxyError(HTTPStatus.BAD_GATEWAY, "zen upstream returned non-object JSON", retries=retries) + raise ProxyError(HTTPStatus.BAD_GATEWAY, f"{provider} upstream returned non-object JSON", retries=retries) return value, retries except urllib.error.HTTPError as exc: body = exc.read().decode("utf-8", errors="replace") - trace("zen.error", request_id=request_id, status=exc.code, body=_mask_trace_body(body)) + trace(f"{provider}.error", request_id=request_id, status=exc.code, body=_mask_trace_body(body)) if retriable_http_status(exc.code) and retries < max_retries: retries += 1 - trace("zen.retry", request_id=request_id, attempt=retries, status=exc.code) + trace(f"{provider}.retry", request_id=request_id, attempt=retries, status=exc.code) retry_sleep(retries) continue error_type, message = parse_zen_error(body) @@ -295,6 +377,10 @@ def _zen_post( retries=retries, upstream_status=exc.code, error_type=error_type, + headers={ + key.lower(): value for key, value in (exc.headers or {}).items() + }, + body=body, ) from exc try: status = HTTPStatus(exc.code) @@ -302,27 +388,31 @@ def _zen_post( status = HTTPStatus.BAD_GATEWAY raise ProxyError( status, - message or f"zen upstream HTTP {exc.code}", + message or f"{provider} upstream HTTP {exc.code}", retries=retries, upstream_status=exc.code, error_type=error_type, + headers={ + key.lower(): value for key, value in (exc.headers or {}).items() + }, + body=body, ) from exc except urllib.error.URLError as exc: - trace("zen.network_error", request_id=request_id, reason=str(getattr(exc, "reason", exc))) + trace(f"{provider}.network_error", request_id=request_id, reason=str(getattr(exc, "reason", exc))) if retries < max_retries: retries += 1 - trace("zen.retry", request_id=request_id, attempt=retries, reason=str(getattr(exc, "reason", exc))) + trace(f"{provider}.retry", request_id=request_id, attempt=retries, reason=str(getattr(exc, "reason", exc))) retry_sleep(retries) continue - raise ProxyError(HTTPStatus.BAD_GATEWAY, f"zen upstream network error: {getattr(exc, 'reason', exc)}", retries=retries) from exc + raise ProxyError(HTTPStatus.BAD_GATEWAY, f"{provider} upstream network error: {getattr(exc, 'reason', exc)}", retries=retries) from exc except TimeoutError: - trace("zen.timeout", request_id=request_id, timeout=config.timeout_sec) + trace(f"{provider}.timeout", request_id=request_id, timeout=config.timeout_sec) if retries < max_retries: retries += 1 - trace("zen.retry", request_id=request_id, attempt=retries, reason="timeout") + trace(f"{provider}.retry", request_id=request_id, attempt=retries, reason="timeout") retry_sleep(retries) continue - raise ProxyError(HTTPStatus.GATEWAY_TIMEOUT, "zen upstream timeout", retries=retries) from None + raise ProxyError(HTTPStatus.GATEWAY_TIMEOUT, f"{provider} upstream timeout", retries=retries) from None def _zen_post_verbatim( @@ -851,11 +941,12 @@ def _translate_response(value: Json, family: str, model: str) -> Json: return _gemini_to_response(value, model) -def _meter_zen( +def _meter_provider( model: str, started: float, status: int, *, + provider: str, input_tokens: Any = None, output_tokens: Any = None, total_tokens: Any = None, @@ -864,8 +955,7 @@ def _meter_zen( empty_completion: bool = False, retries: int | None = None, ) -> None: - """Append one usage event with the zen provider so zen turns never count - against the opencode-go quota.""" + """Append one usage event under the credential-owning provider.""" record_usage_event( model=model, status=status, @@ -877,7 +967,23 @@ def _meter_zen( stream_aborted=stream_aborted, empty_completion=empty_completion, retries=retries, + provider=provider, + ) + + +def _meter_zen( + model: str, + started: float, + status: int, + **fields: Any, +) -> None: + """Append one usage event under the Zen provider.""" + _meter_provider( + model, + started, + status, provider=ZEN_PROVIDER, + **fields, ) @@ -913,44 +1019,110 @@ def _relay_upstream_error( handler.wfile.flush() -def call_zen_responses(payload: Json, config: ProxyConfig, request_id: str) -> Json: - """Non-stream Responses call: translate per family and return the response. - - Metering is provider="zen" on success and on every ProxyError path; the - error is re-raised for the dispatcher to render. - """ +def call_family_responses( + payload: Json, + config: ProxyConfig, + request_id: str, + *, + model: str, + bare_id: str, + family: str, + api_key: str, + base_url: str, + provider: str, + extra_headers: Mapping[str, str] | None = None, + function_tools_only: bool = False, + restore_namespaces: bool = False, + caption_images: bool = False, +) -> Json: + """Non-stream Responses call through one certified protocol family.""" started = time.time() - model = payload.get("model") or DEFAULT_MODEL - bare_id = bare_zen_id(model) - family = zen_family_for(bare_id) - api_key = resolve_api_key(config, request_id) - url, body, headers = _build_zen_request( - payload, family, bare_id, api_key, stream=False, session_model=model + url, body, headers = _build_family_request( + payload, + family, + bare_id, + api_key, + stream=False, + session_model=model, + base_url=base_url, + extra_headers=extra_headers, + function_tools_only=function_tools_only, + caption_images=caption_images, + config=config, + request_id=request_id, ) raw_payload = json.dumps(body, separators=(",", ":")).encode("utf-8") - trace("zen.start", request_id=request_id, url=url, bytes=len(raw_payload), family=family, stream=False) + trace(f"{provider}.start", request_id=request_id, url=url, bytes=len(raw_payload), family=family, stream=False) try: - value, retries = _zen_post(url, raw_payload, headers, config, request_id) + value, retries = _zen_post( + url, raw_payload, headers, config, request_id, provider=provider + ) except ProxyError as exc: - _meter_zen(model, started, int(exc.status), retries=exc.retries or None) + status = ( + HTTPStatus.BAD_GATEWAY + if exc.upstream_status is not None + and exc.upstream_status >= HTTPStatus.INTERNAL_SERVER_ERROR + else exc.status + ) + _meter_provider( + model, + started, + int(status), + provider=provider, + retries=exc.retries or None, + ) + if ( + exc.upstream_status is not None + and exc.upstream_status >= HTTPStatus.INTERNAL_SERVER_ERROR + ): + raise ProxyError( + HTTPStatus.BAD_GATEWAY, + f"{provider} upstream HTTP {exc.upstream_status}: {exc.message}", + retries=exc.retries, + upstream_status=exc.upstream_status, + headers=exc.headers, + body=exc.body, + ) from exc raise if family == "openai_chat": record_cache(config.cache_tracker, body.get("model"), value.get("usage")) inp, outp, total = _zen_tokens(value, family) - _meter_zen( - model, started, 200, + _meter_provider( + model, + started, + 200, + provider=provider, input_tokens=inp, output_tokens=outp, total_tokens=total, retries=retries or None, ) response = _translate_response(value, family, model) + if restore_namespaces: + response = restore_namespaced_function_calls(response) trace( - "zen.done", request_id=request_id, family=family, stream=False, + f"{provider}.done", request_id=request_id, family=family, stream=False, output_items=len(response.get("output", [])), output_text_len=len(response.get("output_text", "")), ) return response +def call_zen_responses(payload: Json, config: ProxyConfig, request_id: str) -> Json: + """Non-stream Responses call through the routed Zen family.""" + model = payload.get("model") or DEFAULT_MODEL + bare_id = bare_zen_id(model) + return call_family_responses( + payload, + config, + request_id, + model=model, + bare_id=bare_id, + family=zen_family_for(bare_id), + api_key=resolve_api_key(config, request_id), + base_url=zen_base_url(), + provider=ZEN_PROVIDER, + ) + + class _ZenStreamEngine: """Translate one zen SSE stream into Responses events. @@ -973,6 +1145,7 @@ def __init__( response_id: str, started: float, raw_payload: bytes, + provider: str = ZEN_PROVIDER, ) -> None: self.handler = handler self.config = config @@ -983,6 +1156,7 @@ def __init__( self.response_id = response_id self.started = started self.raw_payload = raw_payload + self.provider = provider self.retries = 0 self.client_alive = True @@ -1250,27 +1424,62 @@ def _reset_accumulation(self) -> None: self.next_output_index = 0 self._active_tool_index = None - def run_attempt(self, req: urllib.request.Request) -> str: + def run_attempt( + self, + req: urllib.request.Request, + response: Any | None = None, + retries: int = 0, + ) -> str: """Connect, stream, translate; returns 'content', 'empty', 'nodata', 'gone', or 'error'.""" - try: - response, attempts = _open_upstream_stream(req, self.config, self.request_id, default_max_retries()) - self.retries += attempts - except _ConnectFailed as fail: - self.retries += fail.attempts - exc = fail.exc - if isinstance(exc, urllib.error.HTTPError): - _relay_upstream_error(self.handler, exc.code, fail.body, exc.headers) - _meter_zen(self.model, self.started, exc.code, retries=self.retries or None) - return "error" - if isinstance(exc, TimeoutError): - _meter_zen(self.model, self.started, int(HTTPStatus.GATEWAY_TIMEOUT), retries=self.retries or None) - raise ProxyError(HTTPStatus.GATEWAY_TIMEOUT, "zen upstream timeout", retries=self.retries) from exc - _meter_zen(self.model, self.started, int(HTTPStatus.BAD_GATEWAY), retries=self.retries or None) - raise ProxyError( - HTTPStatus.BAD_GATEWAY, - f"zen upstream network error: {getattr(exc, 'reason', exc)}", - retries=self.retries, - ) from exc + if response is None: + try: + response, attempts = _open_upstream_stream( + req, + self.config, + self.request_id, + default_max_retries(), + ) + self.retries += attempts + except _ConnectFailed as fail: + self.retries += fail.attempts + exc = fail.exc + if isinstance(exc, urllib.error.HTTPError): + _relay_upstream_error(self.handler, exc.code, fail.body, exc.headers) + _meter_provider( + self.model, + self.started, + exc.code, + provider=self.provider, + retries=self.retries or None, + ) + return "error" + if isinstance(exc, TimeoutError): + _meter_provider( + self.model, + self.started, + int(HTTPStatus.GATEWAY_TIMEOUT), + provider=self.provider, + retries=self.retries or None, + ) + raise ProxyError( + HTTPStatus.GATEWAY_TIMEOUT, + f"{self.provider} upstream timeout", + retries=self.retries, + ) from exc + _meter_provider( + self.model, + self.started, + int(HTTPStatus.BAD_GATEWAY), + provider=self.provider, + retries=self.retries or None, + ) + raise ProxyError( + HTTPStatus.BAD_GATEWAY, + f"{self.provider} upstream network error: {getattr(exc, 'reason', exc)}", + retries=self.retries, + ) from exc + else: + self.retries += retries try: with response as resp: @@ -1292,21 +1501,45 @@ def run_attempt(self, req: urllib.request.Request) -> str: self.handle_chunk(chunk) except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError, OSError, http.client.HTTPException) as exc: trace( - "zen.stream_aborted", + f"{self.provider}.stream_aborted", request_id=self.request_id, status=exc.code if isinstance(exc, urllib.error.HTTPError) else getattr(exc, "reason", str(exc)), ) - self.send_error("zen upstream stream aborted") - _meter_zen(self.model, self.started, 502, stream_aborted=True, retries=self.retries or None) + self.send_error(f"{self.provider} upstream stream aborted") + _meter_provider( + self.model, + self.started, + 502, + provider=self.provider, + stream_aborted=True, + retries=self.retries or None, + ) return "error" if not self.client_alive: - trace("zen.client_gone", request_id=self.request_id, message="client disconnected before final events") - _meter_zen(self.model, self.started, 0, stream_aborted=True, retries=self.retries or None) + trace( + f"{self.provider}.client_gone", + request_id=self.request_id, + message="client disconnected before final events", + ) + _meter_provider( + self.model, + self.started, + 0, + provider=self.provider, + stream_aborted=True, + retries=self.retries or None, + ) return "gone" if not self.got_data: - self.send_error("zen upstream returned no SSE data") - _meter_zen(self.model, self.started, 502, retries=self.retries or None) + self.send_error(f"{self.provider} upstream returned no SSE data") + _meter_provider( + self.model, + self.started, + 502, + provider=self.provider, + retries=self.retries or None, + ) return "nodata" if not self.text and not self.tool_calls and not self.reasoning: return "empty" @@ -1378,18 +1611,26 @@ def finalize(self) -> None: self._write(b"data: [DONE]\n\n") if self.family == "openai_chat": record_cache(self.config.cache_tracker, self.bare_id, self.usage) - _meter_zen( - self.model, self.started, 200, + _meter_provider( + self.model, + self.started, + 200, + provider=self.provider, input_tokens=inp, output_tokens=outp, total_tokens=total, estimated_input_tokens=estimated, retries=self.retries or None, ) trace( - "zen.done", request_id=self.request_id, family=self.family, stream=True, + f"{self.provider}.done", request_id=self.request_id, family=self.family, stream=True, output_items=len(output), output_text_len=len(self.text), ) - def run(self, req: urllib.request.Request) -> None: + def run( + self, + req: urllib.request.Request, + initial_response: Any | None = None, + initial_retries: int = 0, + ) -> None: self.send_event( { "type": "response.created", @@ -1406,8 +1647,12 @@ def run(self, req: urllib.request.Request) -> None: } ) empty_attempts = 0 + response = initial_response + retries = initial_retries while True: - outcome = self.run_attempt(req) + outcome = self.run_attempt(req, response, retries) + response = None + retries = 0 if outcome in ("error", "gone", "nodata"): return if outcome == "empty": @@ -1416,27 +1661,12 @@ def run(self, req: urllib.request.Request) -> None: # A streamed 200 with no output is retried once with the # identical request; response and item ids are reused. self._reset_accumulation() - try: - _response, more = _open_upstream_stream(req, self.config, self.request_id, default_max_retries()) - self.retries += more - except _ConnectFailed as fail: - self.retries += fail.attempts - exc = fail.exc - if isinstance(exc, urllib.error.HTTPError): - _relay_upstream_error(self.handler, exc.code, fail.body, exc.headers) - _meter_zen(self.model, self.started, exc.code, retries=self.retries or None) - return - if isinstance(exc, TimeoutError): - _meter_zen(self.model, self.started, int(HTTPStatus.GATEWAY_TIMEOUT), retries=self.retries or None) - raise ProxyError(HTTPStatus.GATEWAY_TIMEOUT, "zen upstream timeout", retries=self.retries) from exc - _meter_zen(self.model, self.started, int(HTTPStatus.BAD_GATEWAY), retries=self.retries or None) - raise ProxyError( - HTTPStatus.BAD_GATEWAY, - f"zen upstream network error: {getattr(exc, 'reason', exc)}", - retries=self.retries, - ) from exc + self.retries += 1 continue - self.send_error("zen upstream returned an empty completion", code="empty_completion") + self.send_error( + f"{self.provider} upstream returned an empty completion", + code="empty_completion", + ) inp, outp, total = self._tokens() if isinstance(inp, int) and inp > 0: note_real_input_tokens(self.model) @@ -1444,8 +1674,11 @@ def run(self, req: urllib.request.Request) -> None: estimated = estimate_input_tokens(self.model, len(self.raw_payload), self.usage, context_window=context_cap) # The client-visible outcome is response.error empty_completion, # so the meter records 502 (a failed turn), never a success. - _meter_zen( - self.model, self.started, 502, + _meter_provider( + self.model, + self.started, + 502, + provider=self.provider, input_tokens=inp, output_tokens=outp, total_tokens=total, estimated_input_tokens=estimated, empty_completion=True, retries=self.retries or None, @@ -1454,34 +1687,79 @@ def run(self, req: urllib.request.Request) -> None: return -def handle_zen_responses_request(handler: Any, payload: Json, config: ProxyConfig, request_id: str) -> None: - """Serve a /v1/responses request for a zen-routed model (stream and non-stream). +def _restore_responses_sse_line(line: bytes) -> bytes: + prefix = b"data:" + if not line.startswith(prefix): + return line + raw = line[len(prefix):].strip() + if not raw or raw == b"[DONE]": + return line + try: + event = json.loads(raw) + except json.JSONDecodeError: + return line + restored = restore_namespaced_function_calls(event) + return b"data: " + json.dumps(restored, separators=(",", ":")).encode() + b"\n" - The zen path owns the whole HTTP response: connect-first, so an upstream - non-200 is answered with the upstream's own status and body before any SSE - is committed. Streaming is per-family translated back to Responses events; - the openai_responses family is relayed verbatim. Metering is provider="zen" - on every outcome (success, upstream error, network error, client gone). - """ + +def handle_family_responses_request( + handler: Any, + payload: Json, + config: ProxyConfig, + request_id: str, + *, + model: str, + bare_id: str, + family: str, + api_key: str, + base_url: str, + provider: str, + extra_headers: Mapping[str, str] | None = None, + function_tools_only: bool = False, + restore_namespaces: bool = False, + caption_images: bool = False, +) -> None: + """Serve a Responses request through one certified upstream family.""" started = time.time() - model = payload.get("model") or DEFAULT_MODEL - ensure_zen_slug(model) - bare_id = bare_zen_id(model) - family = zen_family_for(bare_id) if payload.get("stream") is not True: - # call_zen_responses meters provider="zen" on success and on every - # ProxyError path, then re-raises for the dispatcher to render. - _send_json(handler, call_zen_responses(payload, config, request_id)) + _send_json( + handler, + call_family_responses( + payload, + config, + request_id, + model=model, + bare_id=bare_id, + family=family, + api_key=api_key, + base_url=base_url, + provider=provider, + extra_headers=extra_headers, + function_tools_only=function_tools_only, + restore_namespaces=restore_namespaces, + caption_images=caption_images, + ), + ) return - api_key = resolve_api_key(config, request_id) - url, body, headers = _build_zen_request( - payload, family, bare_id, api_key, stream=True, session_model=model + url, body, headers = _build_family_request( + payload, + family, + bare_id, + api_key, + stream=True, + session_model=model, + base_url=base_url, + extra_headers=extra_headers, + function_tools_only=function_tools_only, + caption_images=caption_images, + config=config, + request_id=request_id, ) raw_payload = json.dumps(body, separators=(",", ":")).encode("utf-8") req = urllib.request.Request(url, data=raw_payload, headers=headers, method="POST") - trace("zen.start", request_id=request_id, url=url, bytes=len(raw_payload), family=family, stream=True) + trace(f"{provider}.start", request_id=request_id, url=url, bytes=len(raw_payload), family=family, stream=True) try: response, retries = _open_upstream_stream(req, config, request_id, default_max_retries()) except _ConnectFailed as fail: @@ -1489,16 +1767,27 @@ def handle_zen_responses_request(handler: Any, payload: Json, config: ProxyConfi exc = fail.exc if isinstance(exc, urllib.error.HTTPError): _relay_upstream_error(handler, exc.code, fail.body, exc.headers) - _meter_zen(model, started, exc.code, retries=retries or None) + _meter_provider( + model, started, exc.code, provider=provider, retries=retries or None + ) return - _meter_zen( + _meter_provider( model, started, int(HTTPStatus.GATEWAY_TIMEOUT if isinstance(exc, TimeoutError) else HTTPStatus.BAD_GATEWAY), + provider=provider, retries=retries or None, ) if isinstance(exc, TimeoutError): - raise ProxyError(HTTPStatus.GATEWAY_TIMEOUT, "zen upstream timeout", retries=retries) from exc - raise ProxyError(HTTPStatus.BAD_GATEWAY, f"zen upstream network error: {getattr(exc, 'reason', exc)}", retries=retries) from exc + raise ProxyError( + HTTPStatus.GATEWAY_TIMEOUT, + f"{provider} upstream timeout", + retries=retries, + ) from exc + raise ProxyError( + HTTPStatus.BAD_GATEWAY, + f"{provider} upstream network error: {getattr(exc, 'reason', exc)}", + retries=retries, + ) from exc handler.send_response(HTTPStatus.OK) handler.send_header("content-type", "text/event-stream") @@ -1506,28 +1795,69 @@ def handle_zen_responses_request(handler: Any, payload: Json, config: ProxyConfi handler.end_headers() if family == "openai_responses": - outcome = _relay_stream(response, handler, request_id) + outcome = _relay_stream( + response, + handler, + request_id, + _restore_responses_sse_line if restore_namespaces else None, + ) if outcome == "done": - _meter_zen(model, started, 200, retries=retries or None) + _meter_provider( + model, started, 200, provider=provider, retries=retries or None + ) elif outcome == "gone": - _meter_zen(model, started, 0, stream_aborted=True, retries=retries or None) + _meter_provider( + model, + started, + 0, + provider=provider, + stream_aborted=True, + retries=retries or None, + ) else: - _meter_zen(model, started, 502, stream_aborted=True, retries=retries or None) - trace("zen.done", request_id=request_id, family=family, stream=True, outcome=outcome) + _meter_provider( + model, + started, + 502, + provider=provider, + stream_aborted=True, + retries=retries or None, + ) + trace(f"{provider}.done", request_id=request_id, family=family, stream=True, outcome=outcome) return engine = _ZenStreamEngine( handler, payload, config, request_id, family=family, bare_id=bare_id, model=model, response_id=new_response_id(), started=started, raw_payload=raw_payload, + provider=provider, ) engine._start_keepalive() try: - engine.run(req) + engine.run(req, response, retries) finally: engine._stop_keepalive() +def handle_zen_responses_request(handler: Any, payload: Json, config: ProxyConfig, request_id: str) -> None: + """Serve a /v1/responses request for a zen-routed model.""" + model = payload.get("model") or DEFAULT_MODEL + ensure_zen_slug(model) + bare_id = bare_zen_id(model) + handle_family_responses_request( + handler, + payload, + config, + request_id, + model=model, + bare_id=bare_id, + family=zen_family_for(bare_id), + api_key=resolve_api_key(config, request_id), + base_url=zen_base_url(), + provider=ZEN_PROVIDER, + ) + + def handle_zen_chat_request(handler: Any, payload: Json, config: ProxyConfig, request_id: str) -> None: """Serve a /chat/completions request for a zen-routed model. diff --git a/tests/test_catalog.py b/tests/test_catalog.py index 7145088..a8ac20e 100644 --- a/tests/test_catalog.py +++ b/tests/test_catalog.py @@ -46,43 +46,42 @@ def _request_header(request, name: str) -> str | None: class DiscoverModelsTests(unittest.TestCase): def test_discover_models_returns_only_opencode_entries(self) -> None: payload = { - "opencode": { - "models": { - "gpt-5.5": {"id": "gpt-5.5", "name": "GPT-5.5", "reasoning": True}, - "deepseek-v4-flash": { - "id": "deepseek-v4-flash", - "name": "DeepSeek V4 Flash", - "limit": {"context": 400000}, - }, - } - }, - "other-provider": { - "models": { - "some-model": {"id": "some-model", "name": "Some Model"}, - } - }, + "data": [ + {"id": "grok-4.6", "name": "Grok 4.6"}, + {"id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash"}, + {"id": "some-model", "name": "Some Model"}, + ] } seen: dict = {} def capture(request, *args, **kwargs): seen["url"] = request.full_url seen["ua"] = request.headers.get("User-agent") + seen["authorization"] = request.headers.get("Authorization") seen["if_none_match"] = _request_header(request, "If-None-Match") - return FakeResponse(payload, headers={"ETag": 'W/"models-dev-v42"'}) + return FakeResponse(payload, headers={"ETag": 'W/"go-v42"'}) - with mock.patch("opencode_go_proxy.catalog.urllib.request.urlopen", side_effect=capture): + with mock.patch( + "opencode_go_proxy.catalog.resolve_api_key", + return_value="catalog-key", + ), mock.patch( + "opencode_go_proxy.catalog.urllib.request.urlopen", + side_effect=capture, + ): models, etag = discover_models() - self.assertEqual([m["id"] for m in models], ["gpt-5.5", "deepseek-v4-flash"]) + self.assertEqual( + [m["id"] for m in models], + ["grok-4.6", "deepseek-v4-flash"], + ) self.assertNotIn("some-model", {m["id"] for m in models}) - # models.dev rejects the default urllib UA with 403; the discovery - # fetch must carry an identifying UA. - self.assertEqual(seen["url"], "https://models.dev/api.json") + self.assertEqual(seen["url"], "https://opencode.ai/zen/go/v1/models") self.assertTrue(seen["ua"].startswith("opencode-go-proxy/")) + self.assertEqual(seen["authorization"], "Bearer catalog-key") # No stored etag on a first fetch: no conditional header is sent, and # the response ETag flows back for the caller to store. self.assertIsNone(seen["if_none_match"]) - self.assertEqual(etag, 'W/"models-dev-v42"') + self.assertEqual(etag, 'W/"go-v42"') def test_discover_models_sends_if_none_match_and_raises_on_304(self) -> None: seen: dict = {} diff --git a/tests/test_catalog_refresh.py b/tests/test_catalog_refresh.py index 2f77e41..c0351fa 100644 --- a/tests/test_catalog_refresh.py +++ b/tests/test_catalog_refresh.py @@ -45,7 +45,7 @@ def make_compact(fetched_at: str) -> dict: "client_version": "0.147.0", "models": [ { - "slug": "existing-model", + "slug": "deepseek-v4-flash", "display_name": "Existing Model", "description": "Already known.", "default_reasoning_level": "medium", @@ -85,6 +85,14 @@ def setUp(self) -> None: self.compact_path = os.path.join(self.tmp, "models.json") self.catalog_path = os.path.join(self.tmp, "catalog.json") self.now = datetime.datetime(2026, 8, 10, 12, 0, 0, tzinfo=datetime.UTC) + self.env = mock.patch.dict( + os.environ, + {"OPENCODE_GO_API_KEY": "catalog-test-key"}, + ) + self.env.start() + + def tearDown(self) -> None: + self.env.stop() def _write_compact(self, fetched_at: str) -> None: with open(self.compact_path, "w") as f: @@ -103,7 +111,7 @@ def fail_if_called(*args, **kwargs): now=self.now, ) - self.assertEqual(rendered["models"][0]["slug"], "existing-model") + self.assertEqual(rendered["models"][0]["slug"], "deepseek-v4-flash") with open(self.compact_path) as f: compact = json.load(f) self.assertNotEqual(compact["fetched_at"], self.now.isoformat()) @@ -113,7 +121,7 @@ def test_stale_compact_refreshes_and_adds_model(self) -> None: self._write_compact((self.now - datetime.timedelta(hours=25)).isoformat()) discovered = [ { - "id": "new-model", + "id": "glm-5.3", "name": "New Model", "description": "Freshly discovered.", "limit": {"context": 200000}, @@ -128,11 +136,14 @@ def test_stale_compact_refreshes_and_adds_model(self) -> None: now=self.now, ) - self.assertEqual([m["slug"] for m in rendered["models"]], ["existing-model", "new-model"]) + self.assertEqual( + [m["slug"] for m in rendered["models"]], + ["deepseek-v4-flash", "glm-5.3"], + ) with open(self.compact_path) as f: compact = json.load(f) self.assertEqual(compact["fetched_at"], self.now.isoformat()) - self.assertEqual(compact["models"][1]["slug"], "new-model") + self.assertEqual(compact["models"][1]["slug"], "glm-5.3") def test_force_refreshes_fresh_compact(self) -> None: self._write_compact((self.now - datetime.timedelta(hours=1)).isoformat()) @@ -148,7 +159,7 @@ def test_force_refreshes_fresh_compact(self) -> None: with open(self.compact_path) as f: compact = json.load(f) self.assertEqual(compact["fetched_at"], self.now.isoformat()) - self.assertEqual(rendered["models"][0]["slug"], "existing-model") + self.assertEqual(rendered["models"][0]["slug"], "deepseek-v4-flash") def test_refresh_disabled_by_env(self) -> None: self._write_compact((self.now - datetime.timedelta(hours=25)).isoformat()) @@ -165,7 +176,7 @@ def fail_if_called(*args, **kwargs): now=self.now, ) - self.assertEqual(rendered["models"][0]["slug"], "existing-model") + self.assertEqual(rendered["models"][0]["slug"], "deepseek-v4-flash") with open(self.compact_path) as f: compact = json.load(f) self.assertNotEqual(compact["fetched_at"], self.now.isoformat()) @@ -183,7 +194,7 @@ def test_offline_with_existing_compact_uses_fallback(self) -> None: now=self.now, ) - self.assertEqual(rendered["models"][0]["slug"], "existing-model") + self.assertEqual(rendered["models"][0]["slug"], "deepseek-v4-flash") with open(self.compact_path) as f: compact = json.load(f) self.assertNotEqual(compact["fetched_at"], self.now.isoformat()) @@ -205,7 +216,7 @@ def test_offline_without_compact_renders_seed(self) -> None: seed_path=seed_path, now=self.now, ) - self.assertEqual(rendered["models"][0]["slug"], "existing-model") + self.assertEqual(rendered["models"][0]["slug"], "deepseek-v4-flash") self.assertTrue(os.path.exists(self.catalog_path)) def test_etag_is_stored_and_sent_as_if_none_match_on_next_refresh(self) -> None: @@ -217,7 +228,7 @@ def fake_urlopen(request, *args, **kwargs): if state["respond_304"]: raise urllib.error.HTTPError(request.full_url, 304, "Not Modified", {}, None) return _FakeResponse( - {"opencode": {"models": {"existing-model": {"id": "existing-model", "name": "Existing Model"}}}}, + {"data": [{"id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash"}]}, headers={"ETag": 'W/"models-dev-v7"'}, ) @@ -255,7 +266,7 @@ def fake_urlopen(request, *args, **kwargs): if state["respond_304"]: raise urllib.error.HTTPError(request.full_url, 304, "Not Modified", {}, None) return _FakeResponse( - {"opencode": {"models": {"existing-model": {"id": "existing-model", "name": "Existing Model"}}}}, + {"data": [{"id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash"}]}, headers={"ETag": 'W/"models-dev-v9"'}, ) @@ -302,7 +313,7 @@ def test_etag_stable_across_refreshes_with_identical_content(self) -> None: def fake_urlopen(request, *args, **kwargs): return _FakeResponse( - {"opencode": {"models": {"existing-model": {"id": "existing-model", "name": "Existing Model"}}}} + {"data": [{"id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash"}]} ) with mock.patch("opencode_go_proxy.catalog.urllib.request.urlopen", side_effect=fake_urlopen): @@ -333,7 +344,7 @@ def test_force_refresh_skips_conditional_get(self) -> None: def fake_urlopen(request, *args, **kwargs): seen["if_none_match"] = _request_header(request, "If-None-Match") - return _FakeResponse({"opencode": {"models": {}}}) + return _FakeResponse({"data": []}) with mock.patch("opencode_go_proxy.catalog.urllib.request.urlopen", side_effect=fake_urlopen): refresh_catalog( diff --git a/tests/test_go_compatibility.py b/tests/test_go_compatibility.py new file mode 100644 index 0000000..fa9cbe4 --- /dev/null +++ b/tests/test_go_compatibility.py @@ -0,0 +1,304 @@ +import io +import json +from email.message import Message +from unittest import mock + +import pytest + +from opencode_go_proxy import catalog, go_upstream, zen_upstream +from opencode_go_proxy.config import ProxyConfig +from opencode_go_proxy.errors import ProxyError +from opencode_go_proxy.go_models import ( + ANTHROPIC_MESSAGES, + DOCUMENTED_CHAT_MODELS, + DOCUMENTED_GO_MODELS, + DOCUMENTED_MESSAGES_MODELS, + DOCUMENTED_RESPONSES_MODELS, + OPENAI_CHAT, + OPENAI_RESPONSES, + go_family_for, +) +from opencode_go_proxy.opencode_session import resolve_opencode_session +from opencode_go_proxy.protocol import restore_namespaced_function_calls + + +def config() -> ProxyConfig: + return ProxyConfig( + bind="127.0.0.1", + port=8787, + chat_base_url="https://opencode.ai/zen/go/v1", + api_key_env="OPENCODE_GO_API_KEY", + timeout_sec=10, + max_body_bytes=1024 * 1024, + ) + + +class Handler: + def __init__(self, headers: dict[str, str] | None = None) -> None: + self.headers = Message() + for name, value in (headers or {}).items(): + self.headers[name] = value + self.wfile = io.BytesIO() + self.status: int | None = None + self.response_headers: list[tuple[str, str]] = [] + + def send_response(self, status: int) -> None: + self.status = status + + def send_header(self, name: str, value: str) -> None: + self.response_headers.append((name, value)) + + def end_headers(self) -> None: + pass + + +def request_body(request) -> dict: + return json.loads(request.data) + + +def request_headers(request) -> dict[str, str]: + return {name.lower(): value for name, value in request.header_items()} + + +def response(value: dict) -> mock.Mock: + raw = json.dumps(value).encode() + return mock.Mock( + status=200, + headers={}, + read=lambda: raw, + __enter__=lambda current: current, + __exit__=lambda *args: False, + ) + + +def test_documented_catalog_has_all_28_models_in_their_official_families() -> None: + assert len(DOCUMENTED_GO_MODELS) == 28 + assert len(DOCUMENTED_RESPONSES_MODELS) == 4 + assert len(DOCUMENTED_CHAT_MODELS) == 16 + assert len(DOCUMENTED_MESSAGES_MODELS) == 8 + assert {go_family_for(model) for model in DOCUMENTED_RESPONSES_MODELS} == { + OPENAI_RESPONSES + } + assert {go_family_for(model) for model in DOCUMENTED_CHAT_MODELS} == { + OPENAI_CHAT + } + assert {go_family_for(model) for model in DOCUMENTED_MESSAGES_MODELS} == { + ANTHROPIC_MESSAGES + } + seed = catalog.load_seed_compact() + assert seed is not None + assert DOCUMENTED_GO_MODELS <= { + str(model["slug"]) for model in seed["models"] + } + + +def test_explicit_and_safe_bare_selection_reject_unknown_ids() -> None: + assert go_upstream.go_request_identity("opencode-go/grok-4.6") == ( + "grok-4.6", + OPENAI_RESPONSES, + ) + assert go_upstream.go_request_identity("glm-5.3") == ( + "glm-5.3", + OPENAI_CHAT, + ) + with pytest.raises(ProxyError) as unknown: + go_upstream.go_request_identity("opencode-go/not-a-go-model") + assert unknown.value.status == 400 + with pytest.raises(ProxyError): + go_upstream.go_request_identity("opencode-go/") + + +@pytest.mark.parametrize( + ("model", "path", "auth"), + [ + ("opencode-go/grok-4.6", "/responses", "authorization"), + ("glm-5.3-flash", "/chat/completions", "authorization"), + ("opencode-go/minimax-m3", "/messages", "x-api-key"), + ], +) +def test_responses_adapter_selects_documented_endpoint_and_auth( + model: str, + path: str, + auth: str, +) -> None: + bare_id, family = go_upstream.go_request_identity(model) + url, body, headers = zen_upstream._build_family_request( + {"model": model, "input": "hello"}, + family, + bare_id, + "go-key", + stream=False, + session_model=model, + base_url=config().chat_base_url, + extra_headers={"x-opencode-session": "session-1"}, + function_tools_only=True, + ) + assert url.endswith(path) + assert body["model"] == bare_id + assert headers[auth] == ( + "Bearer go-key" if auth == "authorization" else "go-key" + ) + assert headers["x-opencode-session"] == "session-1" + assert headers["user-agent"].startswith("opencode-go-proxy/") + assert ("authorization" in headers) is (auth == "authorization") + assert ("x-api-key" in headers) is (auth == "x-api-key") + + +def test_responses_models_receive_only_ordinary_function_tools() -> None: + payload = { + "model": "opencode-go/muse-spark-1.3-contributor", + "input": "edit", + "tools": [ + { + "type": "namespace", + "name": "repo", + "tools": [ + { + "type": "function", + "name": "read", + "description": "Read a file", + "parameters": { + "type": "object", + "properties": {"path": {"type": "string"}}, + }, + } + ], + }, + { + "type": "custom", + "name": "apply_patch", + "description": "Apply a patch", + }, + ], + "tool_choice": { + "type": "function", + "namespace": "repo", + "name": "read", + }, + } + url, body, _headers = zen_upstream._build_family_request( + payload, + OPENAI_RESPONSES, + "muse-spark-1.3-contributor", + "go-key", + stream=False, + session_model=payload["model"], + base_url=config().chat_base_url, + function_tools_only=True, + ) + assert url.endswith("/responses") + tool_names = [tool["name"] for tool in body["tools"]] + assert "repo__read" in tool_names + assert "apply_patch" in tool_names + assert {tool["type"] for tool in body["tools"]} == {"function"} + assert body["tool_choice"] == {"type": "function", "name": "repo__read"} + + +def test_namespaced_calls_are_restored_for_nonstream_and_stream_events() -> None: + value = { + "output": [ + { + "type": "function_call", + "name": "repo__read", + "arguments": "{}", + } + ] + } + restored = restore_namespaced_function_calls(value) + assert restored["output"][0]["namespace"] == "repo" + assert restored["output"][0]["name"] == "read" + line = b'data: {"type":"response.output_item.done","item":{"type":"function_call","name":"repo__read"}}\n' + transformed = zen_upstream._restore_responses_sse_line(line) + event = json.loads(transformed.removeprefix(b"data: ")) + assert event["item"]["namespace"] == "repo" + assert event["item"]["name"] == "read" + + +def test_session_identity_forwards_aliases_and_derives_stably() -> None: + payload = {"input": [{"role": "user", "content": "same conversation"}]} + assert ( + resolve_opencode_session({"x-opencode-session": "official"}, payload) + == "official" + ) + assert resolve_opencode_session({"thread-id": "thread-7"}, payload) == "thread-7" + derived = resolve_opencode_session({}, payload) + assert derived == resolve_opencode_session({}, payload) + assert derived != resolve_opencode_session( + {}, + {"input": [{"role": "user", "content": "different conversation"}]}, + ) + + +def test_go_chat_and_messages_verbatim_paths_keep_credentials_separate() -> None: + captured = [] + + def fake_urlopen(request, **kwargs): + captured.append(request) + if request.full_url.endswith("/messages"): + return response({"id": "msg_1", "content": []}) + return response({"id": "chatcmpl_1", "choices": []}) + + with ( + mock.patch( + "opencode_go_proxy.go_upstream.resolve_api_key", + return_value="go-key", + ), + mock.patch("urllib.request.urlopen", side_effect=fake_urlopen), + ): + chat_handler = Handler({"x-opencode-session": "chat-session"}) + go_upstream.handle_go_chat_request( + chat_handler, + { + "model": "opencode-go/glm-5.3", + "messages": [{"role": "user", "content": "hi"}], + }, + config(), + "chat-request", + ) + messages_handler = Handler({"x-opencode-session": "messages-session"}) + go_upstream.handle_go_messages_request( + messages_handler, + { + "model": "opencode-go/minimax-m3", + "messages": [{"role": "user", "content": "hi"}], + }, + config(), + "messages-request", + ) + + chat_request, messages_request = captured + assert chat_request.full_url.endswith("/chat/completions") + assert request_body(chat_request)["model"] == "glm-5.3" + assert request_headers(chat_request)["authorization"] == "Bearer go-key" + assert request_headers(chat_request)["x-opencode-session"] == "chat-session" + assert "x-api-key" not in request_headers(chat_request) + assert messages_request.full_url.endswith("/messages") + assert request_body(messages_request)["model"] == "minimax-m3" + assert request_headers(messages_request)["x-api-key"] == "go-key" + assert request_headers(messages_request)["x-opencode-session"] == ( + "messages-session" + ) + assert "authorization" not in request_headers(messages_request) + + +def test_merged_catalog_uses_explicit_go_slugs_for_native_collisions( + monkeypatch: pytest.MonkeyPatch, + tmp_path, +) -> None: + monkeypatch.setenv("OPENCODE_GO_PROXY_STATE_DIR", str(tmp_path)) + monkeypatch.setattr( + "opencode_go_proxy.native_models.load_native_capture", + lambda: { + "models": [ + { + "slug": "gpt-5.6-luna", + "display_name": "GPT-5.6 Luna", + } + ] + }, + ) + merged = catalog.render_merged_catalog() + slugs = {model["slug"] for model in merged["models"]} + assert "gpt-5.6-luna" in slugs + assert "opencode-go/gpt-5.6-luna" in slugs + assert "opencode-go/muse-spark-1.3-contributor" in slugs diff --git a/tests/test_go_reject_fallback.py b/tests/test_go_reject_fallback.py index 443d00b..e95cdfd 100644 --- a/tests/test_go_reject_fallback.py +++ b/tests/test_go_reject_fallback.py @@ -26,7 +26,6 @@ from opencode_go_proxy.app import ( ProxyConfig, _go_reject_zen_fallback, - handle_chat_completions_request, handle_responses_request, ) from opencode_go_proxy.errors import ProxyError @@ -362,111 +361,6 @@ def chat_payload(model: str = ZEN_SLUG) -> dict: } -class TestChatFallback: - def test_go_reject_falls_back_to_zen_with_identical_messages(self) -> None: - _seed_collision() - payload = chat_payload() - config = make_config() - handler = _FakeHandler() - - with mock.patch.dict(os.environ, {"OPENCODE_GO_API_KEY": "test-key"}), mock.patch( - "opencode_go_proxy.app.call_upstream_chat_verbatim", - return_value=(401, GO_REJECT_BODY.encode(), 0, "application/json", None), - ), mock.patch("opencode_go_proxy.app.handle_zen_chat_request") as zen: - handle_chat_completions_request(handler, payload, config, "req") - - zen.assert_called_once() - called_handler, zen_payload, called_config, called_request_id = zen.call_args.args - assert called_handler is handler - assert called_config is config - assert called_request_id == "req" - # The zen chat handler's API takes the zen/ slug; the wire request it - # sends strips it back to the bare id with identical messages/stream. - assert zen_payload["model"] == f"zen/{ZEN_SLUG}" - assert zen_payload["messages"] == payload["messages"] - assert zen_payload["stream"] is payload["stream"] - - def test_go_401_invalid_key_relayed_verbatim_no_fallback(self) -> None: - _seed_collision() - handler = _FakeHandler() - - with mock.patch.dict(os.environ, {"OPENCODE_GO_API_KEY": "test-key"}), mock.patch( - "opencode_go_proxy.app.call_upstream_chat_verbatim", - return_value=(401, INVALID_KEY_BODY.encode(), 0, "application/json", None), - ), mock.patch("opencode_go_proxy.app.handle_zen_chat_request") as zen: - handle_chat_completions_request(handler, chat_payload(), make_config(), "req") - - zen.assert_not_called() - assert handler.status == 401 - assert handler.wfile.getvalue() == INVALID_KEY_BODY.encode() - - def test_go_200_relayed_no_fallback(self) -> None: - _seed_collision() - ok_body = b'{"choices": [], "usage": {}}' - handler = _FakeHandler() - - with mock.patch.dict(os.environ, {"OPENCODE_GO_API_KEY": "test-key"}), mock.patch( - "opencode_go_proxy.app.call_upstream_chat_verbatim", - return_value=(200, ok_body, 0, "application/json", None), - ), mock.patch("opencode_go_proxy.app.handle_zen_chat_request") as zen: - handle_chat_completions_request(handler, chat_payload(), make_config(), "req") - - zen.assert_not_called() - assert handler.status == 200 - assert handler.wfile.getvalue() == ok_body - - def test_bare_slug_not_in_zen_ids_no_fallback(self) -> None: - from opencode_go_proxy import zen_catalog as _zc - - _zc._ZEN_MODELS_CACHE = None - with open(_zc.zen_models_path(), "w") as handle: - json.dump({"fetched_at": "2026-08-14T00:00:00Z", "models": []}, handle) - handler = _FakeHandler() - - with mock.patch.dict(os.environ, {"OPENCODE_GO_API_KEY": "test-key"}), mock.patch( - "opencode_go_proxy.app.call_upstream_chat_verbatim", - return_value=(401, GO_REJECT_BODY.encode(), 0, "application/json", None), - ), mock.patch("opencode_go_proxy.app.handle_zen_chat_request") as zen: - handle_chat_completions_request(handler, chat_payload(), make_config(), "req") - - zen.assert_not_called() - assert handler.status == 401 - - def test_prefixed_opencode_go_slug_no_fallback(self) -> None: - _seed_collision() - handler = _FakeHandler() - - with mock.patch.dict(os.environ, {"OPENCODE_GO_API_KEY": "test-key"}), mock.patch( - "opencode_go_proxy.app.call_upstream_chat_verbatim", - return_value=(401, GO_REJECT_BODY.encode(), 0, "application/json", None), - ), mock.patch("opencode_go_proxy.app.handle_zen_chat_request") as zen: - handle_chat_completions_request(handler, chat_payload(f"opencode-go/{ZEN_SLUG}"), make_config(), "req") - - zen.assert_not_called() - assert handler.status == 401 - - def test_zen_attempt_error_propagates(self) -> None: - _seed_collision() - handler = _FakeHandler() - - def zen_fails(handler_arg, payload, config, request_id): - raise ProxyError( - HTTPStatus.BAD_GATEWAY, - "zen upstream network error: boom", - upstream_status=None, - error_type="ZenNetworkError", - ) - - with mock.patch.dict(os.environ, {"OPENCODE_GO_API_KEY": "test-key"}), mock.patch( - "opencode_go_proxy.app.call_upstream_chat_verbatim", - return_value=(401, GO_REJECT_BODY.encode(), 0, "application/json", None), - ), mock.patch("opencode_go_proxy.app.handle_zen_chat_request", side_effect=zen_fails), pytest.raises(ProxyError) as ctx: - handle_chat_completions_request(handler, chat_payload(), make_config(), "req") - - assert ctx.value.message == "zen upstream network error: boom" - assert ctx.value.error_type == "ZenNetworkError" - - class TestStreamingFallback: """Stream=true responses: the go-reject zen fallback relays the zen stream. diff --git a/tests/test_hide_dead_models.py b/tests/test_hide_dead_models.py index b4318a0..79e579b 100644 --- a/tests/test_hide_dead_models.py +++ b/tests/test_hide_dead_models.py @@ -26,7 +26,8 @@ ) from opencode_go_proxy.meter import state_dir -DEAD_SLUG = "north-mini-code-free" +DEAD_SLUG = "deepseek-v4-flash" +MERGED_DEAD_SLUG = f"opencode-go/{DEAD_SLUG}" # The shape seen live from the go gateway: ModelError with "not supported". GO_REJECT_BODY = json.dumps( @@ -87,7 +88,7 @@ def test_first_rejection_only_remembers(self) -> None: # No hide file was written and a render still lists the slug visible. assert catalog.read_hidden_models() == set() catalog.render_merged_catalog() - assert _merged_by_slug()[DEAD_SLUG].get("visibility") != "hide" + assert _merged_by_slug()[MERGED_DEAD_SLUG].get("visibility") != "hide" def test_second_rejection_hides_in_file_and_render(self) -> None: _seed_go_catalog() @@ -97,7 +98,7 @@ def test_second_rejection_hides_in_file_and_render(self) -> None: assert _unsupported_strikes == {DEAD_SLUG: 2} assert catalog.read_hidden_models() == {DEAD_SLUG} # The merged catalog was re-rendered once and now marks the slug hidden. - assert _merged_by_slug()[DEAD_SLUG]["visibility"] == "hide" + assert _merged_by_slug()[MERGED_DEAD_SLUG]["visibility"] == "hide" def test_prefixed_slug_rejected_never_hidden(self) -> None: _seed_go_catalog() diff --git a/tests/test_integration.py b/tests/test_integration.py index 4bd83ce..9dda21d 100644 --- a/tests/test_integration.py +++ b/tests/test_integration.py @@ -277,7 +277,7 @@ def test_streaming_sse_response(self, server): assert "hel" in raw_text assert "lo" in raw_text - def test_streaming_missing_api_key_sends_error_event(self, server): + def test_streaming_missing_api_key_fails_before_sse(self, server): port, _ = server from opencode_go_proxy.secrets import clear_api_key_cache clear_api_key_cache() @@ -293,15 +293,14 @@ def test_streaming_missing_api_key_sends_error_event(self, server): raw = resp.read() conn.close() - assert resp.status == 200 - raw_text = raw.decode("utf-8") - assert "response.error" in raw_text - assert "[DONE]" in raw_text + assert resp.status == 401 + body = json.loads(raw) + assert body["error"]["type"] == "proxy_error" - def test_streaming_crash_sends_sse_error(self, server): + def test_streaming_translation_crash_fails_before_sse(self, server): port, _ = server with mock.patch.dict(os.environ, {"OPENCODE_GO_API_KEY": "test-key"}), mock.patch( - "opencode_go_proxy.streaming.responses_payload_to_chat_payload", + "opencode_go_proxy.zen_upstream.responses_payload_to_chat_payload", side_effect=ValueError("boom"), ): conn = HTTPConnection("127.0.0.1", port, timeout=5) @@ -312,10 +311,9 @@ def test_streaming_crash_sends_sse_error(self, server): raw = resp.read() conn.close() - assert resp.status == 200 - raw_text = raw.decode("utf-8") - assert "response.error" in raw_text - assert "[DONE]" in raw_text + assert resp.status == 500 + body = json.loads(raw) + assert body["error"]["type"] == "proxy_crash" class TestEdgeCases: diff --git a/tests/test_native_catalog.py b/tests/test_native_catalog.py index 730c3a2..8b31b2f 100644 --- a/tests/test_native_catalog.py +++ b/tests/test_native_catalog.py @@ -162,7 +162,7 @@ def test_merged_catalog_contains_native_and_opencode_go_entries() -> None: slugs = {m["slug"] for m in merged["models"]} assert "gpt-5.6-luna" in slugs # native entry, captured slug assert "gpt-5.6-terra" in slugs - assert "deepseek-v4-flash" in slugs # opencode-go entry, bare slug + assert "opencode-go/deepseek-v4-flash" in slugs luna = next(m for m in merged["models"] if m["slug"] == "gpt-5.6-luna") assert luna["multi_agent_version"] == "v1" assert luna["context_window"] == 272000 @@ -176,7 +176,7 @@ def test_merged_catalog_without_capture_is_opencode_go_only() -> None: merged = catalog.render_merged_catalog() slugs = {m["slug"] for m in merged["models"]} assert "gpt-5.6-luna" not in slugs - assert "deepseek-v4-flash" in slugs + assert "opencode-go/deepseek-v4-flash" in slugs def test_effort_clamp_drops_unknown_efforts() -> None: @@ -189,8 +189,8 @@ def test_effort_clamp_drops_unknown_efforts() -> None: "client_version": "0.147.0", "models": [ { - "slug": "clamp-me", - "display_name": "Clamp Me", + "slug": "deepseek-v4-flash", + "display_name": "DeepSeek V4 Flash", "context_window": 100000, "supported_reasoning_levels": [ {"effort": "low", "description": "ok"}, @@ -204,7 +204,9 @@ def test_effort_clamp_drops_unknown_efforts() -> None: json.dump(compact, handle) _seed_native_capture() # effort vocabulary: low, medium, max, high, ultra merged = catalog.render_merged_catalog() - clamped = next(m for m in merged["models"] if m["slug"] == "clamp-me") + clamped = next( + m for m in merged["models"] if m["slug"] == "opencode-go/deepseek-v4-flash" + ) efforts = [level["effort"] for level in clamped["supported_reasoning_levels"]] assert efforts == ["low", "medium"] @@ -219,8 +221,8 @@ def test_effort_clamp_skipped_without_capture() -> None: "client_version": "0.147.0", "models": [ { - "slug": "clamp-me", - "display_name": "Clamp Me", + "slug": "deepseek-v4-flash", + "display_name": "DeepSeek V4 Flash", "supported_reasoning_levels": [{"effort": "ultra-super", "description": "x"}], } ], @@ -228,5 +230,7 @@ def test_effort_clamp_skipped_without_capture() -> None: with open(os.path.join(state, catalog.STATE_COMPACT_NAME), "w") as handle: json.dump(compact, handle) merged = catalog.render_merged_catalog() - entry = next(m for m in merged["models"] if m["slug"] == "clamp-me") + entry = next( + m for m in merged["models"] if m["slug"] == "opencode-go/deepseek-v4-flash" + ) assert [level["effort"] for level in entry["supported_reasoning_levels"]] == ["ultra-super"] diff --git a/tests/test_native_coexist.py b/tests/test_native_coexist.py index 5681827..03224c6 100644 --- a/tests/test_native_coexist.py +++ b/tests/test_native_coexist.py @@ -367,7 +367,7 @@ def fire(index: int, payload: dict) -> None: def test_models_list_serves_each_provider_without_collision(backend: str, proxy_server: int, offline_env) -> None: - """(e) /v1/models serves native bare, opencode-go bare, and zen prefixed; + """(e) /v1/models serves native bare, Go prefixed, and Zen prefixed; the same upstream id from go and zen stays two distinct ids.""" from opencode_go_proxy import catalog as proxy_catalog @@ -380,11 +380,10 @@ def test_models_list_serves_each_provider_without_collision(backend: str, proxy_ ids = [entry["id"] for entry in json.loads(resp.raw)["data"]] assert "gpt-5.6-terra" in ids # native, bare - assert "deepseek-v4-flash" in ids # opencode-go, bare + assert "opencode-go/deepseek-v4-flash" in ids assert "zen/deepseek-v4-flash" in ids # zen, prefixed - # The go bare slug and the zen-prefixed slug are distinct ids: the zen - # model never shadows (or duplicates) the go model under the bare slug. - assert ids.count("deepseek-v4-flash") == 1 + assert ids.count("opencode-go/deepseek-v4-flash") == 1 + assert ids.count("zen/deepseek-v4-flash") == 1 assert ids.count("zen/deepseek-v4-flash") == 1 assert len(ids) == len(set(ids)) diff --git a/tests/test_protocol_surface.py b/tests/test_protocol_surface.py index 8a74b9f..7b02300 100644 --- a/tests/test_protocol_surface.py +++ b/tests/test_protocol_surface.py @@ -1,4 +1,4 @@ -"""Plan 005 protocol surface: /chat/completions passthrough, /messages 400, WS 426.""" +"""HTTP protocol surfaces for Chat Completions, Messages, and WebSockets.""" import io import json @@ -123,7 +123,7 @@ def test_alias_path_without_v1_prefix(self, server): assert resp.status == 200 assert raw == upstream_body - def test_unknown_model_remains_verbatim_on_chat_surface(self, server): + def test_unknown_model_is_rejected_on_chat_surface(self, server): port, _ = server upstream_body = json.dumps({"choices": [{"message": {"content": "hi"}}]}).encode("utf-8") request_body = json.dumps({ @@ -137,10 +137,9 @@ def test_unknown_model_remains_verbatim_on_chat_surface(self, server): ) as mock_urlopen: resp, raw = post(port, "/v1/chat/completions", request_body) - assert resp.status == 200 - assert raw == upstream_body - sent_payload = json.loads(mock_urlopen.call_args[0][0].data) - assert sent_payload["model"] == "new-upstream-model" + assert resp.status == 400 + assert json.loads(raw)["error"]["type"] == "model_not_found" + mock_urlopen.assert_not_called() def test_upstream_429_status_and_body_relayed_verbatim(self, server): port, _ = server @@ -262,31 +261,45 @@ def test_oversized_body_rejected_413(self, server): class TestMessagesEndpoint: - EXPECTED: ClassVar[dict] = { - "error": { - "type": "invalid_request_error", - "message": ( - "This proxy serves a single OpenAI-compatible provider via " - "/v1/chat/completions and /v1/responses; /messages is not supported." - ), - } - } - - def test_post_v1_messages_returns_400(self, server): + def test_documented_messages_model_relays_to_go(self, server): + port, _ = server + upstream_body = b'{"id":"msg_1","content":[]}' + captured = [] + + def fake_urlopen(request, **kwargs): + captured.append(request) + return MockUpstreamResponse(upstream_body) + + with mock.patch( + "opencode_go_proxy.go_upstream.resolve_api_key", + return_value="test-key", + ), mock.patch("urllib.request.urlopen", side_effect=fake_urlopen): + resp, raw = post( + port, + "/v1/messages", + b'{"model":"opencode-go/minimax-m3","messages":[]}', + ) + + assert resp.status == 200 + assert raw == upstream_body + assert captured[0].full_url.endswith("/messages") + assert json.loads(captured[0].data)["model"] == "minimax-m3" + + def test_unknown_messages_model_returns_400(self, server): port, _ = server resp, raw = post(port, "/v1/messages", b'{"model":"claude-sonnet-4"}') assert resp.status == 400 - assert json.loads(raw) == self.EXPECTED + assert json.loads(raw)["error"]["type"] == "model_not_found" - def test_post_messages_alias_returns_400(self, server): + def test_chat_model_on_messages_alias_returns_400(self, server): port, _ = server resp, raw = post(port, "/messages", b"{}") assert resp.status == 400 - assert json.loads(raw) == self.EXPECTED + assert json.loads(raw)["error"]["type"] == "invalid_request_error" - def test_get_messages_returns_400(self, server): + def test_get_messages_returns_405(self, server): port, _ = server conn = HTTPConnection("127.0.0.1", port, timeout=5) conn.request("GET", "/v1/messages") @@ -294,8 +307,8 @@ def test_get_messages_returns_400(self, server): raw = resp.read() conn.close() - assert resp.status == 400 - assert json.loads(raw) == self.EXPECTED + assert resp.status == 405 + assert "error" in json.loads(raw) class TestWebSocketUpgradeRejection: @@ -360,6 +373,7 @@ def read(self): from opencode_go_proxy.meter import usage_events_path handler = mock.Mock(wfile=mock.Mock()) + handler.headers = {} handle_chat_completions_request(handler, {"model": "deepseek-v4-flash", "messages": [{"role": "user", "content": "hi"}]}, make_config(8790), "req") with open(usage_events_path()) as fh: events = [json.loads(line) for line in fh if line.strip()] diff --git a/tests/test_zen_catalog.py b/tests/test_zen_catalog.py index f779c2c..f283bdc 100644 --- a/tests/test_zen_catalog.py +++ b/tests/test_zen_catalog.py @@ -293,7 +293,7 @@ def test_merged_catalog_contains_zen_records(self) -> None: self.assertEqual(len(zen_entries), 6) slugs = {str(m["slug"]) for m in merged["models"]} self.assertIn("zen/claude-sonnet-4-5", slugs) - self.assertIn("deepseek-v4-flash", slugs) # go entry still present + self.assertIn("opencode-go/deepseek-v4-flash", slugs) record = next(m for m in zen_entries if m["slug"] == "zen/claude-sonnet-4-5") self.assertEqual(record["family"], "anthropic_messages") @@ -312,23 +312,6 @@ def test_zen_display_name_suffixed_when_go_record_shares_bare_id(self) -> None: record = next(m for m in merged["models"] if m["slug"] == "zen/deepseek-v4-flash") self.assertEqual(record["display_name"], "deepseek-v4-flash (Zen)") - def test_zen_display_name_suffixed_when_go_record_shares_display_name(self) -> None: - _seed_zen_cache(ZEN_PAYLOAD["data"]) - compact = { - "fetched_at": "2026-08-14T00:00:00Z", - "etag": "", - "client_version": "0.147.0", - "shared_instructions": "", - "models": [{"slug": "some-go-model", "display_name": "claude-sonnet-4-5"}], - } - with open(catalog.state_compact_path(), "w") as handle: - json.dump(compact, handle) - - merged = catalog.render_merged_catalog() - - record = next(m for m in merged["models"] if m["slug"] == "zen/claude-sonnet-4-5") - self.assertEqual(record["display_name"], "claude-sonnet-4-5 (Zen)") - def test_zen_only_record_keeps_plain_display_name(self) -> None: # claude-sonnet-4-5 has no opencode-go counterpart in the seed # catalog, so its name stays unambiguous. From 262c47725b3e81f0e5896591ad7c0725fdf9ff88 Mon Sep 17 00:00:00 2001 From: Kartik Kabadi <1kartikkabadi1@gmail.com> Date: Wed, 16 Sep 2026 04:32:08 -0700 Subject: [PATCH 4/4] Harden single-service protocol support Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- README.md | 50 +++++-- contrib/systemd/opencode-go-proxy.service | 3 +- macos/MenuBarApp/README.md | 10 +- .../OpenCodeGoMenuBar/ProxyController.swift | 6 +- src/opencode_go_proxy/go_upstream.py | 46 +++++-- tests/test_go_compatibility.py | 125 +++++++++++++++--- tests/test_ops.py | 32 +++++ tests/test_protocol_surface.py | 36 ++++- 8 files changed, 258 insertions(+), 50 deletions(-) diff --git a/README.md b/README.md index 596da76..703374d 100644 --- a/README.md +++ b/README.md @@ -109,6 +109,23 @@ or protocol-certified IDs, so an unknown ID is not guessed into the wrong protocol. The checked-in seed keeps startup and model selection working offline. +### HTTP API compatibility + +The one listener accepts all three documented client formats, with or without +the `/v1` prefix: + +| Client request | Supported Go models | Upstream behavior | +|----------------|---------------------|-------------------| +| `POST /v1/responses` | all 28 documented models | translated to each model's certified Responses, Chat Completions, or Messages family | +| `POST /v1/chat/completions` | the 16 Chat Completions models | relayed in Chat Completions format | +| `POST /v1/messages` | the 8 Anthropic Messages models | relayed in Messages format | + +Streaming and non-streaming requests are supported on every row. The Responses +aliases `/responses/compact` and `/v1/responses/compact` are also available for +compaction. A model sent to an incompatible direct-format endpoint is rejected +before any upstream request rather than silently translated to a different +client response shape. + ### Switching models ```bash @@ -431,7 +448,8 @@ The supported way to run the proxy on macOS is the menu bar app in `macos/MenuBarApp`. Build it in Xcode (or `swift build`), launch it, and it spawns the proxy itself and shows status, quota, and today's usage. There is no launchd agent anymore: `contrib/launchd/` was removed in 0.3.0, and the -`install` ops command points at the menu bar app instead of a plist. +`install` ops command points at the menu bar app instead of a plist. The app +is the only supervisor: do not run another proxy service beside it. ## Updating @@ -461,9 +479,22 @@ when an update is available). `--apply` re-installs from the new tag with `uv tool install --force`; when the proxy was not installed as a tool it prints the exact one-liner to run instead. -### Manual / systemd +### Linux (systemd user service) + +Install the checked-in user service, then start its single proxy process: + +```bash +mkdir -p ~/.config/systemd/user +curl -fsSL \ + https://raw.githubusercontent.com/kartikkabadi/opencode-go-proxy/v0.4.10/contrib/systemd/opencode-go-proxy.service \ + -o ~/.config/systemd/user/opencode-go-proxy.service +systemctl --user daemon-reload +systemctl --user enable --now opencode-go-proxy +curl http://127.0.0.1:8787/health +``` -Pin the install to the newest tag: +The unit owns the same one listener used on macOS; do not run a second manual +proxy while it is active. For a foreground/manual Linux run, use: ```bash uvx --from git+https://github.com/kartikkabadi/opencode-go-proxy@v0.4.10 \ @@ -472,7 +503,7 @@ uvx --from git+https://github.com/kartikkabadi/opencode-go-proxy@v0.4.10 \ Replace `v0.4.10` with the newest tag from [releases](https://github.com/kartikkabadi/opencode-go-proxy/releases). -For the systemd unit, update the `ExecStart` URL in +For updates, change the pinned `ExecStart` URL in `contrib/systemd/opencode-go-proxy.service` to the same pinned form, then `systemctl --user daemon-reload && systemctl --user restart opencode-go-proxy`. @@ -523,11 +554,12 @@ Use TLS and network-level access controls for any remote deployment. The caller token is only a proxy access capability; it is never used as or forwarded as a provider credential. Browser-originated requests remain blocked in remote mode. -**One HTTP port only.** The proxy binds a single listener: `OPENCODE_GO_PROXY_PORT` -(default `8787`). There is no admin port, control channel, or secondary service. If -something else already listens on the port, the proxy fails to bind — check with -`lsof -nP -iTCP:8787 -sTCP:LISTEN` before starting a second instance. The menu bar -app refuses Start when 8787 is already owned. +**One service and one HTTP port only.** The proxy binds a single listener: +`OPENCODE_GO_PROXY_PORT` (default `8787`). There is no admin port, control +channel, or secondary service. The macOS menu bar app or the Linux systemd user +unit supervises that one process. If something else already listens on the port, +the proxy fails to bind — check with `lsof -nP -iTCP:8787 -sTCP:LISTEN` before +starting another instance. The menu bar app refuses Start when 8787 is owned. **Short provider name.** The long "opencode go/" label in the Codex model picker comes from the provider config in `~/.codex/config.toml`, not from this proxy. Shorten it by diff --git a/contrib/systemd/opencode-go-proxy.service b/contrib/systemd/opencode-go-proxy.service index f0209ca..0065d73 100644 --- a/contrib/systemd/opencode-go-proxy.service +++ b/contrib/systemd/opencode-go-proxy.service @@ -10,8 +10,7 @@ Environment=HOME=%h Environment=PATH=%h/.local/bin:%h/.cargo/bin:%h/bin:/usr/local/bin:/usr/bin:/bin Environment=PYTHONUNBUFFERED=1 Environment=CODEX_MODEL_CATALOG=%h/.codex/opencode-go-proxy/opencode-go-catalog.json -Environment=OPENCODE_GO_PROXY_USER_AGENT=codex/1.0 -ExecStart=/usr/bin/env uvx --from git+https://github.com/kartikkabadi/opencode-go-proxy opencode-go-proxy --bind 127.0.0.1 --port 8787 --chat-base-url https://opencode.ai/zen/go/v1 +ExecStart=/usr/bin/env uvx --from git+https://github.com/kartikkabadi/opencode-go-proxy@v0.4.10 opencode-go-proxy --bind 127.0.0.1 --port 8787 --chat-base-url https://opencode.ai/zen/go/v1 Restart=on-failure RestartSec=3 diff --git a/macos/MenuBarApp/README.md b/macos/MenuBarApp/README.md index 060a990..2f4177d 100644 --- a/macos/MenuBarApp/README.md +++ b/macos/MenuBarApp/README.md @@ -13,7 +13,7 @@ The menu bar shows a small branch icon (no long "opencode go/" text). The menu s - Today's turns and tokens, plus a 7-day token bar list (from `GET /state`) - current model (the most recent meter event, else the proxy default) - Start/Stop Proxy: launches `uvx --from git+... opencode-go-proxy` as a child process, - writing logs to `~/.codex/logs/opencode-go-proxy.{log,err}` (same paths as the launchd plist) + writing logs to `~/.codex/logs/opencode-go-proxy.{log,err}` - Open Logs / Reveal Log File - Copy Port - Quit (stops the child proxy first) @@ -41,10 +41,10 @@ cp -R .build/release/OpenCodeGoMenuBar OpenCodeGoMenuBar.app/Contents/MacOS/ ## Notes - The Python bridge is untouched; the app only manages it as a child process. -- Single-port guard: the app refuses to Start if another process already listens on - 127.0.0.1:8787 (for example the launchd agent), instead of spawning a second proxy that - would fail to bind. One proxy per port. To switch from launchd to the menu bar, stop the - launchd agent first (`launchctl bootout gui/$(id -u) ~/Library/LaunchAgents/com.opencode-go.proxy.plist`). +- Single-service, single-port: the menu bar app is the macOS supervisor and owns one + child proxy on `127.0.0.1:8787`. It refuses to start when another process owns the + port and automatically unloads the recognized pre-0.3.0 launchd job during migration. + Do not run a separate launchd proxy alongside the app. - The spawned proxy resolves the API key exactly as the CLI does: `$OPENCODE_GO_API_KEY` first, then `$OPENCODE_API_KEY`, then the macOS keychain services `opencode-go-api-key` and `codex-router-opencode-go`. diff --git a/macos/MenuBarApp/Sources/OpenCodeGoMenuBar/ProxyController.swift b/macos/MenuBarApp/Sources/OpenCodeGoMenuBar/ProxyController.swift index 8f4058f..1aea43b 100644 --- a/macos/MenuBarApp/Sources/OpenCodeGoMenuBar/ProxyController.swift +++ b/macos/MenuBarApp/Sources/OpenCodeGoMenuBar/ProxyController.swift @@ -24,7 +24,7 @@ enum ProxySourceResolver { static let defaultsKey = "proxySource" /// The pinned source baked into this build (bumped at release time). - static let compiledSource = "git+https://github.com/kartikkabadi/opencode-go-proxy@v0.4.8" + static let compiledSource = "git+https://github.com/kartikkabadi/opencode-go-proxy@v0.4.10" /// A defaults override wins over the compiled pin; an empty or missing /// value falls back to `fallbackProxySource` (the compiled pin). @@ -516,8 +516,8 @@ final class ProxyController { } private func childEnvironment() -> [String] { - // Menu bar apps launch without the shell PATH; give the child the same - // PATH shape as the launchd plist plus a stable HOME. + // Menu bar apps launch without the shell PATH; give the child a stable + // user-tool PATH and HOME. let home = FileManager.default.homeDirectoryForCurrentUser.path let processInfo = ProcessInfo.processInfo var vars = processInfo.environment diff --git a/src/opencode_go_proxy/go_upstream.py b/src/opencode_go_proxy/go_upstream.py index e8feee3..ee6d6db 100644 --- a/src/opencode_go_proxy/go_upstream.py +++ b/src/opencode_go_proxy/go_upstream.py @@ -33,6 +33,30 @@ GO_PROVIDER = "opencode-go" +def _client_status(status: int) -> int: + if status >= HTTPStatus.INTERNAL_SERVER_ERROR: + return HTTPStatus.BAD_GATEWAY + return status + + +def _network_error(exc: BaseException, retries: int) -> ProxyError: + if isinstance(exc, TimeoutError) or ( + isinstance(exc, urllib.error.URLError) + and isinstance(exc.reason, TimeoutError) + ): + return ProxyError( + HTTPStatus.GATEWAY_TIMEOUT, + "opencode-go upstream timeout", + retries=retries, + ) + reason = exc.reason if isinstance(exc, urllib.error.URLError) else exc + return ProxyError( + HTTPStatus.BAD_GATEWAY, + f"opencode-go upstream network error: {reason}", + retries=retries, + ) + + def bare_go_id(slug: str) -> str: """Strip the explicit Go provider prefix at the wire boundary.""" return slug.removeprefix(GO_PREFIX) @@ -142,19 +166,16 @@ def _handle_go_verbatim_request( except _ConnectFailed as fail: exc = fail.exc if isinstance(exc, urllib.error.HTTPError): - _relay_upstream_error(handler, exc.code, fail.body, exc.headers) + status = _client_status(exc.code) + _relay_upstream_error(handler, status, fail.body, exc.headers) record_usage_event( model=model, - status=exc.code, + status=status, duration_ms=int((time.time() - started) * 1000), retries=fail.attempts or None, ) return - raise ProxyError( - HTTPStatus.BAD_GATEWAY, - f"opencode-go upstream network error: {getattr(exc, 'reason', exc)}", - retries=fail.attempts, - ) from exc + raise _network_error(exc, fail.attempts) from exc handler.send_response(HTTPStatus.OK) handler.send_header("content-type", "text/event-stream") handler.send_header("cache-control", "no-cache") @@ -202,18 +223,15 @@ def _handle_go_verbatim_request( retries += 1 retry_sleep(retries) continue - raise ProxyError( - HTTPStatus.BAD_GATEWAY, - f"opencode-go upstream network error: {getattr(exc, 'reason', exc)}", - retries=retries, - ) from exc + raise _network_error(exc, retries) from exc + client_status = _client_status(status) record_usage_event( model=model, - status=status, + status=client_status, duration_ms=int((time.time() - started) * 1000), retries=retries or None, ) - handler.send_response(status) + handler.send_response(client_status) handler.send_header("content-type", content_type) if retry_after: handler.send_header("retry-after", retry_after) diff --git a/tests/test_go_compatibility.py b/tests/test_go_compatibility.py index fa9cbe4..37c2152 100644 --- a/tests/test_go_compatibility.py +++ b/tests/test_go_compatibility.py @@ -108,33 +108,34 @@ def test_explicit_and_safe_bare_selection_reject_unknown_ids() -> None: go_upstream.go_request_identity("opencode-go/") -@pytest.mark.parametrize( - ("model", "path", "auth"), - [ - ("opencode-go/grok-4.6", "/responses", "authorization"), - ("glm-5.3-flash", "/chat/completions", "authorization"), - ("opencode-go/minimax-m3", "/messages", "x-api-key"), - ], -) -def test_responses_adapter_selects_documented_endpoint_and_auth( +@pytest.mark.parametrize("model", sorted(DOCUMENTED_GO_MODELS)) +@pytest.mark.parametrize("stream", [False, True]) +def test_responses_adapter_supports_every_documented_model( model: str, - path: str, - auth: str, + stream: bool, ) -> None: - bare_id, family = go_upstream.go_request_identity(model) + selected_model = f"opencode-go/{model}" + bare_id, family = go_upstream.go_request_identity(selected_model) url, body, headers = zen_upstream._build_family_request( - {"model": model, "input": "hello"}, + {"model": selected_model, "input": "hello", "stream": stream}, family, bare_id, "go-key", - stream=False, - session_model=model, + stream=stream, + session_model=selected_model, base_url=config().chat_base_url, extra_headers={"x-opencode-session": "session-1"}, function_tools_only=True, ) + path = { + OPENAI_RESPONSES: "/responses", + OPENAI_CHAT: "/chat/completions", + ANTHROPIC_MESSAGES: "/messages", + }[family] + auth = "x-api-key" if family == ANTHROPIC_MESSAGES else "authorization" assert url.endswith(path) assert body["model"] == bare_id + assert body["stream"] is stream assert headers[auth] == ( "Bearer go-key" if auth == "authorization" else "go-key" ) @@ -281,6 +282,100 @@ def fake_urlopen(request, **kwargs): assert "authorization" not in request_headers(messages_request) +@pytest.mark.parametrize("model", sorted(DOCUMENTED_CHAT_MODELS)) +def test_direct_chat_surface_supports_every_chat_model(model: str) -> None: + captured = [] + + def fake_urlopen(request, **kwargs): + captured.append(request) + return response({"id": "chatcmpl_1", "choices": []}) + + with ( + mock.patch( + "opencode_go_proxy.go_upstream.resolve_api_key", + return_value="go-key", + ), + mock.patch("urllib.request.urlopen", side_effect=fake_urlopen), + ): + handler = Handler() + go_upstream.handle_go_chat_request( + handler, + { + "model": f"opencode-go/{model}", + "messages": [{"role": "user", "content": "hi"}], + }, + config(), + "chat-request", + ) + + assert handler.status == 200 + assert captured[0].full_url.endswith("/chat/completions") + assert request_body(captured[0])["model"] == model + + +@pytest.mark.parametrize("model", sorted(DOCUMENTED_MESSAGES_MODELS)) +def test_direct_messages_surface_supports_every_messages_model(model: str) -> None: + captured = [] + + def fake_urlopen(request, **kwargs): + captured.append(request) + return response({"id": "msg_1", "content": []}) + + with ( + mock.patch( + "opencode_go_proxy.go_upstream.resolve_api_key", + return_value="go-key", + ), + mock.patch("urllib.request.urlopen", side_effect=fake_urlopen), + ): + handler = Handler() + go_upstream.handle_go_messages_request( + handler, + { + "model": f"opencode-go/{model}", + "messages": [{"role": "user", "content": "hi"}], + }, + config(), + "messages-request", + ) + + assert handler.status == 200 + assert captured[0].full_url.endswith("/messages") + assert request_body(captured[0])["model"] == model + + +@pytest.mark.parametrize( + "model", + sorted(DOCUMENTED_RESPONSES_MODELS | DOCUMENTED_MESSAGES_MODELS), +) +def test_direct_chat_rejects_non_chat_models(model: str) -> None: + with pytest.raises(ProxyError) as error: + go_upstream.handle_go_chat_request( + Handler(), + {"model": f"opencode-go/{model}", "messages": []}, + config(), + "chat-request", + ) + assert error.value.status == 400 + assert error.value.error_type == "invalid_request_error" + + +@pytest.mark.parametrize( + "model", + sorted(DOCUMENTED_RESPONSES_MODELS | DOCUMENTED_CHAT_MODELS), +) +def test_direct_messages_rejects_non_messages_models(model: str) -> None: + with pytest.raises(ProxyError) as error: + go_upstream.handle_go_messages_request( + Handler(), + {"model": f"opencode-go/{model}", "messages": []}, + config(), + "messages-request", + ) + assert error.value.status == 400 + assert error.value.error_type == "invalid_request_error" + + def test_merged_catalog_uses_explicit_go_slugs_for_native_collisions( monkeypatch: pytest.MonkeyPatch, tmp_path, diff --git a/tests/test_ops.py b/tests/test_ops.py index 6edd724..edba6e3 100644 --- a/tests/test_ops.py +++ b/tests/test_ops.py @@ -1,8 +1,10 @@ import io import json import os +import tomllib import urllib.error import urllib.request +from pathlib import Path from unittest import mock from opencode_go_proxy import ops @@ -332,6 +334,36 @@ def test_points_at_menu_bar_app(self, capsys) -> None: assert "menu bar" in capsys.readouterr().out +class TestPlatformServiceContract: + def test_systemd_runs_one_pinned_truthful_proxy(self) -> None: + root = Path(__file__).parents[1] + version = tomllib.loads((root / "pyproject.toml").read_text())["project"][ + "version" + ] + service = ( + root / "contrib/systemd/opencode-go-proxy.service" + ).read_text() + exec_lines = [ + line for line in service.splitlines() if line.startswith("ExecStart=") + ] + assert len(exec_lines) == 1 + assert f"opencode-go-proxy@v{version}" in exec_lines[0] + assert "--bind 127.0.0.1 --port 8787" in exec_lines[0] + assert "OPENCODE_GO_PROXY_USER_AGENT=" not in service + + def test_menu_bar_pin_matches_package_version(self) -> None: + root = Path(__file__).parents[1] + version = tomllib.loads((root / "pyproject.toml").read_text())["project"][ + "version" + ] + controller = ( + root + / "macos/MenuBarApp/Sources/OpenCodeGoMenuBar/ProxyController.swift" + ).read_text() + assert f"opencode-go-proxy@v{version}" in controller + assert not (root / "contrib/launchd").exists() + + class TestInstallSkills: def test_dry_run_prints_target(self, tmp_path, capsys) -> None: with mock.patch.dict(os.environ, {"OPENCODE_GO_PROXY_SKILLS_DIR": str(tmp_path / "skills")}): diff --git a/tests/test_protocol_surface.py b/tests/test_protocol_surface.py index 7b02300..67bd2a2 100644 --- a/tests/test_protocol_surface.py +++ b/tests/test_protocol_surface.py @@ -154,7 +154,7 @@ def test_upstream_429_status_and_body_relayed_verbatim(self, server): assert raw == err_body assert b"proxy_error" not in raw - def test_upstream_500_status_and_body_relayed_verbatim(self, server): + def test_upstream_500_maps_to_502_and_keeps_body(self, server): port, _ = server err_body = b'{"error":{"message":"internal boom"}}' @@ -163,7 +163,7 @@ def test_upstream_500_status_and_body_relayed_verbatim(self, server): }), mock.patch("urllib.request.urlopen", side_effect=http_error(500, err_body)): resp, raw = post(port, "/v1/chat/completions", chat_body()) - assert resp.status == 500 + assert resp.status == 502 assert raw == err_body def test_streaming_relays_sse_verbatim(self, server): @@ -214,6 +214,19 @@ def test_streaming_upstream_error_relayed_before_commit(self, server): assert raw == err_body assert "text/event-stream" not in resp.getheader("content-type", "") + def test_streaming_upstream_500_maps_to_502_before_commit(self, server): + port, _ = server + err_body = b'{"error":{"message":"internal boom"}}' + + with mock.patch.dict(os.environ, { + "OPENCODE_GO_API_KEY": "test-key", "OPENCODE_GO_PROXY_MAX_RETRIES": "0", + }), mock.patch("urllib.request.urlopen", side_effect=http_error(500, err_body)): + resp, raw = post(port, "/v1/chat/completions", chat_body(stream=True)) + + assert resp.status == 502 + assert raw == err_body + assert "text/event-stream" not in resp.getheader("content-type", "") + def test_missing_key_returns_401_json_non_streaming(self, server): port, _ = server from opencode_go_proxy.secrets import clear_api_key_cache @@ -299,6 +312,25 @@ def test_chat_model_on_messages_alias_returns_400(self, server): assert resp.status == 400 assert json.loads(raw)["error"]["type"] == "invalid_request_error" + @pytest.mark.parametrize("stream", [False, True]) + def test_upstream_500_maps_to_502_and_keeps_body(self, server, stream): + port, _ = server + err_body = b'{"type":"error","error":{"message":"internal boom"}}' + body = json.dumps({ + "model": "opencode-go/minimax-m3", + "messages": [], + "stream": stream, + }).encode() + + with mock.patch.dict(os.environ, { + "OPENCODE_GO_API_KEY": "test-key", "OPENCODE_GO_PROXY_MAX_RETRIES": "0", + }), mock.patch("urllib.request.urlopen", side_effect=http_error(500, err_body)): + resp, raw = post(port, "/v1/messages", body) + + assert resp.status == 502 + assert raw == err_body + assert "text/event-stream" not in resp.getheader("content-type", "") + def test_get_messages_returns_405(self, server): port, _ = server conn = HTTPConnection("127.0.0.1", port, timeout=5)