From b31ae6d74c27cf32711d9a12ddc23c57a08b0923 Mon Sep 17 00:00:00 2001 From: Eric Lee Date: Thu, 24 Sep 2026 16:17:28 -0700 Subject: [PATCH 1/2] feat(openai): GPT-6 models + per-model reasoning effort on the ChatGPT subscription - Bump the Codex catalog client_version 0.149.1 -> 0.155.1. It gates the catalog: gpt-6-astra needs 0.153.0, gpt-6-sol/luna need 0.155.0, so subscription logins never saw GPT-6. - Clamp subscription reasoning effort per model from the login's cached catalog (static fallback). Probed live 2026-09-24: gpt-6/gpt-5.6 take max, gpt-5.5 stops at xhigh, 'ultra' 400s everywhere. The old blanket xhigh/max -> high clamp capped GPT-6 two notches low. - Public API: max passes through for GPT-6, still degrades to xhigh elsewhere. - /model picker effort step uses the same per-model ladder, subscription-aware. - Register gpt-6-sol/luna, context rows at 872K (smaller of the 922K API and 872K subscription input limits), list pricing with the 272K long tier, verbosity support; harness effort docs updated. Co-Authored-By: Claude Opus 5.5 (1M context) --- eval/harbor/clawcodex_agent.py | 27 +++--- src/models/configs.py | 33 +++++++ src/providers/__init__.py | 7 +- src/providers/effort_options.py | 50 ++++++++-- src/providers/openai_provider.py | 58 +++++------ src/providers/openai_responses.py | 70 +++++++++++--- src/providers/openai_subscription_models.py | 95 ++++++++++++++++--- src/server/agent_server.py | 13 ++- src/services/pricing.py | 38 ++++++++ tests/providers/test_effort_options.py | 59 ++++++++++-- .../test_openai_subscription_models.py | 71 ++++++++++---- tests/server/test_model_provider_picker.py | 22 +++++ tests/test_gpt6_registration.py | 63 ++++++++++++ tests/test_openai_provider_routing.py | 18 ++++ tests/test_openai_subscription.py | 55 ++++++++--- 15 files changed, 568 insertions(+), 111 deletions(-) create mode 100644 tests/test_gpt6_registration.py diff --git a/eval/harbor/clawcodex_agent.py b/eval/harbor/clawcodex_agent.py index 0d5f915e5..77351879c 100644 --- a/eval/harbor/clawcodex_agent.py +++ b/eval/harbor/clawcodex_agent.py @@ -113,19 +113,22 @@ path ``openrouter/openai/gpt-5.6-luna`` takes, and OpenRouter accepts ``max`` for that model. - **First-party ``--provider openai``** — reasoning models go over the - Responses API, which tops out at ``xhigh``: ``max`` is degraded to - ``xhigh`` rather than sent (the API rejects ``max`` outright with - "Supported values are: 'none', 'low', 'medium', 'high', and 'xhigh'"). - So ``--ak effort=max`` against ``openai/gpt-5.6-luna`` really runs at - ``xhigh``, while the same nominal setting on OpenRouter sends ``max``. - Non-reasoning models (gpt-4o, …) stay on Chat Completions and have the - effort field STRIPPED, since they reject it as an unknown argument. + Responses API. GPT-6 (astra/sol/luna) accepts ``max``; every other + model tops out at ``xhigh``, so ``max`` is degraded to ``xhigh`` rather + than sent (gpt-5.6-luna rejects it outright with "Supported values are: + 'none', 'low', 'medium', 'high', and 'xhigh'"). So ``--ak effort=max`` + against ``openai/gpt-5.6-luna`` really runs at ``xhigh``, while the same + nominal setting on OpenRouter sends ``max``. Non-reasoning models + (gpt-4o, …) stay on Chat Completions and have the effort field + STRIPPED, since they reject it as an unknown argument. - **ChatGPT subscription** (``subscription=true`` with - ``--model openai/…``) clamps ``xhigh`` and ``max`` down to ``high``, - because that backend advertises only low/medium/high. So the same - ``effort=max`` runs at three different levels depending on route: - ``max`` on OpenRouter, ``xhigh`` on an OpenAI key, ``high`` on a - ChatGPT plan. + ``--model openai/…``) clamps PER MODEL to the levels the login's Codex + catalog advertises (static fallback when uncached), probed 2026-09-24: + gpt-6-* and gpt-5.6-* take ``max``; gpt-5.5 stops at ``xhigh``; older + models stop at ``high``. So ``effort=max`` on gpt-5.6-luna runs at + ``max`` on OpenRouter and on a ChatGPT plan, but ``xhigh`` on an OpenAI + key. Builds before 2026-09-24 clamped every subscription model to + ``high``. REQUIRES a clawcodex build from 2026-07-31 or later. Before that, effort was emitted ONLY on the Anthropic branch, so ``--effort`` was a silent diff --git a/src/models/configs.py b/src/models/configs.py index 063dc02bf..d00b3842c 100644 --- a/src/models/configs.py +++ b/src/models/configs.py @@ -722,6 +722,37 @@ class ModelConfig: # entries so they never take that path. As with the Meta entry above, # max_output_tokens is not sent on the wire for OpenAI providers — its # live effect is the auto-compact output reservation (clamped at 20K). + # + # --- GPT-6 ------------------------------------------------------------- + # GPT-6 (Astra / Sol / Luna). developers.openai.com model pages + # (2026-09-24): 1,050,000 context, 922K max INPUT, 128K max output. The + # ChatGPT subscription is tighter still: its Codex catalog gives every + # gpt-6 model ``max_context_window: 872000``. ``context_window`` here + # drives the auto-compact threshold, and OpenAI's context-overflow error + # is not one reactive compaction recognises, so over-estimating is a + # hard failure while under-estimating only compacts early. Hence 872K — + # the smaller of the two real input limits, safe on both backends. + # Placed before every gpt-5.x row for the same prefix-fallback reason as + # GPT-5.6 below: each has base "gpt-6", so an unlisted variant + # (``gpt-6-sol-pro``) lands here rather than on the 272K catch-all. + "gpt-6-astra": ModelConfig( + model_id="gpt-6-astra", + display_name="GPT-6 Astra", + context_window=872_000, + max_output_tokens=128_000, + ), + "gpt-6-sol": ModelConfig( + model_id="gpt-6-sol", + display_name="GPT-6 Sol", + context_window=872_000, + max_output_tokens=128_000, + ), + "gpt-6-luna": ModelConfig( + model_id="gpt-6-luna", + display_name="GPT-6 Luna", + context_window=872_000, + max_output_tokens=128_000, + ), # GPT-5.6 (Sol / Terra / Luna — three durable capability tiers on one # generation, 1.05M context each). These keys sit BEFORE "gpt-5.5" # deliberately: ``get_model_config``'s prefix fallback walks in insertion @@ -730,6 +761,8 @@ class ModelConfig: # ``gpt-5.6-sol-pro`` resolve to 1.05M instead of falling through to the # 272K catch-all. The bare ``gpt-5.6`` alias is deliberately NOT here — # see below. + # KNOWN GAP (not fixed here): the subscription catalog also caps these at + # ``max_context_window: 872000``, so 1.05M over-estimates on that path. "gpt-5.6-sol": ModelConfig( model_id="gpt-5.6-sol", display_name="GPT-5.6 Sol", diff --git a/src/providers/__init__.py b/src/providers/__init__.py index 71385aced..f83639b39 100644 --- a/src/providers/__init__.py +++ b/src/providers/__init__.py @@ -89,8 +89,13 @@ class ProviderInfo(_ProviderInfoOptional): "default_base_url": "https://api.openai.com/v1", "default_model": "gpt-5.4", "available_models": [ - # https://developers.openai.com/api/docs/models (2026-09-06) + # https://developers.openai.com/api/docs/models (2026-09-24) + # GPT-6 — Astra is the frontier tier, Sol the flagship, Luna the + # cheap high-volume tier. All three are also served by the + # ChatGPT subscription (Codex catalog, client_version >= 0.155). "gpt-6-astra", + "gpt-6-sol", + "gpt-6-luna", # GPT-5.6 — Sol / Terra / Luna are # durable capability tiers rather than a size ladder: Sol is the # flagship, Terra balances capability against cost, Luna is the diff --git a/src/providers/effort_options.py b/src/providers/effort_options.py index 6cf854226..98cf3f472 100644 --- a/src/providers/effort_options.py +++ b/src/providers/effort_options.py @@ -31,7 +31,12 @@ AUTO = "auto" -def effort_options(provider_name: str | None, model: str | None) -> dict[str, object]: +def effort_options( + provider_name: str | None, + model: str | None, + *, + openai_subscription: bool | None = None, +) -> dict[str, object]: """The effort levels ``model`` accepts under ``provider_name``. Returns ``{"supported": bool, "levels": [...]}``. ``supported`` False @@ -39,6 +44,10 @@ def effort_options(provider_name: str | None, model: str | None) -> dict[str, ob third step rather than offering a list that cannot be applied. ``levels`` never includes ``auto``; the caller prepends it, since "let the provider decide" is meaningful exactly when some real level is also on offer. + + ``openai_subscription`` says whether OpenAI requests ride the ChatGPT + login rather than an API key — the two backends accept different levels + for the same model. ``None`` infers it from stored credentials. """ canonical = _canonical(provider_name) @@ -46,7 +55,9 @@ def effort_options(provider_name: str | None, model: str | None) -> dict[str, ob return _anthropic_options(model) if canonical == "openai": - return _openai_options(model) + if openai_subscription is None: + openai_subscription = _openai_subscription_in_use() + return _openai_options(model, subscription=openai_subscription) # Every other provider: the codebase carries no per-model effort table, # and the OpenAI-compatible paths pass the value through as a body field @@ -92,18 +103,39 @@ def _anthropic_options(model: str | None) -> dict[str, object]: } -def _openai_options(model: str | None) -> dict[str, object]: +def _openai_subscription_in_use() -> bool: + """Mirror ``OpenAIProvider.__init__``: a configured key wins, otherwise a + stored ChatGPT login is used. (The base-URL guard is not mirrored; a + proxy with no key and a stale login is an edge the wire clamp absorbs.)""" + try: + from src.auth.openai_subscription import load_credentials + from src.providers import resolve_api_key + + return not resolve_api_key("openai") and load_credentials() is not None + except Exception: # noqa: BLE001 — a picker query must not fail on this + return False + + +def _openai_options(model: str | None, *, subscription: bool = False) -> dict[str, object]: """OpenAI: the Responses API rejects a reasoning block outright on - non-reasoning models, so those get no third step at all.""" - from src.providers.openai_responses import OPENAI_REASONING_EFFORTS, supports_reasoning + non-reasoning models, so those get no third step at all. The rest get the + per-model ladder the request path clamps to, so every offered level is + sent as-is rather than silently downgraded.""" + from src.providers.openai_responses import api_effort_levels, supports_reasoning if not supports_reasoning(model or ""): return {"supported": False, "levels": []} - # Intersect rather than pass through: OPENAI_REASONING_EFFORTS carries - # ``none``, which ``/effort`` would reject, and omits ``max``, which - # OpenAI 400s on for the same model OpenRouter tolerates it for. + if subscription: + from src.providers.openai_subscription_models import get_subscription_effort_levels + + accepted = get_subscription_effort_levels(model or "") + else: + accepted = api_effort_levels(model or "") + + # Intersect rather than pass through: the OpenAI lists carry ``none`` + # (and ``minimal``), which ``/effort`` would reject. return { "supported": True, - "levels": [lvl for lvl in LADDER if lvl in OPENAI_REASONING_EFFORTS], + "levels": [lvl for lvl in LADDER if lvl in accepted], } diff --git a/src/providers/openai_provider.py b/src/providers/openai_provider.py index e5ebe9aa5..09730a37e 100644 --- a/src/providers/openai_provider.py +++ b/src/providers/openai_provider.py @@ -61,6 +61,7 @@ RESPONSES_ITEM_BLOCK_TYPE, INCLUDE_ENCRYPTED_REASONING, SUBSCRIPTION_MODELS, + clamp_effort, normalize_openai_effort, supports_reasoning, build_usage_dict, @@ -73,10 +74,9 @@ logger = logging.getLogger(__name__) -_REASONING_EFFORTS = ("minimal", "low", "medium", "high", "xhigh") - - -def _subscription_reasoning_effort(requested: str | None = None) -> str: +def _subscription_reasoning_effort( + requested: str | None = None, model: str = "", +) -> str | None: """Reasoning effort for subscription requests. Precedence: the session's ``/effort`` setting (arrives as @@ -86,25 +86,27 @@ def _subscription_reasoning_effort(requested: str | None = None) -> str: default, transform.ts:1176, and the backend's own default_reasoning_level). - ``xhigh``/``max`` clamp to ``high`` HERE, and only here: this is the - ChatGPT-subscription backend (chatgpt.com/backend-api/codex), whose - general gpt-5.x models advertise low/medium/high and reject higher - tiers (probed 2026-07-25). That is narrower than the public API — - developers.openai.com/api/docs/guides/reasoning lists none | minimal | - low | medium | high | xhigh | max and notes support varies by model — - and narrower than what a gateway may accept (``openai/gpt-5.6-luna`` - via OpenRouter takes both ``xhigh`` and ``max``, probed 2026-07-31, - with reasoning-token counts rising monotonically across the ladder). - So the clamp is a property of THIS backend, not of the level names; - the generic OpenAI-compatible path deliberately does not clamp. + The requested level is clamped to what THIS model takes on the ChatGPT + backend (chatgpt.com/backend-api/codex), which varies per model: probed + 2026-09-24, gpt-6-* and gpt-5.6-* accept ``max``, gpt-5.5 accepts + ``xhigh`` but 400s on ``max``. The per-model list comes from this login's + cached model catalog (``supported_reasoning_levels``), falling back to a + static table — see ``get_subscription_effort_levels``. This used to clamp + ``xhigh``/``max`` to ``high`` for every model (true of the gpt-5.x models + probed 2026-07-25), which silently capped gpt-6 two notches low. + + The generic OpenAI-compatible path deliberately does not clamp: a gateway + (``openai/gpt-5.6-luna`` via OpenRouter) takes both ``xhigh`` and ``max``. """ + from .openai_subscription_models import get_subscription_effort_levels + + levels = get_subscription_effort_levels(model) for candidate in (requested, os.environ.get("CLAWCODEX_OPENAI_REASONING_EFFORT")): - effort = (candidate or "").strip().lower() - if effort in ("xhigh", "max"): - return "high" - if effort in _REASONING_EFFORTS: - return effort - return "medium" + if (candidate or "").strip(): + effort = clamp_effort(candidate, levels) + if effort: + return effort + return clamp_effort("medium", levels) class _HttpxStreamHolder: @@ -487,16 +489,18 @@ def _subscription_request_body( # model rather than only the reasoning ones. if supports_reasoning(model): if self._subscription_active: - # The ChatGPT backend advertises only low/medium/high and - # rejects higher tiers, so it keeps its own clamp. + # The ChatGPT backend's levels vary per model, so it keeps + # its own catalog-driven clamp. effort = _subscription_reasoning_effort( - (kwargs.get("extra_body") or {}).get("reasoning_effort") + (kwargs.get("extra_body") or {}).get("reasoning_effort"), + model, ) else: - # The public API accepts xhigh; only ``max`` is unsupported, - # and it degrades rather than failing the request. + # The public API accepts xhigh everywhere and ``max`` on + # gpt-6 only; elsewhere ``max`` degrades rather than failing. effort = normalize_openai_effort( - (kwargs.get("extra_body") or {}).get("reasoning_effort") + (kwargs.get("extra_body") or {}).get("reasoning_effort"), + model, ) if effort: body["reasoning"] = {"effort": effort, "summary": "auto"} diff --git a/src/providers/openai_responses.py b/src/providers/openai_responses.py index 5d3b979a8..fa0ab0531 100644 --- a/src/providers/openai_responses.py +++ b/src/providers/openai_responses.py @@ -140,29 +140,71 @@ def supports_reasoning(model: str) -> bool: return "codex" in m -def normalize_openai_effort(effort: str | None) -> str | None: +# Every level either OpenAI backend has ever accepted on ``reasoning.effort``, +# lowest to highest. Clamping walks DOWN this ladder, so its order is load- +# bearing. ``ultra`` is deliberately absent: the Codex model catalog +# advertises it for gpt-6 / gpt-5.6, but both the ChatGPT backend and the +# public API 400 on it ("Invalid value: 'ultra'. Supported values are: 'none', +# 'minimal', 'low', 'medium', 'high', 'xhigh', and 'max'." — probed live +# 2026-09-24 on gpt-6-astra/sol/luna and gpt-5.5). It is a Codex-client mode, +# not a wire value. +EFFORT_LADDER = ("none", "minimal", "low", "medium", "high", "xhigh", "max") + + +def _is_max_effort_family(model: str) -> bool: + """GPT-6 accepts ``max`` on the public API (developers.openai.com model + pages for gpt-6-astra/sol/luna, 2026-09-24); gpt-5.6-luna 400s on it + there (probed 2026-08-01, see ``OPENAI_REASONING_EFFORTS``).""" + return (model or "").lower().startswith("gpt-6") + + +def clamp_effort(effort: str | None, supported: tuple[str, ...] | list[str]) -> str | None: + """Degrade ``effort`` to the highest ``supported`` level at or below it. + + A level the model cannot take should cost a notch of depth, not the run: + an unsupported VALUE of reasoning.effort is a hard 400 on both OpenAI + backends. Unknown values (and levels below everything supported) return + ``None`` so the caller omits the block and the backend uses its default. + """ + value = (effort or "").strip().lower() + if value not in EFFORT_LADDER or not supported: + return None + if value in supported: + return value + for level in reversed(EFFORT_LADDER[: EFFORT_LADDER.index(value)]): + # Never degrade INTO ``none``: that switches reasoning off, a + # different behaviour rather than a notch less of the same one. + if level in supported and level != "none": + return level + return None + + +def api_effort_levels(model: str) -> tuple[str, ...]: + """Effort levels the PUBLIC API accepts for ``model`` (reasoning models).""" + if _is_max_effort_family(model): + return (*OPENAI_REASONING_EFFORTS, "max") + return OPENAI_REASONING_EFFORTS + + +def normalize_openai_effort(effort: str | None, model: str = "") -> str | None: """Coerce a cross-provider effort level to one the OpenAI API accepts. - ``max`` degrades to ``xhigh`` (the highest OpenAI level) rather than - erroring, matching how ``resolve_thinking_effort`` degrades an - unsupported ``xhigh`` to ``high`` on the Anthropic wire — a level the - provider cannot take should cost a notch of depth, not the whole run. + ``max`` passes through for GPT-6 and degrades to ``xhigh`` (the highest + level the rest accept) elsewhere, matching how ``resolve_thinking_effort`` + degrades an unsupported ``xhigh`` to ``high`` on the Anthropic wire. Unknown values return ``None`` so the caller omits the block and lets the API apply its own default. """ - value = (effort or "").strip().lower() - if not value: - return None - if value == "max": - return "xhigh" - return value if value in OPENAI_REASONING_EFFORTS else None + return clamp_effort(effort, api_effort_levels(model)) def supports_verbosity(model: str) -> bool: - """gpt-5.x general models accept ``text.verbosity``; codex/chat variants - don't (OpenCode transform.ts:1189-1196).""" + """gpt-5.x / gpt-6 general models accept ``text.verbosity``; codex/chat + variants don't (OpenCode transform.ts:1189-1196). The Codex catalog marks + every gpt-6 model ``support_verbosity: true, default_verbosity: low``, + and ``verbosity: low`` was accepted live on all three (2026-09-24).""" return ( - model.startswith("gpt-5") + model.startswith(("gpt-5", "gpt-6")) and "codex" not in model and "-chat" not in model ) diff --git a/src/providers/openai_subscription_models.py b/src/providers/openai_subscription_models.py index ea28e8c32..bab453f27 100644 --- a/src/providers/openai_subscription_models.py +++ b/src/providers/openai_subscription_models.py @@ -23,12 +23,18 @@ load_credentials, ) -from .openai_responses import SUBSCRIPTION_MODELS +from .openai_responses import EFFORT_LADDER, SUBSCRIPTION_MODELS MODELS_ENDPOINT = "https://chatgpt.com/backend-api/codex/models" # Catalog protocol version verified against the Codex endpoint. This is not # Clawcodex's package version (which the endpoint doesn't understand). -CATALOG_CLIENT_VERSION = "0.149.1" +# +# It is also a GATE, not just a label: the backend hides every model whose +# ``minimal_client_version`` exceeds it. At 0.149.1 the catalog stopped at +# gpt-5.6; gpt-6-astra needs 0.153.0 and gpt-6-sol / gpt-6-luna need 0.155.0 +# (probed 2026-09-24). Bump it only after the new models' wire has been +# checked live — the pin is what keeps an unverified model out of the picker. +CATALOG_CLIENT_VERSION = "0.155.1" TTL_SECONDS = 300 RETRY_SECONDS = 30 VERIFICATION_TTL_SECONDS = 86_400 @@ -119,7 +125,21 @@ def record_subscription_model( _write_entry(path, entry) -def _fetch_models(credentials: SubscriptionCredentials) -> list[str] | None: +def _wire_efforts(raw: object) -> list[str] | None: + """A catalog entry's ``supported_reasoning_levels`` reduced to values the + wire accepts (``ultra`` is advertised but 400s — see EFFORT_LADDER).""" + if not isinstance(raw, list): + return None + levels = [ + entry.get("effort") for entry in raw + if isinstance(entry, dict) and entry.get("effort") in EFFORT_LADDER + ] + return levels or None + + +def _fetch_catalog( + credentials: SubscriptionCredentials, +) -> tuple[list[str], dict[str, list[str]]] | None: import httpx headers = { @@ -143,14 +163,21 @@ def _fetch_models(credentials: SubscriptionCredentials) -> list[str] | None: raw = payload.get("models") if isinstance(payload, dict) else None if not isinstance(raw, list): return None - # Preserve backend order, including subscription-only models (their - # supported_in_api flag is false). Hidden/internal models stay hidden. - return list(dict.fromkeys( - model["slug"] for model in raw + listed = [ + model for model in raw if isinstance(model, dict) and model.get("visibility") == "list" and isinstance(model.get("slug"), str) and model["slug"] - )) + ] + # Preserve backend order, including subscription-only models (their + # supported_in_api flag is false). Hidden/internal models stay hidden. + slugs = list(dict.fromkeys(model["slug"] for model in listed)) + efforts: dict[str, list[str]] = {} + for model in listed: + levels = _wire_efforts(model.get("supported_reasoning_levels")) + if levels: + efforts.setdefault(model["slug"], levels) + return slugs, efforts except (httpx.HTTPError, ValueError): return None @@ -158,14 +185,15 @@ def _fetch_models(credentials: SubscriptionCredentials) -> list[str] | None: def _refresh(path: Path, scope: str, credentials: SubscriptionCredentials) -> None: key = (path, scope) try: - models = _fetch_models(credentials) - if models is None: + catalog = _fetch_catalog(credentials) + if catalog is None: return + models, efforts = catalog with _lock: # Re-read after HTTP so an in-flight successful request cannot # lose its verification when discovery finishes later. entry = _read_entry(path, scope) - entry.update(models=models, fetched_at=time.time()) + entry.update(models=models, efforts=efforts, fetched_at=time.time()) _write_entry(path, entry) finally: with _lock: @@ -209,3 +237,48 @@ def get_subscription_models( rejected = _recent_models(entry, "rejected_models", TTL_SECONDS) listed = list(models) if models is not None else list(SUBSCRIPTION_MODELS) return [model for model in dict.fromkeys([*verified, *listed]) if model not in rejected] + + +def _static_effort_levels(model: str) -> tuple[str, ...]: + """Fallback when this login's catalog has no entry for ``model``. + + Probed live on the ChatGPT backend 2026-09-24: gpt-6-* and gpt-5.6-* take + ``max``; gpt-5.5 takes ``xhigh`` and 400s on ``max``. Anything older keeps + the low/medium/high ceiling measured 2026-07-25, the safe direction — + under-offering hides a level, over-offering 400s every request. Shaped + like the catalog's own lists (which never include ``none``), so a level + behaves the same whether or not this login's catalog is cached yet. + """ + m = (model or "").lower() + if m.startswith(("gpt-6", "gpt-5.6")): + return ("low", "medium", "high", "xhigh", "max") + if m.startswith("gpt-5.5"): + return ("low", "medium", "high", "xhigh") + return ("minimal", "low", "medium", "high") + + +def get_subscription_effort_levels( + model: str, credentials: SubscriptionCredentials | None = None, +) -> tuple[str, ...]: + """The reasoning levels this login's catalog advertises for ``model``. + + Cache-only: it runs on the request path and in the /model picker, so it + must never wait on HTTP. The catalog's list is trusted as-is apart from + ``ultra`` (dropped at fetch time, see ``_wire_efforts``); it matched every + live probe on 2026-09-24. + + The cache is scoped per access token, so after a token refresh the + per-model list is missed until the next catalog refresh and the static + table answers instead. That is safe while the table matches the catalog; + keep the two in step when a new generation lands. + """ + credentials = credentials or load_credentials() + if credentials is not None: + path = credentials_path().with_name("openai-models-cache.json") + efforts = _read_entry(path, _scope(credentials)).get("efforts") + levels = efforts.get(model) if isinstance(efforts, dict) else None + if isinstance(levels, list) and levels: + wire = [lvl for lvl in EFFORT_LADDER if lvl in levels] + if wire: + return tuple(wire) + return _static_effort_levels(model) diff --git a/src/server/agent_server.py b/src/server/agent_server.py index c151a14e0..038ed8cef 100644 --- a/src/server/agent_server.py +++ b/src/server/agent_server.py @@ -2097,7 +2097,18 @@ def _do_effort_options( slug = provider if isinstance(provider, str) and provider else self.provider_name name = model if isinstance(model, str) and model else getattr(self.provider, "model", "") - options = effort_options(slug, name) + # The live provider knows whether it is riding the ChatGPT login; + # for any other provider slug, effort_options infers it. + subscription = None + from src.providers import canonical_provider_name + + if slug == self.provider_name and canonical_provider_name(slug) == "openai": + from src.providers import unwrap_provider + + active = getattr(unwrap_provider(self.provider), "_subscription_active", None) + if isinstance(active, bool): + subscription = active + options = effort_options(slug, name, openai_subscription=subscription) except Exception as exc: # noqa: BLE001 — never break the control channel logger.exception("[agent-server] effort_options failed") self._reply(request_id, {"ok": False, "error": str(exc)}) diff --git a/src/services/pricing.py b/src/services/pricing.py index 8aa4bd4fb..c0aadccea 100644 --- a/src/services/pricing.py +++ b/src/services/pricing.py @@ -324,6 +324,38 @@ } +# OpenAI GPT-6 (developers.openai.com/api/docs/models/gpt-6-{astra,sol,luna}, +# read 2026-09-24). Same long-context rule as GPT-5.6 Luna: prompts above +# 272K input tokens are priced at 2x input and cache rates and 1.5x output +# for the full request. Cache WRITES follow the Luna row's convention of +# 1.25x the uncached input rate; cache READS are the published cached-input +# price. List prices, same policy as Luna (see above). +_GPT_6_INPUT_TIER_LIMIT = 272_000 + + +def _gpt6_tiers(inp: float, cached: float, out: float) -> tuple[dict[str, float], dict[str, float]]: + per_m = 1_000_000 + short = { + "input": inp / per_m, + "output": out / per_m, + "cache_creation": inp * 1.25 / per_m, + "cache_read": cached / per_m, + } + long = { + "input": short["input"] * 2, + "output": short["output"] * 1.5, + "cache_creation": short["cache_creation"] * 2, + "cache_read": short["cache_read"] * 2, + } + return short, long + + +_GPT_6_TIERS: dict[str, tuple[dict[str, float], dict[str, float]]] = { + "gpt-6-astra": _gpt6_tiers(10.00, 1.00, 50.00), + "gpt-6-sol": _gpt6_tiers(2.00, 0.20, 10.00), + "gpt-6-luna": _gpt6_tiers(0.10, 0.01, 0.50), +} + # Exact-match table — keyed by canonical model name. Order DOESN'T matter # for exact match but DOES matter for the prefix fallback below # (more-specific keys must come first). See ``get_pricing``. @@ -389,6 +421,9 @@ # Editing the dicts below changes nothing; edit the tiers instead. "gpt-5.6-luna": _TIER_GPT_56_LUNA, "gpt-5.6-luna-pro": _TIER_GPT_56_LUNA, + # GPT-6 — membership gates like the Luna rows above; the live card is + # picked by prompt size in ``_get_exact_pricing`` from ``_GPT_6_TIERS``. + **{model: tiers[0] for model, tiers in _GPT_6_TIERS.items()}, } @@ -480,6 +515,9 @@ def _get_exact_pricing( if input_tokens > _GPT_56_LUNA_INPUT_TIER_LIMIT else _TIER_GPT_56_LUNA ) + gpt6 = _GPT_6_TIERS.get(model) + if gpt6 is not None: + return gpt6[1] if input_tokens > _GPT_6_INPUT_TIER_LIMIT else gpt6[0] # Time-tiered models: the published rate depends on WHEN the request was # sent and on nothing inside it. Neither existing axis can carry this — # ``input_tokens`` is prompt size, and ``service_tier`` is whatever the diff --git a/tests/providers/test_effort_options.py b/tests/providers/test_effort_options.py index 8eea29ce3..652af5b8b 100644 --- a/tests/providers/test_effort_options.py +++ b/tests/providers/test_effort_options.py @@ -12,6 +12,14 @@ from src.settings.constants import VALID_EFFORT_VALUES +@pytest.fixture(autouse=True) +def _no_real_credentials(tmp_path, monkeypatch): + """OpenAI mode is inferred from stored ChatGPT credentials when a caller + does not say; keep that inference off the developer's real login.""" + monkeypatch.setenv("CLAWCODEX_CONFIG_DIR", str(tmp_path)) + monkeypatch.delenv("OPENAI_API_KEY", raising=False) + + class TestAnthropic: def test_opus_5_carries_the_full_ladder(self): """Wire-probed 2026-07-25: opus-5 accepts xhigh and max.""" @@ -41,26 +49,63 @@ def test_a_model_outside_the_effort_allowlist_gets_no_step(self): class TestOpenAI: + """API-key mode unless a test says otherwise: without the explicit flag + the mode is inferred from stored ChatGPT credentials, which would make + these tests depend on whoever runs them.""" + + @staticmethod + def _api(model): + return effort_options("openai", model, openai_subscription=False) + def test_a_reasoning_model_drops_max(self): - """OPENAI_REASONING_EFFORTS omits max — OpenAI 400s on it for the same - model OpenRouter tolerates it for, so the clamp is provider-scoped.""" - r = effort_options("openai", "gpt-5.6-luna") + """gpt-5.6-luna 400s on max over the public API (probed 2026-08-01) + while OpenRouter tolerates it, so the clamp is provider-scoped.""" + r = self._api("gpt-5.6-luna") assert r["supported"] is True - assert "max" not in r["levels"] assert r["levels"] == ["low", "medium", "high", "xhigh"] + def test_gpt6_offers_max_over_the_api(self): + """developers.openai.com lists max for all three GPT-6 models.""" + for model in ("gpt-6-astra", "gpt-6-sol", "gpt-6-luna"): + assert self._api(model)["levels"] == ["low", "medium", "high", "xhigh", "max"] + def test_none_never_leaks_into_the_ladder(self): """``none`` is an OpenAI level but not a clawcodex one — _do_set_effort would reject it, so an offered row would be unapplicable.""" - assert "none" not in effort_options("openai", "gpt-5.6-luna")["levels"] + assert "none" not in self._api("gpt-5.6-luna")["levels"] + assert "none" not in self._api("gpt-6-sol")["levels"] def test_a_non_reasoning_model_gets_no_step(self): """A reasoning block on gpt-4o is a hard 400, verified live.""" - assert effort_options("openai", "gpt-4o")["supported"] is False + assert self._api("gpt-4o")["supported"] is False def test_chat_variants_are_not_reasoning_models(self): - assert effort_options("openai", "gpt-5-chat-latest")["supported"] is False + assert self._api("gpt-5-chat-latest")["supported"] is False + + +class TestOpenAISubscription: + """The ChatGPT backend's levels differ per model (probed 2026-09-24).""" + + @pytest.fixture(autouse=True) + def _no_login(self, monkeypatch): + from src.providers import openai_subscription_models as catalog + + monkeypatch.setattr(catalog, "load_credentials", lambda: None) + + @staticmethod + def _sub(model): + return effort_options("openai", model, openai_subscription=True) + + def test_gpt6_and_gpt56_offer_max(self): + for model in ("gpt-6-astra", "gpt-6-sol", "gpt-6-luna", "gpt-5.6-luna"): + assert self._sub(model)["levels"] == ["low", "medium", "high", "xhigh", "max"] + + def test_gpt55_stops_at_xhigh(self): + assert self._sub("gpt-5.5")["levels"] == ["low", "medium", "high", "xhigh"] + + def test_unknown_older_models_keep_the_conservative_ceiling(self): + assert self._sub("gpt-5.4")["levels"] == ["low", "medium", "high"] class TestOtherProviders: diff --git a/tests/providers/test_openai_subscription_models.py b/tests/providers/test_openai_subscription_models.py index 70d76e452..a175f6ebb 100644 --- a/tests/providers/test_openai_subscription_models.py +++ b/tests/providers/test_openai_subscription_models.py @@ -13,6 +13,15 @@ from src.providers import openai_subscription_models as catalog +def _patch_fetch(monkeypatch, fetch): + """Stub discovery with a slugs-only fetcher (no per-model efforts).""" + def fetch_catalog(creds): + models = fetch(creds) + return None if models is None else (models, {}) + + monkeypatch.setattr(catalog, "_fetch_catalog", fetch_catalog) + + @pytest.fixture def credentials(tmp_path, monkeypatch): monkeypatch.setenv("CLAWCODEX_CONFIG_DIR", str(tmp_path)) @@ -45,10 +54,36 @@ def test_fetch_uses_this_login_and_only_visible_models(credentials, monkeypatch) assert "secret-token" not in saved and "secret-refresh" not in saved +def test_fetch_keeps_each_models_wire_effort_levels(credentials, monkeypatch): + """``supported_reasoning_levels`` is cached per model minus ``ultra``, + which the catalog advertises but the backend 400s on (2026-09-24).""" + def levels(*names): + return [{"effort": n, "description": "x"} for n in names] + + get = Mock(return_value=httpx.Response(200, request=httpx.Request("GET", catalog.MODELS_ENDPOINT), json={ + "models": [ + {"slug": "gpt-6-astra", "visibility": "list", + "supported_reasoning_levels": levels("low", "medium", "high", "xhigh", "max", "ultra")}, + {"slug": "gpt-5.5", "visibility": "list", + "supported_reasoning_levels": levels("low", "medium", "high", "xhigh")}, + {"slug": "gpt-6-odd", "visibility": "list", + "supported_reasoning_levels": [None, "high", {"effort": "ultra"}]}, + {"slug": "hidden", "visibility": "hide", "supported_reasoning_levels": levels("low")}, + ], + })) + monkeypatch.setattr(httpx, "get", get) + assert catalog.get_subscription_models(background=False) == ["gpt-6-astra", "gpt-5.5", "gpt-6-odd"] + assert catalog.get_subscription_effort_levels("gpt-6-astra") == ("low", "medium", "high", "xhigh", "max") + assert catalog.get_subscription_effort_levels("gpt-5.5") == ("low", "medium", "high", "xhigh") + # No usable level in the catalog: the static table answers instead. + assert catalog.get_subscription_effort_levels("gpt-6-odd") == ("low", "medium", "high", "xhigh", "max") + assert catalog.get_subscription_effort_levels("gpt-5.4") == ("minimal", "low", "medium", "high") + + @pytest.mark.parametrize("changed", [{"account_id": "other"}, {"access_token": "new-token"}]) def test_cache_cannot_leak_models_between_logins(credentials, monkeypatch, changed): fetch = Mock(side_effect=[["gpt-5.6-sol"], ["gpt-5.6-terra"]]) - monkeypatch.setattr(catalog, "_fetch_models", fetch) + _patch_fetch(monkeypatch, fetch) assert catalog.get_subscription_models(credentials, background=False) == ["gpt-5.6-sol"] assert catalog.get_subscription_models(replace(credentials, **changed), background=False) == ["gpt-5.6-terra"] assert fetch.call_count == 2 @@ -56,7 +91,7 @@ def test_cache_cannot_leak_models_between_logins(credentials, monkeypatch, chang def test_stale_cache_survives_failed_refresh_and_throttles_retries(credentials, monkeypatch): fetch = Mock(return_value=["gpt-5.6-terra"]) - monkeypatch.setattr(catalog, "_fetch_models", fetch) + _patch_fetch(monkeypatch, fetch) catalog.get_subscription_models(background=False) path = auth.credentials_path().with_name("openai-models-cache.json") saved = json.loads(path.read_text()) @@ -69,19 +104,19 @@ def test_stale_cache_survives_failed_refresh_and_throttles_retries(credentials, def test_empty_live_catalog_does_not_restore_static_models(credentials, monkeypatch): - monkeypatch.setattr(catalog, "_fetch_models", lambda _: []) + _patch_fetch(monkeypatch, lambda _: []) assert catalog.get_subscription_models(background=False) == [] assert catalog.get_subscription_models() == [] def test_cold_failure_has_conservative_fallback(credentials, monkeypatch): - monkeypatch.setattr(catalog, "_fetch_models", lambda _: None) + _patch_fetch(monkeypatch, lambda _: None) assert catalog.get_subscription_models(background=False) == ["gpt-5.5"] def test_corrupt_cache_is_refetched(credentials, monkeypatch): auth.credentials_path().with_name("openai-models-cache.json").write_text("[]") - monkeypatch.setattr(catalog, "_fetch_models", lambda _: ["gpt-5.6-terra"]) + _patch_fetch(monkeypatch, lambda _: ["gpt-5.6-terra"]) assert catalog.get_subscription_models(background=False) == ["gpt-5.6-terra"] @@ -102,7 +137,7 @@ def refresh(*args): finally: finished.set() - monkeypatch.setattr(catalog, "_fetch_models", fetch) + _patch_fetch(monkeypatch, fetch) monkeypatch.setattr(catalog, "_refresh", refresh) try: assert catalog.get_subscription_models() == ["gpt-5.5"] @@ -117,20 +152,20 @@ def refresh(*args): def test_expired_credentials_do_not_refresh_tokens_on_picker_thread(credentials, monkeypatch): fetch = Mock() - monkeypatch.setattr(catalog, "_fetch_models", fetch) + _patch_fetch(monkeypatch, fetch) assert catalog.get_subscription_models(replace(credentials, expires_at=0), background=False) == ["gpt-5.5"] fetch.assert_not_called() def test_missing_login_does_not_reuse_cache(credentials, monkeypatch): - monkeypatch.setattr(catalog, "_fetch_models", lambda _: ["gpt-5.6-terra"]) + _patch_fetch(monkeypatch, lambda _: ["gpt-5.6-terra"]) catalog.get_subscription_models(background=False) monkeypatch.setattr(catalog, "load_credentials", lambda: None) assert catalog.get_subscription_models() == [] def test_successful_unlisted_model_survives_catalog_refresh(credentials, monkeypatch): - monkeypatch.setattr(catalog, "_fetch_models", lambda _: ["gpt-5.6-sol"]) + _patch_fetch(monkeypatch, lambda _: ["gpt-5.6-sol"]) assert catalog.get_subscription_models(background=False) == ["gpt-5.6-sol"] catalog.record_subscription_model(credentials, "gpt-6-astra") assert catalog.get_subscription_models(background=False, force=True) == ["gpt-6-astra", "gpt-5.6-sol"] @@ -138,14 +173,14 @@ def test_successful_unlisted_model_survives_catalog_refresh(credentials, monkeyp def test_verified_model_is_not_offered_to_another_login(credentials, monkeypatch): - monkeypatch.setattr(catalog, "_fetch_models", lambda _: ["gpt-5.6-terra"]) + _patch_fetch(monkeypatch, lambda _: ["gpt-5.6-terra"]) catalog.record_subscription_model(credentials, "gpt-6-astra") other = replace(credentials, account_id="free-account", access_token="other-token") assert catalog.get_subscription_models(other, background=False) == ["gpt-5.6-terra"] def test_rejected_model_loses_previous_verification(credentials, monkeypatch): - monkeypatch.setattr(catalog, "_fetch_models", lambda _: ["gpt-5.6-sol"]) + _patch_fetch(monkeypatch, lambda _: ["gpt-5.6-sol"]) catalog.get_subscription_models(background=False) catalog.record_subscription_model(credentials, "gpt-6-astra") catalog.record_subscription_model(credentials, "gpt-6-astra", available=False) @@ -154,7 +189,7 @@ def test_rejected_model_loses_previous_verification(credentials, monkeypatch): @pytest.mark.parametrize("discovered", [["gpt-5.5", "gpt-5.6-sol"], None]) def test_rejected_model_stays_hidden_after_refresh_or_fallback(credentials, monkeypatch, discovered): - monkeypatch.setattr(catalog, "_fetch_models", lambda _: discovered) + _patch_fetch(monkeypatch, lambda _: discovered) catalog.get_subscription_models(background=False) catalog.record_subscription_model(credentials, "gpt-5.5", available=False) expected = ["gpt-5.6-sol"] if discovered else [] @@ -163,7 +198,7 @@ def test_rejected_model_stays_hidden_after_refresh_or_fallback(credentials, monk def test_rejection_expires_so_model_can_be_retried(credentials, monkeypatch): - monkeypatch.setattr(catalog, "_fetch_models", lambda _: ["gpt-5.6-sol"]) + _patch_fetch(monkeypatch, lambda _: ["gpt-5.6-sol"]) catalog.get_subscription_models(background=False) catalog.record_subscription_model(credentials, "gpt-5.6-sol", available=False) assert catalog.get_subscription_models() == [] @@ -173,7 +208,7 @@ def test_rejection_expires_so_model_can_be_retried(credentials, monkeypatch): def test_success_clears_previous_rejection(credentials, monkeypatch): - monkeypatch.setattr(catalog, "_fetch_models", lambda _: []) + _patch_fetch(monkeypatch, lambda _: []) catalog.get_subscription_models(background=False) catalog.record_subscription_model(credentials, "gpt-6-astra", available=False) catalog.record_subscription_model(credentials, "gpt-6-astra") @@ -185,12 +220,12 @@ def fetch(creds): catalog.record_subscription_model(creds, "gpt-5.6-sol", available=False) return ["gpt-5.6-sol", "gpt-5.6-terra"] - monkeypatch.setattr(catalog, "_fetch_models", fetch) + _patch_fetch(monkeypatch, fetch) assert catalog.get_subscription_models(background=False) == ["gpt-5.6-terra"] def test_expired_verification_is_not_offered(credentials, monkeypatch): - monkeypatch.setattr(catalog, "_fetch_models", lambda _: ["gpt-5.6-sol"]) + _patch_fetch(monkeypatch, lambda _: ["gpt-5.6-sol"]) catalog.get_subscription_models(background=False) catalog.record_subscription_model(credentials, "gpt-6-astra") path = auth.credentials_path().with_name("openai-models-cache.json") @@ -205,7 +240,7 @@ def fetch(creds): catalog.record_subscription_model(creds, "gpt-6-astra") return ["gpt-5.6-sol"] - monkeypatch.setattr(catalog, "_fetch_models", fetch) + _patch_fetch(monkeypatch, fetch) assert catalog.get_subscription_models(background=False) == ["gpt-6-astra", "gpt-5.6-sol"] @@ -214,7 +249,7 @@ def test_subscription_login_saves_discovered_default(credentials, monkeypatch): monkeypatch.setattr(auth, "has_codex_cli_credentials", lambda: True) monkeypatch.setattr(auth, "import_codex_cli_credentials", lambda: credentials) - monkeypatch.setattr(catalog, "_fetch_models", lambda _: ["gpt-5.6-terra", "gpt-5.6-luna"]) + _patch_fetch(monkeypatch, lambda _: ["gpt-5.6-terra", "gpt-5.6-luna"]) save = Mock() monkeypatch.setattr("src.config.set_api_key", save) monkeypatch.setattr("src.config.set_default_provider", Mock()) diff --git a/tests/server/test_model_provider_picker.py b/tests/server/test_model_provider_picker.py index 171f2fd1d..11d2aaaaa 100644 --- a/tests/server/test_model_provider_picker.py +++ b/tests/server/test_model_provider_picker.py @@ -702,6 +702,28 @@ def test_falls_back_to_the_session_when_arguments_are_missing(self): assert sess.last["model"] == "claude-opus-5" assert sess.last["supported"] is True + def test_live_openai_provider_decides_subscription_mode(self): + """The session's own OpenAI provider knows whether it rides the + ChatGPT login; that wins over inference from stored credentials (the + sandbox has none, so inference alone would answer API-key mode). + gpt-5.6-luna is where the two backends differ: max on the plan, + xhigh over an API key.""" + sess = _StubSession("openai") + sess.provider = SimpleNamespace(model="gpt-5.6-luna", _subscription_active=True) + _AgentSession._do_effort_options(sess, "r1", "openai", "gpt-5.6-luna") + assert sess.last["levels"][-1] == "max" + + sess.provider = SimpleNamespace(model="gpt-5.6-luna", _subscription_active=False) + _AgentSession._do_effort_options(sess, "r1", "openai", "gpt-5.6-luna") + assert sess.last["levels"][-1] == "xhigh" + + def test_another_providers_subscription_flag_is_not_borrowed(self): + """A Claude-subscription session's flag says nothing about OpenAI.""" + sess = _StubSession("anthropic") + sess.provider = SimpleNamespace(model="claude-opus-5", _subscription_active=True) + _AgentSession._do_effort_options(sess, "r1", "openai", "gpt-5.6-luna") + assert sess.last["levels"][-1] == "xhigh" + # ── the cross-provider signal ──────────────────────────────────────────────── diff --git a/tests/test_gpt6_registration.py b/tests/test_gpt6_registration.py new file mode 100644 index 000000000..cd1174c98 --- /dev/null +++ b/tests/test_gpt6_registration.py @@ -0,0 +1,63 @@ +"""GPT-6 (Astra / Sol / Luna) registration: context windows, pricing, lists. + +Numbers are from developers.openai.com/api/docs/models/gpt-6-{astra,sol,luna} +read 2026-09-24; pinned here so a later edit cannot drift silently. +""" + +from __future__ import annotations + +import pytest + +from src.models.configs import get_model_config +from src.providers import PROVIDER_INFO +from src.providers.openai_responses import supports_reasoning, supports_verbosity +from src.services.pricing import get_pricing + +GPT6 = ("gpt-6-astra", "gpt-6-sol", "gpt-6-luna") + +# (input, cached input, output) per 1M tokens, list price. +PUBLISHED = { + "gpt-6-astra": (10.00, 1.00, 50.00), + "gpt-6-sol": (2.00, 0.20, 10.00), + "gpt-6-luna": (0.10, 0.01, 0.50), +} + + +@pytest.mark.parametrize("model", GPT6) +def test_listed_for_the_openai_provider(model): + assert model in PROVIDER_INFO["openai"]["available_models"] + + +@pytest.mark.parametrize("model", GPT6) +def test_context_window_is_the_smaller_real_input_limit(model): + """922K max input on the API, 872K on the ChatGPT backend; over-estimating + makes auto-compact fire past the limit, so the row holds the smaller.""" + config = get_model_config(model) + assert config is not None and config.model_id == model + assert config.context_window == 872_000 + assert config.max_output_tokens == 128_000 + + +def test_unlisted_gpt6_variant_does_not_fall_to_the_272k_catch_all(): + assert get_model_config("gpt-6-sol-pro").context_window == 872_000 + + +@pytest.mark.parametrize("model", GPT6) +def test_wire_capabilities(model): + assert supports_reasoning(model) + assert supports_verbosity(model) + + +@pytest.mark.parametrize("model", GPT6) +def test_published_rates_and_the_272k_long_context_tier(model): + inp, cached, out = PUBLISHED[model] + short = get_pricing(model, input_tokens=1_000) + assert short["input"] * 1e6 == pytest.approx(inp) + assert short["cache_read"] * 1e6 == pytest.approx(cached) + assert short["output"] * 1e6 == pytest.approx(out) + # 2x input and cache rates, 1.5x output for the full request above 272K. + long = get_pricing(model, input_tokens=272_001) + assert long["input"] == pytest.approx(short["input"] * 2) + assert long["cache_read"] == pytest.approx(short["cache_read"] * 2) + assert long["output"] == pytest.approx(short["output"] * 1.5) + assert get_pricing(model, input_tokens=272_000) == short diff --git a/tests/test_openai_provider_routing.py b/tests/test_openai_provider_routing.py index eed37df3e..4b50fb4c4 100644 --- a/tests/test_openai_provider_routing.py +++ b/tests/test_openai_provider_routing.py @@ -70,6 +70,24 @@ def test_max_is_clamped_to_xhigh_for_openai() -> None: assert normalize_openai_effort("MAX") == "xhigh" +def test_max_passes_through_for_gpt6_on_the_api() -> None: + """developers.openai.com lists max for gpt-6-astra/sol/luna (2026-09-24); + gpt-5.6-luna still 400s on it, so only GPT-6 keeps it.""" + for model in ("gpt-6-astra", "gpt-6-sol", "gpt-6-luna"): + assert normalize_openai_effort("max", model) == "max" + assert normalize_openai_effort("max", "gpt-5.6-luna") == "xhigh" + + +def test_clamp_never_degrades_into_none() -> None: + """``none`` turns reasoning OFF — a different behaviour, not less of it.""" + from src.providers.openai_responses import clamp_effort + + assert clamp_effort("minimal", ("none", "low", "medium")) is None + assert clamp_effort("max", ("none", "low", "medium", "high")) == "high" + assert clamp_effort("none", ("none", "low")) == "none" + assert clamp_effort("ultra", ("low", "max")) is None + + def test_known_efforts_pass_through_and_junk_is_dropped() -> None: for effort in ("none", "low", "medium", "high", "xhigh"): assert normalize_openai_effort(effort) == effort diff --git a/tests/test_openai_subscription.py b/tests/test_openai_subscription.py index 6aed987ec..97d0acfd5 100644 --- a/tests/test_openai_subscription.py +++ b/tests/test_openai_subscription.py @@ -630,20 +630,53 @@ def _abort_after_first_chunk(): def test_effort_setting_reaches_reasoning_body(monkeypatch) -> None: """/effort (injected as extra_body.reasoning_effort by the agent-server - wrapper) wins over the default; xhigh clamps to high.""" - provider = _subscription_provider(monkeypatch) - body = provider._subscription_request_body( - [{"role": "user", "content": "hi"}], None, - extra_body={"reasoning_effort": "low"}, - ) - assert body["reasoning"]["effort"] == "low" - body = provider._subscription_request_body( - [{"role": "user", "content": "hi"}], None, - extra_body={"reasoning_effort": "xhigh"}, + wrapper) wins over the default and is clamped PER MODEL to what the + ChatGPT backend accepts (probed live 2026-09-24): gpt-5.5 takes xhigh + but 400s on max; gpt-6 takes max; ``ultra`` is catalog-only and 400s.""" + monkeypatch.setattr( + "src.providers.openai_subscription_models.load_credentials", lambda: None, ) - assert body["reasoning"]["effort"] == "high" + provider = _subscription_provider(monkeypatch) + + def effort(model, requested=None): + kwargs = {"model": model} + if requested is not None: + kwargs["extra_body"] = {"reasoning_effort": requested} + body = provider._subscription_request_body( + [{"role": "user", "content": "hi"}], None, **kwargs, + ) + return body["reasoning"]["effort"] + + assert effort("gpt-5.5", "low") == "low" + assert effort("gpt-5.5", "xhigh") == "xhigh" + assert effort("gpt-5.5", "max") == "xhigh" + assert effort("gpt-6-astra", "max") == "max" + assert effort("gpt-6-sol", "ultra") == "medium" # not a wire value: default + assert effort("gpt-5.4", "xhigh") == "high" + assert effort("gpt-6-luna") == "medium" + + +def test_subscription_effort_follows_the_cached_catalog(monkeypatch, tmp_path) -> None: + """A login's catalog ``supported_reasoning_levels`` outranks the static + table, minus ``ultra`` (advertised, but a 400 on the wire).""" + from src.providers import openai_subscription_models as catalog + + monkeypatch.setenv("CLAWCODEX_CONFIG_DIR", str(tmp_path)) + creds = _credentials() + monkeypatch.setattr(catalog, "load_credentials", lambda: creds) + monkeypatch.setattr(catalog, "_fetch_catalog", lambda _c: ( + ["gpt-6-nova"], + {"gpt-6-nova": catalog._wire_efforts([ + {"effort": "low"}, {"effort": "medium"}, {"effort": "ultra"}, + ])}, + )) + assert catalog.get_subscription_models(background=False, force=True) == ["gpt-6-nova"] + assert catalog.get_subscription_effort_levels("gpt-6-nova") == ("low", "medium") + + provider = _subscription_provider(monkeypatch) body = provider._subscription_request_body( [{"role": "user", "content": "hi"}], None, + model="gpt-6-nova", extra_body={"reasoning_effort": "max"}, ) assert body["reasoning"]["effort"] == "medium" From 4bf9762b9f3a1acc0918b875786fcb7c14f76840 Mon Sep 17 00:00:00 2001 From: Eric Lee Date: Thu, 24 Sep 2026 16:31:21 -0700 Subject: [PATCH 2/2] fix(models): size gpt-5.6 context at 872K, not the advertised 1.05M The public API caps gpt-5.6 input at 922K and the ChatGPT subscription catalog caps it at 872K (max_context_window). At 1.05M, auto-compact fired near 998K, past both limits, into an overflow error reactive compaction does not recognise. Same reasoning as the GPT-6 rows. Co-Authored-By: Claude Opus 5.5 (1M context) --- eval/harbor/RUN_LUNA_TB21.md | 2 +- src/models/configs.py | 24 +++++++++++++--------- tests/providers/test_openrouter_catalog.py | 6 +++--- tests/test_query_openai_compat_effort.py | 10 ++++----- 4 files changed, 23 insertions(+), 19 deletions(-) diff --git a/eval/harbor/RUN_LUNA_TB21.md b/eval/harbor/RUN_LUNA_TB21.md index dcb721c1c..29d7bce9a 100644 --- a/eval/harbor/RUN_LUNA_TB21.md +++ b/eval/harbor/RUN_LUNA_TB21.md @@ -8,7 +8,7 @@ clawcodex Harbor adapter. | | | |---|---| | id | `openai/gpt-5.6-luna` (a `-pro` variant exists with identical specs) | -| context | 1,050,000 tokens (registered as 1,048,576 = 2^20, matching the sibling gpt-5.6 rows; under-reading is the safe direction) | +| context | 1,050,000 advertised; 922K max input (registered as 872,000 since 2026-09-24, the smaller of the API and ChatGPT-subscription input limits; runs past ~822K now compact where older builds overflowed, so such runs are not like-for-like with earlier baselines) | | max output | 128,000 tokens | | price | $0.10/M in, $0.60/M out — doubling to $0.20 / $0.90 above 272K prompt tokens | | reasoning | `reasoning_effort` supported; verified honored, not just accepted — reasoning tokens rise low 148 → medium 154 → high 266 → max 516 on a fixed prompt | diff --git a/src/models/configs.py b/src/models/configs.py index d00b3842c..60841f959 100644 --- a/src/models/configs.py +++ b/src/models/configs.py @@ -754,31 +754,35 @@ class ModelConfig: max_output_tokens=128_000, ), # GPT-5.6 (Sol / Terra / Luna — three durable capability tiers on one - # generation, 1.05M context each). These keys sit BEFORE "gpt-5.5" + # generation, 1.05M advertised context each). These keys sit BEFORE "gpt-5.5" # deliberately: ``get_model_config``'s prefix fallback walks in insertion # order and each of these has base "gpt-5.6" (rsplit on the last "-"), so # putting them first is what lets an unlisted variant like - # ``gpt-5.6-sol-pro`` resolve to 1.05M instead of falling through to the + # ``gpt-5.6-sol-pro`` resolve to the 5.6 window instead of falling through to the # 272K catch-all. The bare ``gpt-5.6`` alias is deliberately NOT here — # see below. - # KNOWN GAP (not fixed here): the subscription catalog also caps these at - # ``max_context_window: 872000``, so 1.05M over-estimates on that path. + # Window is 872K, not the advertised 1.05M, for the same reason as the + # GPT-6 rows above: the public API caps INPUT at 922K + # (developers.openai.com/api/docs/models/gpt-5.6-sol) and the ChatGPT + # subscription catalog caps it at ``max_context_window: 872000``. At + # 1.05M auto-compact fired near 998K — past both limits, into an + # overflow error that reactive compaction does not recognise. "gpt-5.6-sol": ModelConfig( model_id="gpt-5.6-sol", display_name="GPT-5.6 Sol", - context_window=1_048_576, + context_window=872_000, max_output_tokens=128_000, ), "gpt-5.6-terra": ModelConfig( model_id="gpt-5.6-terra", display_name="GPT-5.6 Terra", - context_window=1_048_576, + context_window=872_000, max_output_tokens=128_000, ), "gpt-5.6-luna": ModelConfig( model_id="gpt-5.6-luna", display_name="GPT-5.6 Luna", - context_window=1_048_576, + context_window=872_000, max_output_tokens=128_000, ), # The same model as the row above, under its OpenRouter id. This table is @@ -803,7 +807,7 @@ class ModelConfig: "openai/gpt-5.6-luna": ModelConfig( model_id="openai/gpt-5.6-luna", display_name="GPT-5.6 Luna", - context_window=1_048_576, + context_window=872_000, max_output_tokens=128_000, ), "gpt-5.5": ModelConfig( @@ -814,14 +818,14 @@ class ModelConfig: ), # ``gpt-5.6`` is OpenAI's alias for Sol. Its base is "gpt" (not # "gpt-5.6"), so it would become the catch-all for EVERY unknown gpt id if - # it preceded "gpt-5.5" — handing them a 1.05M window. Over-estimating + # it preceded "gpt-5.5" — handing them an 872K window. Over-estimating # overflows the context; under-estimating only compacts early, so the # catch-all must stay on the 272K entry. Exact lookups are unaffected by # position: ``get_model_config`` checks for an exact key first. "gpt-5.6": ModelConfig( model_id="gpt-5.6", display_name="GPT-5.6 (Sol)", - context_window=1_048_576, + context_window=872_000, max_output_tokens=128_000, ), "gpt-5.4": ModelConfig( diff --git a/tests/providers/test_openrouter_catalog.py b/tests/providers/test_openrouter_catalog.py index abe01ca85..907f4ec8b 100644 --- a/tests/providers/test_openrouter_catalog.py +++ b/tests/providers/test_openrouter_catalog.py @@ -121,16 +121,16 @@ def test_carries_the_full_context_window(self, model): config = get_model_config(model) assert config is not None, model - assert config.context_window == 1_048_576, model + assert config.context_window == 872_000, model def test_an_unlisted_5_6_variant_inherits_the_5_6_window(self): from src.models.configs import get_model_config - assert get_model_config("gpt-5.6-sol-pro").context_window == 1_048_576 + assert get_model_config("gpt-5.6-sol-pro").context_window == 872_000 def test_an_unknown_gpt_id_still_gets_the_conservative_catch_all(self): """The ordering invariant. ``gpt-5.6``'s prefix base is "gpt", so - placing it above ``gpt-5.5`` would hand every unknown gpt id a 1.05M + placing it above ``gpt-5.5`` would hand every unknown gpt id an 872K window. Over-estimating overflows the context; under-estimating only compacts early, so the catch-all must stay on the 272K entry.""" from src.models.configs import get_model_config diff --git a/tests/test_query_openai_compat_effort.py b/tests/test_query_openai_compat_effort.py index 5220773f1..b154e9643 100644 --- a/tests/test_query_openai_compat_effort.py +++ b/tests/test_query_openai_compat_effort.py @@ -227,7 +227,7 @@ class TestLunaOpenRouterIdRegistration(unittest.TestCase): reaches the provider when the terminal-bench harness runs this model.""" def test_openrouter_id_resolves_to_the_real_window(self): - self.assertEqual(get_context_window_for_model(LUNA), 1_048_576) + self.assertEqual(get_context_window_for_model(LUNA), 872_000) self.assertEqual(get_model_max_output_tokens(LUNA), 128_000) def test_not_the_200k_default(self): @@ -249,7 +249,7 @@ def test_pro_variant_resolves_through_the_prefix_fallback(self): for model_id in ("openai/gpt-5.6-luna-pro", "gpt-5.6-luna-pro"): with self.subTest(model=model_id): self.assertEqual( - get_context_window_for_model(model_id), 1_048_576 + get_context_window_for_model(model_id), 872_000 ) def test_decision_1_upheld_no_vendor_prefix_stripping(self): @@ -272,8 +272,8 @@ def test_the_new_row_does_not_perturb_other_ids(self): # Inside the claimed prefix — these DO now resolve, to the same # window their bare equivalents get from #773's rows. Pinned so # the size of the claim is a fact rather than a comment. - "openai/gpt-5.6-mini": 1_048_576, - "openai/gpt-5.6-sol": 1_048_576, + "openai/gpt-5.6-mini": 872_000, + "openai/gpt-5.6-sol": 872_000, # Outside it — must stay unresolved. "openai/gpt-4o": None, "openai/gpt-5.5": None, @@ -302,7 +302,7 @@ def test_user_model_limits_override_still_reachable(self): The shadowed surface is the row's PREFIX, not just its key: the added row's derived base is ``openai/gpt-5.6``, so a ``modelLimits`` entry for any ``openai/gpt-5.6*`` id now loses to the table (they all get - 1,048,576, matching what their bare equivalents already get — so the + 872,000, matching what their bare equivalents already get — so the qualified namespace mirrors the bare one rather than diverging). What must NOT happen is that claim spreading further, which is what this pins: an override outside the prefix still wins."""