From 4e991d8c9212b42983630bc5e29c1c494bbf54a9 Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Mon, 5 Oct 2026 16:04:42 +0800 Subject: [PATCH 1/3] feat(providers): centralize typed model capability constraints --- .github/workflows/ci.yml | 8 +- agent_core/__init__.py | 3 + agent_core/model_capabilities.py | 174 ++++++++++++++++++++++ agent_core/providers/anthropic.py | 16 +- agent_core/providers/aux_builder.py | 26 +++- agent_core/providers/protocol_client.py | 77 ++++++---- agent_core/runtime/loop/model_profile.py | 11 +- changes/model-capabilities.feature.md | 1 + docs/provider-substrate-boundary.md | 67 +++++++++ tests/test_model_capabilities.py | 178 +++++++++++++++++++++++ 10 files changed, 528 insertions(+), 33 deletions(-) create mode 100644 agent_core/model_capabilities.py create mode 100644 changes/model-capabilities.feature.md create mode 100644 tests/test_model_capabilities.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index aae2f4c..2640115 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -4,7 +4,7 @@ on: push: branches: [main] pull_request: - branches: [main] + branches: ['**'] types: [opened, synchronize, reopened, labeled, unlabeled] jobs: @@ -25,7 +25,11 @@ jobs: - run: uv run pytest -q # Keep current-model request/response handling usable at our SDK floor, # including SDK 0.x (httpx) versus 1.x (httpx2) parsing and transport. - - run: uv run --isolated --with 'anthropic[bedrock]==0.69.0' --extra dev pytest -q tests/test_anthropic_latest_models.py tests/test_provider_native_clients.py tests/test_provider_protocol_reasoning.py + - run: >- + uv run --isolated --with 'anthropic[bedrock]==0.69.0' --extra dev pytest -q + tests/test_anthropic_latest_models.py tests/test_provider_native_clients.py + tests/test_provider_protocol_reasoning.py tests/test_model_capabilities.py + tests/test_aux_builder.py # The gate pins tiktoken's own cache key and content hash, and only a real # tiktoken can falsify them. Scope that optional dependency to this contract # test; the isolated run uses the tokenizer version pinned in uv.lock. diff --git a/agent_core/__init__.py b/agent_core/__init__.py index c182b4e..0aea91c 100644 --- a/agent_core/__init__.py +++ b/agent_core/__init__.py @@ -12,6 +12,7 @@ tool_msg, user_msg, ) +from agent_core.model_capabilities import ModelCapabilities, resolve_model_capabilities from agent_core.types import TaskStatus __all__ = [ @@ -21,11 +22,13 @@ "LLMClient", "LLMResponse", "Message", + "ModelCapabilities", "StreamDelta", "TaskStatus", "ToolCall", "assistant_msg", "build_execution_scope", + "resolve_model_capabilities", "system_msg", "tool_msg", "user_msg", diff --git a/agent_core/model_capabilities.py b/agent_core/model_capabilities.py new file mode 100644 index 0000000..551ee49 --- /dev/null +++ b/agent_core/model_capabilities.py @@ -0,0 +1,174 @@ +"""Typed model request constraints, separate from credentials and deployment catalogs. + +None means unknown, an empty set means unsupported. Resolution is explicit and +local to a client/profile: protocol defaults < exact model facts < host overrides. +No network access or mutable process-wide registration is performed here. +""" + +from __future__ import annotations + +import re +from collections.abc import Iterable, Mapping +from dataclasses import dataclass, replace +from types import MappingProxyType +from typing import Literal, cast + +WireProtocol = Literal["chat_completions", "anthropic", "responses", "bedrock"] +ThinkingMode = Literal["adaptive", "enabled", "disabled"] +SignatureBinding = Literal["none", "model", "conversation_prefix"] + + +@dataclass(frozen=True) +class ModelCapabilities: + thinking_modes: frozenset[str] | None = None + thinking_required: bool | None = None + effort_levels: frozenset[str] | None = None + default_effort: str | None = None + sampling_parameters: frozenset[str] | None = None + tool_choice_modes: frozenset[str] | None = None + thinking_signature_binding: SignatureBinding | None = None + max_input_tokens: int | None = None + max_output_tokens: int | None = None + source_urls: tuple[str, ...] = () + verified_on: str = "" + overridden_fields: frozenset[str] = frozenset() + + def validate_request( + self, *, model: str, thinking: Mapping[str, object] | None, + effort: str = "", max_tokens: int | None = None, + ) -> None: + """Reject known unsupported settings; leave unknown capabilities alone.""" + mode = thinking.get("type") if thinking is not None else None + if mode is not None and self.thinking_modes is not None and (not isinstance(mode, str) or mode not in self.thinking_modes): + raise ValueError(f"{model}: thinking mode {mode!r} is unsupported; use {sorted(self.thinking_modes)}") + if mode == "disabled" and self.thinking_required is True: + raise ValueError(f"{model}: thinking is always enabled") + if effort and self.effort_levels is not None and effort not in self.effort_levels: + raise ValueError(f"{model}: effort {effort!r} is unsupported; use {sorted(self.effort_levels)}") + if max_tokens is not None and self.max_output_tokens is not None and max_tokens > self.max_output_tokens: + raise ValueError(f"{model}: max_tokens={max_tokens} exceeds the model limit {self.max_output_tokens}") + + +_CURRENT_CLAUDE = ModelCapabilities( + thinking_modes=frozenset({"adaptive"}), thinking_required=True, + effort_levels=frozenset({"low", "medium", "high", "xhigh", "max"}), + default_effort="high", sampling_parameters=frozenset(), + tool_choice_modes=frozenset({"auto", "none"}), + thinking_signature_binding="conversation_prefix", + max_input_tokens=1_000_000, max_output_tokens=128_000, + source_urls=( + "https://platform.claude.com/docs/en/models/fable-5-1/migration-guide", + "https://platform.claude.com/docs/en/models/fable-5-1/overview", + "https://platform.claude.com/docs/en/build-with-claude/effort", + ), verified_on="2026-10-05", +) +_LEGACY_CLAUDE = ModelCapabilities( + thinking_modes=frozenset({"enabled", "disabled"}), thinking_required=False, + effort_levels=frozenset(), + source_urls=("https://platform.claude.com/docs/en/build-with-claude/extended-thinking",), + verified_on="2026-10-05", +) + +# These are request facts, not an endpoint, credential, price, or routing catalog. +# Only documented exact IDs/aliases match; a new version never inherits a guessed +# capability merely because its model name starts with an older one. +MODEL_CAPABILITIES: Mapping[str, ModelCapabilities] = MappingProxyType({ + "claude-fable-5-1": _CURRENT_CLAUDE, + "claude-opus-5-5": replace( + _CURRENT_CLAUDE, default_effort="medium", source_urls=( + "https://platform.claude.com/docs/en/models/opus-5-5/migration-guide", + "https://platform.claude.com/docs/en/models/opus-5-5/overview", + "https://platform.claude.com/docs/en/build-with-claude/effort", + ), + ), + "claude-sonnet-4-5": _LEGACY_CLAUDE, + "claude-sonnet-4-5-20250929": _LEGACY_CLAUDE, + "claude-haiku-4-5": _LEGACY_CLAUDE, + "claude-haiku-4-5-20251001": _LEGACY_CLAUDE, + "claude-opus-4-5": replace(_LEGACY_CLAUDE, + effort_levels=frozenset({"low", "medium", "high"}), default_effort="high", + source_urls=(*_LEGACY_CLAUDE.source_urls, "https://platform.claude.com/docs/en/build-with-claude/effort"), + ), + "claude-opus-4-5-20251101": replace(_LEGACY_CLAUDE, + effort_levels=frozenset({"low", "medium", "high"}), default_effort="high", + source_urls=(*_LEGACY_CLAUDE.source_urls, "https://platform.claude.com/docs/en/build-with-claude/effort"), + ), +}) + +_SET_FIELDS = frozenset({"thinking_modes", "effort_levels", "sampling_parameters", "tool_choice_modes"}) +_LIMIT_FIELDS = frozenset({"max_input_tokens", "max_output_tokens"}) +_OVERRIDE_FIELDS = _SET_FIELDS | _LIMIT_FIELDS | { + "thinking_required", "default_effort", "thinking_signature_binding", +} + + +def _string_set(key: str, value: object) -> frozenset[str]: + if not isinstance(value, (list, tuple, set, frozenset)): + raise ValueError(f"{key} must be a collection of strings or null") + members = cast(Iterable[object], value) + strings: set[str] = set() + for member in members: + if not isinstance(member, str): + raise ValueError(f"{key} must be a collection of strings or null") + strings.add(member) + return frozenset(strings) + + +def _canonical_model_id(model_id: str, protocol: WireProtocol) -> str: + if protocol != "bedrock": + return model_id + # Bedrock inference-profile regional prefixes and documented version suffix. + # Arbitrary proxy aliases and ARNs require explicit host overrides. + match = re.fullmatch(r"(?:(?:us|eu|apac|global)\.)?anthropic\.(claude-[a-z0-9-]+?)(?:-v\d+:\d+)?", model_id) + return match.group(1) if match else model_id + + +def resolve_model_capabilities( + model_id: str, *, protocol: WireProtocol, + overrides: Mapping[str, object] | None = None, +) -> ModelCapabilities: + """Resolve native model facts and a validated, per-deployment override. + + A Claude name served over Chat Completions does not imply native Messages + capabilities. Proxies and custom model aliases must declare their overrides. + Explicit None clears a known fact back to unknown. + """ + capabilities = ModelCapabilities() + if protocol in ("anthropic", "bedrock"): + capabilities = MODEL_CAPABILITIES.get(_canonical_model_id(model_id, protocol), capabilities) + if overrides is None: + return capabilities + unknown = set(overrides) - _OVERRIDE_FIELDS + if unknown: + raise ValueError(f"unknown model capability fields: {sorted(unknown)}") + changes: dict[str, object] = {} + for key, value in overrides.items(): + if value is None: + changes[key] = None + elif key in _SET_FIELDS: + changes[key] = _string_set(key, value) + elif key in _LIMIT_FIELDS: + if not isinstance(value, int) or isinstance(value, bool) or value <= 0: + raise ValueError(f"{key} must be a positive integer or null") + changes[key] = value + elif key == "thinking_required": + if not isinstance(value, bool): + raise ValueError("thinking_required must be a boolean or null") + changes[key] = value + elif key == "thinking_signature_binding": + if value not in ("none", "model", "conversation_prefix"): + raise ValueError("thinking_signature_binding must be none, model, conversation_prefix or null") + changes[key] = value + elif key == "default_effort": + if not isinstance(value, str): + raise ValueError("default_effort must be a string or null") + changes[key] = value + result = replace(capabilities, **changes, overridden_fields=frozenset(overrides)) + if result.default_effort is not None and result.effort_levels is not None and result.default_effort not in result.effort_levels: + raise ValueError("default_effort must belong to effort_levels; override both fields together") + if result.thinking_required is True and result.thinking_modes is not None and "disabled" in result.thinking_modes: + raise ValueError("thinking_required conflicts with disabled thinking mode") + return result + + +__all__ = ["MODEL_CAPABILITIES", "ModelCapabilities", "SignatureBinding", "ThinkingMode", "WireProtocol", "resolve_model_capabilities"] diff --git a/agent_core/providers/anthropic.py b/agent_core/providers/anthropic.py index af48724..2f74401 100644 --- a/agent_core/providers/anthropic.py +++ b/agent_core/providers/anthropic.py @@ -29,6 +29,7 @@ from agent_core.llm import LLMClient, LLMResponse, StreamDelta from agent_core.messages import Message, ToolCall, text_of +from agent_core.model_capabilities import ModelCapabilities, resolve_model_capabilities from agent_core.providers.finish_reason import normalize_finish_reason logger = logging.getLogger(__name__) @@ -54,6 +55,7 @@ def __init__( effort: str = "", bedrock: bool = False, default_headers: dict[str, str] | None = None, + capabilities: ModelCapabilities | None = None, ) -> None: self.model = model self.default_temperature = temperature @@ -66,6 +68,13 @@ def __init__( # (low|medium|high|xhigh|max) → ``output_config.effort`` via extra_body. self._thinking = thinking or None self._effort = (effort or "").strip() + self.capabilities = capabilities or resolve_model_capabilities( + model, protocol="bedrock" if bedrock else "anthropic", + ) + self.capabilities.validate_request( + model=model, thinking=self._thinking, effort=self._effort, + max_tokens=max_tokens, + ) # Transport: ``bedrock`` swaps AsyncAnthropic (``/v1/messages`` + # ``x-api-key``) for the AWS Bedrock runtime (``/model/{id}/invoke`` + # ``anthropic_version`` body stamp) authenticated with a Bedrock API Key @@ -97,6 +106,11 @@ def _build_kwargs( timeout: float | None, ) -> dict[str, Any]: """Shared request-shape builder for :meth:`chat` and :meth:`stream`.""" + output_limit = max_tokens or self.default_max_tokens or 4096 + self.capabilities.validate_request( + model=self.model, thinking=self._thinking, effort=self._effort, + max_tokens=output_limit, + ) system, msgs = _split_system(messages) # ``_to_anthropic_msg`` returns None for a message with nothing # sendable (a contentless assistant turn); those are dropped. @@ -114,7 +128,7 @@ def _build_kwargs( kwargs: dict[str, Any] = { "model": self.model, "messages": [converted for converted, _ in pairs], - "max_tokens": max_tokens or self.default_max_tokens or 4096, + "max_tokens": output_limit, } if system: kwargs["system"] = system diff --git a/agent_core/providers/aux_builder.py b/agent_core/providers/aux_builder.py index 44a752e..910a73e 100644 --- a/agent_core/providers/aux_builder.py +++ b/agent_core/providers/aux_builder.py @@ -7,6 +7,8 @@ from collections.abc import Callable, Mapping from typing import Any +from agent_core.model_capabilities import resolve_model_capabilities + logger = logging.getLogger(__name__) type ClientFactory = Callable[..., Any] @@ -143,12 +145,34 @@ def _build_anthropic( "effort": str(section.get("effort") or ""), "bedrock": bedrock, } + overrides = section.get("model_capabilities") + if overrides is not None and not isinstance(overrides, Mapping): + raise ValueError("model_capabilities must be a mapping") + capabilities = resolve_model_capabilities( + str(section["model"]), protocol="bedrock" if bedrock else "anthropic", + overrides=overrides, + ) + # Preserve the caller's intent for validation before legacy conversion + # maps an explicit disabled/off configuration to no thinking field. + raw_thinking = section.get("thinking") + validation_thinking = kwargs["thinking"] + if isinstance(raw_thinking, Mapping): + kind = str(raw_thinking.get("type") or "").strip().lower() + if kind in {"disabled", "off", "none", "false"}: + validation_thinking = {"type": "disabled"} + maximum = section.get("max_completion_tokens") or section.get("max_tokens") + capabilities.validate_request( + model=str(section["model"]), thinking=validation_thinking, + effort=kwargs["effort"].strip(), + max_tokens=int(maximum) if maximum is not None else None, + ) + if overrides is not None: + kwargs["capabilities"] = capabilities if section.get("base_url"): kwargs["base_url"] = section["base_url"] default_headers = _headers(section.get("extra_headers")) if default_headers: kwargs["default_headers"] = default_headers - maximum = section.get("max_completion_tokens") or section.get("max_tokens") if maximum is not None: kwargs["max_tokens"] = int(maximum) return self._anthropic_factory(**kwargs) diff --git a/agent_core/providers/protocol_client.py b/agent_core/providers/protocol_client.py index a6cce77..ecceea9 100644 --- a/agent_core/providers/protocol_client.py +++ b/agent_core/providers/protocol_client.py @@ -23,6 +23,7 @@ from typing import Any, get_args from agent_core.llm import LLMClient +from agent_core.model_capabilities import resolve_model_capabilities # Single source of truth for the valid ``llm.protocol`` values. Duplicating the # set here would let the two drift, which is how a protocol becomes buildable @@ -124,47 +125,66 @@ def _build_anthropic( ``output_config.effort``. ``default_headers`` (gateway routing / auth headers) is merged over ``X-Title``, as in the Responses builder. - ``thinking_type`` selects the request shape (default ``adaptive``). Live- - verified against api.anthropic.com + Bedrock 2026-07-09 (see - ``temp/2026-07-09_reasoning-protocol-live-verification.md``); matches the - official matrix at platform.claude.com/docs/en/build-with-claude/adaptive-thinking: - - - ``adaptive`` (DEFAULT) — the RECOMMENDED mode for all current Claude - (Opus 4.6/4.7/4.8, Sonnet 4.6/5, Fable/Mythos), and the ONLY mode on the - newest (Opus 4.7/4.8, Sonnet 5) — ``enabled`` is rejected there with 400. - Emits ``thinking={"type":"adaptive"}`` + ``thinking_display`` (default - ``summarized`` so the readable thinking text is captured; the newest - models default ``display`` to ``omitted`` = empty ``thinking`` field with - the ``signature`` still present for replay). ``effort`` is forwarded only - in this mode (it is an adaptive-only knob; the oldest models 400 on - ``enabled``+effort). - - ``enabled`` — LEGACY opt-in for models older than Opus 4.6 / Sonnet 4.6 - (Sonnet 4.5, Opus 4.5, …), which reject ``adaptive`` with 400. Emits - ``thinking={"type":"enabled","budget_tokens":N}`` (``N`` from - ``thinking_budget_tokens``, default 8192, clamped to ``[1024, max_tokens-1]`` - since Anthropic requires ``budget_tokens < max_tokens``). When the - configured ``max_tokens`` is too small for the 1024 floor, ``max_tokens`` - is RAISED to ``budget + 1`` rather than emitting an invalid pair — see - :func:`_enabled_thinking_budget`. ``effort`` is - NOT sent (budget_tokens is the control knob here; oldest models 400 on it). - Deprecated on Opus 4.6 / Sonnet 4.6 per Anthropic. + ``thinking_type`` overrides the model-aware default: prefer adaptive, use + enabled for known manual-thinking-only models, or omit thinking when the + host declares no supported mode. Unknown models retain the protocol's + previous adaptive default. Adaptive display defaults to summarized so + readable progress is retained even when the API defaults to omitted. + + Enabled thinking uses :func:`_enabled_thinking_budget` to keep the legacy + 1024-token floor below max_tokens. It forwards effort only when the model + capability record establishes support (for example Opus 4.5); unknown + legacy models retain the previous behavior of omitting effort. Unsupported + explicit settings raise before creating the SDK client. Hosts can override + facts through the profile's ``model_capabilities`` mapping. """ from agent_core.providers.anthropic import AnthropicClient max_tokens = int(cfg.get("max_tokens", 32768)) - ttype = str(cfg.get("thinking_type", "adaptive")).strip().lower() + overrides = cfg.get("model_capabilities") + if overrides is not None and not isinstance(overrides, dict): + raise ValueError("model_capabilities must be a mapping") + capabilities = resolve_model_capabilities( + cfg["model"], protocol="bedrock" if bedrock else "anthropic", overrides=overrides, + ) + default_mode: str | None = "adaptive" + if capabilities.thinking_modes is not None and "adaptive" not in capabilities.thinking_modes: + default_mode = ( + "enabled" if "enabled" in capabilities.thinking_modes + else "disabled" if "disabled" in capabilities.thinking_modes else None + ) + raw_mode = cfg.get("thinking_type") + if raw_mode is None or (isinstance(raw_mode, str) and not raw_mode.strip()): + ttype = default_mode + elif isinstance(raw_mode, str): + ttype = raw_mode.strip().lower() + else: + raise ValueError("thinking_type must be a string or null") + if ttype is not None and ttype not in ("adaptive", "enabled", "disabled"): + raise ValueError(f"unknown thinking_type {ttype!r}; use adaptive, enabled, or disabled") + capabilities.validate_request( + model=cfg["model"], thinking={"type": ttype} if ttype else None, + effort=_effort_str(cfg), + ) + thinking: dict[str, Any] | None if ttype == "enabled": budget, max_tokens = _enabled_thinking_budget( int(cfg.get("thinking_budget_tokens", 8192)), max_tokens, ) - thinking: dict[str, Any] = {"type": "enabled", "budget_tokens": budget} - effort = "" - else: + thinking = {"type": "enabled", "budget_tokens": budget} + effort = _effort_str(cfg) if capabilities.effort_levels else "" + elif ttype == "adaptive": thinking = {"type": "adaptive"} display = cfg.get("thinking_display", "summarized") if isinstance(display, str) and display.strip(): thinking["display"] = display.strip() effort = _effort_str(cfg) + elif ttype == "disabled": + thinking = {"type": "disabled"} + effort = _effort_str(cfg) + else: + thinking = None + effort = _effort_str(cfg) return AnthropicClient( model=cfg["model"], api_key=cfg.get("api_key", "dummy"), @@ -174,6 +194,7 @@ def _build_anthropic( effort=effort, default_headers={"X-Title": title, **(cfg.get("default_headers") or {})}, bedrock=bedrock, + capabilities=capabilities, ) diff --git a/agent_core/runtime/loop/model_profile.py b/agent_core/runtime/loop/model_profile.py index 583c73d..f2cc0d9 100644 --- a/agent_core/runtime/loop/model_profile.py +++ b/agent_core/runtime/loop/model_profile.py @@ -19,6 +19,8 @@ assistant_msg, assistant_msg_with_reasoning, ) +from agent_core.model_capabilities import ModelCapabilities, resolve_model_capabilities +from agent_core.model_capabilities import WireProtocol as WireProtocol logger = logging.getLogger(__name__) @@ -41,7 +43,6 @@ # Wire protocol the client speaks. Named alias so config readers can declare it # instead of returning a bare str that every ModelProfile call site then rejects. -WireProtocol = Literal["chat_completions", "anthropic", "responses", "bedrock"] _VALID_PROTOCOLS: frozenset[str] = frozenset(get_args(WireProtocol)) @@ -209,6 +210,14 @@ class ModelProfile: # ``content_block`` so the parser keeps the verbatim blocks (signatures / # encrypted_content) for faithful multi-turn replay + trajectory. protocol: WireProtocol = "chat_completions" + # Request constraints use the same records as provider construction. + # context_window above remains the host's operational context budget; + # capabilities.max_input_tokens describes the provider's maximum. + capabilities: ModelCapabilities | None = None + + @property + def request_capabilities(self) -> ModelCapabilities: + return self.capabilities or resolve_model_capabilities(self.model_id, protocol=self.protocol) @dataclass diff --git a/changes/model-capabilities.feature.md b/changes/model-capabilities.feature.md new file mode 100644 index 0000000..7a506c7 --- /dev/null +++ b/changes/model-capabilities.feature.md @@ -0,0 +1 @@ +Add typed, source-documented model capability records shared by model profiles and Anthropic request construction. Known models select compatible thinking defaults and reject unsupported thinking, effort, or output-token settings before API calls; unknown models retain pass-through behavior and hosts can override deployment-specific facts without global mutable state. Enable CI for stacked pull requests. diff --git a/docs/provider-substrate-boundary.md b/docs/provider-substrate-boundary.md index 91d315b..b234c7a 100644 --- a/docs/provider-substrate-boundary.md +++ b/docs/provider-substrate-boundary.md @@ -107,3 +107,70 @@ Unknown detail fields/categories pass through, including with older SDKs. Observers receive `TurnContext.finish_reason` and `TurnContext.stop_details`; the trajectory observer writes them to JSON snapshots and JSONL events. This telemetry does not change loop termination or choose a fallback model. + +## Model capabilities and deployment overrides + +`agent_core.model_capabilities` holds immutable request facts keyed by exact +model IDs and documented aliases. Each record has source URLs and a verification +date. `ModelProfile.request_capabilities` and the Anthropic client use this same +resolver; the host still owns endpoints, credentials, routing, prices, and model +catalog discovery. No network request occurs during resolution. + +`None` means unknown; an empty set means a feature has no supported values. +Unknown IDs, newer versions, and Claude names served over Chat Completions do +not inherit native Anthropic restrictions. Bedrock's documented regional model +prefixes and version suffixes resolve to the same underlying model facts; the +outbound model ID is never rewritten. Arbitrary aliases and ARNs require host +overrides. This first table covers Fable 5.1, Opus 5.5, and the legacy 4.5 models; +other providers and models remain unknown until verified facts are added. + +Known thinking modes select builder defaults and reject unsupported explicit +modes. Effort and output limits are validated in direct clients, native profile +clients, and per-call overrides. No parameter is silently clamped. The existing +legacy thinking-budget adjustment remains in place. Empty thinking support +omits the thinking field; unknown native models keep the prior adaptive default. +Sampling, tool-choice, and signature-binding facts are available to hosts; +protocol field conversion, SDK transport limitations, and error recovery stay +in adapters. The Anthropic adapter still omits sampling fields for SDK 1.x +compatibility. This PR does not add forced tool-choice parameters. + +Resolution applies verified model facts after unknown protocol defaults, then +applies a per-client host override. The `model_capabilities` profile key is a +mapping with the capability field names; omitted fields inherit, explicit null +clears a fact to unknown, and empty lists declare unsupported values. Unknown +keys, malformed values, and contradictory defaults raise `ValueError`. +`overridden_fields` identifies which facts the host supplied; `source_urls` +describes inherited facts, not proof of a host override. Hosts own the evidence +for their deployment overrides. Records never mutate a global registry. + +```yaml +llm: + protocol: anthropic + model: claude-opus-5-5 + effort: medium + model_capabilities: + max_output_tokens: 32768 # gateway limit overrides the model maximum +``` + +For a custom alias, specify its supported modes and effort levels explicitly: + +```python +from agent_core import resolve_model_capabilities +from agent_core.runtime.loop.model_profile import ModelProfile + +caps = resolve_model_capabilities( + "gateway-alias", protocol="anthropic", + overrides={"thinking_modes": ["adaptive"], "effort_levels": ["low", "high"]}, +) +profile = ModelProfile( + model_id="gateway-alias", provider="gateway", protocol="anthropic", + capabilities=caps, context_window=64000, +) +``` + +`context_window` remains a host-selected operational budget. The capability's +`max_input_tokens` is the provider's maximum, and does not overwrite that budget. +Defaults such as `default_effort` are descriptive; callers who omit effort keep +the provider's own default. Beta/platform-specific limits require a host override +paired with the appropriate headers. Models API discovery/caching can be added +by hosts later; this PR does not introduce a background synchronization service. diff --git a/tests/test_model_capabilities.py b/tests/test_model_capabilities.py new file mode 100644 index 0000000..e7d60d5 --- /dev/null +++ b/tests/test_model_capabilities.py @@ -0,0 +1,178 @@ +from __future__ import annotations + +from dataclasses import FrozenInstanceError + +import pytest + +from agent_core.model_capabilities import ( + MODEL_CAPABILITIES, + ModelCapabilities, + resolve_model_capabilities, +) +from agent_core.providers.anthropic import AnthropicClient +from agent_core.providers.protocol_client import build_protocol_client +from agent_core.runtime.loop.model_profile import ModelProfile + + +@pytest.mark.parametrize("model,effort", [("claude-fable-5-1", "high"), ("claude-opus-5-5", "medium")]) +def test_current_model_facts_are_shared_with_profiles(model, effort): + capabilities = resolve_model_capabilities(model, protocol="anthropic") + profile = ModelProfile(model_id=model, provider="anthropic", protocol="anthropic", context_window=64_000) + assert profile.request_capabilities is capabilities + assert profile.context_window == 64_000 # operational budget stays host-owned + assert capabilities.thinking_required is True + assert capabilities.default_effort == effort + assert capabilities.source_urls and capabilities.verified_on == "2026-10-05" + assert capabilities.max_output_tokens == 128_000 + assert capabilities.thinking_signature_binding == "conversation_prefix" + + +@pytest.mark.parametrize("model,protocol", [ + ("claude-opus-5-5-next", "anthropic"), + ("gateway-alias", "anthropic"), + ("claude-opus-5-5", "chat_completions"), + ("gpt-5.5", "responses"), +]) +def test_unknown_and_other_protocols_do_not_inherit_native_rules(model, protocol): + capabilities = resolve_model_capabilities(model, protocol=protocol) + assert capabilities == ModelCapabilities() + capabilities.validate_request(model=model, thinking={"type": "future"}, effort="custom", max_tokens=500_000) + + +@pytest.mark.parametrize("model", [ + "anthropic.claude-opus-5-5", "us.anthropic.claude-opus-5-5", + "eu.anthropic.claude-opus-5-5-v1:0", "global.anthropic.claude-opus-5-5", +]) +def test_bedrock_model_ids_resolve_without_rewriting_request_model(model): + assert resolve_model_capabilities(model, protocol="bedrock") is MODEL_CAPABILITIES["claude-opus-5-5"] + assert resolve_model_capabilities(model, protocol="anthropic") == ModelCapabilities() + + +def test_overrides_distinguish_unknown_from_unsupported_and_are_local(): + baseline = resolve_model_capabilities("claude-opus-5-5", protocol="anthropic") + overrides = {"thinking_modes": ["enabled", "disabled"], "thinking_required": False, + "effort_levels": [], "default_effort": None, "max_output_tokens": None} + patched = resolve_model_capabilities("claude-opus-5-5", protocol="anthropic", overrides=overrides) + assert patched.thinking_modes == frozenset({"enabled", "disabled"}) + assert patched.effort_levels == frozenset() + assert patched.max_output_tokens is None + assert patched.overridden_fields == frozenset(overrides) + assert resolve_model_capabilities("claude-opus-5-5", protocol="anthropic") is baseline + overrides["thinking_modes"].append("adaptive") + assert "adaptive" not in patched.thinking_modes + with pytest.raises(FrozenInstanceError): + patched.thinking_required = True + with pytest.raises(TypeError): + MODEL_CAPABILITIES["custom"] = patched + + +@pytest.mark.parametrize("overrides", [ + {"typo": True}, {"effort_levels": "high"}, {"effort_levels": [1]}, + {"thinking_required": "false"}, {"max_output_tokens": True}, + {"max_input_tokens": 0}, {"thinking_signature_binding": "typo"}, + {"default_effort": 5}, {"effort_levels": []}, + {"thinking_modes": ["disabled"]}, +]) +def test_invalid_overrides_are_rejected(overrides): + with pytest.raises(ValueError): + resolve_model_capabilities("claude-opus-5-5", protocol="anthropic", overrides=overrides) + + +@pytest.mark.parametrize("thinking,effort,limit", [ + ({"type": "disabled"}, "", 4096), ({"type": "enabled", "budget_tokens": 1024}, "", 4096), + ({"type": ["adaptive"]}, "", 4096), (None, "minimal", 4096), + (None, "", 128_001), +]) +def test_direct_client_rejects_known_invalid_requests_before_sdk_creation(thinking, effort, limit, monkeypatch): + import anthropic + + def unexpected_client(**kwargs): + raise AssertionError("invalid request must fail before opening an SDK client") + + monkeypatch.setattr(anthropic, "AsyncAnthropic", unexpected_client) + with pytest.raises(ValueError): + AnthropicClient("claude-opus-5-5", api_key="x", thinking=thinking, effort=effort, max_tokens=limit) + + +@pytest.mark.parametrize("model,mode", [ + ("claude-fable-5-1", "adaptive"), ("claude-opus-5-5", "adaptive"), + ("claude-sonnet-4-5-20250929", "enabled"), ("claude-haiku-4-5", "enabled"), + ("claude-x", "adaptive"), +]) +async def test_native_builder_uses_model_aware_defaults(model, mode): + client = build_protocol_client({"protocol": "anthropic", "model": model, "max_tokens": 4096}, title="test") + try: + kwargs = client._build_kwargs([{"role": "user", "content": "hi"}], tools=None, temperature=None, max_tokens=None, extra_headers=None, timeout=None) + assert kwargs["thinking"]["type"] == mode + if mode == "enabled": + assert 1024 <= kwargs["thinking"]["budget_tokens"] < kwargs["max_tokens"] + finally: + await client._client.close() + + +async def test_per_call_limit_is_validated_as_well_as_constructor_default(): + client = AnthropicClient("claude-opus-5-5", api_key="x") + try: + with pytest.raises(ValueError, match="max_tokens"): + client._build_kwargs([], tools=None, temperature=None, max_tokens=128_001, extra_headers=None, timeout=None) + finally: + await client._client.close() + + +async def test_builder_override_can_disable_thinking_configuration_entirely(): + client = build_protocol_client({ + "protocol": "anthropic", "model": "claude-opus-5-5", + "model_capabilities": {"thinking_modes": [], "thinking_required": False}, + }, title="test") + try: + assert client._thinking is None + finally: + await client._client.close() + + +async def test_manual_thinking_can_forward_effort_when_model_supports_it(): + client = build_protocol_client({"protocol": "anthropic", "model": "claude-opus-4-5", "effort": "medium"}, title="test") + try: + kwargs = client._build_kwargs([], tools=None, temperature=None, max_tokens=None, extra_headers=None, timeout=None) + assert kwargs["thinking"]["type"] == "enabled" + assert kwargs["extra_body"]["output_config"]["effort"] == "medium" + finally: + await client._client.close() + + +@pytest.mark.parametrize("mode", [None, "", " "]) +async def test_unset_thinking_type_uses_model_default(mode): + client = build_protocol_client({"protocol": "anthropic", "model": "claude-haiku-4-5", "thinking_type": mode}, title="test") + try: + assert client._thinking["type"] == "enabled" + finally: + await client._client.close() + + +def test_auxiliary_factory_forwards_overrides_to_the_same_capability_resolver(): + from agent_core.providers.aux_builder import AuxLLMFactory + + factory = AuxLLMFactory(openai_factory=lambda **kw: kw, anthropic_factory=lambda **kw: kw, + provider_type=lambda _: "anthropic") + kwargs = factory.build({"provider": "gateway", "model": "claude-opus-5-5", "api_key": "x", + "model_capabilities": {"max_output_tokens": 32_000}}) + assert kwargs["capabilities"].max_output_tokens == 32_000 + assert MODEL_CAPABILITIES["claude-opus-5-5"].max_output_tokens == 128_000 + + +@pytest.mark.parametrize("thinking", [{"type": "disabled"}, {"type": "off"}, {"type": "enabled", "budget_tokens": 1024}]) +def test_auxiliary_client_checks_explicit_modes_before_legacy_conversion(thinking): + from agent_core.providers.aux_builder import AuxLLMFactory + + def unexpected_factory(**kwargs): + raise AssertionError("configuration must fail before client construction") + + factory = AuxLLMFactory(openai_factory=unexpected_factory, anthropic_factory=unexpected_factory, + provider_type=lambda _: "anthropic") + with pytest.raises(ValueError, match="thinking mode"): + factory.build({"provider": "anthropic", "model": "claude-opus-5-5", "api_key": "x", "thinking": thinking}) + + +def test_known_legacy_model_rejects_effort_instead_of_silently_dropping_it(): + with pytest.raises(ValueError, match="effort"): + build_protocol_client({"protocol": "anthropic", "model": "claude-sonnet-4-5", "effort": "high"}, title="test") From c48a3e0bdf1517f05c737773a329e6ba7230c1ed Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Mon, 5 Oct 2026 16:25:30 +0800 Subject: [PATCH 2/3] fix(capabilities): share thinking-mode parsing and profile overrides Follow-up to 4e991d8 (centralize typed model capability constraints). - ModelProfile gains `model_capabilities`, the same override mapping the client config uses, and `request_capabilities` resolves with it. Before, the profile ignored deployment overrides, so it could disagree with the client built from the same config. - New `normalize_thinking_mode` is used by both `protocol_client` and the auxiliary builder: off/none/false all mean disabled, blank means unset, anything else raises. Previously `off` was disabled on the aux path but an "unknown thinking_type" error on the main path. - `_build_kwargs` only re-validates the per-call output limit; thinking and effort are fixed and already validated at construction. - Deduplicate the Opus 4.5 record, accept us-gov/jp/au Bedrock prefixes, and fix the resolver docstring (there is no protocol-default layer). - Changelog spells out the behavior changes for existing configs. Co-Authored-By: Claude Opus 5.5 (1M context) --- agent_core/model_capabilities.py | 48 +++++++++++++++++------ agent_core/providers/anthropic.py | 5 ++- agent_core/providers/aux_builder.py | 16 ++++---- agent_core/providers/protocol_client.py | 12 +----- agent_core/runtime/loop/model_profile.py | 10 ++++- changes/model-capabilities.feature.md | 2 +- docs/provider-substrate-boundary.md | 13 +++++-- tests/test_model_capabilities.py | 49 ++++++++++++++++++++++++ 8 files changed, 120 insertions(+), 35 deletions(-) diff --git a/agent_core/model_capabilities.py b/agent_core/model_capabilities.py index 551ee49..33d0e38 100644 --- a/agent_core/model_capabilities.py +++ b/agent_core/model_capabilities.py @@ -1,7 +1,7 @@ """Typed model request constraints, separate from credentials and deployment catalogs. None means unknown, an empty set means unsupported. Resolution is explicit and -local to a client/profile: protocol defaults < exact model facts < host overrides. +local to a client/profile: exact model facts, then per-deployment host overrides. No network access or mutable process-wide registration is performed here. """ @@ -68,6 +68,11 @@ def validate_request( source_urls=("https://platform.claude.com/docs/en/build-with-claude/extended-thinking",), verified_on="2026-10-05", ) +# The only manual-thinking model with effort; it combines with budget_tokens. +_OPUS_4_5 = replace( + _LEGACY_CLAUDE, effort_levels=frozenset({"low", "medium", "high"}), default_effort="high", + source_urls=(*_LEGACY_CLAUDE.source_urls, "https://platform.claude.com/docs/en/build-with-claude/effort"), +) # These are request facts, not an endpoint, credential, price, or routing catalog. # Only documented exact IDs/aliases match; a new version never inherits a guessed @@ -85,16 +90,34 @@ def validate_request( "claude-sonnet-4-5-20250929": _LEGACY_CLAUDE, "claude-haiku-4-5": _LEGACY_CLAUDE, "claude-haiku-4-5-20251001": _LEGACY_CLAUDE, - "claude-opus-4-5": replace(_LEGACY_CLAUDE, - effort_levels=frozenset({"low", "medium", "high"}), default_effort="high", - source_urls=(*_LEGACY_CLAUDE.source_urls, "https://platform.claude.com/docs/en/build-with-claude/effort"), - ), - "claude-opus-4-5-20251101": replace(_LEGACY_CLAUDE, - effort_levels=frozenset({"low", "medium", "high"}), default_effort="high", - source_urls=(*_LEGACY_CLAUDE.source_urls, "https://platform.claude.com/docs/en/build-with-claude/effort"), - ), + "claude-opus-4-5": _OPUS_4_5, + "claude-opus-4-5-20251101": _OPUS_4_5, }) +_DISABLED_ALIASES = frozenset({"disabled", "off", "none", "false"}) + + +def normalize_thinking_mode(value: object) -> ThinkingMode | None: + """Normalize a configured thinking type; None/blank means "not configured". + + Shared by every native builder so one spelling means the same thing on the + main, auxiliary, and direct paths. Unknown values raise instead of silently + falling back to a mode the operator did not choose. + """ + if value is None: + return None + if not isinstance(value, str): + raise ValueError("thinking type must be a string or null") + mode = value.strip().lower() + if not mode: + return None + if mode in _DISABLED_ALIASES: + return "disabled" + if mode in ("adaptive", "enabled"): + return cast(ThinkingMode, mode) + raise ValueError(f"unknown thinking type {value!r}; use adaptive, enabled, or disabled") + + _SET_FIELDS = frozenset({"thinking_modes", "effort_levels", "sampling_parameters", "tool_choice_modes"}) _LIMIT_FIELDS = frozenset({"max_input_tokens", "max_output_tokens"}) _OVERRIDE_FIELDS = _SET_FIELDS | _LIMIT_FIELDS | { @@ -119,7 +142,7 @@ def _canonical_model_id(model_id: str, protocol: WireProtocol) -> str: return model_id # Bedrock inference-profile regional prefixes and documented version suffix. # Arbitrary proxy aliases and ARNs require explicit host overrides. - match = re.fullmatch(r"(?:(?:us|eu|apac|global)\.)?anthropic\.(claude-[a-z0-9-]+?)(?:-v\d+:\d+)?", model_id) + match = re.fullmatch(r"(?:(?:us|us-gov|eu|apac|jp|au|global)\.)?anthropic\.(claude-[a-z0-9-]+?)(?:-v\d+:\d+)?", model_id) return match.group(1) if match else model_id @@ -171,4 +194,7 @@ def resolve_model_capabilities( return result -__all__ = ["MODEL_CAPABILITIES", "ModelCapabilities", "SignatureBinding", "ThinkingMode", "WireProtocol", "resolve_model_capabilities"] +__all__ = [ + "MODEL_CAPABILITIES", "ModelCapabilities", "SignatureBinding", "ThinkingMode", + "WireProtocol", "normalize_thinking_mode", "resolve_model_capabilities", +] diff --git a/agent_core/providers/anthropic.py b/agent_core/providers/anthropic.py index 2f74401..3b7eccc 100644 --- a/agent_core/providers/anthropic.py +++ b/agent_core/providers/anthropic.py @@ -107,9 +107,10 @@ def _build_kwargs( ) -> dict[str, Any]: """Shared request-shape builder for :meth:`chat` and :meth:`stream`.""" output_limit = max_tokens or self.default_max_tokens or 4096 + # Thinking and effort are fixed and validated at construction; only + # the per-call output limit can change here. self.capabilities.validate_request( - model=self.model, thinking=self._thinking, effort=self._effort, - max_tokens=output_limit, + model=self.model, thinking=None, max_tokens=output_limit, ) system, msgs = _split_system(messages) # ``_to_anthropic_msg`` returns None for a message with nothing diff --git a/agent_core/providers/aux_builder.py b/agent_core/providers/aux_builder.py index 910a73e..ef5055f 100644 --- a/agent_core/providers/aux_builder.py +++ b/agent_core/providers/aux_builder.py @@ -7,7 +7,7 @@ from collections.abc import Callable, Mapping from typing import Any -from agent_core.model_capabilities import resolve_model_capabilities +from agent_core.model_capabilities import normalize_thinking_mode, resolve_model_capabilities logger = logging.getLogger(__name__) @@ -65,9 +65,10 @@ def _anthropic_thinking(section: Mapping[str, Any]) -> dict[str, Any] | None: if not isinstance(raw, Mapping): return None thinking = {str(key): value for key, value in raw.items()} - kind = str(thinking.get("type") or "").strip().lower() - if kind in {"", "disabled", "off", "none", "false"}: + kind = normalize_thinking_mode(thinking.get("type")) + if kind is None or kind == "disabled": return None + thinking["type"] = kind return thinking @@ -156,10 +157,11 @@ def _build_anthropic( # maps an explicit disabled/off configuration to no thinking field. raw_thinking = section.get("thinking") validation_thinking = kwargs["thinking"] - if isinstance(raw_thinking, Mapping): - kind = str(raw_thinking.get("type") or "").strip().lower() - if kind in {"disabled", "off", "none", "false"}: - validation_thinking = {"type": "disabled"} + if ( + isinstance(raw_thinking, Mapping) + and normalize_thinking_mode(raw_thinking.get("type")) == "disabled" + ): + validation_thinking = {"type": "disabled"} maximum = section.get("max_completion_tokens") or section.get("max_tokens") capabilities.validate_request( model=str(section["model"]), thinking=validation_thinking, diff --git a/agent_core/providers/protocol_client.py b/agent_core/providers/protocol_client.py index ecceea9..879c292 100644 --- a/agent_core/providers/protocol_client.py +++ b/agent_core/providers/protocol_client.py @@ -23,7 +23,7 @@ from typing import Any, get_args from agent_core.llm import LLMClient -from agent_core.model_capabilities import resolve_model_capabilities +from agent_core.model_capabilities import normalize_thinking_mode, resolve_model_capabilities # Single source of truth for the valid ``llm.protocol`` values. Duplicating the # set here would let the two drift, which is how a protocol becomes buildable @@ -153,15 +153,7 @@ def _build_anthropic( "enabled" if "enabled" in capabilities.thinking_modes else "disabled" if "disabled" in capabilities.thinking_modes else None ) - raw_mode = cfg.get("thinking_type") - if raw_mode is None or (isinstance(raw_mode, str) and not raw_mode.strip()): - ttype = default_mode - elif isinstance(raw_mode, str): - ttype = raw_mode.strip().lower() - else: - raise ValueError("thinking_type must be a string or null") - if ttype is not None and ttype not in ("adaptive", "enabled", "disabled"): - raise ValueError(f"unknown thinking_type {ttype!r}; use adaptive, enabled, or disabled") + ttype = normalize_thinking_mode(cfg.get("thinking_type")) or default_mode capabilities.validate_request( model=cfg["model"], thinking={"type": ttype} if ttype else None, effort=_effort_str(cfg), diff --git a/agent_core/runtime/loop/model_profile.py b/agent_core/runtime/loop/model_profile.py index f2cc0d9..324c145 100644 --- a/agent_core/runtime/loop/model_profile.py +++ b/agent_core/runtime/loop/model_profile.py @@ -213,11 +213,19 @@ class ModelProfile: # Request constraints use the same records as provider construction. # context_window above remains the host's operational context budget; # capabilities.max_input_tokens describes the provider's maximum. + # ``capabilities`` takes a resolved record (e.g. ``client.capabilities``); + # otherwise ``model_capabilities`` carries the same per-deployment override + # mapping the client config uses, so both sides resolve identically. capabilities: ModelCapabilities | None = None + model_capabilities: Mapping[str, object] | None = None @property def request_capabilities(self) -> ModelCapabilities: - return self.capabilities or resolve_model_capabilities(self.model_id, protocol=self.protocol) + if self.capabilities is not None: + return self.capabilities + return resolve_model_capabilities( + self.model_id, protocol=self.protocol, overrides=self.model_capabilities, + ) @dataclass diff --git a/changes/model-capabilities.feature.md b/changes/model-capabilities.feature.md index 7a506c7..d12985c 100644 --- a/changes/model-capabilities.feature.md +++ b/changes/model-capabilities.feature.md @@ -1 +1 @@ -Add typed, source-documented model capability records shared by model profiles and Anthropic request construction. Known models select compatible thinking defaults and reject unsupported thinking, effort, or output-token settings before API calls; unknown models retain pass-through behavior and hosts can override deployment-specific facts without global mutable state. Enable CI for stacked pull requests. +Add typed, source-documented model capability records shared by model profiles and Anthropic request construction. Known models select compatible thinking defaults and reject unsupported thinking, effort, or output-token settings before API calls; unknown models retain pass-through behavior and hosts can override deployment-specific facts without global mutable state. Enable CI for stacked pull requests. Behavior changes: an unrecognized `thinking_type` now raises instead of falling back to adaptive (`off`/`none`/`false` mean disabled on every builder); configuring `effort` for Sonnet 4.5 or Haiku 4.5 now raises instead of being dropped; and Sonnet/Haiku/Opus 4.5 without an explicit `thinking_type` now default to `enabled` (budget 8192, which may raise `max_tokens`) instead of the rejected adaptive mode. diff --git a/docs/provider-substrate-boundary.md b/docs/provider-substrate-boundary.md index b234c7a..8ff23ea 100644 --- a/docs/provider-substrate-boundary.md +++ b/docs/provider-substrate-boundary.md @@ -125,7 +125,9 @@ overrides. This first table covers Fable 5.1, Opus 5.5, and the legacy 4.5 model other providers and models remain unknown until verified facts are added. Known thinking modes select builder defaults and reject unsupported explicit -modes. Effort and output limits are validated in direct clients, native profile +modes. Every native builder normalizes the configured thinking type the same +way: `adaptive`, `enabled`, and `disabled` (also spelled `off`, `none`, or +`false`); blank means unset, and any other value raises `ValueError`. Effort and output limits are validated in direct clients, native profile clients, and per-call overrides. No parameter is silently clamped. The existing legacy thinking-budget adjustment remains in place. Empty thinking support omits the thinking field; unknown native models keep the prior adaptive default. @@ -134,8 +136,8 @@ protocol field conversion, SDK transport limitations, and error recovery stay in adapters. The Anthropic adapter still omits sampling fields for SDK 1.x compatibility. This PR does not add forced tool-choice parameters. -Resolution applies verified model facts after unknown protocol defaults, then -applies a per-client host override. The `model_capabilities` profile key is a +Resolution starts from verified model facts (or unknown), then applies a +per-client host override. The `model_capabilities` profile key is a mapping with the capability field names; omitted fields inherit, explicit null clears a fact to unknown, and empty lists declare unsupported values. Unknown keys, malformed values, and contradictory defaults raise `ValueError`. @@ -152,6 +154,11 @@ llm: max_output_tokens: 32768 # gateway limit overrides the model maximum ``` +A profile must see the same overrides as its client. Pass the profile the same +mapping (`ModelProfile(..., model_capabilities=cfg.get("model_capabilities"))`) +or the client's resolved record (`capabilities=client.capabilities`); a profile +built with neither resolves the unoverridden model facts. + For a custom alias, specify its supported modes and effort levels explicitly: ```python diff --git a/tests/test_model_capabilities.py b/tests/test_model_capabilities.py index e7d60d5..55ed077 100644 --- a/tests/test_model_capabilities.py +++ b/tests/test_model_capabilities.py @@ -176,3 +176,52 @@ def unexpected_factory(**kwargs): def test_known_legacy_model_rejects_effort_instead_of_silently_dropping_it(): with pytest.raises(ValueError, match="effort"): build_protocol_client({"protocol": "anthropic", "model": "claude-sonnet-4-5", "effort": "high"}, title="test") + + +async def test_profile_and_client_resolve_the_same_deployment_overrides(): + overrides = {"max_output_tokens": 32_000} + client = build_protocol_client({ + "protocol": "anthropic", "model": "claude-opus-5-5", "max_tokens": 16_000, + "model_capabilities": overrides, + }, title="test") + try: + profile = ModelProfile(model_id="claude-opus-5-5", provider="anthropic", + protocol="anthropic", model_capabilities=overrides) + assert profile.request_capabilities == client.capabilities + assert profile.request_capabilities.max_output_tokens == 32_000 + finally: + await client._client.close() + + +@pytest.mark.parametrize("spelling", ["disabled", "off", "none", "False"]) +async def test_main_and_auxiliary_builders_share_disabled_spellings(spelling): + from agent_core.providers.aux_builder import AuxLLMFactory + + client = build_protocol_client({"protocol": "anthropic", "model": "claude-haiku-4-5", "thinking_type": spelling}, title="test") + try: + assert client._thinking == {"type": "disabled"} + finally: + await client._client.close() + factory = AuxLLMFactory(openai_factory=lambda **kw: kw, anthropic_factory=lambda **kw: kw, + provider_type=lambda _: "anthropic") + kwargs = factory.build({"provider": "anthropic", "model": "claude-haiku-4-5", "api_key": "x", + "thinking": {"type": spelling}}) + assert kwargs["thinking"] is None + + +def test_main_and_auxiliary_builders_reject_the_same_unknown_mode(): + from agent_core.providers.aux_builder import AuxLLMFactory + + with pytest.raises(ValueError, match="unknown thinking type"): + build_protocol_client({"protocol": "anthropic", "model": "claude-x", "thinking_type": "auto"}, title="test") + factory = AuxLLMFactory(openai_factory=lambda **kw: kw, anthropic_factory=lambda **kw: kw, + provider_type=lambda _: "anthropic") + with pytest.raises(ValueError, match="unknown thinking type"): + factory.build({"provider": "anthropic", "model": "claude-x", "api_key": "x", "thinking": {"type": "auto"}}) + + +@pytest.mark.parametrize("model_id", [ + "jp.anthropic.claude-opus-5-5", "au.anthropic.claude-opus-5-5-v1:0", "us-gov.anthropic.claude-opus-5-5", +]) +def test_additional_bedrock_regional_prefixes_resolve(model_id): + assert resolve_model_capabilities(model_id, protocol="bedrock") is MODEL_CAPABILITIES["claude-opus-5-5"] From cdc36fbd139de579956a9770bac17990b970a3ac Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Mon, 5 Oct 2026 16:29:21 +0800 Subject: [PATCH 3/3] test(capabilities): cover normalization, overrides, and per-call limits - Document on ModelCapabilities (and in the boundary doc) which fields validate_request enforces and which are descriptive until an adapter consumes them, so hosts do not assume e.g. tool_choice_modes is applied. - Tests: normalize_thinking_mode spellings, unknown and non-string values; the shared Opus 4.5 record; all documented Bedrock prefixes plus forms that must stay unknown (unlisted prefix, ARN, Bedrock ID over the native protocol); per-call max_tokens checked against a deployment override and the record maximum; ModelProfile precedence of an explicit record over the override mapping, malformed overrides failing like the client, and overrides on non-native protocols; descriptive fields being overridable without affecting request validation. Co-Authored-By: Claude Opus 5.5 (1M context) --- agent_core/model_capabilities.py | 13 ++++ docs/provider-substrate-boundary.md | 5 +- tests/test_model_capabilities.py | 111 ++++++++++++++++++++++++++++ 3 files changed, 128 insertions(+), 1 deletion(-) diff --git a/agent_core/model_capabilities.py b/agent_core/model_capabilities.py index 33d0e38..18bcc0a 100644 --- a/agent_core/model_capabilities.py +++ b/agent_core/model_capabilities.py @@ -20,6 +20,19 @@ @dataclass(frozen=True) class ModelCapabilities: + """Request facts for one model, resolved for one client/profile. + + Enforced before API calls by :meth:`validate_request`: ``thinking_modes``, + ``thinking_required``, ``effort_levels`` and ``max_output_tokens``. + Descriptive only, for hosts to read: ``default_effort`` (an omitted effort + keeps the provider default), ``sampling_parameters`` (the Anthropic adapter + omits sampling for every model regardless), ``tool_choice_modes`` (no + adapter sends ``tool_choice`` yet), ``thinking_signature_binding`` and + ``max_input_tokens`` (``ModelProfile.context_window`` stays the operational + budget). A field moves to the enforced list only together with the adapter + code that consumes it. + """ + thinking_modes: frozenset[str] | None = None thinking_required: bool | None = None effort_levels: frozenset[str] | None = None diff --git a/docs/provider-substrate-boundary.md b/docs/provider-substrate-boundary.md index 8ff23ea..4fdd31c 100644 --- a/docs/provider-substrate-boundary.md +++ b/docs/provider-substrate-boundary.md @@ -131,7 +131,10 @@ way: `adaptive`, `enabled`, and `disabled` (also spelled `off`, `none`, or clients, and per-call overrides. No parameter is silently clamped. The existing legacy thinking-budget adjustment remains in place. Empty thinking support omits the thinking field; unknown native models keep the prior adaptive default. -Sampling, tool-choice, and signature-binding facts are available to hosts; +Only thinking modes, required thinking, effort levels, and output limits are +enforced. `default_effort`, `sampling_parameters`, `tool_choice_modes`, +`thinking_signature_binding`, and `max_input_tokens` are descriptive until an +adapter consumes them. Sampling, tool-choice, and signature-binding facts are available to hosts; protocol field conversion, SDK transport limitations, and error recovery stay in adapters. The Anthropic adapter still omits sampling fields for SDK 1.x compatibility. This PR does not add forced tool-choice parameters. diff --git a/tests/test_model_capabilities.py b/tests/test_model_capabilities.py index 55ed077..eb7c1ed 100644 --- a/tests/test_model_capabilities.py +++ b/tests/test_model_capabilities.py @@ -7,6 +7,7 @@ from agent_core.model_capabilities import ( MODEL_CAPABILITIES, ModelCapabilities, + normalize_thinking_mode, resolve_model_capabilities, ) from agent_core.providers.anthropic import AnthropicClient @@ -225,3 +226,113 @@ def test_main_and_auxiliary_builders_reject_the_same_unknown_mode(): ]) def test_additional_bedrock_regional_prefixes_resolve(model_id): assert resolve_model_capabilities(model_id, protocol="bedrock") is MODEL_CAPABILITIES["claude-opus-5-5"] + + +@pytest.mark.parametrize("value,expected", [ + (None, None), ("", None), (" ", None), + (" Adaptive ", "adaptive"), ("ENABLED", "enabled"), + ("disabled", "disabled"), ("Off", "disabled"), ("none", "disabled"), ("false", "disabled"), +]) +def test_normalize_thinking_mode_accepts_documented_spellings(value, expected): + assert normalize_thinking_mode(value) == expected + + +@pytest.mark.parametrize("value", ["auto", "on", "true", "budget"]) +def test_normalize_thinking_mode_rejects_unknown_values(value): + with pytest.raises(ValueError, match="unknown thinking type"): + normalize_thinking_mode(value) + + +@pytest.mark.parametrize("value", [False, 0, {"type": "adaptive"}]) +def test_normalize_thinking_mode_rejects_non_strings(value): + with pytest.raises(ValueError, match="must be a string"): + normalize_thinking_mode(value) + + +def test_opus_4_5_alias_and_dated_id_share_one_record(): + alias = MODEL_CAPABILITIES["claude-opus-4-5"] + assert alias is MODEL_CAPABILITIES["claude-opus-4-5-20251101"] + assert alias.effort_levels == frozenset({"low", "medium", "high"}) + assert alias.thinking_modes == frozenset({"enabled", "disabled"}) + assert "https://platform.claude.com/docs/en/build-with-claude/effort" in alias.source_urls + + +@pytest.mark.parametrize("model_id", [ + "anthropic.claude-opus-5-5", "us.anthropic.claude-opus-5-5-v1:0", + "eu.anthropic.claude-opus-5-5", "apac.anthropic.claude-opus-5-5", + "global.anthropic.claude-opus-5-5", +]) +def test_documented_bedrock_prefixes_still_resolve(model_id): + assert resolve_model_capabilities(model_id, protocol="bedrock") is MODEL_CAPABILITIES["claude-opus-5-5"] + + +@pytest.mark.parametrize("model_id", [ + "xx.anthropic.claude-opus-5-5", + "arn:aws:bedrock:us-east-1:123:inference-profile/us.anthropic.claude-opus-5-5-v1:0", + "jp.anthropic.claude-opus-5-5", # bedrock form, but resolved over the native protocol +]) +def test_unrecognized_bedrock_forms_stay_unknown(model_id): + protocol = "anthropic" if model_id.startswith("jp.") else "bedrock" + assert resolve_model_capabilities(model_id, protocol=protocol) == ModelCapabilities() + + +async def test_per_call_limit_uses_the_deployment_override(): + client = build_protocol_client({ + "protocol": "anthropic", "model": "claude-opus-5-5", "max_tokens": 16_000, + "model_capabilities": {"max_output_tokens": 32_000}, + }, title="test") + try: + ok = client._build_kwargs([], tools=None, temperature=None, max_tokens=32_000, extra_headers=None, timeout=None) + assert ok["max_tokens"] == 32_000 + with pytest.raises(ValueError, match="max_tokens=32001"): + client._build_kwargs([], tools=None, temperature=None, max_tokens=32_001, extra_headers=None, timeout=None) + finally: + await client._client.close() + + +async def test_per_call_default_limit_is_validated_against_the_record(): + client = AnthropicClient("claude-opus-5-5", api_key="x", max_tokens=128_000) + try: + kwargs = client._build_kwargs([], tools=None, temperature=None, max_tokens=None, extra_headers=None, timeout=None) + assert kwargs["max_tokens"] == 128_000 + finally: + await client._client.close() + + +def test_profile_prefers_an_explicit_record_over_override_mapping(): + record = resolve_model_capabilities("claude-opus-5-5", protocol="anthropic", overrides={"max_output_tokens": 8_000}) + profile = ModelProfile(model_id="claude-opus-5-5", provider="anthropic", protocol="anthropic", + capabilities=record, model_capabilities={"max_output_tokens": 64_000}) + assert profile.request_capabilities is record + + +def test_profile_rejects_malformed_overrides_like_the_client(): + profile = ModelProfile(model_id="claude-opus-5-5", provider="anthropic", protocol="anthropic", + model_capabilities={"max_output_tokens": -1}) + with pytest.raises(ValueError, match="positive integer"): + _ = profile.request_capabilities + with pytest.raises(ValueError, match="positive integer"): + build_protocol_client({"protocol": "anthropic", "model": "claude-opus-5-5", + "model_capabilities": {"max_output_tokens": -1}}, title="test") + + +def test_profile_overrides_do_not_apply_to_other_protocols_facts(): + profile = ModelProfile(model_id="claude-opus-5-5", provider="gateway", protocol="chat_completions", + model_capabilities={"max_output_tokens": 32_000}) + caps = profile.request_capabilities + assert caps.max_output_tokens == 32_000 + assert caps.thinking_modes is None # native facts are not inherited over chat completions + assert caps.overridden_fields == frozenset({"max_output_tokens"}) + + +def test_descriptive_fields_are_exposed_and_overridable(): + caps = resolve_model_capabilities("claude-fable-5-1", protocol="anthropic", overrides={ + "tool_choice_modes": ["auto", "none", "any"], "sampling_parameters": None, + "thinking_signature_binding": "model", "max_input_tokens": 200_000, + }) + assert caps.tool_choice_modes == frozenset({"auto", "none", "any"}) + assert caps.sampling_parameters is None + assert caps.thinking_signature_binding == "model" + assert caps.max_input_tokens == 200_000 + # Descriptive facts never affect request validation. + caps.validate_request(model="claude-fable-5-1", thinking={"type": "adaptive"}, effort="high", max_tokens=128_000)