diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index aae2f4c..2640115 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -4,7 +4,7 @@ on: push: branches: [main] pull_request: - branches: [main] + branches: ['**'] types: [opened, synchronize, reopened, labeled, unlabeled] jobs: @@ -25,7 +25,11 @@ jobs: - run: uv run pytest -q # Keep current-model request/response handling usable at our SDK floor, # including SDK 0.x (httpx) versus 1.x (httpx2) parsing and transport. - - run: uv run --isolated --with 'anthropic[bedrock]==0.69.0' --extra dev pytest -q tests/test_anthropic_latest_models.py tests/test_provider_native_clients.py tests/test_provider_protocol_reasoning.py + - run: >- + uv run --isolated --with 'anthropic[bedrock]==0.69.0' --extra dev pytest -q + tests/test_anthropic_latest_models.py tests/test_provider_native_clients.py + tests/test_provider_protocol_reasoning.py tests/test_model_capabilities.py + tests/test_aux_builder.py # The gate pins tiktoken's own cache key and content hash, and only a real # tiktoken can falsify them. Scope that optional dependency to this contract # test; the isolated run uses the tokenizer version pinned in uv.lock. diff --git a/agent_core/__init__.py b/agent_core/__init__.py index c182b4e..0aea91c 100644 --- a/agent_core/__init__.py +++ b/agent_core/__init__.py @@ -12,6 +12,7 @@ tool_msg, user_msg, ) +from agent_core.model_capabilities import ModelCapabilities, resolve_model_capabilities from agent_core.types import TaskStatus __all__ = [ @@ -21,11 +22,13 @@ "LLMClient", "LLMResponse", "Message", + "ModelCapabilities", "StreamDelta", "TaskStatus", "ToolCall", "assistant_msg", "build_execution_scope", + "resolve_model_capabilities", "system_msg", "tool_msg", "user_msg", diff --git a/agent_core/model_capabilities.py b/agent_core/model_capabilities.py new file mode 100644 index 0000000..18bcc0a --- /dev/null +++ b/agent_core/model_capabilities.py @@ -0,0 +1,213 @@ +"""Typed model request constraints, separate from credentials and deployment catalogs. + +None means unknown, an empty set means unsupported. Resolution is explicit and +local to a client/profile: exact model facts, then per-deployment host overrides. +No network access or mutable process-wide registration is performed here. +""" + +from __future__ import annotations + +import re +from collections.abc import Iterable, Mapping +from dataclasses import dataclass, replace +from types import MappingProxyType +from typing import Literal, cast + +WireProtocol = Literal["chat_completions", "anthropic", "responses", "bedrock"] +ThinkingMode = Literal["adaptive", "enabled", "disabled"] +SignatureBinding = Literal["none", "model", "conversation_prefix"] + + +@dataclass(frozen=True) +class ModelCapabilities: + """Request facts for one model, resolved for one client/profile. + + Enforced before API calls by :meth:`validate_request`: ``thinking_modes``, + ``thinking_required``, ``effort_levels`` and ``max_output_tokens``. + Descriptive only, for hosts to read: ``default_effort`` (an omitted effort + keeps the provider default), ``sampling_parameters`` (the Anthropic adapter + omits sampling for every model regardless), ``tool_choice_modes`` (no + adapter sends ``tool_choice`` yet), ``thinking_signature_binding`` and + ``max_input_tokens`` (``ModelProfile.context_window`` stays the operational + budget). A field moves to the enforced list only together with the adapter + code that consumes it. + """ + + thinking_modes: frozenset[str] | None = None + thinking_required: bool | None = None + effort_levels: frozenset[str] | None = None + default_effort: str | None = None + sampling_parameters: frozenset[str] | None = None + tool_choice_modes: frozenset[str] | None = None + thinking_signature_binding: SignatureBinding | None = None + max_input_tokens: int | None = None + max_output_tokens: int | None = None + source_urls: tuple[str, ...] = () + verified_on: str = "" + overridden_fields: frozenset[str] = frozenset() + + def validate_request( + self, *, model: str, thinking: Mapping[str, object] | None, + effort: str = "", max_tokens: int | None = None, + ) -> None: + """Reject known unsupported settings; leave unknown capabilities alone.""" + mode = thinking.get("type") if thinking is not None else None + if mode is not None and self.thinking_modes is not None and (not isinstance(mode, str) or mode not in self.thinking_modes): + raise ValueError(f"{model}: thinking mode {mode!r} is unsupported; use {sorted(self.thinking_modes)}") + if mode == "disabled" and self.thinking_required is True: + raise ValueError(f"{model}: thinking is always enabled") + if effort and self.effort_levels is not None and effort not in self.effort_levels: + raise ValueError(f"{model}: effort {effort!r} is unsupported; use {sorted(self.effort_levels)}") + if max_tokens is not None and self.max_output_tokens is not None and max_tokens > self.max_output_tokens: + raise ValueError(f"{model}: max_tokens={max_tokens} exceeds the model limit {self.max_output_tokens}") + + +_CURRENT_CLAUDE = ModelCapabilities( + thinking_modes=frozenset({"adaptive"}), thinking_required=True, + effort_levels=frozenset({"low", "medium", "high", "xhigh", "max"}), + default_effort="high", sampling_parameters=frozenset(), + tool_choice_modes=frozenset({"auto", "none"}), + thinking_signature_binding="conversation_prefix", + max_input_tokens=1_000_000, max_output_tokens=128_000, + source_urls=( + "https://platform.claude.com/docs/en/models/fable-5-1/migration-guide", + "https://platform.claude.com/docs/en/models/fable-5-1/overview", + "https://platform.claude.com/docs/en/build-with-claude/effort", + ), verified_on="2026-10-05", +) +_LEGACY_CLAUDE = ModelCapabilities( + thinking_modes=frozenset({"enabled", "disabled"}), thinking_required=False, + effort_levels=frozenset(), + source_urls=("https://platform.claude.com/docs/en/build-with-claude/extended-thinking",), + verified_on="2026-10-05", +) +# The only manual-thinking model with effort; it combines with budget_tokens. +_OPUS_4_5 = replace( + _LEGACY_CLAUDE, effort_levels=frozenset({"low", "medium", "high"}), default_effort="high", + source_urls=(*_LEGACY_CLAUDE.source_urls, "https://platform.claude.com/docs/en/build-with-claude/effort"), +) + +# These are request facts, not an endpoint, credential, price, or routing catalog. +# Only documented exact IDs/aliases match; a new version never inherits a guessed +# capability merely because its model name starts with an older one. +MODEL_CAPABILITIES: Mapping[str, ModelCapabilities] = MappingProxyType({ + "claude-fable-5-1": _CURRENT_CLAUDE, + "claude-opus-5-5": replace( + _CURRENT_CLAUDE, default_effort="medium", source_urls=( + "https://platform.claude.com/docs/en/models/opus-5-5/migration-guide", + "https://platform.claude.com/docs/en/models/opus-5-5/overview", + "https://platform.claude.com/docs/en/build-with-claude/effort", + ), + ), + "claude-sonnet-4-5": _LEGACY_CLAUDE, + "claude-sonnet-4-5-20250929": _LEGACY_CLAUDE, + "claude-haiku-4-5": _LEGACY_CLAUDE, + "claude-haiku-4-5-20251001": _LEGACY_CLAUDE, + "claude-opus-4-5": _OPUS_4_5, + "claude-opus-4-5-20251101": _OPUS_4_5, +}) + +_DISABLED_ALIASES = frozenset({"disabled", "off", "none", "false"}) + + +def normalize_thinking_mode(value: object) -> ThinkingMode | None: + """Normalize a configured thinking type; None/blank means "not configured". + + Shared by every native builder so one spelling means the same thing on the + main, auxiliary, and direct paths. Unknown values raise instead of silently + falling back to a mode the operator did not choose. + """ + if value is None: + return None + if not isinstance(value, str): + raise ValueError("thinking type must be a string or null") + mode = value.strip().lower() + if not mode: + return None + if mode in _DISABLED_ALIASES: + return "disabled" + if mode in ("adaptive", "enabled"): + return cast(ThinkingMode, mode) + raise ValueError(f"unknown thinking type {value!r}; use adaptive, enabled, or disabled") + + +_SET_FIELDS = frozenset({"thinking_modes", "effort_levels", "sampling_parameters", "tool_choice_modes"}) +_LIMIT_FIELDS = frozenset({"max_input_tokens", "max_output_tokens"}) +_OVERRIDE_FIELDS = _SET_FIELDS | _LIMIT_FIELDS | { + "thinking_required", "default_effort", "thinking_signature_binding", +} + + +def _string_set(key: str, value: object) -> frozenset[str]: + if not isinstance(value, (list, tuple, set, frozenset)): + raise ValueError(f"{key} must be a collection of strings or null") + members = cast(Iterable[object], value) + strings: set[str] = set() + for member in members: + if not isinstance(member, str): + raise ValueError(f"{key} must be a collection of strings or null") + strings.add(member) + return frozenset(strings) + + +def _canonical_model_id(model_id: str, protocol: WireProtocol) -> str: + if protocol != "bedrock": + return model_id + # Bedrock inference-profile regional prefixes and documented version suffix. + # Arbitrary proxy aliases and ARNs require explicit host overrides. + match = re.fullmatch(r"(?:(?:us|us-gov|eu|apac|jp|au|global)\.)?anthropic\.(claude-[a-z0-9-]+?)(?:-v\d+:\d+)?", model_id) + return match.group(1) if match else model_id + + +def resolve_model_capabilities( + model_id: str, *, protocol: WireProtocol, + overrides: Mapping[str, object] | None = None, +) -> ModelCapabilities: + """Resolve native model facts and a validated, per-deployment override. + + A Claude name served over Chat Completions does not imply native Messages + capabilities. Proxies and custom model aliases must declare their overrides. + Explicit None clears a known fact back to unknown. + """ + capabilities = ModelCapabilities() + if protocol in ("anthropic", "bedrock"): + capabilities = MODEL_CAPABILITIES.get(_canonical_model_id(model_id, protocol), capabilities) + if overrides is None: + return capabilities + unknown = set(overrides) - _OVERRIDE_FIELDS + if unknown: + raise ValueError(f"unknown model capability fields: {sorted(unknown)}") + changes: dict[str, object] = {} + for key, value in overrides.items(): + if value is None: + changes[key] = None + elif key in _SET_FIELDS: + changes[key] = _string_set(key, value) + elif key in _LIMIT_FIELDS: + if not isinstance(value, int) or isinstance(value, bool) or value <= 0: + raise ValueError(f"{key} must be a positive integer or null") + changes[key] = value + elif key == "thinking_required": + if not isinstance(value, bool): + raise ValueError("thinking_required must be a boolean or null") + changes[key] = value + elif key == "thinking_signature_binding": + if value not in ("none", "model", "conversation_prefix"): + raise ValueError("thinking_signature_binding must be none, model, conversation_prefix or null") + changes[key] = value + elif key == "default_effort": + if not isinstance(value, str): + raise ValueError("default_effort must be a string or null") + changes[key] = value + result = replace(capabilities, **changes, overridden_fields=frozenset(overrides)) + if result.default_effort is not None and result.effort_levels is not None and result.default_effort not in result.effort_levels: + raise ValueError("default_effort must belong to effort_levels; override both fields together") + if result.thinking_required is True and result.thinking_modes is not None and "disabled" in result.thinking_modes: + raise ValueError("thinking_required conflicts with disabled thinking mode") + return result + + +__all__ = [ + "MODEL_CAPABILITIES", "ModelCapabilities", "SignatureBinding", "ThinkingMode", + "WireProtocol", "normalize_thinking_mode", "resolve_model_capabilities", +] diff --git a/agent_core/providers/anthropic.py b/agent_core/providers/anthropic.py index af48724..3b7eccc 100644 --- a/agent_core/providers/anthropic.py +++ b/agent_core/providers/anthropic.py @@ -29,6 +29,7 @@ from agent_core.llm import LLMClient, LLMResponse, StreamDelta from agent_core.messages import Message, ToolCall, text_of +from agent_core.model_capabilities import ModelCapabilities, resolve_model_capabilities from agent_core.providers.finish_reason import normalize_finish_reason logger = logging.getLogger(__name__) @@ -54,6 +55,7 @@ def __init__( effort: str = "", bedrock: bool = False, default_headers: dict[str, str] | None = None, + capabilities: ModelCapabilities | None = None, ) -> None: self.model = model self.default_temperature = temperature @@ -66,6 +68,13 @@ def __init__( # (low|medium|high|xhigh|max) → ``output_config.effort`` via extra_body. self._thinking = thinking or None self._effort = (effort or "").strip() + self.capabilities = capabilities or resolve_model_capabilities( + model, protocol="bedrock" if bedrock else "anthropic", + ) + self.capabilities.validate_request( + model=model, thinking=self._thinking, effort=self._effort, + max_tokens=max_tokens, + ) # Transport: ``bedrock`` swaps AsyncAnthropic (``/v1/messages`` + # ``x-api-key``) for the AWS Bedrock runtime (``/model/{id}/invoke`` + # ``anthropic_version`` body stamp) authenticated with a Bedrock API Key @@ -97,6 +106,12 @@ def _build_kwargs( timeout: float | None, ) -> dict[str, Any]: """Shared request-shape builder for :meth:`chat` and :meth:`stream`.""" + output_limit = max_tokens or self.default_max_tokens or 4096 + # Thinking and effort are fixed and validated at construction; only + # the per-call output limit can change here. + self.capabilities.validate_request( + model=self.model, thinking=None, max_tokens=output_limit, + ) system, msgs = _split_system(messages) # ``_to_anthropic_msg`` returns None for a message with nothing # sendable (a contentless assistant turn); those are dropped. @@ -114,7 +129,7 @@ def _build_kwargs( kwargs: dict[str, Any] = { "model": self.model, "messages": [converted for converted, _ in pairs], - "max_tokens": max_tokens or self.default_max_tokens or 4096, + "max_tokens": output_limit, } if system: kwargs["system"] = system diff --git a/agent_core/providers/aux_builder.py b/agent_core/providers/aux_builder.py index 44a752e..ef5055f 100644 --- a/agent_core/providers/aux_builder.py +++ b/agent_core/providers/aux_builder.py @@ -7,6 +7,8 @@ from collections.abc import Callable, Mapping from typing import Any +from agent_core.model_capabilities import normalize_thinking_mode, resolve_model_capabilities + logger = logging.getLogger(__name__) type ClientFactory = Callable[..., Any] @@ -63,9 +65,10 @@ def _anthropic_thinking(section: Mapping[str, Any]) -> dict[str, Any] | None: if not isinstance(raw, Mapping): return None thinking = {str(key): value for key, value in raw.items()} - kind = str(thinking.get("type") or "").strip().lower() - if kind in {"", "disabled", "off", "none", "false"}: + kind = normalize_thinking_mode(thinking.get("type")) + if kind is None or kind == "disabled": return None + thinking["type"] = kind return thinking @@ -143,12 +146,35 @@ def _build_anthropic( "effort": str(section.get("effort") or ""), "bedrock": bedrock, } + overrides = section.get("model_capabilities") + if overrides is not None and not isinstance(overrides, Mapping): + raise ValueError("model_capabilities must be a mapping") + capabilities = resolve_model_capabilities( + str(section["model"]), protocol="bedrock" if bedrock else "anthropic", + overrides=overrides, + ) + # Preserve the caller's intent for validation before legacy conversion + # maps an explicit disabled/off configuration to no thinking field. + raw_thinking = section.get("thinking") + validation_thinking = kwargs["thinking"] + if ( + isinstance(raw_thinking, Mapping) + and normalize_thinking_mode(raw_thinking.get("type")) == "disabled" + ): + validation_thinking = {"type": "disabled"} + maximum = section.get("max_completion_tokens") or section.get("max_tokens") + capabilities.validate_request( + model=str(section["model"]), thinking=validation_thinking, + effort=kwargs["effort"].strip(), + max_tokens=int(maximum) if maximum is not None else None, + ) + if overrides is not None: + kwargs["capabilities"] = capabilities if section.get("base_url"): kwargs["base_url"] = section["base_url"] default_headers = _headers(section.get("extra_headers")) if default_headers: kwargs["default_headers"] = default_headers - maximum = section.get("max_completion_tokens") or section.get("max_tokens") if maximum is not None: kwargs["max_tokens"] = int(maximum) return self._anthropic_factory(**kwargs) diff --git a/agent_core/providers/protocol_client.py b/agent_core/providers/protocol_client.py index a6cce77..879c292 100644 --- a/agent_core/providers/protocol_client.py +++ b/agent_core/providers/protocol_client.py @@ -23,6 +23,7 @@ from typing import Any, get_args from agent_core.llm import LLMClient +from agent_core.model_capabilities import normalize_thinking_mode, resolve_model_capabilities # Single source of truth for the valid ``llm.protocol`` values. Duplicating the # set here would let the two drift, which is how a protocol becomes buildable @@ -124,47 +125,58 @@ def _build_anthropic( ``output_config.effort``. ``default_headers`` (gateway routing / auth headers) is merged over ``X-Title``, as in the Responses builder. - ``thinking_type`` selects the request shape (default ``adaptive``). Live- - verified against api.anthropic.com + Bedrock 2026-07-09 (see - ``temp/2026-07-09_reasoning-protocol-live-verification.md``); matches the - official matrix at platform.claude.com/docs/en/build-with-claude/adaptive-thinking: - - - ``adaptive`` (DEFAULT) — the RECOMMENDED mode for all current Claude - (Opus 4.6/4.7/4.8, Sonnet 4.6/5, Fable/Mythos), and the ONLY mode on the - newest (Opus 4.7/4.8, Sonnet 5) — ``enabled`` is rejected there with 400. - Emits ``thinking={"type":"adaptive"}`` + ``thinking_display`` (default - ``summarized`` so the readable thinking text is captured; the newest - models default ``display`` to ``omitted`` = empty ``thinking`` field with - the ``signature`` still present for replay). ``effort`` is forwarded only - in this mode (it is an adaptive-only knob; the oldest models 400 on - ``enabled``+effort). - - ``enabled`` — LEGACY opt-in for models older than Opus 4.6 / Sonnet 4.6 - (Sonnet 4.5, Opus 4.5, …), which reject ``adaptive`` with 400. Emits - ``thinking={"type":"enabled","budget_tokens":N}`` (``N`` from - ``thinking_budget_tokens``, default 8192, clamped to ``[1024, max_tokens-1]`` - since Anthropic requires ``budget_tokens < max_tokens``). When the - configured ``max_tokens`` is too small for the 1024 floor, ``max_tokens`` - is RAISED to ``budget + 1`` rather than emitting an invalid pair — see - :func:`_enabled_thinking_budget`. ``effort`` is - NOT sent (budget_tokens is the control knob here; oldest models 400 on it). - Deprecated on Opus 4.6 / Sonnet 4.6 per Anthropic. + ``thinking_type`` overrides the model-aware default: prefer adaptive, use + enabled for known manual-thinking-only models, or omit thinking when the + host declares no supported mode. Unknown models retain the protocol's + previous adaptive default. Adaptive display defaults to summarized so + readable progress is retained even when the API defaults to omitted. + + Enabled thinking uses :func:`_enabled_thinking_budget` to keep the legacy + 1024-token floor below max_tokens. It forwards effort only when the model + capability record establishes support (for example Opus 4.5); unknown + legacy models retain the previous behavior of omitting effort. Unsupported + explicit settings raise before creating the SDK client. Hosts can override + facts through the profile's ``model_capabilities`` mapping. """ from agent_core.providers.anthropic import AnthropicClient max_tokens = int(cfg.get("max_tokens", 32768)) - ttype = str(cfg.get("thinking_type", "adaptive")).strip().lower() + overrides = cfg.get("model_capabilities") + if overrides is not None and not isinstance(overrides, dict): + raise ValueError("model_capabilities must be a mapping") + capabilities = resolve_model_capabilities( + cfg["model"], protocol="bedrock" if bedrock else "anthropic", overrides=overrides, + ) + default_mode: str | None = "adaptive" + if capabilities.thinking_modes is not None and "adaptive" not in capabilities.thinking_modes: + default_mode = ( + "enabled" if "enabled" in capabilities.thinking_modes + else "disabled" if "disabled" in capabilities.thinking_modes else None + ) + ttype = normalize_thinking_mode(cfg.get("thinking_type")) or default_mode + capabilities.validate_request( + model=cfg["model"], thinking={"type": ttype} if ttype else None, + effort=_effort_str(cfg), + ) + thinking: dict[str, Any] | None if ttype == "enabled": budget, max_tokens = _enabled_thinking_budget( int(cfg.get("thinking_budget_tokens", 8192)), max_tokens, ) - thinking: dict[str, Any] = {"type": "enabled", "budget_tokens": budget} - effort = "" - else: + thinking = {"type": "enabled", "budget_tokens": budget} + effort = _effort_str(cfg) if capabilities.effort_levels else "" + elif ttype == "adaptive": thinking = {"type": "adaptive"} display = cfg.get("thinking_display", "summarized") if isinstance(display, str) and display.strip(): thinking["display"] = display.strip() effort = _effort_str(cfg) + elif ttype == "disabled": + thinking = {"type": "disabled"} + effort = _effort_str(cfg) + else: + thinking = None + effort = _effort_str(cfg) return AnthropicClient( model=cfg["model"], api_key=cfg.get("api_key", "dummy"), @@ -174,6 +186,7 @@ def _build_anthropic( effort=effort, default_headers={"X-Title": title, **(cfg.get("default_headers") or {})}, bedrock=bedrock, + capabilities=capabilities, ) diff --git a/agent_core/runtime/loop/model_profile.py b/agent_core/runtime/loop/model_profile.py index 583c73d..324c145 100644 --- a/agent_core/runtime/loop/model_profile.py +++ b/agent_core/runtime/loop/model_profile.py @@ -19,6 +19,8 @@ assistant_msg, assistant_msg_with_reasoning, ) +from agent_core.model_capabilities import ModelCapabilities, resolve_model_capabilities +from agent_core.model_capabilities import WireProtocol as WireProtocol logger = logging.getLogger(__name__) @@ -41,7 +43,6 @@ # Wire protocol the client speaks. Named alias so config readers can declare it # instead of returning a bare str that every ModelProfile call site then rejects. -WireProtocol = Literal["chat_completions", "anthropic", "responses", "bedrock"] _VALID_PROTOCOLS: frozenset[str] = frozenset(get_args(WireProtocol)) @@ -209,6 +210,22 @@ class ModelProfile: # ``content_block`` so the parser keeps the verbatim blocks (signatures / # encrypted_content) for faithful multi-turn replay + trajectory. protocol: WireProtocol = "chat_completions" + # Request constraints use the same records as provider construction. + # context_window above remains the host's operational context budget; + # capabilities.max_input_tokens describes the provider's maximum. + # ``capabilities`` takes a resolved record (e.g. ``client.capabilities``); + # otherwise ``model_capabilities`` carries the same per-deployment override + # mapping the client config uses, so both sides resolve identically. + capabilities: ModelCapabilities | None = None + model_capabilities: Mapping[str, object] | None = None + + @property + def request_capabilities(self) -> ModelCapabilities: + if self.capabilities is not None: + return self.capabilities + return resolve_model_capabilities( + self.model_id, protocol=self.protocol, overrides=self.model_capabilities, + ) @dataclass diff --git a/changes/model-capabilities.feature.md b/changes/model-capabilities.feature.md new file mode 100644 index 0000000..d12985c --- /dev/null +++ b/changes/model-capabilities.feature.md @@ -0,0 +1 @@ +Add typed, source-documented model capability records shared by model profiles and Anthropic request construction. Known models select compatible thinking defaults and reject unsupported thinking, effort, or output-token settings before API calls; unknown models retain pass-through behavior and hosts can override deployment-specific facts without global mutable state. Enable CI for stacked pull requests. Behavior changes: an unrecognized `thinking_type` now raises instead of falling back to adaptive (`off`/`none`/`false` mean disabled on every builder); configuring `effort` for Sonnet 4.5 or Haiku 4.5 now raises instead of being dropped; and Sonnet/Haiku/Opus 4.5 without an explicit `thinking_type` now default to `enabled` (budget 8192, which may raise `max_tokens`) instead of the rejected adaptive mode. diff --git a/docs/provider-substrate-boundary.md b/docs/provider-substrate-boundary.md index 91d315b..4fdd31c 100644 --- a/docs/provider-substrate-boundary.md +++ b/docs/provider-substrate-boundary.md @@ -107,3 +107,80 @@ Unknown detail fields/categories pass through, including with older SDKs. Observers receive `TurnContext.finish_reason` and `TurnContext.stop_details`; the trajectory observer writes them to JSON snapshots and JSONL events. This telemetry does not change loop termination or choose a fallback model. + +## Model capabilities and deployment overrides + +`agent_core.model_capabilities` holds immutable request facts keyed by exact +model IDs and documented aliases. Each record has source URLs and a verification +date. `ModelProfile.request_capabilities` and the Anthropic client use this same +resolver; the host still owns endpoints, credentials, routing, prices, and model +catalog discovery. No network request occurs during resolution. + +`None` means unknown; an empty set means a feature has no supported values. +Unknown IDs, newer versions, and Claude names served over Chat Completions do +not inherit native Anthropic restrictions. Bedrock's documented regional model +prefixes and version suffixes resolve to the same underlying model facts; the +outbound model ID is never rewritten. Arbitrary aliases and ARNs require host +overrides. This first table covers Fable 5.1, Opus 5.5, and the legacy 4.5 models; +other providers and models remain unknown until verified facts are added. + +Known thinking modes select builder defaults and reject unsupported explicit +modes. Every native builder normalizes the configured thinking type the same +way: `adaptive`, `enabled`, and `disabled` (also spelled `off`, `none`, or +`false`); blank means unset, and any other value raises `ValueError`. Effort and output limits are validated in direct clients, native profile +clients, and per-call overrides. No parameter is silently clamped. The existing +legacy thinking-budget adjustment remains in place. Empty thinking support +omits the thinking field; unknown native models keep the prior adaptive default. +Only thinking modes, required thinking, effort levels, and output limits are +enforced. `default_effort`, `sampling_parameters`, `tool_choice_modes`, +`thinking_signature_binding`, and `max_input_tokens` are descriptive until an +adapter consumes them. Sampling, tool-choice, and signature-binding facts are available to hosts; +protocol field conversion, SDK transport limitations, and error recovery stay +in adapters. The Anthropic adapter still omits sampling fields for SDK 1.x +compatibility. This PR does not add forced tool-choice parameters. + +Resolution starts from verified model facts (or unknown), then applies a +per-client host override. The `model_capabilities` profile key is a +mapping with the capability field names; omitted fields inherit, explicit null +clears a fact to unknown, and empty lists declare unsupported values. Unknown +keys, malformed values, and contradictory defaults raise `ValueError`. +`overridden_fields` identifies which facts the host supplied; `source_urls` +describes inherited facts, not proof of a host override. Hosts own the evidence +for their deployment overrides. Records never mutate a global registry. + +```yaml +llm: + protocol: anthropic + model: claude-opus-5-5 + effort: medium + model_capabilities: + max_output_tokens: 32768 # gateway limit overrides the model maximum +``` + +A profile must see the same overrides as its client. Pass the profile the same +mapping (`ModelProfile(..., model_capabilities=cfg.get("model_capabilities"))`) +or the client's resolved record (`capabilities=client.capabilities`); a profile +built with neither resolves the unoverridden model facts. + +For a custom alias, specify its supported modes and effort levels explicitly: + +```python +from agent_core import resolve_model_capabilities +from agent_core.runtime.loop.model_profile import ModelProfile + +caps = resolve_model_capabilities( + "gateway-alias", protocol="anthropic", + overrides={"thinking_modes": ["adaptive"], "effort_levels": ["low", "high"]}, +) +profile = ModelProfile( + model_id="gateway-alias", provider="gateway", protocol="anthropic", + capabilities=caps, context_window=64000, +) +``` + +`context_window` remains a host-selected operational budget. The capability's +`max_input_tokens` is the provider's maximum, and does not overwrite that budget. +Defaults such as `default_effort` are descriptive; callers who omit effort keep +the provider's own default. Beta/platform-specific limits require a host override +paired with the appropriate headers. Models API discovery/caching can be added +by hosts later; this PR does not introduce a background synchronization service. diff --git a/tests/test_model_capabilities.py b/tests/test_model_capabilities.py new file mode 100644 index 0000000..eb7c1ed --- /dev/null +++ b/tests/test_model_capabilities.py @@ -0,0 +1,338 @@ +from __future__ import annotations + +from dataclasses import FrozenInstanceError + +import pytest + +from agent_core.model_capabilities import ( + MODEL_CAPABILITIES, + ModelCapabilities, + normalize_thinking_mode, + resolve_model_capabilities, +) +from agent_core.providers.anthropic import AnthropicClient +from agent_core.providers.protocol_client import build_protocol_client +from agent_core.runtime.loop.model_profile import ModelProfile + + +@pytest.mark.parametrize("model,effort", [("claude-fable-5-1", "high"), ("claude-opus-5-5", "medium")]) +def test_current_model_facts_are_shared_with_profiles(model, effort): + capabilities = resolve_model_capabilities(model, protocol="anthropic") + profile = ModelProfile(model_id=model, provider="anthropic", protocol="anthropic", context_window=64_000) + assert profile.request_capabilities is capabilities + assert profile.context_window == 64_000 # operational budget stays host-owned + assert capabilities.thinking_required is True + assert capabilities.default_effort == effort + assert capabilities.source_urls and capabilities.verified_on == "2026-10-05" + assert capabilities.max_output_tokens == 128_000 + assert capabilities.thinking_signature_binding == "conversation_prefix" + + +@pytest.mark.parametrize("model,protocol", [ + ("claude-opus-5-5-next", "anthropic"), + ("gateway-alias", "anthropic"), + ("claude-opus-5-5", "chat_completions"), + ("gpt-5.5", "responses"), +]) +def test_unknown_and_other_protocols_do_not_inherit_native_rules(model, protocol): + capabilities = resolve_model_capabilities(model, protocol=protocol) + assert capabilities == ModelCapabilities() + capabilities.validate_request(model=model, thinking={"type": "future"}, effort="custom", max_tokens=500_000) + + +@pytest.mark.parametrize("model", [ + "anthropic.claude-opus-5-5", "us.anthropic.claude-opus-5-5", + "eu.anthropic.claude-opus-5-5-v1:0", "global.anthropic.claude-opus-5-5", +]) +def test_bedrock_model_ids_resolve_without_rewriting_request_model(model): + assert resolve_model_capabilities(model, protocol="bedrock") is MODEL_CAPABILITIES["claude-opus-5-5"] + assert resolve_model_capabilities(model, protocol="anthropic") == ModelCapabilities() + + +def test_overrides_distinguish_unknown_from_unsupported_and_are_local(): + baseline = resolve_model_capabilities("claude-opus-5-5", protocol="anthropic") + overrides = {"thinking_modes": ["enabled", "disabled"], "thinking_required": False, + "effort_levels": [], "default_effort": None, "max_output_tokens": None} + patched = resolve_model_capabilities("claude-opus-5-5", protocol="anthropic", overrides=overrides) + assert patched.thinking_modes == frozenset({"enabled", "disabled"}) + assert patched.effort_levels == frozenset() + assert patched.max_output_tokens is None + assert patched.overridden_fields == frozenset(overrides) + assert resolve_model_capabilities("claude-opus-5-5", protocol="anthropic") is baseline + overrides["thinking_modes"].append("adaptive") + assert "adaptive" not in patched.thinking_modes + with pytest.raises(FrozenInstanceError): + patched.thinking_required = True + with pytest.raises(TypeError): + MODEL_CAPABILITIES["custom"] = patched + + +@pytest.mark.parametrize("overrides", [ + {"typo": True}, {"effort_levels": "high"}, {"effort_levels": [1]}, + {"thinking_required": "false"}, {"max_output_tokens": True}, + {"max_input_tokens": 0}, {"thinking_signature_binding": "typo"}, + {"default_effort": 5}, {"effort_levels": []}, + {"thinking_modes": ["disabled"]}, +]) +def test_invalid_overrides_are_rejected(overrides): + with pytest.raises(ValueError): + resolve_model_capabilities("claude-opus-5-5", protocol="anthropic", overrides=overrides) + + +@pytest.mark.parametrize("thinking,effort,limit", [ + ({"type": "disabled"}, "", 4096), ({"type": "enabled", "budget_tokens": 1024}, "", 4096), + ({"type": ["adaptive"]}, "", 4096), (None, "minimal", 4096), + (None, "", 128_001), +]) +def test_direct_client_rejects_known_invalid_requests_before_sdk_creation(thinking, effort, limit, monkeypatch): + import anthropic + + def unexpected_client(**kwargs): + raise AssertionError("invalid request must fail before opening an SDK client") + + monkeypatch.setattr(anthropic, "AsyncAnthropic", unexpected_client) + with pytest.raises(ValueError): + AnthropicClient("claude-opus-5-5", api_key="x", thinking=thinking, effort=effort, max_tokens=limit) + + +@pytest.mark.parametrize("model,mode", [ + ("claude-fable-5-1", "adaptive"), ("claude-opus-5-5", "adaptive"), + ("claude-sonnet-4-5-20250929", "enabled"), ("claude-haiku-4-5", "enabled"), + ("claude-x", "adaptive"), +]) +async def test_native_builder_uses_model_aware_defaults(model, mode): + client = build_protocol_client({"protocol": "anthropic", "model": model, "max_tokens": 4096}, title="test") + try: + kwargs = client._build_kwargs([{"role": "user", "content": "hi"}], tools=None, temperature=None, max_tokens=None, extra_headers=None, timeout=None) + assert kwargs["thinking"]["type"] == mode + if mode == "enabled": + assert 1024 <= kwargs["thinking"]["budget_tokens"] < kwargs["max_tokens"] + finally: + await client._client.close() + + +async def test_per_call_limit_is_validated_as_well_as_constructor_default(): + client = AnthropicClient("claude-opus-5-5", api_key="x") + try: + with pytest.raises(ValueError, match="max_tokens"): + client._build_kwargs([], tools=None, temperature=None, max_tokens=128_001, extra_headers=None, timeout=None) + finally: + await client._client.close() + + +async def test_builder_override_can_disable_thinking_configuration_entirely(): + client = build_protocol_client({ + "protocol": "anthropic", "model": "claude-opus-5-5", + "model_capabilities": {"thinking_modes": [], "thinking_required": False}, + }, title="test") + try: + assert client._thinking is None + finally: + await client._client.close() + + +async def test_manual_thinking_can_forward_effort_when_model_supports_it(): + client = build_protocol_client({"protocol": "anthropic", "model": "claude-opus-4-5", "effort": "medium"}, title="test") + try: + kwargs = client._build_kwargs([], tools=None, temperature=None, max_tokens=None, extra_headers=None, timeout=None) + assert kwargs["thinking"]["type"] == "enabled" + assert kwargs["extra_body"]["output_config"]["effort"] == "medium" + finally: + await client._client.close() + + +@pytest.mark.parametrize("mode", [None, "", " "]) +async def test_unset_thinking_type_uses_model_default(mode): + client = build_protocol_client({"protocol": "anthropic", "model": "claude-haiku-4-5", "thinking_type": mode}, title="test") + try: + assert client._thinking["type"] == "enabled" + finally: + await client._client.close() + + +def test_auxiliary_factory_forwards_overrides_to_the_same_capability_resolver(): + from agent_core.providers.aux_builder import AuxLLMFactory + + factory = AuxLLMFactory(openai_factory=lambda **kw: kw, anthropic_factory=lambda **kw: kw, + provider_type=lambda _: "anthropic") + kwargs = factory.build({"provider": "gateway", "model": "claude-opus-5-5", "api_key": "x", + "model_capabilities": {"max_output_tokens": 32_000}}) + assert kwargs["capabilities"].max_output_tokens == 32_000 + assert MODEL_CAPABILITIES["claude-opus-5-5"].max_output_tokens == 128_000 + + +@pytest.mark.parametrize("thinking", [{"type": "disabled"}, {"type": "off"}, {"type": "enabled", "budget_tokens": 1024}]) +def test_auxiliary_client_checks_explicit_modes_before_legacy_conversion(thinking): + from agent_core.providers.aux_builder import AuxLLMFactory + + def unexpected_factory(**kwargs): + raise AssertionError("configuration must fail before client construction") + + factory = AuxLLMFactory(openai_factory=unexpected_factory, anthropic_factory=unexpected_factory, + provider_type=lambda _: "anthropic") + with pytest.raises(ValueError, match="thinking mode"): + factory.build({"provider": "anthropic", "model": "claude-opus-5-5", "api_key": "x", "thinking": thinking}) + + +def test_known_legacy_model_rejects_effort_instead_of_silently_dropping_it(): + with pytest.raises(ValueError, match="effort"): + build_protocol_client({"protocol": "anthropic", "model": "claude-sonnet-4-5", "effort": "high"}, title="test") + + +async def test_profile_and_client_resolve_the_same_deployment_overrides(): + overrides = {"max_output_tokens": 32_000} + client = build_protocol_client({ + "protocol": "anthropic", "model": "claude-opus-5-5", "max_tokens": 16_000, + "model_capabilities": overrides, + }, title="test") + try: + profile = ModelProfile(model_id="claude-opus-5-5", provider="anthropic", + protocol="anthropic", model_capabilities=overrides) + assert profile.request_capabilities == client.capabilities + assert profile.request_capabilities.max_output_tokens == 32_000 + finally: + await client._client.close() + + +@pytest.mark.parametrize("spelling", ["disabled", "off", "none", "False"]) +async def test_main_and_auxiliary_builders_share_disabled_spellings(spelling): + from agent_core.providers.aux_builder import AuxLLMFactory + + client = build_protocol_client({"protocol": "anthropic", "model": "claude-haiku-4-5", "thinking_type": spelling}, title="test") + try: + assert client._thinking == {"type": "disabled"} + finally: + await client._client.close() + factory = AuxLLMFactory(openai_factory=lambda **kw: kw, anthropic_factory=lambda **kw: kw, + provider_type=lambda _: "anthropic") + kwargs = factory.build({"provider": "anthropic", "model": "claude-haiku-4-5", "api_key": "x", + "thinking": {"type": spelling}}) + assert kwargs["thinking"] is None + + +def test_main_and_auxiliary_builders_reject_the_same_unknown_mode(): + from agent_core.providers.aux_builder import AuxLLMFactory + + with pytest.raises(ValueError, match="unknown thinking type"): + build_protocol_client({"protocol": "anthropic", "model": "claude-x", "thinking_type": "auto"}, title="test") + factory = AuxLLMFactory(openai_factory=lambda **kw: kw, anthropic_factory=lambda **kw: kw, + provider_type=lambda _: "anthropic") + with pytest.raises(ValueError, match="unknown thinking type"): + factory.build({"provider": "anthropic", "model": "claude-x", "api_key": "x", "thinking": {"type": "auto"}}) + + +@pytest.mark.parametrize("model_id", [ + "jp.anthropic.claude-opus-5-5", "au.anthropic.claude-opus-5-5-v1:0", "us-gov.anthropic.claude-opus-5-5", +]) +def test_additional_bedrock_regional_prefixes_resolve(model_id): + assert resolve_model_capabilities(model_id, protocol="bedrock") is MODEL_CAPABILITIES["claude-opus-5-5"] + + +@pytest.mark.parametrize("value,expected", [ + (None, None), ("", None), (" ", None), + (" Adaptive ", "adaptive"), ("ENABLED", "enabled"), + ("disabled", "disabled"), ("Off", "disabled"), ("none", "disabled"), ("false", "disabled"), +]) +def test_normalize_thinking_mode_accepts_documented_spellings(value, expected): + assert normalize_thinking_mode(value) == expected + + +@pytest.mark.parametrize("value", ["auto", "on", "true", "budget"]) +def test_normalize_thinking_mode_rejects_unknown_values(value): + with pytest.raises(ValueError, match="unknown thinking type"): + normalize_thinking_mode(value) + + +@pytest.mark.parametrize("value", [False, 0, {"type": "adaptive"}]) +def test_normalize_thinking_mode_rejects_non_strings(value): + with pytest.raises(ValueError, match="must be a string"): + normalize_thinking_mode(value) + + +def test_opus_4_5_alias_and_dated_id_share_one_record(): + alias = MODEL_CAPABILITIES["claude-opus-4-5"] + assert alias is MODEL_CAPABILITIES["claude-opus-4-5-20251101"] + assert alias.effort_levels == frozenset({"low", "medium", "high"}) + assert alias.thinking_modes == frozenset({"enabled", "disabled"}) + assert "https://platform.claude.com/docs/en/build-with-claude/effort" in alias.source_urls + + +@pytest.mark.parametrize("model_id", [ + "anthropic.claude-opus-5-5", "us.anthropic.claude-opus-5-5-v1:0", + "eu.anthropic.claude-opus-5-5", "apac.anthropic.claude-opus-5-5", + "global.anthropic.claude-opus-5-5", +]) +def test_documented_bedrock_prefixes_still_resolve(model_id): + assert resolve_model_capabilities(model_id, protocol="bedrock") is MODEL_CAPABILITIES["claude-opus-5-5"] + + +@pytest.mark.parametrize("model_id", [ + "xx.anthropic.claude-opus-5-5", + "arn:aws:bedrock:us-east-1:123:inference-profile/us.anthropic.claude-opus-5-5-v1:0", + "jp.anthropic.claude-opus-5-5", # bedrock form, but resolved over the native protocol +]) +def test_unrecognized_bedrock_forms_stay_unknown(model_id): + protocol = "anthropic" if model_id.startswith("jp.") else "bedrock" + assert resolve_model_capabilities(model_id, protocol=protocol) == ModelCapabilities() + + +async def test_per_call_limit_uses_the_deployment_override(): + client = build_protocol_client({ + "protocol": "anthropic", "model": "claude-opus-5-5", "max_tokens": 16_000, + "model_capabilities": {"max_output_tokens": 32_000}, + }, title="test") + try: + ok = client._build_kwargs([], tools=None, temperature=None, max_tokens=32_000, extra_headers=None, timeout=None) + assert ok["max_tokens"] == 32_000 + with pytest.raises(ValueError, match="max_tokens=32001"): + client._build_kwargs([], tools=None, temperature=None, max_tokens=32_001, extra_headers=None, timeout=None) + finally: + await client._client.close() + + +async def test_per_call_default_limit_is_validated_against_the_record(): + client = AnthropicClient("claude-opus-5-5", api_key="x", max_tokens=128_000) + try: + kwargs = client._build_kwargs([], tools=None, temperature=None, max_tokens=None, extra_headers=None, timeout=None) + assert kwargs["max_tokens"] == 128_000 + finally: + await client._client.close() + + +def test_profile_prefers_an_explicit_record_over_override_mapping(): + record = resolve_model_capabilities("claude-opus-5-5", protocol="anthropic", overrides={"max_output_tokens": 8_000}) + profile = ModelProfile(model_id="claude-opus-5-5", provider="anthropic", protocol="anthropic", + capabilities=record, model_capabilities={"max_output_tokens": 64_000}) + assert profile.request_capabilities is record + + +def test_profile_rejects_malformed_overrides_like_the_client(): + profile = ModelProfile(model_id="claude-opus-5-5", provider="anthropic", protocol="anthropic", + model_capabilities={"max_output_tokens": -1}) + with pytest.raises(ValueError, match="positive integer"): + _ = profile.request_capabilities + with pytest.raises(ValueError, match="positive integer"): + build_protocol_client({"protocol": "anthropic", "model": "claude-opus-5-5", + "model_capabilities": {"max_output_tokens": -1}}, title="test") + + +def test_profile_overrides_do_not_apply_to_other_protocols_facts(): + profile = ModelProfile(model_id="claude-opus-5-5", provider="gateway", protocol="chat_completions", + model_capabilities={"max_output_tokens": 32_000}) + caps = profile.request_capabilities + assert caps.max_output_tokens == 32_000 + assert caps.thinking_modes is None # native facts are not inherited over chat completions + assert caps.overridden_fields == frozenset({"max_output_tokens"}) + + +def test_descriptive_fields_are_exposed_and_overridable(): + caps = resolve_model_capabilities("claude-fable-5-1", protocol="anthropic", overrides={ + "tool_choice_modes": ["auto", "none", "any"], "sampling_parameters": None, + "thinking_signature_binding": "model", "max_input_tokens": 200_000, + }) + assert caps.tool_choice_modes == frozenset({"auto", "none", "any"}) + assert caps.sampling_parameters is None + assert caps.thinking_signature_binding == "model" + assert caps.max_input_tokens == 200_000 + # Descriptive facts never affect request validation. + caps.validate_request(model="claude-fable-5-1", thinking={"type": "adaptive"}, effort="high", max_tokens=128_000)