Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
77 changes: 72 additions & 5 deletions agent_core/providers/anthropic.py
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,7 @@ def __init__(
bedrock: bool = False,
default_headers: dict[str, str] | None = None,
capabilities: ModelCapabilities | None = None,
prompt_cache_ttl: str | None = "",
) -> None:
self.model = model
self.default_temperature = temperature
Expand All @@ -68,6 +69,9 @@ def __init__(
# (low|medium|high|xhigh|max) → ``output_config.effort`` via extra_body.
self._thinking = thinking or None
self._effort = (effort or "").strip()
# Rejected here rather than per call: a bad value is a config error and
# should fail at construction, not on the first request of a long run.
self._prompt_cache_ttl = normalize_prompt_cache_ttl(prompt_cache_ttl)
self.capabilities = capabilities or resolve_model_capabilities(
model, protocol="bedrock" if bedrock else "anthropic",
)
Expand Down Expand Up @@ -147,7 +151,9 @@ def _build_kwargs(
kwargs["timeout"] = timeout
elif self.default_timeout is not None:
kwargs["timeout"] = self.default_timeout
_add_prompt_cache(kwargs, transient_tail=transient_tail)
_add_prompt_cache(
kwargs, transient_tail=transient_tail, ttl=self._prompt_cache_ttl,
)
# After the cache breakpoint is placed, so it stays on the last
# persistent block rather than moving onto the per-call text.
kwargs["messages"] = _fold_transient_tail(kwargs["messages"], transient_tail)
Expand Down Expand Up @@ -547,7 +553,54 @@ def _split_system(messages: list[Message]) -> tuple[str, list[Message]]:
return "", list(messages)


def _add_prompt_cache(kwargs: dict[str, Any], *, transient_tail: int = 0) -> None:
#: Cache lifetimes Anthropic accepts on ``cache_control``. The empty string is
#: this adapter's "not configured", which omits the field and takes the API
#: default of five minutes.
PROMPT_CACHE_TTLS = frozenset({"5m", "1h"})

#: Deployment-wide default when no client was configured with one, so an
#: operator can turn the longer lifetime on for a whole run without a config
#: path reaching every construction site. A client's own value wins.
PROMPT_CACHE_TTL_ENV = "ANTHROPIC_PROMPT_CACHE_TTL"


def normalize_prompt_cache_ttl(value: object) -> str:
"""Normalize a configured cache TTL; ``""`` means "not configured".

Unknown values raise rather than falling back to the default, matching
:func:`agent_core.model_capabilities.normalize_thinking_mode`: a typo that
silently reverts to five minutes is the failure this whole knob exists to
fix, and it would only surface as a cache-hit-rate regression nobody is
watching.
"""
if value is None:
return ""
if not isinstance(value, str):
raise ValueError("prompt cache ttl must be a string or null")
ttl = value.strip().lower()
if not ttl:
return ""
if ttl not in PROMPT_CACHE_TTLS:
raise ValueError(
f"unsupported prompt cache ttl {value!r}; use {sorted(PROMPT_CACHE_TTLS)}",
)
return ttl


def _cache_control(ttl: str) -> dict[str, str]:
"""The ``cache_control`` value for a breakpoint, with TTL when configured."""
resolved = ttl or normalize_prompt_cache_ttl(os.getenv(PROMPT_CACHE_TTL_ENV))
if not resolved or resolved == "5m":
# Omitted rather than sent explicitly: "5m" is the API default, and not
# sending the field keeps the request shape of every existing consumer
# byte-identical.
return {"type": "ephemeral"}
return {"type": "ephemeral", "ttl": resolved}


def _add_prompt_cache(
kwargs: dict[str, Any], *, transient_tail: int = 0, ttl: str = "",
) -> None:
"""Set Anthropic prompt-cache breakpoints on ``kwargs`` in place.

Anthropic caching is opt-in per content block (unlike OpenAI's automatic
Expand All @@ -565,16 +618,30 @@ def _add_prompt_cache(kwargs: dict[str, Any], *, transient_tail: int = 0) -> Non
so a cached prefix ending on one never matches again and every turn would
re-write the whole conversation at the cache-write rate while reading only
the static head.

``ttl`` ("5m" default, or "1h") applies to BOTH breakpoints. A cached entry
expires that long after its last use, so the lifetime that matters is not
how long a run takes but how long one TURN takes: a gap longer than the TTL
loses the whole prefix and re-writes it. On a slow model that is a routine
event rather than an edge case — in ApodexHarness's 2026-10-05 GDPval batch,
12.9% of claude-opus-5-5 turn gaps exceeded five minutes (p90 355s) and
those calls missed the cache 35.7% of the time against 9.5% for the rest,
re-writing a median 81k tokens each. The trade is the write rate: 2x base
input for an hour against 1.25x for five minutes, so this pays off exactly
when turns are slow enough to straddle the shorter window and costs extra
when they are not. Writes land in ``cache_creation.ephemeral_1h_input_tokens``,
which :func:`_anthropic_cache_write_tokens` already counts.
"""
if os.getenv("ANTHROPIC_PROMPT_CACHE", "1") == "0":
return
cache_control = _cache_control(ttl)
# System prefix (a plain string) -> one cache-controlled text block.
system = kwargs.get("system")
if isinstance(system, str) and system:
kwargs["system"] = [{
"type": "text",
"text": system,
"cache_control": {"type": "ephemeral"},
"cache_control": cache_control,
}]
# Rolling tail: mark the last persistent message's final content block.
msgs = kwargs.get("messages")
Expand All @@ -587,10 +654,10 @@ def _add_prompt_cache(kwargs: dict[str, Any], *, transient_tail: int = 0) -> Non
last["content"] = [{
"type": "text",
"text": content,
"cache_control": {"type": "ephemeral"},
"cache_control": cache_control,
}]
elif isinstance(content, list) and content and isinstance(content[-1], dict):
content[-1] = {**content[-1], "cache_control": {"type": "ephemeral"}}
content[-1] = {**content[-1], "cache_control": cache_control}


def _to_anthropic_msg(m: Message) -> dict[str, Any] | None:
Expand Down
5 changes: 5 additions & 0 deletions agent_core/providers/protocol_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -187,6 +187,11 @@ def _build_anthropic(
default_headers={"X-Title": title, **(cfg.get("default_headers") or {})},
bedrock=bedrock,
capabilities=capabilities,
# A profile-level key, so one model's declared cache lifetime travels
# with the model it was chosen for rather than with the deployment.
# Preserve invalid falsey values so the client rejects them instead of
# silently selecting the environment's default lifetime.
prompt_cache_ttl=cfg.get("prompt_cache_ttl"),
)


Expand Down
3 changes: 3 additions & 0 deletions changes/61.feature.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
`AnthropicClient` and `build_protocol_client` accept `prompt_cache_ttl` (`5m`, the API default, or `1h`), which sets the lifetime of both prompt-cache breakpoints the adapter places. Blank and `5m` omit the field, so an unconfigured consumer's requests are byte-identical to before; `ANTHROPIC_PROMPT_CACHE_TTL` provides a deployment-wide default for construction sites a profile does not reach, and a client's own value outranks it. Unsupported values (`30m`, numbers, a stray unit) raise `ValueError` at construction rather than silently reverting to five minutes, which would only surface as a cache-hit-rate regression. The existing `ANTHROPIC_PROMPT_CACHE=0` kill switch still wins, and 1h writes were already counted by the adapter's cache-write accounting.

Consumers should set this per model, not globally: an hour costs 2x base input to write against 1.25x for five minutes, and it pays off only when one TURN can take longer than the shorter window — a cached entry expires relative to its last use. In ApodexHarness's 2026-10-05 GDPval batch, 12.9% of `claude-opus-5-5` turn gaps at `effort=max` exceeded 300s and those calls missed the cache 35.7% of the time against 9.5% for the rest, rewriting a median 81k tokens each; a fast model has no such rewrites to avoid and would only pay the higher write rate. A gateway has to forward the field: llm-hub was verified on 2026-10-05 (HTTP 200, write billed to the 1h bucket, a read at 548s still hitting while the 5m control had expired), and the `extended-cache-ttl-2025-04-11` beta header was not required.
42 changes: 42 additions & 0 deletions docs/provider-substrate-boundary.md
Original file line number Diff line number Diff line change
Expand Up @@ -184,3 +184,45 @@ Defaults such as `default_effort` are descriptive; callers who omit effort keep
the provider's own default. Beta/platform-specific limits require a host override
paired with the appropriate headers. Models API discovery/caching can be added
by hosts later; this PR does not introduce a background synchronization service.

## Anthropic prompt-cache lifetime

Anthropic caching is opt-in per content block, so `AnthropicClient` places two
`ephemeral` breakpoints itself: the system prefix and the last persistent
message's final block. `ANTHROPIC_PROMPT_CACHE=0` disables both.

The lifetime of those breakpoints is configurable with the `prompt_cache_ttl`
profile key (`5m`, the API default, or `1h`), forwarded by
`build_protocol_client`; `normalize_prompt_cache_ttl` accepts a blank value as
"not configured" and raises `ValueError` on anything else, including `30m` and
numbers. `5m` and blank both OMIT the field, so an unconfigured client's request
is unchanged. `ANTHROPIC_PROMPT_CACHE_TTL` sets a deployment-wide default for
construction sites a profile does not reach; a client's own value wins. A TTL
applies to both breakpoints and never places one — with the cache disabled it
has no effect.

```yaml
llm:
protocol: anthropic
model: claude-opus-5-5
prompt_cache_ttl: 1h # turns are slower than the 5m default window
```

Which lifetime pays off is a property of TURN duration, not run duration: a
cached entry expires that long after its last use, so a gap longer than the TTL
loses the whole prefix and rewrites it at the cache-write rate. An hour costs 2x
base input to write against 1.25x for five minutes, so it wins only where turns
are slow enough to straddle the shorter window. Measured on ApodexHarness's
2026-10-05 GDPval batch (6056 `claude-opus-5-5` calls at `effort=max`): 12.9% of
turn gaps exceeded 300s (p90 355s), and those calls missed the cache 35.7% of the
time against 9.5% for the rest, rewriting a median 81k tokens each. A fast model
whose gaps sit well inside five minutes has no such rewrites to avoid and only
pays the higher write rate.

Writes land in `cache_creation.ephemeral_1h_input_tokens`, which the adapter's
cache-write accounting already includes, so `cache_write_tokens` stays correct
across both lifetimes. Gateways must forward the field for it to take effect:
verified against llm-hub on 2026-10-05 — `ttl: "1h"` returned HTTP 200, the
write was billed to the 1h bucket rather than the 5m one, and a read 548s later
still hit in full while the unconfigured control had expired and rewritten. The
`extended-cache-ttl-2025-04-11` beta header was not required.
152 changes: 152 additions & 0 deletions tests/test_provider_native_clients.py
Original file line number Diff line number Diff line change
Expand Up @@ -1500,3 +1500,155 @@ async def gen():
assert terminal.stop_details == {
"type": "refusal", "category": "cyber", "explanation": "declined",
}


# ── prompt-cache TTL (5m default, opt-in 1h) ──


def _cache_controls(kwargs):
"""Every ``cache_control`` value in the request, system block first."""
found = [
b["cache_control"] for b in kwargs.get("system", [])
if isinstance(b, dict) and "cache_control" in b
]
for m in kwargs["messages"]:
if isinstance(m["content"], list):
found += [
b["cache_control"] for b in m["content"]
if isinstance(b, dict) and "cache_control" in b
]
return found


def _build_cached(monkeypatch, **client_kwargs):
monkeypatch.delenv("ANTHROPIC_PROMPT_CACHE", raising=False)
c = ac.AnthropicClient("claude-x", api_key="x", **client_kwargs)
return c._build_kwargs(
[system_msg("s"), user_msg("q")],
tools=None, temperature=None, max_tokens=None,
extra_headers=None, timeout=None,
)


def test_prompt_cache_omits_ttl_by_default(monkeypatch):
"""Five minutes is the API default, so an unconfigured client must send a
request byte-identical to the one it sent before this knob existed."""
monkeypatch.delenv(ac.PROMPT_CACHE_TTL_ENV, raising=False)
kwargs = _build_cached(monkeypatch)
assert _cache_controls(kwargs) == [{"type": "ephemeral"}] * 2


def test_prompt_cache_ttl_applies_to_both_breakpoints(monkeypatch):
"""The static head AND the rolling tail, or the conversation prefix — the
part that actually grows — still expires after five minutes."""
monkeypatch.delenv(ac.PROMPT_CACHE_TTL_ENV, raising=False)
kwargs = _build_cached(monkeypatch, prompt_cache_ttl="1h")
assert _cache_controls(kwargs) == [{"type": "ephemeral", "ttl": "1h"}] * 2


def test_explicit_five_minutes_stays_off_the_wire(monkeypatch):
"""Naming the default must not change the request shape."""
monkeypatch.delenv(ac.PROMPT_CACHE_TTL_ENV, raising=False)
assert _cache_controls(_build_cached(monkeypatch, prompt_cache_ttl="5m")) == [
{"type": "ephemeral"},
] * 2


def test_env_sets_the_ttl_and_the_client_outranks_it(monkeypatch):
"""The env var is the deployment-wide escape hatch for a run whose config
path does not reach every construction site; a client that states its own
lifetime is not overridden by the environment it happens to run in."""
monkeypatch.setenv(ac.PROMPT_CACHE_TTL_ENV, "1h")
assert _cache_controls(_build_cached(monkeypatch))[0] == {
"type": "ephemeral", "ttl": "1h",
}
assert _cache_controls(_build_cached(monkeypatch, prompt_cache_ttl="5m")) == [
{"type": "ephemeral"},
] * 2


@pytest.mark.parametrize("raw", ["30m", "1 h", "3600", "forever", 60])
def test_unsupported_ttl_is_refused_at_construction(raw):
"""A typo that silently reverted to five minutes would show up only as a
cache-hit-rate regression nobody is watching."""
with pytest.raises(ValueError):
ac.AnthropicClient("claude-x", api_key="x", prompt_cache_ttl=raw)


def test_unsupported_ttl_in_env_fails_loudly(monkeypatch):
monkeypatch.setenv(ac.PROMPT_CACHE_TTL_ENV, "90m")
with pytest.raises(ValueError):
_build_cached(monkeypatch)


def test_ttl_does_not_resurrect_a_disabled_cache(monkeypatch):
"""``ANTHROPIC_PROMPT_CACHE=0`` is the kill switch; a TTL is a lifetime for
breakpoints that exist, not a second way to place them."""
monkeypatch.setenv("ANTHROPIC_PROMPT_CACHE", "0")
monkeypatch.setenv(ac.PROMPT_CACHE_TTL_ENV, "1h")
c = ac.AnthropicClient("claude-x", api_key="x", prompt_cache_ttl="1h")
kwargs = c._build_kwargs(
[system_msg("s"), user_msg("q")],
tools=None, temperature=None, max_tokens=None,
extra_headers=None, timeout=None,
)
assert _cache_controls(kwargs) == []


def test_protocol_builder_forwards_the_profile_ttl(monkeypatch):
"""The lifetime belongs to the model a profile chose, so it has to survive
the builder rather than only being reachable by constructing by hand."""
monkeypatch.delenv(ac.PROMPT_CACHE_TTL_ENV, raising=False)
from agent_core.providers.protocol_client import build_protocol_client

client = build_protocol_client({
"protocol": "anthropic", "model": "claude-x", "api_key": "k",
"prompt_cache_ttl": "1h",
}, title="t")
assert client._prompt_cache_ttl == "1h"
plain = build_protocol_client({
"protocol": "anthropic", "model": "claude-x", "api_key": "k",
}, title="t")
assert plain._prompt_cache_ttl == ""


@pytest.mark.parametrize("protocol", ["anthropic", "bedrock"])
@pytest.mark.parametrize("env_ttl", [None, "1h"])
@pytest.mark.parametrize("raw", [0, False, [], {}, "30m", 60])
def test_protocol_builder_rejects_invalid_profile_ttl(monkeypatch, protocol, env_ttl, raw):
from agent_core.providers.protocol_client import build_protocol_client

if env_ttl is None:
monkeypatch.delenv(ac.PROMPT_CACHE_TTL_ENV, raising=False)
else:
monkeypatch.setenv(ac.PROMPT_CACHE_TTL_ENV, env_ttl)
with pytest.raises(ValueError):
build_protocol_client({
"protocol": protocol, "model": "claude-x", "api_key": "k",
"prompt_cache_ttl": raw,
}, title="t")


@pytest.mark.parametrize("raw", [None, "", "5m", "1h"])
@pytest.mark.asyncio
async def test_protocol_builder_ttl_request_smoke(monkeypatch, raw):
from agent_core.providers.protocol_client import build_protocol_client

monkeypatch.setenv(ac.PROMPT_CACHE_TTL_ENV, "1h")
monkeypatch.delenv("ANTHROPIC_PROMPT_CACHE", raising=False)
client = build_protocol_client({
"protocol": "anthropic", "model": "claude-x", "api_key": "k",
"prompt_cache_ttl": raw,
}, title="t")
try:
kwargs = client._build_kwargs(
[system_msg("s"), user_msg("q")],
tools=None, temperature=None, max_tokens=None,
extra_headers=None, timeout=None,
)
expected = {"type": "ephemeral"}
if raw != "5m":
expected["ttl"] = "1h"
assert _cache_controls(kwargs) == [expected] * 2
finally:
await client._client.close()
Loading