From bf3ecda46dd7040b99db84cae8bbc159371b829a Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Mon, 5 Oct 2026 19:11:05 +0800 Subject: [PATCH 1/2] =?UTF-8?q?feat(anthropic):=20prompt=20cache=20?= =?UTF-8?q?=E7=9A=84=20TTL=20=E5=8F=AF=E9=85=8D,=E7=BC=BA=E7=9C=81?= =?UTF-8?q?=E4=BB=8D=E6=98=AF=205m?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `AnthropicClient` 自己放两个 ephemeral 断点(system 前缀 + 最后一条持久消息), 但一直只用 API 缺省的 5 分钟。那个窗口的参照物不是任务多长,而是**一轮**多长: 缓存项在最后一次使用后才开始计时,所以只要相邻两次调用的间隔超过 TTL,整条 前缀就作废并按写价重写。 ApodexHarness 2026-10-05 的 GDPval 批次量到了代价(6056 次 claude-opus-5-5 effort=max 调用):相邻调用间隔 p90 = 355s、**12.9% 超过 300s**,这些调用冷率 **35.7%**,而间隔 ≤300s 的只有 9.5%;turn>3 且写入 >30k 的冷调用 498 次、 中位 81k,合计 61.6M 写 token。 - `prompt_cache_ttl`("5m" | "1h")加到构造器,并由 `build_protocol_client` 从 cfg 透传,所以这是**profile 级**的键 —— 一个模型的缓存寿命跟着选它的 profile 走,而不是跟着部署走。 - 空值与 "5m" 都**不发** ttl 字段:5m 本就是 API 缺省,不发能让所有现有消费者 的请求逐字节不变。 - `ANTHROPIC_PROMPT_CACHE_TTL` 作为整次运行的兜底(给 profile 够不到的构造点), client 自己声明的值优先。 - 非法值(`30m`、数字、带空格的单位)在构造期抛 `ValueError`,不回落到 5m —— 回落只会表现为"命中率下降",而命中率没人盯着,正是这个旋钮要修的那种静默。 - TTL 只决定已有断点的寿命,不会放断点:`ANTHROPIC_PROMPT_CACHE=0` 依然通杀。 - 计费不用改:`_anthropic_cache_write_tokens` 早就把 `cache_creation.ephemeral_1h_input_tokens` 算进去了。 **该按模型开,不要全局开**:1h 的写价是基础输入的 2×,5m 是 1.25×,所以只有 "一轮可能比 5 分钟还长"的慢模型才赚;快模型没有冷重写可省,只会多付写价。 文档里把这条判据和实测数字都写下来了。 网关必须真的转发该字段,所以 llm-hub 是实测过的(2026-10-05, temp/2026-10-05_llmhub-1h-ttl-实测.md):带 `ttl:"1h"` 返回 HTTP 200,服务端把 write 记进 **1h 桶**而非 5m 桶(43215 vs 对照组 40615 进 5m),548s 后回读 **完整命中**(creation 0),而未配置的对照组已失效并整条重写。 `anthropic-beta: extended-cache-ttl-2025-04-11` **不需要**。这里两项都查了: ttl 被网关悄悄剥掉时,5 秒回读的表现和成功完全一样,只有计费桶和跨窗口回读 能区分。 测试:新增 8 个(缺省不发字段、1h 落在两个断点上、显式 5m 不上线、env 兜底与 client 优先、非法值在构造期/env 下抛错、kill switch 仍通杀、builder 透传)。 全量 1913 passed / 2 skipped;ruff 与 pyright(改动模块)清。 Co-Authored-By: Claude Opus 5 (1M context) --- agent_core/providers/anthropic.py | 77 +++++++++++++++-- agent_core/providers/protocol_client.py | 3 + changes/61.feature.md | 3 + docs/provider-substrate-boundary.md | 42 +++++++++ tests/test_provider_native_clients.py | 110 ++++++++++++++++++++++++ 5 files changed, 230 insertions(+), 5 deletions(-) create mode 100644 changes/61.feature.md diff --git a/agent_core/providers/anthropic.py b/agent_core/providers/anthropic.py index 3b7eccc..cb8fdf1 100644 --- a/agent_core/providers/anthropic.py +++ b/agent_core/providers/anthropic.py @@ -56,6 +56,7 @@ def __init__( bedrock: bool = False, default_headers: dict[str, str] | None = None, capabilities: ModelCapabilities | None = None, + prompt_cache_ttl: str = "", ) -> None: self.model = model self.default_temperature = temperature @@ -68,6 +69,9 @@ def __init__( # (low|medium|high|xhigh|max) → ``output_config.effort`` via extra_body. self._thinking = thinking or None self._effort = (effort or "").strip() + # Rejected here rather than per call: a bad value is a config error and + # should fail at construction, not on the first request of a long run. + self._prompt_cache_ttl = normalize_prompt_cache_ttl(prompt_cache_ttl) self.capabilities = capabilities or resolve_model_capabilities( model, protocol="bedrock" if bedrock else "anthropic", ) @@ -147,7 +151,9 @@ def _build_kwargs( kwargs["timeout"] = timeout elif self.default_timeout is not None: kwargs["timeout"] = self.default_timeout - _add_prompt_cache(kwargs, transient_tail=transient_tail) + _add_prompt_cache( + kwargs, transient_tail=transient_tail, ttl=self._prompt_cache_ttl, + ) # After the cache breakpoint is placed, so it stays on the last # persistent block rather than moving onto the per-call text. kwargs["messages"] = _fold_transient_tail(kwargs["messages"], transient_tail) @@ -547,7 +553,54 @@ def _split_system(messages: list[Message]) -> tuple[str, list[Message]]: return "", list(messages) -def _add_prompt_cache(kwargs: dict[str, Any], *, transient_tail: int = 0) -> None: +#: Cache lifetimes Anthropic accepts on ``cache_control``. The empty string is +#: this adapter's "not configured", which omits the field and takes the API +#: default of five minutes. +PROMPT_CACHE_TTLS = frozenset({"5m", "1h"}) + +#: Deployment-wide default when no client was configured with one, so an +#: operator can turn the longer lifetime on for a whole run without a config +#: path reaching every construction site. A client's own value wins. +PROMPT_CACHE_TTL_ENV = "ANTHROPIC_PROMPT_CACHE_TTL" + + +def normalize_prompt_cache_ttl(value: object) -> str: + """Normalize a configured cache TTL; ``""`` means "not configured". + + Unknown values raise rather than falling back to the default, matching + :func:`agent_core.model_capabilities.normalize_thinking_mode`: a typo that + silently reverts to five minutes is the failure this whole knob exists to + fix, and it would only surface as a cache-hit-rate regression nobody is + watching. + """ + if value is None: + return "" + if not isinstance(value, str): + raise ValueError("prompt cache ttl must be a string or null") + ttl = value.strip().lower() + if not ttl: + return "" + if ttl not in PROMPT_CACHE_TTLS: + raise ValueError( + f"unsupported prompt cache ttl {value!r}; use {sorted(PROMPT_CACHE_TTLS)}", + ) + return ttl + + +def _cache_control(ttl: str) -> dict[str, str]: + """The ``cache_control`` value for a breakpoint, with TTL when configured.""" + resolved = ttl or normalize_prompt_cache_ttl(os.getenv(PROMPT_CACHE_TTL_ENV)) + if not resolved or resolved == "5m": + # Omitted rather than sent explicitly: "5m" is the API default, and not + # sending the field keeps the request shape of every existing consumer + # byte-identical. + return {"type": "ephemeral"} + return {"type": "ephemeral", "ttl": resolved} + + +def _add_prompt_cache( + kwargs: dict[str, Any], *, transient_tail: int = 0, ttl: str = "", +) -> None: """Set Anthropic prompt-cache breakpoints on ``kwargs`` in place. Anthropic caching is opt-in per content block (unlike OpenAI's automatic @@ -565,16 +618,30 @@ def _add_prompt_cache(kwargs: dict[str, Any], *, transient_tail: int = 0) -> Non so a cached prefix ending on one never matches again and every turn would re-write the whole conversation at the cache-write rate while reading only the static head. + + ``ttl`` ("5m" default, or "1h") applies to BOTH breakpoints. A cached entry + expires that long after its last use, so the lifetime that matters is not + how long a run takes but how long one TURN takes: a gap longer than the TTL + loses the whole prefix and re-writes it. On a slow model that is a routine + event rather than an edge case — in ApodexHarness's 2026-10-05 GDPval batch, + 12.9% of claude-opus-5-5 turn gaps exceeded five minutes (p90 355s) and + those calls missed the cache 35.7% of the time against 9.5% for the rest, + re-writing a median 81k tokens each. The trade is the write rate: 2x base + input for an hour against 1.25x for five minutes, so this pays off exactly + when turns are slow enough to straddle the shorter window and costs extra + when they are not. Writes land in ``cache_creation.ephemeral_1h_input_tokens``, + which :func:`_anthropic_cache_write_tokens` already counts. """ if os.getenv("ANTHROPIC_PROMPT_CACHE", "1") == "0": return + cache_control = _cache_control(ttl) # System prefix (a plain string) -> one cache-controlled text block. system = kwargs.get("system") if isinstance(system, str) and system: kwargs["system"] = [{ "type": "text", "text": system, - "cache_control": {"type": "ephemeral"}, + "cache_control": cache_control, }] # Rolling tail: mark the last persistent message's final content block. msgs = kwargs.get("messages") @@ -587,10 +654,10 @@ def _add_prompt_cache(kwargs: dict[str, Any], *, transient_tail: int = 0) -> Non last["content"] = [{ "type": "text", "text": content, - "cache_control": {"type": "ephemeral"}, + "cache_control": cache_control, }] elif isinstance(content, list) and content and isinstance(content[-1], dict): - content[-1] = {**content[-1], "cache_control": {"type": "ephemeral"}} + content[-1] = {**content[-1], "cache_control": cache_control} def _to_anthropic_msg(m: Message) -> dict[str, Any] | None: diff --git a/agent_core/providers/protocol_client.py b/agent_core/providers/protocol_client.py index 879c292..04d0d30 100644 --- a/agent_core/providers/protocol_client.py +++ b/agent_core/providers/protocol_client.py @@ -187,6 +187,9 @@ def _build_anthropic( default_headers={"X-Title": title, **(cfg.get("default_headers") or {})}, bedrock=bedrock, capabilities=capabilities, + # A profile-level key, so one model's declared cache lifetime travels + # with the model it was chosen for rather than with the deployment. + prompt_cache_ttl=cfg.get("prompt_cache_ttl") or "", ) diff --git a/changes/61.feature.md b/changes/61.feature.md new file mode 100644 index 0000000..42d9d5d --- /dev/null +++ b/changes/61.feature.md @@ -0,0 +1,3 @@ +`AnthropicClient` and `build_protocol_client` accept `prompt_cache_ttl` (`5m`, the API default, or `1h`), which sets the lifetime of both prompt-cache breakpoints the adapter places. Blank and `5m` omit the field, so an unconfigured consumer's requests are byte-identical to before; `ANTHROPIC_PROMPT_CACHE_TTL` provides a deployment-wide default for construction sites a profile does not reach, and a client's own value outranks it. Unsupported values (`30m`, numbers, a stray unit) raise `ValueError` at construction rather than silently reverting to five minutes, which would only surface as a cache-hit-rate regression. The existing `ANTHROPIC_PROMPT_CACHE=0` kill switch still wins, and 1h writes were already counted by the adapter's cache-write accounting. + +Consumers should set this per model, not globally: an hour costs 2x base input to write against 1.25x for five minutes, and it pays off only when one TURN can take longer than the shorter window — a cached entry expires relative to its last use. In ApodexHarness's 2026-10-05 GDPval batch, 12.9% of `claude-opus-5-5` turn gaps at `effort=max` exceeded 300s and those calls missed the cache 35.7% of the time against 9.5% for the rest, rewriting a median 81k tokens each; a fast model has no such rewrites to avoid and would only pay the higher write rate. A gateway has to forward the field: llm-hub was verified on 2026-10-05 (HTTP 200, write billed to the 1h bucket, a read at 548s still hitting while the 5m control had expired), and the `extended-cache-ttl-2025-04-11` beta header was not required. diff --git a/docs/provider-substrate-boundary.md b/docs/provider-substrate-boundary.md index 4fdd31c..963aeaf 100644 --- a/docs/provider-substrate-boundary.md +++ b/docs/provider-substrate-boundary.md @@ -184,3 +184,45 @@ Defaults such as `default_effort` are descriptive; callers who omit effort keep the provider's own default. Beta/platform-specific limits require a host override paired with the appropriate headers. Models API discovery/caching can be added by hosts later; this PR does not introduce a background synchronization service. + +## Anthropic prompt-cache lifetime + +Anthropic caching is opt-in per content block, so `AnthropicClient` places two +`ephemeral` breakpoints itself: the system prefix and the last persistent +message's final block. `ANTHROPIC_PROMPT_CACHE=0` disables both. + +The lifetime of those breakpoints is configurable with the `prompt_cache_ttl` +profile key (`5m`, the API default, or `1h`), forwarded by +`build_protocol_client`; `normalize_prompt_cache_ttl` accepts a blank value as +"not configured" and raises `ValueError` on anything else, including `30m` and +numbers. `5m` and blank both OMIT the field, so an unconfigured client's request +is unchanged. `ANTHROPIC_PROMPT_CACHE_TTL` sets a deployment-wide default for +construction sites a profile does not reach; a client's own value wins. A TTL +applies to both breakpoints and never places one — with the cache disabled it +has no effect. + +```yaml +llm: + protocol: anthropic + model: claude-opus-5-5 + prompt_cache_ttl: 1h # turns are slower than the 5m default window +``` + +Which lifetime pays off is a property of TURN duration, not run duration: a +cached entry expires that long after its last use, so a gap longer than the TTL +loses the whole prefix and rewrites it at the cache-write rate. An hour costs 2x +base input to write against 1.25x for five minutes, so it wins only where turns +are slow enough to straddle the shorter window. Measured on ApodexHarness's +2026-10-05 GDPval batch (6056 `claude-opus-5-5` calls at `effort=max`): 12.9% of +turn gaps exceeded 300s (p90 355s), and those calls missed the cache 35.7% of the +time against 9.5% for the rest, rewriting a median 81k tokens each. A fast model +whose gaps sit well inside five minutes has no such rewrites to avoid and only +pays the higher write rate. + +Writes land in `cache_creation.ephemeral_1h_input_tokens`, which the adapter's +cache-write accounting already includes, so `cache_write_tokens` stays correct +across both lifetimes. Gateways must forward the field for it to take effect: +verified against llm-hub on 2026-10-05 — `ttl: "1h"` returned HTTP 200, the +write was billed to the 1h bucket rather than the 5m one, and a read 548s later +still hit in full while the unconfigured control had expired and rewritten. The +`extended-cache-ttl-2025-04-11` beta header was not required. diff --git a/tests/test_provider_native_clients.py b/tests/test_provider_native_clients.py index d55fc58..510fd4a 100644 --- a/tests/test_provider_native_clients.py +++ b/tests/test_provider_native_clients.py @@ -1500,3 +1500,113 @@ async def gen(): assert terminal.stop_details == { "type": "refusal", "category": "cyber", "explanation": "declined", } + + +# ── prompt-cache TTL (5m default, opt-in 1h) ── + + +def _cache_controls(kwargs): + """Every ``cache_control`` value in the request, system block first.""" + found = [ + b["cache_control"] for b in kwargs.get("system", []) + if isinstance(b, dict) and "cache_control" in b + ] + for m in kwargs["messages"]: + if isinstance(m["content"], list): + found += [ + b["cache_control"] for b in m["content"] + if isinstance(b, dict) and "cache_control" in b + ] + return found + + +def _build_cached(monkeypatch, **client_kwargs): + monkeypatch.delenv("ANTHROPIC_PROMPT_CACHE", raising=False) + c = ac.AnthropicClient("claude-x", api_key="x", **client_kwargs) + return c._build_kwargs( + [system_msg("s"), user_msg("q")], + tools=None, temperature=None, max_tokens=None, + extra_headers=None, timeout=None, + ) + + +def test_prompt_cache_omits_ttl_by_default(monkeypatch): + """Five minutes is the API default, so an unconfigured client must send a + request byte-identical to the one it sent before this knob existed.""" + monkeypatch.delenv(ac.PROMPT_CACHE_TTL_ENV, raising=False) + kwargs = _build_cached(monkeypatch) + assert _cache_controls(kwargs) == [{"type": "ephemeral"}] * 2 + + +def test_prompt_cache_ttl_applies_to_both_breakpoints(monkeypatch): + """The static head AND the rolling tail, or the conversation prefix — the + part that actually grows — still expires after five minutes.""" + monkeypatch.delenv(ac.PROMPT_CACHE_TTL_ENV, raising=False) + kwargs = _build_cached(monkeypatch, prompt_cache_ttl="1h") + assert _cache_controls(kwargs) == [{"type": "ephemeral", "ttl": "1h"}] * 2 + + +def test_explicit_five_minutes_stays_off_the_wire(monkeypatch): + """Naming the default must not change the request shape.""" + monkeypatch.delenv(ac.PROMPT_CACHE_TTL_ENV, raising=False) + assert _cache_controls(_build_cached(monkeypatch, prompt_cache_ttl="5m")) == [ + {"type": "ephemeral"}, + ] * 2 + + +def test_env_sets_the_ttl_and_the_client_outranks_it(monkeypatch): + """The env var is the deployment-wide escape hatch for a run whose config + path does not reach every construction site; a client that states its own + lifetime is not overridden by the environment it happens to run in.""" + monkeypatch.setenv(ac.PROMPT_CACHE_TTL_ENV, "1h") + assert _cache_controls(_build_cached(monkeypatch))[0] == { + "type": "ephemeral", "ttl": "1h", + } + assert _cache_controls(_build_cached(monkeypatch, prompt_cache_ttl="5m")) == [ + {"type": "ephemeral"}, + ] * 2 + + +@pytest.mark.parametrize("raw", ["30m", "1 h", "3600", "forever", 60]) +def test_unsupported_ttl_is_refused_at_construction(raw): + """A typo that silently reverted to five minutes would show up only as a + cache-hit-rate regression nobody is watching.""" + with pytest.raises(ValueError): + ac.AnthropicClient("claude-x", api_key="x", prompt_cache_ttl=raw) + + +def test_unsupported_ttl_in_env_fails_loudly(monkeypatch): + monkeypatch.setenv(ac.PROMPT_CACHE_TTL_ENV, "90m") + with pytest.raises(ValueError): + _build_cached(monkeypatch) + + +def test_ttl_does_not_resurrect_a_disabled_cache(monkeypatch): + """``ANTHROPIC_PROMPT_CACHE=0`` is the kill switch; a TTL is a lifetime for + breakpoints that exist, not a second way to place them.""" + monkeypatch.setenv("ANTHROPIC_PROMPT_CACHE", "0") + monkeypatch.setenv(ac.PROMPT_CACHE_TTL_ENV, "1h") + c = ac.AnthropicClient("claude-x", api_key="x", prompt_cache_ttl="1h") + kwargs = c._build_kwargs( + [system_msg("s"), user_msg("q")], + tools=None, temperature=None, max_tokens=None, + extra_headers=None, timeout=None, + ) + assert _cache_controls(kwargs) == [] + + +def test_protocol_builder_forwards_the_profile_ttl(monkeypatch): + """The lifetime belongs to the model a profile chose, so it has to survive + the builder rather than only being reachable by constructing by hand.""" + monkeypatch.delenv(ac.PROMPT_CACHE_TTL_ENV, raising=False) + from agent_core.providers.protocol_client import build_protocol_client + + client = build_protocol_client({ + "protocol": "anthropic", "model": "claude-x", "api_key": "k", + "prompt_cache_ttl": "1h", + }, title="t") + assert client._prompt_cache_ttl == "1h" + plain = build_protocol_client({ + "protocol": "anthropic", "model": "claude-x", "api_key": "k", + }, title="t") + assert plain._prompt_cache_ttl == "" From addbabc4b94e18d93b2304f60922004b49883376 Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Tue, 6 Oct 2026 06:41:04 +0800 Subject: [PATCH 2/2] fix(anthropic): reject invalid falsey profile cache TTLs --- agent_core/providers/anthropic.py | 2 +- agent_core/providers/protocol_client.py | 4 ++- tests/test_provider_native_clients.py | 42 +++++++++++++++++++++++++ 3 files changed, 46 insertions(+), 2 deletions(-) diff --git a/agent_core/providers/anthropic.py b/agent_core/providers/anthropic.py index cb8fdf1..6b29771 100644 --- a/agent_core/providers/anthropic.py +++ b/agent_core/providers/anthropic.py @@ -56,7 +56,7 @@ def __init__( bedrock: bool = False, default_headers: dict[str, str] | None = None, capabilities: ModelCapabilities | None = None, - prompt_cache_ttl: str = "", + prompt_cache_ttl: str | None = "", ) -> None: self.model = model self.default_temperature = temperature diff --git a/agent_core/providers/protocol_client.py b/agent_core/providers/protocol_client.py index 04d0d30..1bd5e13 100644 --- a/agent_core/providers/protocol_client.py +++ b/agent_core/providers/protocol_client.py @@ -189,7 +189,9 @@ def _build_anthropic( capabilities=capabilities, # A profile-level key, so one model's declared cache lifetime travels # with the model it was chosen for rather than with the deployment. - prompt_cache_ttl=cfg.get("prompt_cache_ttl") or "", + # Preserve invalid falsey values so the client rejects them instead of + # silently selecting the environment's default lifetime. + prompt_cache_ttl=cfg.get("prompt_cache_ttl"), ) diff --git a/tests/test_provider_native_clients.py b/tests/test_provider_native_clients.py index 510fd4a..884d1b3 100644 --- a/tests/test_provider_native_clients.py +++ b/tests/test_provider_native_clients.py @@ -1610,3 +1610,45 @@ def test_protocol_builder_forwards_the_profile_ttl(monkeypatch): "protocol": "anthropic", "model": "claude-x", "api_key": "k", }, title="t") assert plain._prompt_cache_ttl == "" + + +@pytest.mark.parametrize("protocol", ["anthropic", "bedrock"]) +@pytest.mark.parametrize("env_ttl", [None, "1h"]) +@pytest.mark.parametrize("raw", [0, False, [], {}, "30m", 60]) +def test_protocol_builder_rejects_invalid_profile_ttl(monkeypatch, protocol, env_ttl, raw): + from agent_core.providers.protocol_client import build_protocol_client + + if env_ttl is None: + monkeypatch.delenv(ac.PROMPT_CACHE_TTL_ENV, raising=False) + else: + monkeypatch.setenv(ac.PROMPT_CACHE_TTL_ENV, env_ttl) + with pytest.raises(ValueError): + build_protocol_client({ + "protocol": protocol, "model": "claude-x", "api_key": "k", + "prompt_cache_ttl": raw, + }, title="t") + + +@pytest.mark.parametrize("raw", [None, "", "5m", "1h"]) +@pytest.mark.asyncio +async def test_protocol_builder_ttl_request_smoke(monkeypatch, raw): + from agent_core.providers.protocol_client import build_protocol_client + + monkeypatch.setenv(ac.PROMPT_CACHE_TTL_ENV, "1h") + monkeypatch.delenv("ANTHROPIC_PROMPT_CACHE", raising=False) + client = build_protocol_client({ + "protocol": "anthropic", "model": "claude-x", "api_key": "k", + "prompt_cache_ttl": raw, + }, title="t") + try: + kwargs = client._build_kwargs( + [system_msg("s"), user_msg("q")], + tools=None, temperature=None, max_tokens=None, + extra_headers=None, timeout=None, + ) + expected = {"type": "ephemeral"} + if raw != "5m": + expected["ttl"] = "1h" + assert _cache_controls(kwargs) == [expected] * 2 + finally: + await client._client.close()