diff --git a/core/providers/catalog.py b/core/providers/catalog.py index f304f0e58..a211c86d1 100644 --- a/core/providers/catalog.py +++ b/core/providers/catalog.py @@ -249,6 +249,28 @@ class ModelInfo: "deepseek-r1": ModelInfo( "deepseek-r1", 128_000, 65_536, 0.55, 2.19, reasoning=_REASONING_ALWAYS_ON ), + # DeepSeek V4 — source: https://api-docs.deepseek.com/quick_start/pricing + # (read 2026-09-27). Every V4 id used to fall through to the ``deepseek`` + # family rule below, which got two things wrong: + # * window — V4 is 1M with 384K max output, not V3's 128K / 8K, so + # anything compacting against this catalog fired ~8x too early. + # * price — V3's 0.27/1.10 was inherited by *both* tiers, so a pro/flash + # split was invisible to anything that accounts by model id. + # V4 is priced by time of day and a row holds one number, so the **peak** + # rate is used: the upper bound is the safe direction for a budget guard. + # Off-peak is exactly half. + # ``deepseek-flash`` is the vendor's current name for V4.1-Flash; + # ``deepseek-v4-flash`` is a retired alias that still resolves and is billed + # at the Flash price (vendor footnote 1) — same row, both ids. + "deepseek-flash": ModelInfo( + "deepseek-flash", 1_000_000, 384_000, 0.30, 1.20, reasoning=_REASONING_TOGGLE + ), + "deepseek-v4-flash": ModelInfo( + "deepseek-v4-flash", 1_000_000, 384_000, 0.30, 1.20, reasoning=_REASONING_TOGGLE + ), + "deepseek-v4-pro": ModelInfo( + "deepseek-v4-pro", 1_000_000, 384_000, 1.32, 3.96, reasoning=_REASONING_TOGGLE + ), # Alibaba Qwen. "qwen3-max": ModelInfo( "qwen3-max", 256_000, 32_768, 1.2, 6.0, reasoning=_REASONING_TOGGLE @@ -325,6 +347,13 @@ class ModelInfo: ("kimi-latest", _SEED["kimi-k3"]), ("kimi", _SEED["kimi-k2.6"]), ("deepseek-r", _SEED["deepseek-r1"]), + # V4 tiers are seeded now, so an unseen point release inherits the + # **cheapest** tier rather than silently taking v3's rate — over-charging a + # cheap model is the failure mode that matters for a budget guard. The + # vendor's current name gets its own rule so ``deepseek-flash-*`` cannot + # fall to v3 either. + ("deepseek-flash", _SEED["deepseek-flash"]), + ("deepseek-v4", _SEED["deepseek-v4-flash"]), ("deepseek", _SEED["deepseek-v3"]), ("qwen", _SEED["qwen3-max"]), ("grok", _SEED["grok-4"]), diff --git a/tests/test_catalog.py b/tests/test_catalog.py index 6cd9ab701..1a40ed1da 100644 --- a/tests/test_catalog.py +++ b/tests/test_catalog.py @@ -6,6 +6,8 @@ import sys from pathlib import Path +import pytest + ROOT = Path(__file__).resolve().parents[1] if str(ROOT) not in sys.path: sys.path.insert(0, str(ROOT)) @@ -90,3 +92,66 @@ def test_snapshot_skips_entries_without_context(tmp_path, monkeypatch): snap.write_text(json.dumps({"weird-model": {"cost": {"input": 1.0}}})) monkeypatch.setattr(catalog, "_SNAPSHOT", {}) assert catalog.load_catalog_snapshot(snap) == 0 + + +# Vendor page read 2026-09-27: https://api-docs.deepseek.com/quick_start/pricing +# ``deepseek-flash`` is the current name for V4.1-Flash; ``deepseek-v4-flash`` is +# a retired alias the vendor still accepts and still bills at the Flash price. +# Both must carry the vendor row instead of falling through to the +# ``deepseek`` family rule, which also handed V4 a 128K window. +@pytest.mark.parametrize("model_id", ["deepseek-flash", "deepseek-v4-flash"]) +def test_deepseek_v4_flash_ids_carry_the_vendor_row(model_id): + info = catalog.resolve_model_info(model_id) + + assert info.source == "seed" + assert info.context_window == 1_000_000 + assert info.max_output_tokens == 384_000 + + +def test_deepseek_v4_tiers_are_priced_apart(): + # Regression: both V4 tiers used to fall through to the ``deepseek`` family + # rule and take ``deepseek-v3``'s price, so pro and flash were one row in + # the cost ledger (found by the layer-④ cost census). V4 is priced by time of + # day; the seeded numbers are the peak rate, i.e. the upper bound. + flash = catalog.resolve_model_info("deepseek-v4-flash") + pro = catalog.resolve_model_info("deepseek-v4-pro") + v3 = catalog.resolve_model_info("deepseek-v3") + + assert flash.source == "seed" + assert pro.source == "seed" + assert pro.input_cost_per_1m > flash.input_cost_per_1m + assert pro.output_cost_per_1m > flash.output_cost_per_1m + assert flash.input_cost_per_1m != v3.input_cost_per_1m + + +@pytest.mark.parametrize( + "spelling", + [ + "deepseek-v4-flash", + "deepseek/deepseek-v4-flash", + "deepseek-ai/DeepSeek-V4-Flash", + ], +) +def test_observed_gateway_spellings_fold_onto_one_seed_row(spelling): + # All three spellings appear in the gateway log for the same logical model. + # Prefix-stripping must fold them onto a single id, or cost accounting + # fragments by spelling instead of by model. + info = catalog.resolve_model_info(spelling) + + assert info.id == "deepseek-v4-flash" + assert info.source == "seed" + + +@pytest.mark.parametrize( + ("model_id", "expected_source"), + [ + ("deepseek-v4-turbo", "family:deepseek-v4"), + ("deepseek-flash-turbo", "family:deepseek-flash"), + ], +) +def test_unseeded_point_release_does_not_inherit_v3(model_id, expected_source): + info = catalog.resolve_model_info(model_id) + v3 = catalog.resolve_model_info("deepseek-v3") + + assert info.source == expected_source + assert info.input_cost_per_1m != v3.input_cost_per_1m