From f8d7b66318e3f199eb6ce97d208bb884f9870c18 Mon Sep 17 00:00:00 2001 From: raymondginger2018-sudo Date: Sun, 27 Sep 2026 14:30:51 +0800 Subject: [PATCH] fix(catalog): give DeepSeek V4 its own window and price Every DeepSeek V4 id used to miss the seed table and fall through to the ``deepseek`` family rule, i.e. inherit the deepseek-v3 row. Two numbers were wrong as a result: * window: V4 is 1M with 384K max output, not V3's 128K / 8K, so anything sizing a prompt against this catalog budgeted ~8x too small; * price: V3's 0.27/1.10 was applied to both tiers, so a pro/flash split was invisible to any accounting keyed on model id. Numbers come from the vendor's published table, read 2026-09-27: https://api-docs.deepseek.com/quick_start/pricing Two details on that page shape the rows: * ``deepseek-flash`` is the vendor's current name for V4.1-Flash. The id ``deepseek-v4-flash`` is a retired alias that still resolves and is still billed at the Flash price (page footnote 1), so both ids carry one row. * V4 is priced by time of day and a catalog row holds a single number, so the peak rate is seeded: it is the upper bound, which is the safe direction for a budget guard. Off-peak is exactly half of it. A ``deepseek-flash`` family rule is added as well, so a future ``deepseek-flash-*`` point release cannot fall back to V3 either. tests/test_catalog.py gains coverage for the current and retired Flash ids resolving to the vendor row, for the tiers staying priced apart, for gateway spelling folding, and for unseen point releases not inheriting V3. --- core/providers/catalog.py | 29 +++++++++++++++++ tests/test_catalog.py | 65 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 94 insertions(+) diff --git a/core/providers/catalog.py b/core/providers/catalog.py index f304f0e58..a211c86d1 100644 --- a/core/providers/catalog.py +++ b/core/providers/catalog.py @@ -249,6 +249,28 @@ class ModelInfo: "deepseek-r1": ModelInfo( "deepseek-r1", 128_000, 65_536, 0.55, 2.19, reasoning=_REASONING_ALWAYS_ON ), + # DeepSeek V4 — source: https://api-docs.deepseek.com/quick_start/pricing + # (read 2026-09-27). Every V4 id used to fall through to the ``deepseek`` + # family rule below, which got two things wrong: + # * window — V4 is 1M with 384K max output, not V3's 128K / 8K, so + # anything compacting against this catalog fired ~8x too early. + # * price — V3's 0.27/1.10 was inherited by *both* tiers, so a pro/flash + # split was invisible to anything that accounts by model id. + # V4 is priced by time of day and a row holds one number, so the **peak** + # rate is used: the upper bound is the safe direction for a budget guard. + # Off-peak is exactly half. + # ``deepseek-flash`` is the vendor's current name for V4.1-Flash; + # ``deepseek-v4-flash`` is a retired alias that still resolves and is billed + # at the Flash price (vendor footnote 1) — same row, both ids. + "deepseek-flash": ModelInfo( + "deepseek-flash", 1_000_000, 384_000, 0.30, 1.20, reasoning=_REASONING_TOGGLE + ), + "deepseek-v4-flash": ModelInfo( + "deepseek-v4-flash", 1_000_000, 384_000, 0.30, 1.20, reasoning=_REASONING_TOGGLE + ), + "deepseek-v4-pro": ModelInfo( + "deepseek-v4-pro", 1_000_000, 384_000, 1.32, 3.96, reasoning=_REASONING_TOGGLE + ), # Alibaba Qwen. "qwen3-max": ModelInfo( "qwen3-max", 256_000, 32_768, 1.2, 6.0, reasoning=_REASONING_TOGGLE @@ -325,6 +347,13 @@ class ModelInfo: ("kimi-latest", _SEED["kimi-k3"]), ("kimi", _SEED["kimi-k2.6"]), ("deepseek-r", _SEED["deepseek-r1"]), + # V4 tiers are seeded now, so an unseen point release inherits the + # **cheapest** tier rather than silently taking v3's rate — over-charging a + # cheap model is the failure mode that matters for a budget guard. The + # vendor's current name gets its own rule so ``deepseek-flash-*`` cannot + # fall to v3 either. + ("deepseek-flash", _SEED["deepseek-flash"]), + ("deepseek-v4", _SEED["deepseek-v4-flash"]), ("deepseek", _SEED["deepseek-v3"]), ("qwen", _SEED["qwen3-max"]), ("grok", _SEED["grok-4"]), diff --git a/tests/test_catalog.py b/tests/test_catalog.py index 6cd9ab701..1a40ed1da 100644 --- a/tests/test_catalog.py +++ b/tests/test_catalog.py @@ -6,6 +6,8 @@ import sys from pathlib import Path +import pytest + ROOT = Path(__file__).resolve().parents[1] if str(ROOT) not in sys.path: sys.path.insert(0, str(ROOT)) @@ -90,3 +92,66 @@ def test_snapshot_skips_entries_without_context(tmp_path, monkeypatch): snap.write_text(json.dumps({"weird-model": {"cost": {"input": 1.0}}})) monkeypatch.setattr(catalog, "_SNAPSHOT", {}) assert catalog.load_catalog_snapshot(snap) == 0 + + +# Vendor page read 2026-09-27: https://api-docs.deepseek.com/quick_start/pricing +# ``deepseek-flash`` is the current name for V4.1-Flash; ``deepseek-v4-flash`` is +# a retired alias the vendor still accepts and still bills at the Flash price. +# Both must carry the vendor row instead of falling through to the +# ``deepseek`` family rule, which also handed V4 a 128K window. +@pytest.mark.parametrize("model_id", ["deepseek-flash", "deepseek-v4-flash"]) +def test_deepseek_v4_flash_ids_carry_the_vendor_row(model_id): + info = catalog.resolve_model_info(model_id) + + assert info.source == "seed" + assert info.context_window == 1_000_000 + assert info.max_output_tokens == 384_000 + + +def test_deepseek_v4_tiers_are_priced_apart(): + # Regression: both V4 tiers used to fall through to the ``deepseek`` family + # rule and take ``deepseek-v3``'s price, so pro and flash were one row in + # the cost ledger (found by the layer-④ cost census). V4 is priced by time of + # day; the seeded numbers are the peak rate, i.e. the upper bound. + flash = catalog.resolve_model_info("deepseek-v4-flash") + pro = catalog.resolve_model_info("deepseek-v4-pro") + v3 = catalog.resolve_model_info("deepseek-v3") + + assert flash.source == "seed" + assert pro.source == "seed" + assert pro.input_cost_per_1m > flash.input_cost_per_1m + assert pro.output_cost_per_1m > flash.output_cost_per_1m + assert flash.input_cost_per_1m != v3.input_cost_per_1m + + +@pytest.mark.parametrize( + "spelling", + [ + "deepseek-v4-flash", + "deepseek/deepseek-v4-flash", + "deepseek-ai/DeepSeek-V4-Flash", + ], +) +def test_observed_gateway_spellings_fold_onto_one_seed_row(spelling): + # All three spellings appear in the gateway log for the same logical model. + # Prefix-stripping must fold them onto a single id, or cost accounting + # fragments by spelling instead of by model. + info = catalog.resolve_model_info(spelling) + + assert info.id == "deepseek-v4-flash" + assert info.source == "seed" + + +@pytest.mark.parametrize( + ("model_id", "expected_source"), + [ + ("deepseek-v4-turbo", "family:deepseek-v4"), + ("deepseek-flash-turbo", "family:deepseek-flash"), + ], +) +def test_unseeded_point_release_does_not_inherit_v3(model_id, expected_source): + info = catalog.resolve_model_info(model_id) + v3 = catalog.resolve_model_info("deepseek-v3") + + assert info.source == expected_source + assert info.input_cost_per_1m != v3.input_cost_per_1m