Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
29 changes: 29 additions & 0 deletions core/providers/catalog.py
Original file line number Diff line number Diff line change
Expand Up @@ -249,6 +249,28 @@ class ModelInfo:
"deepseek-r1": ModelInfo(
"deepseek-r1", 128_000, 65_536, 0.55, 2.19, reasoning=_REASONING_ALWAYS_ON
),
# DeepSeek V4 — source: https://api-docs.deepseek.com/quick_start/pricing
# (read 2026-09-27). Every V4 id used to fall through to the ``deepseek``
# family rule below, which got two things wrong:
# * window — V4 is 1M with 384K max output, not V3's 128K / 8K, so
# anything compacting against this catalog fired ~8x too early.
# * price — V3's 0.27/1.10 was inherited by *both* tiers, so a pro/flash
# split was invisible to anything that accounts by model id.
# V4 is priced by time of day and a row holds one number, so the **peak**
# rate is used: the upper bound is the safe direction for a budget guard.
# Off-peak is exactly half.
# ``deepseek-flash`` is the vendor's current name for V4.1-Flash;
# ``deepseek-v4-flash`` is a retired alias that still resolves and is billed
# at the Flash price (vendor footnote 1) — same row, both ids.
"deepseek-flash": ModelInfo(
"deepseek-flash", 1_000_000, 384_000, 0.30, 1.20, reasoning=_REASONING_TOGGLE
),
"deepseek-v4-flash": ModelInfo(
"deepseek-v4-flash", 1_000_000, 384_000, 0.30, 1.20, reasoning=_REASONING_TOGGLE
),
"deepseek-v4-pro": ModelInfo(
"deepseek-v4-pro", 1_000_000, 384_000, 1.32, 3.96, reasoning=_REASONING_TOGGLE
),
# Alibaba Qwen.
"qwen3-max": ModelInfo(
"qwen3-max", 256_000, 32_768, 1.2, 6.0, reasoning=_REASONING_TOGGLE
Expand Down Expand Up @@ -325,6 +347,13 @@ class ModelInfo:
("kimi-latest", _SEED["kimi-k3"]),
("kimi", _SEED["kimi-k2.6"]),
("deepseek-r", _SEED["deepseek-r1"]),
# V4 tiers are seeded now, so an unseen point release inherits the
# **cheapest** tier rather than silently taking v3's rate — over-charging a
# cheap model is the failure mode that matters for a budget guard. The
# vendor's current name gets its own rule so ``deepseek-flash-*`` cannot
# fall to v3 either.
("deepseek-flash", _SEED["deepseek-flash"]),
("deepseek-v4", _SEED["deepseek-v4-flash"]),
("deepseek", _SEED["deepseek-v3"]),
("qwen", _SEED["qwen3-max"]),
("grok", _SEED["grok-4"]),
Expand Down
65 changes: 65 additions & 0 deletions tests/test_catalog.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,8 @@
import sys
from pathlib import Path

import pytest

ROOT = Path(__file__).resolve().parents[1]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
Expand Down Expand Up @@ -90,3 +92,66 @@ def test_snapshot_skips_entries_without_context(tmp_path, monkeypatch):
snap.write_text(json.dumps({"weird-model": {"cost": {"input": 1.0}}}))
monkeypatch.setattr(catalog, "_SNAPSHOT", {})
assert catalog.load_catalog_snapshot(snap) == 0


# Vendor page read 2026-09-27: https://api-docs.deepseek.com/quick_start/pricing
# ``deepseek-flash`` is the current name for V4.1-Flash; ``deepseek-v4-flash`` is
# a retired alias the vendor still accepts and still bills at the Flash price.
# Both must carry the vendor row instead of falling through to the
# ``deepseek`` family rule, which also handed V4 a 128K window.
@pytest.mark.parametrize("model_id", ["deepseek-flash", "deepseek-v4-flash"])
def test_deepseek_v4_flash_ids_carry_the_vendor_row(model_id):
info = catalog.resolve_model_info(model_id)

assert info.source == "seed"
assert info.context_window == 1_000_000
assert info.max_output_tokens == 384_000


def test_deepseek_v4_tiers_are_priced_apart():
# Regression: both V4 tiers used to fall through to the ``deepseek`` family
# rule and take ``deepseek-v3``'s price, so pro and flash were one row in
# the cost ledger (found by the layer-④ cost census). V4 is priced by time of
# day; the seeded numbers are the peak rate, i.e. the upper bound.
flash = catalog.resolve_model_info("deepseek-v4-flash")
pro = catalog.resolve_model_info("deepseek-v4-pro")
v3 = catalog.resolve_model_info("deepseek-v3")

assert flash.source == "seed"
assert pro.source == "seed"
assert pro.input_cost_per_1m > flash.input_cost_per_1m
assert pro.output_cost_per_1m > flash.output_cost_per_1m
assert flash.input_cost_per_1m != v3.input_cost_per_1m


@pytest.mark.parametrize(
"spelling",
[
"deepseek-v4-flash",
"deepseek/deepseek-v4-flash",
"deepseek-ai/DeepSeek-V4-Flash",
],
)
def test_observed_gateway_spellings_fold_onto_one_seed_row(spelling):
# All three spellings appear in the gateway log for the same logical model.
# Prefix-stripping must fold them onto a single id, or cost accounting
# fragments by spelling instead of by model.
info = catalog.resolve_model_info(spelling)

assert info.id == "deepseek-v4-flash"
assert info.source == "seed"


@pytest.mark.parametrize(
("model_id", "expected_source"),
[
("deepseek-v4-turbo", "family:deepseek-v4"),
("deepseek-flash-turbo", "family:deepseek-flash"),
],
)
def test_unseeded_point_release_does_not_inherit_v3(model_id, expected_source):
info = catalog.resolve_model_info(model_id)
v3 = catalog.resolve_model_info("deepseek-v3")

assert info.source == expected_source
assert info.input_cost_per_1m != v3.input_cost_per_1m
Loading