diff --git a/package.json b/package.json index fc57eb2e..9b3780d1 100644 --- a/package.json +++ b/package.json @@ -2,7 +2,7 @@ "name": "thinkwatch-lite", "private": true, "description": "Desktop app for a local AI API gateway on macOS, Windows and Linux", - "version": "2026.10.5", + "version": "2026.10.6", "type": "module", "packageManager": "pnpm@11.13.0", "scripts": { diff --git a/release-notes/2026.10.6.md b/release-notes/2026.10.6.md new file mode 100644 index 00000000..ee6096a9 --- /dev/null +++ b/release-notes/2026.10.6.md @@ -0,0 +1,30 @@ +**Upgrade notes:** +- The bundled core is now 0.63.0, which speaks control-plane protocol 39. A remote server has to run core 0.63.0 too: this version does not connect to 0.62.0, and 2026.10.5 and earlier do not connect to 0.63.0. +- Request history is kept. +- A configuration edited by hand that lists an upstream twice in a strategy group, or a name that is not an upstream, no longer loads; safe mode names the group and the member. +- A key's concurrency limit now counts streamed answers until they end, so a key at its limit waits more often than before. + +**Strategy groups:** +- Load-balance groups take a weight per member, from 1 to 100, and a choice of how to distribute: by ratio, by speed, by reliability, or both. Speed is the time to first token, reliability the recent success rate; both adjust the weights, and an upstream without enough measurements counts as average. +- Ongoing conversations stay on their upstream and count toward its share. The route test shows each member's weight, speed, success rate and share, and predicts the next upstream. + +**Upstreams:** +- An upstream can have a concurrency limit. A conversation in progress waits for a free slot; other requests go to the next upstream, and when every upstream is full the request waits, then fails with a clear message. +- A model's context window and output limit can be set by hand under "Specs…" in the upstream's model list, for relays the price table does not know or gets wrong. A manual value is marked and replaces the price table's. +- Check-up findings (a different model name in answers, input reported high or low, few cache reads) now appear on the upstream's row; the separate Check-up tab is gone. + +**Failover settings:** +- "Move to the next upstream when the start times out" (off by default): a streamed answer that has sent nothing within the wait moves to the next upstream, if one can take it. The last upstream keeps waiting. +- "Wait for a free slot at most" (30 seconds by default) is the total time a request waits for a full upstream or a key's per-minute or per-hour limit. + +**Keys:** +- A key can have usage limits: requests, tokens or cost (USD) per minute, hour, day, week or month, all of which must hold. A token limit can count cache reads. Day, week and month reset at local midnight, on Monday and on the 1st. +- The key dialog shows each limit's usage and reset time, and lists the models the key can use that have no price; they count as $0. Keys at a limit are marked in the list, and a notification comes at 80% and when a limit is reached. + +**Traffic:** +- The attempt chain shows an upstream given up for a slow start, with the input tokens it may have billed, upstreams skipped at their concurrency limit, and time spent waiting for a slot. +- Requests refused by a key's usage limit or because every upstream was full are labelled as such. +- On a Responses WebSocket connection, each turn is now a request of its own, with its usage and cost. + +**Also:** +- The menu bar's menu follows the appearance chosen in the app instead of the menu bar's tint, so it no longer opens light in Dark Mode over a bright wallpaper. diff --git a/scripts/shots/core/en/keys.json b/scripts/shots/core/en/keys.json index d53cbaee..bbaf468f 100644 --- a/scripts/shots/core/en/keys.json +++ b/scripts/shots/core/en/keys.json @@ -3,6 +3,7 @@ "allow": null, "default": true, "key": "tw-m9EXAMPLEj7ak", + "limits": [], "max_concurrent": null, "name": "default", "route": null @@ -11,6 +12,7 @@ "allow": null, "client": "claude-code", "key": "tw-5aEXAMPLEqmg8", + "limits": [], "max_concurrent": null, "name": "claude-code", "route": null @@ -19,6 +21,7 @@ "allow": null, "client": "codex", "key": "tw-h5EXAMPLEyrqn", + "limits": [], "max_concurrent": 4, "name": "codex", "route": "codex" @@ -30,6 +33,7 @@ ], "client": "cursor", "key": "tw-e6EXAMPLE6td6", + "limits": [], "max_concurrent": null, "name": "cursor", "route": "cursor" diff --git a/scripts/shots/core/en/overview.json b/scripts/shots/core/en/overview.json index 158330cf..f80c2019 100644 --- a/scripts/shots/core/en/overview.json +++ b/scripts/shots/core/en/overview.json @@ -26,6 +26,7 @@ "allow": null, "default": true, "key": "tw-m9…j7ak", + "limits": [], "max_concurrent": null, "name": "default", "route": null @@ -34,6 +35,7 @@ "allow": null, "client": "claude-code", "key": "tw-5a…qmg8", + "limits": [], "max_concurrent": null, "name": "claude-code", "route": null @@ -42,6 +44,7 @@ "allow": null, "client": "codex", "key": "tw-h5…yrqn", + "limits": [], "max_concurrent": 4, "name": "codex", "route": "codex" @@ -53,6 +56,7 @@ ], "client": "cursor", "key": "tw-e6…6td6", + "limits": [], "max_concurrent": null, "name": "cursor", "route": "cursor" @@ -63,32 +67,39 @@ "failover": { "failures_to_pause": 3, "max_pause_secs": 600, + "next_on_slow_start": false, "no_balance_pause_secs": 1800, "pause_secs": 60, "quota_pause_secs": 3600, "rate_limit_max_pause_secs": 3600, + "slot_wait_secs": 30, "stream_start_wait_secs": 15 }, "groups": [ { + "balance_by": "weights", "builtin": false, "kind": "fallback", "name": "main", "providers": [ "anthropic", "relay" - ] + ], + "weights": {} }, { + "balance_by": "weights", "builtin": false, "kind": "cheapest", "name": "budget", "providers": [ "anthropic", "relay" - ] + ], + "weights": {} }, { + "balance_by": "weights", "builtin": true, "kind": "fallback", "name": "__all__", @@ -100,7 +111,8 @@ "deepseek", "gemini", "ollama" - ] + ], + "weights": {} } ], "listen": { diff --git a/scripts/shots/core/en/status.json b/scripts/shots/core/en/status.json index 6d6548f8..a5b7f59b 100644 --- a/scripts/shots/core/en/status.json +++ b/scripts/shots/core/en/status.json @@ -14,5 +14,5 @@ "reachable": [] }, "uptime_secs": 0, - "version": "0.61.0" + "version": "0.62.0" } diff --git a/scripts/shots/core/oracle.sh b/scripts/shots/core/oracle.sh index 9173ad43..255f4192 100755 --- a/scripts/shots/core/oracle.sh +++ b/scripts/shots/core/oracle.sh @@ -8,15 +8,16 @@ # 钉住的 core 之后。截图页的配置类数据(上游、密钥、路由、安全规则、试算) # 直接用这几份文件,所以它们必须是 core 真的答出来的,不是照着样子手写的。 # -# 用的是 src-tauri/Cargo.toml 钉住的那个 tag:从检出里 `git archive` 一份到临时 -# 目录,放进 oracle.rs 跑一次。**不改那个检出。**要先在那边 `git fetch --tags`。 +# 用的是 src-tauri/Cargo.toml 钉住的那个 tag(core 还没发版时临时钉的 rev 也认):从检出里 +# `git archive` 一份到临时目录,放进 oracle.rs 跑一次。**不改那个检出。**要先在那边 +# `git fetch --tags`。 set -euo pipefail core=${1:?用法:oracle.sh } here=$(cd "$(dirname "$0")" && pwd) root=$(cd "$here/../../.." && pwd) -tag=$(sed -n 's/^tw-api = .*tag = "\([^"]*\)".*/\1/p' "$root/src-tauri/Cargo.toml") -[ -n "$tag" ] || { echo "src-tauri/Cargo.toml 里找不到 tw-api 的 tag" >&2; exit 1; } +tag=$(sed -nE 's/^tw-api = .*(tag|rev) = "([^"]*)".*/\2/p' "$root/src-tauri/Cargo.toml") +[ -n "$tag" ] || { echo "src-tauri/Cargo.toml 里找不到 tw-api 的 tag 或 rev" >&2; exit 1; } src=$(mktemp -d) trap 'rm -rf "$src"' EXIT diff --git a/scripts/shots/core/zh/keys.json b/scripts/shots/core/zh/keys.json index d53cbaee..bbaf468f 100644 --- a/scripts/shots/core/zh/keys.json +++ b/scripts/shots/core/zh/keys.json @@ -3,6 +3,7 @@ "allow": null, "default": true, "key": "tw-m9EXAMPLEj7ak", + "limits": [], "max_concurrent": null, "name": "default", "route": null @@ -11,6 +12,7 @@ "allow": null, "client": "claude-code", "key": "tw-5aEXAMPLEqmg8", + "limits": [], "max_concurrent": null, "name": "claude-code", "route": null @@ -19,6 +21,7 @@ "allow": null, "client": "codex", "key": "tw-h5EXAMPLEyrqn", + "limits": [], "max_concurrent": 4, "name": "codex", "route": "codex" @@ -30,6 +33,7 @@ ], "client": "cursor", "key": "tw-e6EXAMPLE6td6", + "limits": [], "max_concurrent": null, "name": "cursor", "route": "cursor" diff --git a/scripts/shots/core/zh/overview.json b/scripts/shots/core/zh/overview.json index 8665b2eb..f64795d9 100644 --- a/scripts/shots/core/zh/overview.json +++ b/scripts/shots/core/zh/overview.json @@ -26,6 +26,7 @@ "allow": null, "default": true, "key": "tw-m9…j7ak", + "limits": [], "max_concurrent": null, "name": "default", "route": null @@ -34,6 +35,7 @@ "allow": null, "client": "claude-code", "key": "tw-5a…qmg8", + "limits": [], "max_concurrent": null, "name": "claude-code", "route": null @@ -42,6 +44,7 @@ "allow": null, "client": "codex", "key": "tw-h5…yrqn", + "limits": [], "max_concurrent": 4, "name": "codex", "route": "codex" @@ -53,6 +56,7 @@ ], "client": "cursor", "key": "tw-e6…6td6", + "limits": [], "max_concurrent": null, "name": "cursor", "route": "cursor" @@ -63,32 +67,39 @@ "failover": { "failures_to_pause": 3, "max_pause_secs": 600, + "next_on_slow_start": false, "no_balance_pause_secs": 1800, "pause_secs": 60, "quota_pause_secs": 3600, "rate_limit_max_pause_secs": 3600, + "slot_wait_secs": 30, "stream_start_wait_secs": 15 }, "groups": [ { + "balance_by": "weights", "builtin": false, "kind": "fallback", "name": "主力", "providers": [ "anthropic", "relay" - ] + ], + "weights": {} }, { + "balance_by": "weights", "builtin": false, "kind": "cheapest", "name": "低价", "providers": [ "anthropic", "relay" - ] + ], + "weights": {} }, { + "balance_by": "weights", "builtin": true, "kind": "fallback", "name": "__all__", @@ -100,7 +111,8 @@ "deepseek", "gemini", "ollama" - ] + ], + "weights": {} } ], "listen": { diff --git a/scripts/shots/core/zh/status.json b/scripts/shots/core/zh/status.json index 6d6548f8..a5b7f59b 100644 --- a/scripts/shots/core/zh/status.json +++ b/scripts/shots/core/zh/status.json @@ -14,5 +14,5 @@ "reachable": [] }, "uptime_secs": 0, - "version": "0.61.0" + "version": "0.62.0" } diff --git a/scripts/shots/mock/core.ts b/scripts/shots/mock/core.ts index 31105ea0..ac542204 100644 --- a/scripts/shots/mock/core.ts +++ b/scripts/shots/mock/core.ts @@ -121,6 +121,7 @@ export const CORE: { [N in WebviewEndpoint]: Handler } = { UpdateProvider: refuse, DeleteProvider: refuse, ProviderModels: (_req, [name]) => providerModels(name!) ?? notFound(`Upstream ${name}`), + SetModelSpec: refuse, RefreshProviderModels: (_req, [name]) => providerModels(name!) ?? notFound(`Upstream ${name}`), RefreshStaleModels: () => ({ providers: [] }), diff --git a/src-tauri/Cargo.lock b/src-tauri/Cargo.lock index dffa0c2c..7984f4bc 100644 --- a/src-tauri/Cargo.lock +++ b/src-tauri/Cargo.lock @@ -4766,7 +4766,7 @@ dependencies = [ [[package]] name = "thinkwatch-lite" -version = "2026.10.5" +version = "2026.10.6" dependencies = [ "anyhow", "block2", @@ -5268,8 +5268,8 @@ dependencies = [ [[package]] name = "tw-api" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "serde", "serde_json", @@ -5280,8 +5280,8 @@ dependencies = [ [[package]] name = "tw-dialect" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "serde", "serde_json", @@ -5289,8 +5289,8 @@ dependencies = [ [[package]] name = "tw-guard" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "base64 0.22.1", "bytes", @@ -5305,8 +5305,8 @@ dependencies = [ [[package]] name = "tw-link" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "serde", "serde_json", @@ -5332,8 +5332,8 @@ dependencies = [ [[package]] name = "tw-types" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "serde", "ts-rs", @@ -5341,8 +5341,8 @@ dependencies = [ [[package]] name = "tw-watch" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "notify", "thiserror 2.0.21", @@ -5351,8 +5351,8 @@ dependencies = [ [[package]] name = "tw-yaml" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "saphyr-parser", "thiserror 2.0.21", diff --git a/src-tauri/Cargo.toml b/src-tauri/Cargo.toml index 05be5b2e..1a130ac5 100644 --- a/src-tauri/Cargo.toml +++ b/src-tauri/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "thinkwatch-lite" -version = "2026.10.5" +version = "2026.10.6" edition = "2024" # **这个数决定依赖能升到哪一版。**edition 2024 的解析器只挑声明的 Rust # 版本编得动的依赖:写 1.85 的时候,`cargo update` 一直停在旧的 time 和 @@ -26,12 +26,12 @@ license = "MIT" # twcore 二进制必须和这里编译进去的协议镜像来自同一个 core 版本 —— 打包脚本正是 # 从 tw-api 锁到的 tag 去取二进制的(见 scripts/fetch-core.sh)。 [workspace.dependencies] -tw-api = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } -tw-types = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } -tw-yaml = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } -tw-guard = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } -tw-watch = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } -tw-link = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } +tw-api = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } +tw-types = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } +tw-yaml = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } +tw-guard = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } +tw-watch = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } +tw-link = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } [lib] name = "thinkwatch_lite_lib" diff --git a/src-tauri/src/call.rs b/src-tauri/src/call.rs index afa29d7f..7befdab4 100644 --- a/src-tauri/src/call.rs +++ b/src-tauri/src/call.rs @@ -97,6 +97,8 @@ webview_endpoints![ UpdateProvider, DeleteProvider, ProviderModels, + // 手写一个模型的上下文窗口、输出上限 + SetModelSpec, RefreshProviderModels, RefreshStaleModels, CreateProxy, diff --git a/src-tauri/src/clients/desktop_rule.rs b/src-tauri/src/clients/desktop_rule.rs index a781a23e..3a1fd3be 100644 --- a/src-tauri/src/clients/desktop_rule.rs +++ b/src-tauri/src/clients/desktop_rule.rs @@ -477,7 +477,8 @@ mod tests { "security": {"redact": "observe", "inspect_tools": "observe", "content": "observe"}, "retention": {"body_days": 7, "row_days": 90, "body_max_bytes": 0, "body_bytes_now": 0}, "failover": {"failures_to_pause": 3, "pause_secs": 60, "max_pause_secs": 600, "no_balance_pause_secs": 1800, - "quota_pause_secs": 3600, "rate_limit_max_pause_secs": 3600, "stream_start_wait_secs": 15} + "quota_pause_secs": 3600, "rate_limit_max_pause_secs": 3600, "stream_start_wait_secs": 15, + "next_on_slow_start": false, "slot_wait_secs": 30} }) } @@ -506,8 +507,8 @@ mod tests { "rules": [{"name": "rest", "conditions": [], "to": "chatgpt", "catch_all": true, "phase_two": false, "shadowed": false}]} ]), serde_json::json!([ - {"name": "claude-desktop", "key": "tw-x", "client": "claude-desktop", "disabled": false, "default": false}, - {"name": "cd-2", "key": "tw-y", "client": "claude-desktop@Ubuntu", "route": "codex", "disabled": false, "default": false} + {"name": "claude-desktop", "key": "tw-x", "client": "claude-desktop", "disabled": false, "default": false, "limits": []}, + {"name": "cd-2", "key": "tw-y", "client": "claude-desktop@Ubuntu", "route": "codex", "disabled": false, "default": false, "limits": []} ]), ); Snapshot { diff --git a/src-tauri/src/clients/ops.rs b/src-tauri/src/clients/ops.rs index 6a6a4259..71af8053 100644 --- a/src-tauri/src/clients/ops.rs +++ b/src-tauri/src/clients/ops.rs @@ -981,6 +981,8 @@ pub(crate) mod tests { disabled: false, default, last_seen_ms: None, + limits: vec![], + unpriced_models: vec![], } } diff --git a/src-tauri/src/notices/fixtures/snapshot.json b/src-tauri/src/notices/fixtures/snapshot.json index 1b5e5d94..eddfc802 100644 --- a/src-tauri/src/notices/fixtures/snapshot.json +++ b/src-tauri/src/notices/fixtures/snapshot.json @@ -101,7 +101,9 @@ "kind": "fallback", "providers": [ "官方" - ] + ], + "weights": {}, + "balance_by": "weights" } ], "clients": [ @@ -111,7 +113,8 @@ "max_concurrent": null, "route": null, "allow": null, - "default": true + "default": true, + "limits": [] } ], "listen": { @@ -172,7 +175,9 @@ "no_balance_pause_secs": 1800, "quota_pause_secs": 3600, "rate_limit_max_pause_secs": 3600, - "stream_start_wait_secs": 15 + "stream_start_wait_secs": 15, + "next_on_slow_start": false, + "slot_wait_secs": 30 }, "price_sheets": [] }, diff --git a/src-tauri/src/notices/rules.rs b/src-tauri/src/notices/rules.rs index 00350099..c3b094c4 100644 --- a/src-tauri/src/notices/rules.rs +++ b/src-tauri/src/notices/rules.rs @@ -48,6 +48,7 @@ const UPSTREAMS: &str = "upstreams"; const SECURITY: &str = "security"; const MCP: &str = "mcp"; const PLUGINS: &str = "plugins"; +const KEYS: &str = "keys"; const SETTINGS: &str = "settings"; /// 设置页的「网关监听」一节(`settings:<节>`,界面滚到那一节) const LISTEN_SETTINGS: &str = "settings:listen"; @@ -60,6 +61,7 @@ pub fn default_view(key: &str) -> &'static str { // 客户端配置里的可疑内容在 MCP 页:服务器、技能、钩子和扫描发现都在那儿 "scan" => MCP, "plugin" => PLUGINS, + "keylimit" => KEYS, // 网关、配置文件、监听,以及认不出来的:设置页至少能看到网关在不在跑 _ => SETTINGS, } @@ -396,6 +398,7 @@ pub fn from_event(ev: &Event) -> Vec { .event(), ] } + Event::KeyLimitAlert { .. } => key_limit(ev), Event::PluginFailed { plugin_id, plugin_name, @@ -407,6 +410,134 @@ pub fn from_event(ev: &Event) -> Vec { } } +/// 一把网关密钥这一期(天、周、月)的用量到了一条上限的八成,或者到了上限。 +/// +/// **每一期都是新的一件**(`event`):core 每一期、每一档只报一次,昨天看过的那一条不该 +/// 压住今天的。**同一条上限只留一条**:到顶时收起这一期八成的那条;新的一期到了八成, +/// 上一期到顶的那条也不再是现状,一并收起。正文写密钥的名字(不写它的值)、哪一条、 +/// 什么时候重置 +fn key_limit(ev: &Event) -> Vec { + let &Event::KeyLimitAlert { + ref key, + per, + measure, + max, + used, + cache_reads, + reached, + resets_at_ms, + .. + } = ev + else { + return Vec::new(); + }; + // 同一条上限:同一个周期、同一种量、缓存读取算法相同(和 core 认重复的规矩一样) + let limit = format!( + "keylimit:{key}:{per}-{measure}{}", + if cache_reads { "-cache" } else { "" } + ); + let near = format!("{limit}:near"); + let phrase = limit_phrase(per, measure, max, cache_reads); + let secs = resets_at_ms.saturating_sub(super::now_ms()) / 1000; + let reset = after(secs); + if reached { + return vec![ + Signal::cleared(near), + Signal::raised( + limit, + Level::Warning, + tr!( + format!("密钥「{key}」已达用量上限"), + format!("Key “{key}” Reached Its Usage Limit") + ), + ) + .body(tr!( + format!("上限「{phrase}」已用满{reset}。使用此密钥的请求会被拒绝。"), + format!( + "The limit of {phrase} has been reached{reset}. Requests with this key will be rejected." + ) + )) + .view(KEYS) + .event(), + ]; + } + let percent = (used.saturating_mul(100) / max.max(1)).min(99); + let n = limit_amount(measure, used); + let spent = match measure { + tw_api::LimitMeasure::Requests => tr!(format!("{n} 次"), format!("{n} requests")), + tw_api::LimitMeasure::Tokens => tr!(format!("{n} token"), format!("{n} tokens")), + tw_api::LimitMeasure::Cost => n, + }; + vec![ + Signal::cleared(limit.clone()), + Signal::raised( + near, + Level::Warning, + tr!( + format!("密钥「{key}」的用量接近上限"), + format!("Key “{key}” Is Nearing Its Usage Limit") + ), + ) + .body(tr!( + format!( + "上限「{phrase}」已使用 {percent}%({spent}){reset}。达到上限后,使用此密钥的请求会被拒绝。" + ), + format!( + "The limit of {phrase} is {percent}% used ({spent}){reset}. Once it is reached, requests with this key will be rejected." + ) + )) + .view(KEYS) + .event(), + ] +} + +/// 一条上限说成一句:「每天 $5.00 费用」「每天 1,000,000 token(含缓存读取)」;英文是 +/// 「$5.00 per day」「1,000,000 tokens per day (cache reads included)」。和密钥对话框里的同一种说法 +fn limit_phrase( + per: tw_api::LimitPer, + measure: tw_api::LimitMeasure, + max: u64, + cache_reads: bool, +) -> String { + use tw_api::{LimitMeasure, LimitPer}; + let (zh, en) = match per { + LimitPer::Minute => ("分钟", "minute"), + LimitPer::Hour => ("小时", "hour"), + LimitPer::Day => ("天", "day"), + LimitPer::Week => ("周", "week"), + LimitPer::Month => ("月", "month"), + }; + let n = limit_amount(measure, max); + match measure { + LimitMeasure::Requests => tr!( + format!("每{zh} {n} 次请求"), + format!("{n} requests per {en}") + ), + LimitMeasure::Tokens => { + let cache = if cache_reads { + tr!("(含缓存读取)", " (cache reads included)") + } else { + "" + }; + tr!( + format!("每{zh} {n} token{cache}"), + format!("{n} tokens per {en}{cache}") + ) + } + LimitMeasure::Cost => tr!(format!("每{zh} {n} 费用"), format!("{n} per {en}")), + } +} + +/// 用量或上限写成字:费用是微分,写到分;请求数、token 数带千分位 +fn limit_amount(measure: tw_api::LimitMeasure, n: u64) -> String { + use crate::menubar::model::{cost_long, grouped}; + let n = i64::try_from(n).unwrap_or(i64::MAX); + match measure { + tw_api::LimitMeasure::Cost => cost_long(n), + _ => grouped(n), + } +} + /// 一个插件没能把事情做成(core 的 `plugin_failed`):在一个请求上运行出错(`request` 有), /// 或者文件变了、加载不了,从此不再运行(没有)。 /// diff --git a/src-tauri/src/notices/tests.rs b/src-tauri/src/notices/tests.rs index 3a93f012..a0faa09b 100644 --- a/src-tauri/src/notices/tests.rs +++ b/src-tauri/src/notices/tests.rs @@ -608,6 +608,7 @@ fn every_key_lands_on_the_page_that_handles_it() { ("toolwall:relay", "security"), ("scan", "mcp"), ("plugin:add-date", "plugins"), + ("keylimit:codex:day-cost", "keys"), ] { assert_eq!(rules::default_view(key), view, "{key}"); } @@ -646,6 +647,9 @@ async fn an_unreachable_upstream_is_only_listed_and_needs_real_evidence_to_clear status: Some(200), error: None, ms: 800, + usage: None, + queued_ms: None, + skipped: None, }], billing: tw_api::Billing::PerToken, }); @@ -1528,3 +1532,145 @@ async fn reconciling_fills_in_what_was_missed_without_saying_anything_twice() { "{shown:?}" ); } + +/// 一把网关密钥的一条上限到了八成(`reached` 为假)或者到顶,`secs` 秒之后重置 +fn key_limit_alert( + measure: tw_api::LimitMeasure, + max: u64, + used: u64, + cache_reads: bool, + reached: bool, + secs: u64, +) -> tw_api::Event { + tw_api::Event::KeyLimitAlert { + id: 1, + key: "codex".into(), + per: tw_api::LimitPer::Day, + measure, + max, + used, + cache_reads, + reached, + resets_at_ms: resets_in(secs).unwrap(), + at_ms: T0, + } +} + +/// 通知写明是哪把密钥(名字,不是值)、哪一条上限、什么时候重置 +#[test] +fn a_key_limit_names_the_key_the_limit_and_the_reset() { + use tw_api::LimitMeasure::{Cost, Requests, Tokens}; + let said = |ev: &tw_api::Event| { + let s = rules::from_event(ev) + .into_iter() + .find(|s| s.change == Change::Raised) + .expect("要说"); + (s.title, s.body) + }; + with_lang(Lang::Zh, || { + assert_eq!( + said(&key_limit_alert( + Cost, + 5_000_000, + 5_020_000, + false, + true, + 5 * 3600 + )), + ( + "密钥「codex」已达用量上限".to_string(), + "上限「每天 $5.00 费用」已用满,约 5 小时后重置。使用此密钥的请求会被拒绝。" + .to_string() + ) + ); + assert_eq!( + said(&key_limit_alert( + Tokens, 1_000_000, 812_345, true, false, 600 + )) + .1, + "上限「每天 1,000,000 token(含缓存读取)」已使用 81%(812,345 token),约 10 分钟后重置。\ + 达到上限后,使用此密钥的请求会被拒绝。" + ); + assert_eq!( + said(&key_limit_alert( + Requests, + 500, + 400, + false, + false, + 3 * 86_400 + )) + .1, + "上限「每天 500 次请求」已使用 80%(400 次),约 3 天后重置。达到上限后,使用此密钥的请求会被拒绝。" + ); + }); + with_lang(Lang::En, || { + assert_eq!( + said(&key_limit_alert( + Cost, + 5_000_000, + 5_020_000, + false, + true, + 5 * 3600 + )), + ( + "Key “codex” Reached Its Usage Limit".to_string(), + "The limit of $5.00 per day has been reached and resets in about 5 hours. \ + Requests with this key will be rejected." + .to_string() + ) + ); + let (title, body) = said(&key_limit_alert( + Tokens, 1_000_000, 812_345, true, false, 600, + )); + assert_eq!(title, "Key “codex” Is Nearing Its Usage Limit"); + assert_eq!( + body, + "The limit of 1,000,000 tokens per day (cache reads included) is 81% used \ + (812,345 tokens) and resets in about 10 minutes. Once it is reached, requests with \ + this key will be rejected." + ); + }); +} + +/// 八成说一次、到顶再说一次;**同一条上限只留一条**。新的一期到八成时,上一期到顶的那条 +/// 收起、八成那条重新说(每一期都是新的一件,昨天看过的不压住今天的) +#[tokio::test] +async fn a_key_limit_is_told_at_80_percent_and_at_the_limit_once_per_period() { + use tw_api::LimitMeasure::Cost; + let b = bed(); + let near = key_limit_alert(Cost, 5_000_000, 4_100_000, false, false, 3600); + let reached = key_limit_alert(Cost, 5_000_000, 5_000_000, false, true, 3600); + b.bus.on_event(&near); + b.bus.on_event(&reached); + assert_eq!(b.titles().len(), 2, "两档都弹系统通知:{:?}", b.titles()); + let keys = |b: &Bed| b.bus.list().into_iter().map(|n| n.key).collect::>(); + assert_eq!(keys(&b), ["keylimit:codex:day-cost"], "到顶时收起八成那条"); + assert_eq!( + *b.withdrawn.lock().unwrap(), + ["keylimit:codex:day-cost:near"] + ); + + b.bus.mark_all_read(); + // 第二天 + b.bus.on_event(&near); + assert_eq!( + keys(&b), + ["keylimit:codex:day-cost:near"], + "上一期到顶的那条收起" + ); + assert!(!b.bus.list()[0].read, "新的一期是新的一件"); + assert_eq!(b.titles().len(), 3); + + // 算缓存读取的 token 上限和不算的是两条 + b.bus.on_event(&key_limit_alert( + tw_api::LimitMeasure::Tokens, + 100, + 100, + true, + true, + 3600, + )); + assert!(keys(&b).contains(&"keylimit:codex:day-tokens-cache".to_string())); +} diff --git a/src-tauri/tauri.conf.json b/src-tauri/tauri.conf.json index 121160ad..8cf7b534 100644 --- a/src-tauri/tauri.conf.json +++ b/src-tauri/tauri.conf.json @@ -1,7 +1,7 @@ { "$schema": "https://schema.tauri.app/config/2", "productName": "ThinkWatch Lite", - "version": "2026.10.5", + "version": "2026.10.6", "identifier": "app.thinkwatch.lite", "build": { "beforeDevCommand": "pnpm dev", diff --git a/src-tauri/tests/control_plane.rs b/src-tauri/tests/control_plane.rs index e08a626b..f4d35907 100644 --- a/src-tauri/tests/control_plane.rs +++ b/src-tauri/tests/control_plane.rs @@ -390,6 +390,7 @@ async fn a_bedrock_upstream_is_saved_the_way_the_dialog_sends_it() { models_only: None, billing: None, pricing: None, + max_concurrent: None, disabled: false, }, base_version: None, diff --git a/src/RequestDrawer.i18n.tsx b/src/RequestDrawer.i18n.tsx index 948e22a6..4f31e508 100644 --- a/src/RequestDrawer.i18n.tsx +++ b/src/RequestDrawer.i18n.tsx @@ -98,6 +98,12 @@ export const requestDrawerText = messages( /** 规则写的拒绝理由,或者选中的上游为何都无法服务 */ reason: "原因", attempts: "尝试链", + /** 这一跳等空位等了多久(上游满着)。秒数已经按一位小数写好 */ + queued: (s: string) => `排队 ${s} 秒`, + /** 放弃了的一跳(开头超时)下面那一行:上游没报用量时,网关估的输入 */ + abandonedEstimate: (n: string) => `输入约 ${n} token`, + mayBeBilled: "上游可能已计费", + mayBeBilledTip: "上游是否收取这部分费用无法得知,此请求的费用不含这部分。", // 尝试链里一跳发出的模型名和客户端写的不同:悬停按原因说 sentModel: (model: string) => `规则改写了模型名:这一跳发给上游的是 ${model},费用按它计算`, sentByAlias: (upstream: string, model: string) => `别名:这一跳发给 ${upstream} 的是 ${model},费用按它计算`, @@ -112,6 +118,11 @@ export const requestDrawerText = messages( deniedAfterPick: (rule: string) => `选定上游后,规则「${rule}」拒绝了此请求,未发往任何上游。`, deniedBeforePick: (rule: string) => `选定上游之前,规则「${rule}」已拒绝此请求,未发往任何上游。`, unavailable: "规则选中的上游均无法服务此请求,未发往任何上游。", + /** 前面几跳里有满着跳过的、开头超时放弃的:它们不是上游的失败 */ + switched: (n: number) => `已自动切换上游:前 ${n} 次尝试未接下此请求。`, + limited: "网关密钥已达到用量上限,此请求未发往任何上游。", + busy: "上游均已达到并发上限,等待期间没有空出位置,此请求未发往任何上游。", + busyAfterTries: (n: number) => `发出的 ${n} 次尝试未成功,其余上游均已达到并发上限,等待期间没有空出位置。`, noRouting: "此请求由网关本地应答,未经过路由。", routingPending: "路由尚未完成", noAttempts: "此请求没有上游尝试记录。", @@ -245,6 +256,10 @@ export const requestDrawerText = messages( deniedBy: "Denied by", reason: "Reason", attempts: "Attempts", + queued: (s: string) => `Queued ${s} s`, + abandonedEstimate: (n: string) => `About ${n} input tokens`, + mayBeBilled: "may have been billed by the upstream", + mayBeBilledTip: "Whether the upstream charged for these tokens is unknown; they are not included in this request's cost.", sentModel: (model: string) => `A rule rewrote the model: this attempt sent ${model}, and the cost is priced by it`, sentByAlias: (upstream: string, model: string) => `Alias: this attempt sent ${model} to ${upstream}, and the cost is priced by it`, @@ -265,6 +280,16 @@ export const requestDrawerText = messages( deniedBeforePick: (rule: string) => `Rule “${rule}” denied this request before an upstream was chosen; it was not sent to any upstream.`, unavailable: "No upstream the rule selected can serve this request; it was not sent to any upstream.", + switched: (n: number) => + n === 1 + ? "Switched upstreams automatically: the first attempt did not take this request." + : `Switched upstreams automatically: the first ${n} attempts did not take this request.`, + limited: "The gateway key had reached a usage limit; this request was not sent to any upstream.", + busy: "Every upstream was at its concurrency limit and none freed up in time; this request was not sent to any upstream.", + busyAfterTries: (n: number) => + n === 1 + ? "The attempt sent did not succeed, and the other upstreams were at their concurrency limits with none freeing up in time." + : `The ${n} attempts sent did not succeed, and the other upstreams were at their concurrency limits with none freeing up in time.`, noRouting: "The gateway answered this request locally; it did not go through routing.", routingPending: "Routing has not finished yet", noAttempts: "No upstream attempts were recorded for this request.", diff --git a/src/RequestDrawer.tsx b/src/RequestDrawer.tsx index 888439a8..dfd5b648 100644 --- a/src/RequestDrawer.tsx +++ b/src/RequestDrawer.tsx @@ -39,10 +39,11 @@ import { } from "./labels"; import { prettyJson } from "./prettyJson"; import { requestDrawerText } from "./RequestDrawer.i18n"; -import { notSent, routingFacts, type RoutingNote } from "./requestRouting"; +import { notSent, routingFacts, skippedHop, type RoutingNote } from "./requestRouting"; import { ActionBadge, byCodepoints, EventDetail, ruleName, whereOf } from "./security/labels"; import { usd, + type AttemptUsage, type AttemptView, type BodyView, type CoreEvent, @@ -682,35 +683,51 @@ function Routing({ r, plugins, running }: { r: HistoryRow; plugins: PluginRunVie
    {f.hops.map(({ attempt: a, denied }, i) => { const outcome = attemptText(a); + // 没有发给这个上游的一跳:被规则拒绝,或者它满着、换了下一家 + const unsent = denied || skippedHop(a); return ( -
  1. - {i + 1} - - - {a.provider} - - {/* 这一跳发出的模型名:有一跳改了名、或者用了别名、指定模型时每一跳都写, - 费用也按它算 */} - {m.show && m.hops[i] && } - {denied ? ( - // 选定上游之后的规则在这一跳拒绝了它:没有发给这个上游,不是上游的失败 - - {deniedHopText(f.deniedBy ?? "")} - - ) : ( - /* **失败的原因要留着** —— 一条说「试过 A → B → C」的链和一条还说清 - 每一跳为什么失败的链,排查价值差得远 */ - - {outcome.text} - - )} - {/* 没有发出的那一跳没有耗时可言 */} - {denied ? "—" : ms(a.ms)} +
  2. +
    + {i + 1} + + + {a.provider} + + {/* 这一跳发出的模型名:有一跳改了名、或者用了别名、指定模型时每一跳都写, + 费用也按它算 */} + {m.show && m.hops[i] && } + {denied ? ( + // 选定上游之后的规则在这一跳拒绝了它:没有发给这个上游,不是上游的失败 + + {deniedHopText(f.deniedBy ?? "")} + + ) : ( + /* **失败的原因要留着** —— 一条说「试过 A → B → C」的链和一条还说清 + 每一跳为什么失败的链,排查价值差得远。短名(开头超时、并发已满)悬停 + 是 core 的原话 */ + + {outcome.tip ? ( + + {outcome.text} + + ) : ( + outcome.text + )} + + )} + {/* 等空位的时间不算在这一跳的耗时里,另写一项 */} + {a.queued_ms != null && a.queued_ms > 0 && ( + + {t.queued(seconds(a.queued_ms))} + + )} + {/* 没有发出的那一跳没有耗时可言 */} + {unsent ? "—" : ms(a.ms)} +
    + {/* 放弃了的这一跳(开头超时)上游可能已经按输入收了钱:不在这个请求的费用里 */} + {a.usage && }
  3. ); })} @@ -726,6 +743,38 @@ function Routing({ r, plugins, running }: { r: HistoryRow; plugins: PluginRunVie ); } +/** + * 放弃了的一跳(开头超时)上游可能已经收了钱的输入,写在那一跳下面一行。上游在流开头 + * 报了的写它报的几种 token;没报的是网关估的输入,写「约」。**输出不知道**,不写。 + * + * 这部分不进这个请求的费用:上游收没收、收了多少,网关看不到。悬停说这一点 + */ +function AbandonedUsage({ usage: u }: { usage: AttemptUsage }) { + const t = useText(requestDrawerText); + const parts = u.estimated + ? [t.abandonedEstimate(u.input.toLocaleString())] + : [ + `${t.input} ${u.input.toLocaleString()}`, + ...(u.cache_read > 0 ? [`${t.cacheReads} ${u.cache_read.toLocaleString()}`] : []), + ...(u.cache_write > 0 ? [`${t.cacheWrites} ${u.cache_write.toLocaleString()}`] : []), + ]; + return ( + // 和上游名对齐:序号那一格 16px 加间距 12px +

    + {parts.join(" · ")} ·{" "} + + {t.mayBeBilled} + +

    + ); +} + +/** 排队等了多久:不到 10 秒的留一位小数,最少写 0.1 */ +function seconds(ms: number): string { + const s = ms / 1000; + return String(s < 10 ? Math.max(0.1, Math.round(s * 10) / 10) : Math.round(s)); +} + /** * 尝试链里一跳发出的模型名。和客户端写的不同时带虚线下划线,悬停按原因说(别名、规则改名、 * 指定模型、插件);原因对不上现在的配置时只说发出的是什么。 @@ -766,6 +815,12 @@ function noteText(n: RoutingNote, t: (typeof requestDrawerText)["zh"]): string { switch (n.kind) { case "failover": return t.failover(n.failed); + case "switched": + return t.switched(n.count); + case "limited": + return t.limited; + case "busy": + return n.tried > 0 ? t.busyAfterTries(n.tried) : t.busy; case "failover_denied": return t.failoverDenied(n.failed, n.rule); case "denied_after_pick": diff --git a/src/control.ts b/src/control.ts index ef4084f7..935cb219 100644 --- a/src/control.ts +++ b/src/control.ts @@ -59,6 +59,7 @@ export const WEBVIEW_ENDPOINTS = [ "UpdateProvider", "DeleteProvider", "ProviderModels", + "SetModelSpec", "RefreshProviderModels", "RefreshStaleModels", "CreateProxy", diff --git a/src/generated/tw-api.ts b/src/generated/tw-api.ts index 9dbdd002..50bf80bc 100644 --- a/src/generated/tw-api.ts +++ b/src/generated/tw-api.ts @@ -1,6 +1,6 @@ // Generated by tw-api (`tw_api::ts::export_all`). Do not edit by hand. -export const CONTROL_API_VERSION = 38; +export const CONTROL_API_VERSION = 39; /** * 一个账号上游登的是哪个账号。 @@ -174,7 +174,7 @@ served_by: Array, */ shadows: Array, /** - * 上下文窗口,来自默认价目表:第一家能服务它的上游发出的那个模型的 + * 上下文窗口:第一家能服务它的上游发出的那个模型的,这一家手写的优先于价目表 */ context_window?: number | null, /** @@ -219,7 +219,26 @@ suggestions: Array, }; /** * 尝试链里一跳的结果。 */ -export type AttemptOutcome = "served" | "status" | "error" | "estimated"; +export type AttemptOutcome = "served" | "status" | "error" | "estimated" | "slow_start"; + +/** + * 放弃了的一跳([`AttemptOutcome::SlowStart`])上游可能已经收了钱的输入。 + * + * 上游在流开头报了的(Anthropic 的 `message_start`)是它报的数;没报的只有 `input`,是网关 + * 估的(`estimated`,和 [`Event::RequestStarted`] 的 `input_estimate` 同一个数)。**输出不知道**: + * 先想好再输出的模型,放弃之前可能已经想了一阵,上游不说就看不到。 + * + * **不算进这个请求的费用**:上游收没收、收了多少,网关看不到 + */ +export type AttemptUsage = { +/** + * 输入 token,不含缓存读写 + */ +input: number, cache_read: number, cache_write: number, +/** + * `input` 是网关估的,上游什么都没报 + */ +estimated: boolean, }; /** * 尝试链里的一跳。 @@ -237,8 +256,9 @@ export type AttemptView = { provider: string, */ model?: string | null, /** - * WebSocket 的那一跳是一次握手:上游同意升级(101)是 `served`,回了别的 - * 状态码是 `status`,连不上是 `error`。 + * WebSocket 连接的那一跳是一次握手:上游同意升级(101)是 `served`,回了别的 + * 状态码是 `status`,连不上是 `error`。Responses 的连接上每一轮是一个请求,那一跳是这条 + * 已经接下的连接:发出去了是 `served`,状态码记 200。 */ outcome: AttemptOutcome, /** @@ -246,9 +266,24 @@ outcome: AttemptOutcome, */ status?: number | null, /** - * `error` 时的说明。和这一跳报给客户端的那条错误是同一句 + * `error` 时的说明。和这一跳报给客户端的那条错误是同一句。`slow_start` 时说等了多久 + */ +error?: Msg | null, ms: number, +/** + * 放弃了的这一跳(`slow_start`)上游可能已经收了钱的输入(见 [`AttemptUsage`])。估不 + * 出来的(请求解不开)没有。别的结果都没有:接下请求的那一跳的用量在结局里 + */ +usage?: AttemptUsage | null, +/** + * 这一跳等了多少毫秒才轮到一个空位:这家设了 `max_concurrent` 而它满着。不算在 `ms` + * 里。没等的没有 + */ +queued_ms?: number | null, +/** + * 这一跳为什么没发出去:`busy`(这家满着,换了下一家;等过它的话 `queued_ms` 是等了 + * 多久)。这时 `outcome` 是 `error`,`error` 是同一件事的那句话。发出去了的没有 */ -error?: Msg | null, ms: number, }; +skipped?: ServeSkip | null, }; /** * 上游接不接受凭据。 @@ -290,6 +325,15 @@ profile?: string | null, */ region?: string | null, }; +/** + * `load-balance` 组按什么分请求:配置里 `balance_by` 写的那个词。 + * + * 成员的权重永远是底数,快慢、成败算出的系数乘在上面 + * ([`DryRunCandidate::balance_factor`]),长期看各家分到的请求是乘出来的比例;进行中的 + * 对话照旧留在回答它的那一家,记在那一家的份额里。没有测到的上游算中等。 + */ +export type BalanceBy = "weights" | "latency" | "health" | "latency-health"; + /** * 删除时带上的版本。 */ @@ -550,7 +594,16 @@ default?: boolean, * 最后一次被用在什么时候。**按密钥算,不是按客户端自报的标识** —— * 那个可以伪造。从来没被用过时没有 */ -last_seen_ms?: number | null, }; +last_seen_ms?: number | null, +/** + * 用量上限,按配置里的顺序,各带此刻用了多少。没设的是空的 + */ +limits: Array, +/** + * 这把密钥用得到、却没有价格的模型。**只有设了费用上限的密钥才算**:这些模型的 + * 请求费用记 0,费用上限管不住它们,对话框里要提醒一句。没有就不带 + */ +unpriced_models?: Array, }; /** * 路由规则 `when` 里的键。 @@ -894,7 +947,26 @@ sent_model?: string | null, * 发出的名字为什么和请求里写的不一样:`alias`(别名对到这一家的名称)、`rule`(规则 * 改写了模型)、`pinned`(规则指定了这一家发什么模型)。一样时没有 */ -model_via?: string | null, }; +model_via?: string | null, +/** + * 经过的是 `load-balance` 组时,它在组里的权重(没写权重的是 1)。别的时候没有 + */ +weight?: number | null, +/** + * 它典型的快慢:从发出去到回答的第一段内容(最近样本的中位数),毫秒。只在顺序看它时 + * 有:`url-test`,按快慢分的 `load-balance`。样本不够时没有 + */ +ttfb_ms?: number | null, +/** + * 它最近的成功率,0 到 1(最近 50 次、30 分钟以内)。只在按成败分的 `load-balance` + * 里有;不到 5 次时没有 + */ +success_rate?: number | null, +/** + * 按快慢、成败算出的系数,乘在权重([`Self::weight`])上:大于 1 分得多,小于 1 + * 分得少,没有样本的那一项算 1。只在 `balance_by` 不是 `weights` 的 `load-balance` 里有 + */ +balance_factor?: number | null, }; /** * 一次试算的结论。 @@ -944,6 +1016,10 @@ route: string, * 写在第一个」。直指 provider 时是 None。 */ strategy?: GroupKind | null, +/** + * 经过的是 `load-balance` 组时,它按什么分请求。别的时候没有 + */ +balance_by?: BalanceBy | null, /** * `route` | `deny` | `no_match` | `unavailable`(选中的上游都服务不了, * 见 `skipped`)| `intercepted` @@ -1015,7 +1091,8 @@ client_hint?: string | null, * 的指纹)、离这段对话的上一个请求不超过半小时,就还是那一次;隔久了算 * 新的一次。**开始时就给出来**,界面才能把一个还在跑的请求放进它的会话、 * 把那次会话标成进行中。认不出会话的没有:正文里没有任何能认人的东西, - * 或者是 WebSocket 升级(升级请求没有正文) + * 或者是整条连接一行的 WebSocket(升级请求没有正文)。Responses 连接上的每一轮 + * 按那一帧认,和 HTTP 的请求一样 */ session?: string | null, /** @@ -1093,7 +1170,8 @@ session_log_bytes?: number | null, at_ms: number, } | { "kind": "request_headers * 在结局里才到的。模型名只在开始事件里的话,一个开始时没人在听、 * 结束时有人在听的请求,它的用量就不知道该记在哪个模型上。 * - * WebSocket 那条路是空串:升级请求里没有模型名(和开始事件一样)。 + * 整条连接一行的 WebSocket 和开始事件一样:Realtime 是查询串里的那个,别的连接 + * 升级时还不知道,是空串。 */ model: string, status: number, bytes: number, duration_ms: number, /** @@ -1114,7 +1192,7 @@ tokens_per_sec?: number | null, * 上游在回答里写的模型名:Anthropic 和 Chat 的 `model`、Responses 的 * `response.model`、Gemini 的 `modelVersion`。**原样,不归一。** * - * 回答里没写的没有:Bedrock 的 Converse 不写,WebSocket 那条路不看。和 + * 回答里没写的没有:Bedrock 的 Converse 不写,整条连接一行的 WebSocket 不看。和 * `model` 不是一回事 —— 那是客户端要的,这是上游说它用的 */ answered_model?: string | null, } | { "kind": "request_failed", id: number, @@ -1333,7 +1411,23 @@ window: string, /** * 什么时候重置(见 `QuotaWindow::resets_at_ms`)。上游没说就没有 */ -resets_at_ms?: number | null, at_ms: number, } | { "kind": "listen_changed", id: number, +resets_at_ms?: number | null, at_ms: number, } | { "kind": "key_limit_alert", id: number, +/** + * 密钥的名字 + */ +key: string, per: LimitPer, measure: LimitMeasure, +/** + * 上限,单位同 `KeyLimitView::max` + */ +max: number, +/** + * 报的时候用了多少,同上 + */ +used: number, cache_reads?: boolean, +/** + * `true` = 到了上限,之后的请求被拒到 `resets_at_ms`;`false` = 到了八成 + */ +reached: boolean, resets_at_ms: number, at_ms: number, } | { "kind": "listen_changed", id: number, /** * 此刻在听的那个地址,同 `Status::gateway_addr` */ @@ -1431,7 +1525,16 @@ rate_limit_max_pause_secs: number, /** * 流式回答的开头最多等多少秒 */ -stream_start_wait_secs: number, }; +stream_start_wait_secs: number, +/** + * 等过 `stream_start_wait_secs` 还没有内容就换下一家(最后一家照常等) + */ +next_on_slow_start: boolean, +/** + * 一个请求合计最多等多少秒:等密钥的分钟、小时上限空出名额,和等满着(`max_concurrent`) + * 的上游空出位置,共用这一段。0 是不等 + */ +slot_wait_secs: number, }; /** * 一个请求失败在哪一方。和 HTTP 响应里的 `x-thinkwatch-error` 同一个词表, @@ -1450,7 +1553,16 @@ providers: Array, /** * `select` 组优先使用的成员 */ -selected?: string | null, }; +selected?: string | null, +/** + * `load-balance` 组成员的权重,1 到 100。不给 = 都是 1;给了的话没写到的成员是 1。 + * 别的类型只能不给、或者都是 1 + */ +weights?: { [key in string]: number } | null, +/** + * `load-balance` 组按什么分请求。不给 = `weights`;别的类型只能是 `weights` + */ +balance_by?: BalanceBy | null, }; /** * 策略组按什么排候选:配置里 `type` 写的那个词。 @@ -1484,7 +1596,17 @@ kind: GroupKind, * **界面要能切它** —— 这个策略本身就是「UI 上点选或托盘里切」, * 而切不了的话它等于一个只能改 YAML 才能用的功能。 */ -selected?: string | null, providers: Array, }; +selected?: string | null, providers: Array, +/** + * `load-balance` 组每个成员的权重,**每个成员都在**,没写权重的是 1:长期看各家分到的 + * 请求就是这个比例。进行中的对话留在回答它的那一家,那一轮记在那一家的份额里,新对话把 + * 差的补回去。别的类型不用权重,是空的 + */ +weights: { [key in string]: number }, +/** + * `load-balance` 按什么分请求。别的类型永远是 `weights` + */ +balance_by: BalanceBy, }; /** * 哪一项防护。配置里 `security` 下的那个键,也是管理接口路径里的那一段。 @@ -1602,7 +1724,7 @@ translated?: TranslatedView | null, * 任务;看着一次很贵的任务,也回不到具体是哪一条。库里这一列一直 * 都在(`requests.session`,还建了索引),只是没有交出来。 * - * 认不出会话的请求(拼不出指纹的,比如 WebSocket、本地应答)是 `None`。 + * 认不出会话的请求(拼不出指纹的,比如整条连接一行的 WebSocket、本地应答)是 `None`。 */ session?: string | null, /** @@ -1818,7 +1940,51 @@ route?: string | null, /** * 三态:不写 / 写非空 / 写 `[]`(一个都不给) */ -allow?: Array | null, disabled?: boolean, }; +allow?: Array | null, disabled?: boolean, +/** + * 用量上限,整份替换。不带 = 一条都没有 + */ +limits?: Array, }; + +/** + * 新建、保存密钥时的一条用量上限。 + */ +export type KeyLimitInput = { per: LimitPer, measure: LimitMeasure, +/** + * 上限:请求数、token 数,费用是微分。要大于 0 + */ +max: number, +/** + * 只有 token 上限能开 + */ +cache_reads?: boolean, }; + +/** + * 一条用量上限,和它此刻用了多少。 + */ +export type KeyLimitView = { per: LimitPer, measure: LimitMeasure, +/** + * 上限:请求数、token 数,费用是微分 + */ +max: number, +/** + * token 上限把从缓存读的也算进去 + */ +cache_reads: boolean, +/** + * 用了多少,单位同 `max`。**在跑的请求也算**:按它们的输入估算占着,结束时换成 + * 记下的实数 —— 准入看的就是这个数。滚动的是最近那一段时间里的,重启之后从空的 + * 开始;自然的是这一期的,重启之后从请求记录里加回来 + */ +used: number, +/** + * 这一期什么时候结束、重新算。只有天、周、月有 + */ +resets_at_ms?: number | null, +/** + * 到了:`used` 不小于 `max`,新的请求此刻会被拒(滚动的会先等一会儿) + */ +reached: boolean, }; /** * 换哪把密钥(`POST /keys/{name}/rotate`)。 @@ -1949,6 +2115,17 @@ export type LatencyView = { model: string, p50: number, p95: number, */ samples: number, }; +/** + * 一条用量上限数的是什么。 + */ +export type LimitMeasure = "requests" | "tokens" | "cost"; + +/** + * 用量上限按多长一段时间算。分钟、小时是**滚动的**(最近 60 秒、最近 60 分钟); + * 天、周、月是**自然的**,按 core 所在机器的本地时区:零点、周一零点、一号零点重新算。 + */ +export type LimitPer = "minute" | "hour" | "day" | "week" | "month"; + /** * 一张列表要的两样:看哪一段,最多几条(`GET /history`、`/sessions`)。 * @@ -2138,9 +2315,21 @@ export type ModelRow = { id: string, */ enabled: boolean, /** - * 上下文窗口,来自默认价目表 + * 上下文窗口:这一家手写的(`model_specs`),没写时来自价目表 */ context_window?: number | null, +/** + * `context_window` 从哪儿来。不知道上下文窗口时没有 + */ +context_window_source?: SpecSource | null, +/** + * 一次最多输出多少 token:这一家手写的,没写时来自价目表 + */ +max_output_tokens?: number | null, +/** + * `max_output_tokens` 从哪儿来。不知道输出上限时没有 + */ +max_output_tokens_source?: SpecSource | null, /** * 按这个上游选的价目表查到的价格。空 = 无法计价 */ @@ -2160,6 +2349,28 @@ aliases: Array, }; */ export type ModelSource = "discovered" | "manual" | "none"; +/** + * 设一家上游的一个模型的规格(`PUT /provider-model-spec`):价目表不认识这个模型、 + * 或者写错了时手写。**两项都空就是删掉这一项**,回到价目表。 + */ +export type ModelSpecSave = { provider: string, +/** + * 模型 ID,和这家的清单里写的完全相等。去掉首尾空白 + */ +model: string, +/** + * 上下文窗口(token)。空 = 用价目表的 + */ +context_window?: number | null, +/** + * 输出上限(token)。空 = 用价目表的 + */ +max_output_tokens?: number | null, +/** + * 你基于哪一版。**对不上就是 409** + */ +base_version?: string | null, }; + /** * 页面打开时补问模型清单:开始问的是哪几家。答案随 `models_changed` 到。 */ @@ -2843,6 +3054,10 @@ billing?: Billing | null, * 按哪张价目表计价。不给就是默认价目表 */ pricing?: string | null, +/** + * 同时最多发给这家几个请求,1 到 1000。不给就是不限 + */ +max_concurrent?: number | null, /** * 停用 */ @@ -3057,7 +3272,11 @@ references: Array, /** * 选的价目表。空 = 默认价目表 */ -pricing: string | null, }; +pricing: string | null, +/** + * 同时最多发给这家几个请求。不限是空 + */ +max_concurrent?: number | null, }; /** * 代理的用户名和密码。 @@ -4014,7 +4233,7 @@ export type SecurityView = { redact: GuardMode, inspect_tools: GuardMode, conten /** * 一个上游为什么服务不了这个模型。 */ -export type ServeSkip = "disabled" | "out_of_scope" | "not_offered" | "not_allowed"; +export type ServeSkip = "disabled" | "out_of_scope" | "not_offered" | "not_allowed" | "busy"; export type SessionDetail = { session: SessionView, turns: Array, }; @@ -4106,6 +4325,11 @@ export type SkippedView = { provider: string, */ reason: ServeSkip, }; +/** + * 上下文窗口、输出上限这样的模型规格从哪儿来。 + */ +export type SpecSource = "price_table" | "manual"; + /** * L3 测速要花多少。 * @@ -4278,7 +4502,7 @@ cost_micros_exact: number, cost_micros_estimated: number, unpriced_requests: number, /** * 有多少条请求**没有拿到用量**,所以同样算不出钱:上游没报,或者连接 - * 在它报之前就结束了(客户端取消、WebSocket 会话)。 + * 在它报之前就结束了(客户端取消、整条连接一行的 WebSocket 会话)。 * * 和 `unpriced_requests` 一样让金额合计偏低,但配价格解决不了它 —— * 界面上是两句不同的话。上游确实接下了的才算:成功的响应和客户端 @@ -4648,6 +4872,7 @@ export const ENDPOINTS = { UpdateProvider: { method: "PUT", path: "/providers/{name}", params: ["name"], format: "json" }, DeleteProvider: { method: "DELETE", path: "/providers/{name}", params: ["name"], format: "json" }, ProviderModels: { method: "GET", path: "/providers/{name}/models", params: ["name"], format: "json" }, + SetModelSpec: { method: "PUT", path: "/provider-model-spec", params: [], format: "json" }, RefreshProviderModels: { method: "POST", path: "/providers/{name}/models/refresh", params: ["name"], format: "json" }, RefreshStaleModels: { method: "POST", path: "/models/refresh", params: [], format: "json" }, CreateProxy: { method: "POST", path: "/proxies", params: [], format: "json" }, @@ -4769,6 +4994,7 @@ export type Endpoints = { UpdateProvider: { req: ProviderSave; res: ConfigWritten }; DeleteProvider: { req: BaseVersion; res: ConfigWritten }; ProviderModels: { req: null; res: ProviderModelsView }; + SetModelSpec: { req: ModelSpecSave; res: ConfigWritten }; RefreshProviderModels: { req: null; res: ProviderModelsView }; RefreshStaleModels: { req: null; res: ModelsRefreshing }; CreateProxy: { req: ProxySave; res: ConfigWritten }; diff --git a/src/i18n/core.zh.cases.json b/src/i18n/core.zh.cases.json index 4cca53d3..0a7750e9 100644 --- a/src/i18n/core.zh.cases.json +++ b/src/i18n/core.zh.cases.json @@ -301,5 +301,68 @@ "text": "There is no plugin `wsl-paths`." }, "zh": "插件「wsl-paths」不存在。" + }, + { + "msg": { + "code": "gw.key_limit.cost_per_period", + "args": { + "key": "claude-code", + "max": "$5.00", + "per": "day", + "used": "$5.02", + "resets": "2026-10-06 00:00 +08:00", + "resets_at_ms": "1791216000000" + }, + "text": "Gateway key `claude-code` has reached its limit of $5.00 per day: $5.02 spent so far. It resets at 2026-10-06 00:00 +08:00." + }, + "zh": "网关密钥「claude-code」已达到每天 $5.00 的费用上限,已产生费用 $5.02,将于 2026-10-06 00:00 +08:00 重置。" + }, + { + "msg": { + "code": "gw.key_limit.requests_rolling", + "args": { + "key": "codex", + "max": "30", + "per": "minute", + "used": "30", + "retry": "12" + }, + "text": "Gateway key `codex` has reached its limit of 30 requests per minute: 30 in the last minute. Try again in 12 s." + }, + "zh": "网关密钥「codex」已达到每分钟 30 次请求的上限,最近一分钟内已有 30 次,请于 12 秒后重试。" + }, + { + "msg": { + "code": "config.key_limit_not_positive", + "args": { + "key": "codex", + "per": "week", + "measure": "tokens", + "value": "0" + }, + "text": "the tokens limit per week of gateway key `codex` is 0; it has to be more than 0" + }, + "zh": "网关密钥「codex」的一条用量上限(token 数,每周)为 0,须大于 0。" + }, + { + "msg": { + "code": "config.key_limit_two_measures", + "args": { + "key": "codex", + "measures": "tokens and cost" + }, + "text": "a limit of gateway key `codex` names tokens and cost together. Each entry takes one of requests, tokens or cost; write one entry for each" + }, + "zh": "网关密钥「codex」的一条用量上限同时写了 tokens 和 cost。每一条只能写 requests、tokens、cost 之一,请分成多条。" + }, + { + "msg": { + "code": "gw.busy_all", + "args": { + "upstreams": "`官方`, `中转`" + }, + "text": "Every upstream that can serve this request is at its concurrency limit (max_concurrent): `官方`, `中转`. None had a free slot in time; try again shortly." + }, + "zh": "可处理此请求的上游均已达到并发上限(max_concurrent):「官方」、「中转」。等待期间没有空出位置,请稍后重试。" } ] diff --git a/src/i18n/core.zh.json b/src/i18n/core.zh.json index f252d9ec..41888daf 100644 --- a/src/i18n/core.zh.json +++ b/src/i18n/core.zh.json @@ -193,6 +193,24 @@ "list_keys": "Listing the account's API keys", "create_key": "Creating an API key", "read_key": "Reading the API key" + }, + "limit_per": { + "minute": "分钟", + "hour": "小时", + "day": "天", + "week": "周", + "month": "月" + }, + "limit_measure": { + "requests": "请求数", + "tokens": "token 数", + "cost": "费用" + }, + "limit_measures": { + "requests and tokens": "requests 和 tokens", + "requests and cost": "requests 和 cost", + "tokens and cost": "tokens 和 cost", + "requests and tokens and cost": "requests、tokens 和 cost" } }, "contexts": [ @@ -304,6 +322,12 @@ "gw.auth.key_disabled": "网关密钥「{key}」已停用。在应用的密钥页启用它即可恢复。", "gw.auth.source_not_allowed": "{peer} 不在允许的来源地址中。", "gw.auth.source_not_allowed_hint": "{peer} 不在允许的来源地址中。请修改 listen.gateway.allow_from,或将 bind 改为 loopback。", + "gw.key_limit.requests_per_period": "网关密钥「{key}」已达到每{per:limit_per} {max} 次请求的上限,已使用 {used} 次,将于 {resets} 重置。", + "gw.key_limit.tokens_per_period": "网关密钥「{key}」已达到每{per:limit_per} {max} 个 token 的上限,已使用 {used} 个 token,将于 {resets} 重置。", + "gw.key_limit.cost_per_period": "网关密钥「{key}」已达到每{per:limit_per} {max} 的费用上限,已产生费用 {used},将于 {resets} 重置。", + "gw.key_limit.requests_rolling": "网关密钥「{key}」已达到每{per:limit_per} {max} 次请求的上限,最近一{per:limit_per}内已有 {used} 次,请于 {retry} 秒后重试。", + "gw.key_limit.tokens_rolling": "网关密钥「{key}」已达到每{per:limit_per} {max} 个 token 的上限,最近一{per:limit_per}内已使用 {used} 个 token,请于 {retry} 秒后重试。", + "gw.key_limit.cost_rolling": "网关密钥「{key}」已达到每{per:limit_per} {max} 的费用上限,最近一{per:limit_per}内已产生费用 {used},请于 {retry} 秒后重试。", "gw.config.no_upstreams": "尚未配置任何上游。请在 ThinkWatch Lite 中添加上游,或在 config.yaml 的 providers 中添加。", "gw.config.proxy_undefined": "上游「{upstream}」使用的代理「{proxy}」未在 proxies 中定义,内置选项只有 direct 和 system。", "gw.config.proxy_unusable": "上游「{upstream}」的代理「{proxy}」不可用:{detail}", @@ -388,6 +412,8 @@ "gw.route.rule_failed": "规则求值失败:{detail}", "gw.route.denied": "规则「{rule}」拒绝了此请求:{reason}", "gw.route.no_upstream_alive": "没有可用的上游。", + "gw.busy_upstream": "上游「{upstream}」已有 {limit} 个请求在进行,达到其并发上限(max_concurrent)。", + "gw.busy_all": "可处理此请求的上游均已达到并发上限(max_concurrent):{upstreams!names}。等待期间没有空出位置,请稍后重试。", "gw.route.upstream_missing": "配置中不存在「{upstream}」。", "gw.route.selected_upstream_missing": "规则「{rule}」选中的上游「{upstream}」在配置中不存在。", "gw.route.all_selected_disabled": "路由选中的上游均已停用:{detail}", @@ -406,6 +432,7 @@ "gw.upstream.status": "上游「{upstream}」返回 {status}。", "gw.upstream.status_message": "上游「{upstream}」返回 {status}:{message}", "gw.upstream.stream_opening_error": "上游「{upstream}」开始回答后、给出任何内容之前报错({kind}):{message}", + "gw.slow_start": "上游「{upstream}」在 {secs} 秒内没有返回内容,请求已转到下一个上游。", "gw.upstream.stream_error": "上游「{upstream}」在回答过程中报错:{message}", "gw.upstream.stream_exception": "上游「{upstream}」以 {kind} 结束了响应流:{message}", "gw.upstream.eventstream_broken": "上游「{upstream}」发来了损坏的 AWS eventstream 帧:{detail}", @@ -424,6 +451,7 @@ "gw.ws.connect_failed": "无法连接上游的 WebSocket:{detail}", "gw.ws.send_failed": "向上游发送数据失败:{detail}", "gw.ws.upstream_broke": "上游连接中断:{detail}", + "gw.ws.upstream_closed": "上游在回答完成前关闭了连接。", "gw.ws.proxy_unsupported": "上游「{upstream}」配置了代理({proxy}),WebSocket 连接暂不支持经代理转发,仅支持直连的上游。", "gw.toolcall.connection_cut": "回答中的 {tool} 调用命中规则「{?why:{rule:scan_rule}|{name}}」{?why:({rule:rule_why})},已切断连接。", "// ── control:控制面的 HTTP 错误 ──────────────────────────────────": "", @@ -555,6 +583,10 @@ "control.group.upstream_twice": "上游「{upstream}」重复。", "control.group.empty": "策略组至少需要一个上游。", "control.group.preferred_not_member": "优先使用的上游「{upstream}」不在该策略组中。", + "control.group.weight_not_member": "为「{upstream}」设置了权重,但它不是该策略组的成员。", + "control.group.weight_not_load_balance": "上游「{upstream}」的权重为 {weight},只有轮询策略组使用权重。", + "control.group.weight_out_of_range": "上游「{upstream}」的权重为 {weight},权重须为 1 到 100 之间的整数。", + "control.group.balance_not_load_balance": "balance_by「{balance_by}」只适用于轮询策略组。", "// ── control.plugin:装插件、改插件、批准文件、试运行。{plugin} 在 ID 上是 id,在确认上是插件的名字 ──": "", "control.plugin.not_found": "插件「{plugin}」不存在。", "control.plugin.bad_id": "「{plugin}」不是有效的插件 ID:只能使用小写字母、数字和连字符,1 到 {max} 个字符。", @@ -597,6 +629,14 @@ "config.bad_base_url": "上游「{upstream}」的接口地址既不是 http 也不是 https:{url}", "config.empty_key": "网关密钥「{key}」的值为空。", "config.zero_concurrency": "网关密钥「{key}」的 max_concurrent 为 0,使用它的请求会一直等待。不限制并发时请删除 max_concurrent。", + "config.provider_concurrency_range": "上游「{upstream}」的 max_concurrent 为 {value},须在 1 到 {max} 之间。不限制并发时请删除 max_concurrent。", + "config.key_limit_empty": "网关密钥「{key}」的一条用量上限没有写明计量。limits 中的每一条须写 requests、tokens、cost 之一。", + "config.key_limit_two_measures": "网关密钥「{key}」的一条用量上限同时写了 {measures:limit_measures}。每一条只能写 requests、tokens、cost 之一,请分成多条。", + "config.key_limit_not_positive": "网关密钥「{key}」的一条用量上限({measure:limit_measure},每{per:limit_per})为 {value},须大于 0。", + "config.key_limit_duplicate": "网关密钥「{key}」有两条相同的用量上限({measure:limit_measure},每{per:limit_per}),请只保留一条。", + "config.key_limit_cache_reads": "网关密钥「{key}」的一条用量上限({measure:limit_measure})设置了 cache_reads,只有 token 数上限可以设置该项。", + "config.key_limit_retention": "网关密钥「{key}」设置了每{per:limit_per}的用量上限,而 retention.row_days 为 {days}。重启后当{per:limit_per}用量要从请求记录中重新累计,因此 row_days 须至少为 {min}。", + "config.key_limit_cost_too_small": "网关密钥「{key}」每{per:limit_per}的费用上限为 {value},须至少为 0.01。", "config.name_collision": "「{name}」同时是上游和策略组的名称,规则的 to 无法区分指的是哪一个。请重命名其中一个。", "config.bad_allow_from": "listen.gateway.allow_from 中的 {entry} 不是有效的 IP 地址或 CIDR,应写成 192.168.0.0/16 的形式。", "config.unknown_price_sheet": "上游「{upstream}」使用的价目表「{sheet}」不存在。", @@ -610,6 +650,10 @@ "config.alias_blank_model": "别名「{alias}」列出的模型中有空白项。", "config.alias_chained": "别名「{alias}」列出的 {model} 本身也是别名。别名列出的应是上游使用的模型名称,不能是其他别名。", "config.alias_only_itself": "别名「{alias}」只列出了它自己,不起任何作用。请列出各上游使用的名称,或删除该别名。", + "config.model_spec_blank_model": "上游「{upstream}」的模型规格(model_specs)中有一项的模型 ID 为空。", + "config.model_spec_wildcard": "上游「{upstream}」的模型规格「{model}」含有 * 或 ?。模型规格只能对应一个确切的模型 ID。", + "config.model_spec_empty": "上游「{upstream}」的模型规格「{model}」既未设置 context_window,也未设置 max_output_tokens。请至少设置一项,或删除该规格。", + "config.model_spec_zero": "上游「{upstream}」的模型规格「{model}」中 {field} 为 0,须为大于 0 的 token 数。要使用价目表中的值,请删除该项。", "config.reserved_name": "{what:kind}名称「{name}」以 __ 开头,该前缀保留给内置项,请使用其他名称。", "config.rule_name_empty": "有一条自定义{what:rule_line}规则没有名称。", "config.rule_name_taken": "自定义{what:rule_line}规则名称「{name}」重复。", @@ -619,6 +663,7 @@ "config.rule_label_bad": "自定义出站脱敏规则「{name}」的占位符名称为「{label}」,须以大写字母开头,由 1 到 24 个大写字母、数字或下划线组成。", "config.unknown_rule": "security.{guard} 中的「{rule}」不是内置规则。", "config.failover_range": "failover.{field} 为 {value},须在 {min} 到 {max} 之间。", + "config.slow_start_too_short": "已开启 failover.next_on_slow_start,而 failover.stream_start_wait_secs 为 {secs}。该值须至少为 {min},否则正常的回答会在开始之前被切断。", "config.plugin.bad_id": "插件 ID「{plugin}」写法有误:只能使用小写字母、数字和连字符,1 到 {max} 个字符。", "config.plugin.duplicate": "插件 ID「{plugin}」重复。", "config.plugin.file": "插件「{plugin}」的文件为 {file},应为 plugins/{plugin}.js。", @@ -692,6 +737,11 @@ "engine.unknown_target": "规则「{rule}」指向的「{target}」既不是上游也不是策略组。", "engine.empty_group": "策略组「{group}」中没有上游。", "engine.duplicate_group": "策略组名称「{group}」重复。规则按名称引用策略组,名称必须唯一。", + "engine.group_upstream_twice": "策略组「{group}」中上游「{upstream}」出现了不止一次。每个上游在策略组中只能出现一次。", + "engine.group_unknown_upstream": "策略组「{group}」列出的「{upstream}」不是上游。策略组的成员须为上游的名称。", + "engine.group_weight_not_load_balance": "策略组「{group}」为上游「{upstream}」设置了权重 {weight},而只有轮询(load-balance)策略组使用权重。请删除权重,或将该策略组改为轮询。", + "engine.group_weight_out_of_range": "策略组「{group}」为上游「{upstream}」设置的权重为 {weight}。权重须为 1 到 100 之间的整数。", + "engine.group_balance_not_load_balance": "策略组「{group}」设置了 balance_by: {balance_by},而只有轮询(load-balance)策略组使用该项。请删除该项,或将该策略组改为轮询。", "engine.duplicate_route": "路由名称「{route}」重复。网关密钥按名称绑定路由,名称必须唯一。", "engine.unknown_default_route": "default_route 指向的路由「{route}」不存在,未绑定路由的网关密钥将无法命中任何规则。", "engine.unknown_route": "网关密钥「{key}」绑定的路由「{route}」不存在。", diff --git a/src/i18n/terminology.md b/src/i18n/terminology.md index c69625e2..d16bff36 100644 --- a/src/i18n/terminology.md +++ b/src/i18n/terminology.md @@ -93,7 +93,8 @@ known colloquialisms. | 积分 | credits | the unit of a GLM Coding Plan billed in credits: 剩余 1,976 / 2,000 积分 = 1,976 / 2,000 credits left; not the ChatGPT reset credits | | 会话日志 | session log | the whole conversation DeepSeek Harness attaches to each request | | 按量计费 / 不计费 | Per token / Free | billing: the only two modes; subscription accounts are billed per token | -| 首字节 | time to first byte (TTFB) | column headers may use "TTFB" | +| 首字节 | time to first byte (TTFB) | column headers may use "TTFB"; the response headers arriving, not the answer | +| 首 token / 首 token 时间 | first token / time to first token | from sending to the first content of the answer; group ordering by speed (url-test, load-balance by speed) uses this | | 延迟 / 总耗时 / 生成用时 | latency / total time / generation time | | | 故障转移 | failover | | | 尝试链 | attempts | | diff --git a/src/keys/KeyDialog.i18n.ts b/src/keys/KeyDialog.i18n.ts index d2152e83..36db35bf 100644 --- a/src/keys/KeyDialog.i18n.ts +++ b/src/keys/KeyDialog.i18n.ts @@ -5,6 +5,8 @@ export const keyDialogText = messages( nameRequired: "请填写名称", nameTaken: "这个名称已被占用", patternsRequired: "请至少添加一条规则或选中一个模型", + limitRequired: "请填写用量上限的数值", + limitsInvalid: "请修正用量上限", editTitle: "编辑密钥", newTitle: "新建密钥", newDescription: "新密钥立即可用。客户端把它填进请求头即可连接网关。", @@ -28,6 +30,8 @@ export const keyDialogText = messages( nameRequired: "A name is required", nameTaken: "This name is already in use", patternsRequired: "At least one pattern or one model is required", + limitRequired: "Enter an amount for each usage limit", + limitsInvalid: "Fix the usage limits", editTitle: "Edit key", newTitle: "New key", newDescription: diff --git a/src/keys/KeyDialog.tsx b/src/keys/KeyDialog.tsx index 38f3c99c..275759e0 100644 --- a/src/keys/KeyDialog.tsx +++ b/src/keys/KeyDialog.tsx @@ -21,6 +21,8 @@ import { api } from "./api"; import type { KeyUse } from "./data"; import { keyDialogText } from "./KeyDialog.i18n"; import { errorText, routeLabel, takeoverOf } from "./labels"; +import { inputsOf, limitProblems, rowsOf, type LimitRow } from "./limits"; +import { LimitsEditor } from "./LimitsEditor"; import { TakeoverBadge } from "./KeysTable"; import { ModelScope } from "./ModelScope"; import { CopyButton, focusSelf, useDialogFocus } from "./parts"; @@ -42,6 +44,7 @@ export function KeyDialog({ routes, defaultRoute, catalog, + rowDays, version, onClose, onSaved, @@ -59,6 +62,8 @@ export function KeyDialog({ defaultRoute: string; /** 网关知道的全部模型,用来勾选可见范围。取不到时为空 */ catalog: KnownModel[]; + /** 请求记录留几天(概览里的)。每月的用量上限要求至少 31 天 */ + rowDays: number | null; /** 这一页最后知道的配置版本。**打开时读一次**,保存带的是那一个(见下面的 `base`) */ version: { get: () => string }; onClose: () => void; @@ -74,6 +79,7 @@ export function KeyDialog({ const [scope, setScope] = useState(scopeOf(editing?.allow)); const [entries, setEntries] = useState(editing?.allow ?? []); const [limit, setLimit] = useState(editing?.max_concurrent != null ? String(editing.max_concurrent) : ""); + const [limits, setLimits] = useState(() => rowsOf(editing?.limits)); const [enabled, setEnabled] = useState(!editing?.disabled); const [saving, setSaving] = useState(false); const [error, setError] = useState(null); @@ -86,14 +92,19 @@ export function KeyDialog({ const owner = editing ? takeoverOf(editing, clients, manual) : null; const taken = keys.some((k) => k.name === name.trim() && k.name !== editing?.name); + const problems = limitProblems(limits, rowDays); const missing = name.trim().length === 0 ? t.nameRequired : taken ? t.nameTaken - : scope === "some" && entries.length === 0 - ? t.patternsRequired - : null; + : problems.size > 0 + ? [...problems.values()].every((p) => p === "required") + ? t.limitRequired + : t.limitsInvalid + : scope === "some" && entries.length === 0 + ? t.patternsRequired + : null; async function save() { setSaving(true); @@ -105,6 +116,7 @@ export function KeyDialog({ allow: allowOf(scope, entries), max_concurrent: limit.trim() ? Number(limit.trim()) : null, disabled: !enabled, + limits: inputsOf(limits), }, base_version: base, }; @@ -214,6 +226,15 @@ export function KeyDialog({ /> + +
    diff --git a/src/keys/KeysPage.tsx b/src/keys/KeysPage.tsx index 59f4651c..935ad9db 100644 --- a/src/keys/KeysPage.tsx +++ b/src/keys/KeysPage.tsx @@ -14,6 +14,7 @@ import type { ClientView, KeyInput, Overview } from "@/types"; import { CostFigure } from "@/CostFigure"; import { useText } from "@/i18n"; import { useClients } from "@/clients/data"; +import { useCoreEvent } from "@/useCoreEvent"; import { writeQueue } from "@/lib/writeQueue"; import { api } from "./api"; import { CreatedDialog } from "./CreatedDialog"; @@ -23,9 +24,13 @@ import { KeyDialog } from "./KeyDialog"; import { KeysTable } from "./KeysTable"; import { keysPageText } from "./KeysPage.i18n"; import { takeoverOf } from "./labels"; +import { inputOfView, nextReset } from "./limits"; import { RowsSkeleton } from "./parts"; import { RotateDialog } from "./RotateDialog"; +/** setTimeout 能等的最长时间(2^31 − 1 毫秒):再长会当成 0,立刻就响 */ +const MAX_TIMER_MS = 2_147_483_647; + type DialogState = | null | { kind: "edit"; name: string | null } @@ -102,6 +107,19 @@ export default function KeysPage({ return () => clearTimeout(h); }, [highlight]); + /* + 用量上限按天、周、月重新算:到了那一刻重取一次,「已达上限」和对话框里的用量跟着 + 换(平时请求落地就会重取,这一次是给一直没有请求的时候)。定时器量的钟睡着时不走: + 睡醒、改了时钟(`clock_changed`)也重取。setTimeout 最多等 24.8 天,再远的到时候再排 + */ + const reset = nextReset(list); + useEffect(() => { + if (reset == null) return; + const h = setTimeout(() => void keys.reload(), Math.min(MAX_TIMER_MS, Math.max(1_000, reset - Date.now() + 1_000))); + return () => clearTimeout(h); + }, [reset]); + useCoreEvent(["clock_changed"], () => void keys.reload()); + /** 写完一次:记下新版本,重读列表,告诉外壳(概览跟着重读) */ function wrote(v: string) { version.set(v); @@ -252,6 +270,7 @@ export default function KeysPage({ routes={ov.routes} defaultRoute={ov.default_route} catalog={catalog.data ?? []} + rowDays={ov.retention.row_days} version={version} onClose={() => setDialog(null)} onSaved={(name, v) => { @@ -372,6 +391,8 @@ function inputOf(k: ClientView, patch: Partial): KeyInput { allow: k.allow ?? null, max_concurrent: k.max_concurrent, disabled: k.disabled ?? false, + // 上限是整份替换的:不带就是一条都不要了 + limits: k.limits.map(inputOfView), ...patch, }; } diff --git a/src/keys/KeysTable.i18n.ts b/src/keys/KeysTable.i18n.ts index 7af85b3c..152f8cfd 100644 --- a/src/keys/KeysTable.i18n.ts +++ b/src/keys/KeysTable.i18n.ts @@ -10,6 +10,8 @@ export const keysTableText = messages( actions: "操作", default: "默认", disabled: "已停用", + limitReached: "已达上限", + reachedUntil: (limit: string, resets: string) => `${limit},${resets}`, actionsFor: (name: string) => `${name} 的操作`, copyKey: "复制密钥", edit: "编辑…", @@ -31,6 +33,8 @@ export const keysTableText = messages( actions: "Actions", default: "Default", disabled: "Disabled", + limitReached: "Limit reached", + reachedUntil: (limit: string, resets: string) => `${limit}, ${resets}`, actionsFor: (name: string) => `Actions for ${name}`, copyKey: "Copy key", edit: "Edit…", diff --git a/src/keys/KeysTable.tsx b/src/keys/KeysTable.tsx index fa3a47b9..f5bc935b 100644 --- a/src/keys/KeysTable.tsx +++ b/src/keys/KeysTable.tsx @@ -13,6 +13,7 @@ import type { KeyUse } from "./data"; import { keysTableText } from "./KeysTable.i18n"; import { labelsText } from "./labels.i18n"; import { routeLabel, scopeLabel, takeoverOf, type KeyOwner } from "./labels"; +import { limitPhrase, resetText } from "./limits"; import { ClientMark, CopyIconButton, CostCell, OPENABLE_ROW, Tile, UsageCell, openable, stop } from "./parts"; export interface KeyActions { @@ -123,6 +124,7 @@ export function KeysTable({ {t.disabled} )} + {owner && }
    actions.copy(k.name, true)} /> @@ -153,6 +155,37 @@ export function KeysTable({ ); } +/** + * 用到上限的那几把:「已达上限」,悬停写是哪一条、什么时候重置。**没到就不出现** —— + * 状态只在异常时出现 + */ +function LimitReached({ limits }: { limits: ClientView["limits"] }) { + const t = useText(keysTableText); + const reached = limits.filter((l) => l.reached); + if (reached.length === 0) return null; + const now = Date.now(); + return ( + + {reached.map((l) => ( +

    + {l.resets_at_ms != null && l.resets_at_ms > now + ? t.reachedUntil(limitPhrase(l), resetText(l.resets_at_ms, now)) + : limitPhrase(l)} +

    + ))} + + } + > + + + {t.limitReached} + +
    + ); +} + /** * 为某个客户端生成的那几把,**单独一个标记**,写出是为谁生成的 —— 接管时生成的写 * 「接管 · Claude Code」,手动配置时生成的写「手动配置 · Cursor」。 diff --git a/src/keys/LimitsEditor.i18n.ts b/src/keys/LimitsEditor.i18n.ts new file mode 100644 index 00000000..a244fdb1 --- /dev/null +++ b/src/keys/LimitsEditor.i18n.ts @@ -0,0 +1,54 @@ +import { messages } from "@/i18n"; + +export const limitsEditorText = messages( + { + title: "用量上限", + hint: "任一上限用满后,此密钥的请求会被拒绝", + none: "未设上限,用量不限", + add: "添加上限", + every: "每", + atMost: "最多", + perLabel: (n: number) => `第 ${n} 条上限的周期`, + maxLabel: (n: number) => `第 ${n} 条上限的数值`, + measureLabel: (n: number) => `第 ${n} 条上限的计量`, + cacheReads: "计入缓存读取", + remove: "删除", + removeLabel: (n: number) => `删除第 ${n} 条上限`, + required: "请填写上限", + notPositive: (cost: boolean): string => (cost ? "须为大于 0 的金额" : "须为大于 0 的整数"), + costTooSmall: "金额须至少为 $0.01", + duplicate: "与前面的一条上限重复", + /** `per` 是周期的字(天、周、月),`need` 是它要求的天数 */ + retention: (per: string, need: number, days: number) => + `每${per}的上限要求请求记录至少保留 ${need} 天,当前为 ${days} 天`, + unpriced: (n: number) => `${n} 个可用模型没有价格,其费用按 0 计入上限:`, + more: (n: number) => `另有 ${n} 个`, + fewer: "收起", + }, + { + title: "Usage limits", + hint: "Once any limit is used up, requests with this key are rejected", + none: "No limits set: usage is unlimited", + add: "Add limit", + every: "Per", + atMost: "at most", + perLabel: (n: number) => `Period of limit ${n}`, + maxLabel: (n: number) => `Amount of limit ${n}`, + measureLabel: (n: number) => `Measure of limit ${n}`, + cacheReads: "Count cache reads", + remove: "Remove", + removeLabel: (n: number) => `Remove limit ${n}`, + required: "Enter a limit", + notPositive: (cost: boolean): string => (cost ? "Must be an amount above 0" : "Must be a whole number above 0"), + costTooSmall: "Must be at least $0.01", + duplicate: "Same as a limit above", + retention: (per: string, need: number, days: number) => + `A limit per ${per} needs request records kept for at least ${need === 1 ? "1 day" : `${need} days`}; they are kept for ${days === 1 ? "1 day" : `${days} days`}`, + unpriced: (n: number) => + n === 1 + ? "1 model this key can use has no price and counts as $0 toward the limit:" + : `${n} models this key can use have no price and count as $0 toward the limit:`, + more: (n: number) => `${n} more`, + fewer: "Show less", + }, +); diff --git a/src/keys/LimitsEditor.tsx b/src/keys/LimitsEditor.tsx new file mode 100644 index 00000000..f266ad2c --- /dev/null +++ b/src/keys/LimitsEditor.tsx @@ -0,0 +1,289 @@ +import { useState } from "react"; +import { PlusIcon } from "lucide-react"; +import { Button } from "@/ui/button"; +import { Checkbox } from "@/ui/checkbox"; +import { Input } from "@/ui/input"; +import { rowMotion, usePresentList } from "@/ui/motion"; +import { NativeSelect, NativeSelectOption } from "@/ui/native-select"; +import { StatusLabel } from "@/ui/status-dot"; +import { cn } from "@/lib/utils"; +import { useText } from "@/i18n"; +import { useNow } from "@/useNow"; +import { Boxed, Note } from "@/upstreams/parts"; +import type { KeyLimitView } from "@/types"; +import { + MEASURES, + PERS, + ROW_DAYS_NEEDED, + amount, + cleanMax, + newRow, + parseMax, + resetText, + usageOf, + type LimitProblem, + type LimitRow, +} from "./limits"; +import { limitsText } from "./limits.i18n"; +import { limitsEditorText } from "./LimitsEditor.i18n"; + +/** 没有价格的模型多于这么多个时先收起,只列前面几个 */ +const UNPRICED_SHOWN = 6; + +/** + * 对话框里的「用量上限」:一条一行,读起来是一句话 ——「每 [天] 最多 [5] [费用 (USD)]」, + * token 上限再加一个「计入缓存读取」。 + * + * 编辑一把已有的密钥时,每一行下面写着此刻用了多少(core 的 `KeyLimitView`):天、周、月 + * 是这一期的,带重置的时刻;分钟、小时是最近这一段的。到了的那一行标出来。 + * + * 填错的当场说,写在那一行下面。**空着的那一格等离开它才说**:刚加的一行一出现就标红, + * 说的是一件用户正要去做的事。删一条写成字,× 在这个应用里只表示关闭。 + */ +export function LimitsEditor({ + rows, + views, + unpriced, + problems, + rowDays, + onChange, +}: { + rows: LimitRow[]; + /** core 给的这把密钥的上限和用量。新建时是空的 */ + views: readonly KeyLimitView[]; + /** 这把密钥用得到、却没有价格的模型。core 只在已经存了费用上限时才算 */ + unpriced: readonly string[]; + problems: ReadonlyMap; + /** 此刻请求记录留几天,「每月」那一条的提示用 */ + rowDays: number | null; + onChange: (rows: LimitRow[]) => void; +}) { + const t = useText(limitsEditorText); + // 「今天」「00:00 重置」随时间走:开着对话框过了零点,这一行要跟着换 + const now = useNow(60_000); + const [focus, setFocus] = useState(null); + /** 离开过数值那一格的行:空着的从这时起才说「请填写」 */ + const [left, setLeft] = useState>(() => new Set()); + const shown = usePresentList(rows, (r) => r.id); + + function update(id: number, patch: Partial) { + onChange(rows.map((r) => (r.id === id ? { ...r, ...patch } : r))); + } + + function add() { + const r = newRow(rows); + setFocus(r.id); + onChange([...rows, r]); + } + + const costLimited = rows.some((r) => r.measure === "cost"); + + return ( +
    +
    + {t.title} + {t.hint} +
    + {rows.length > 0 && ( + + {shown.map(({ item: r, key, presence }) => { + const n = rows.indexOf(r) + 1; + const problem = problems.get(r.id); + return ( + update(r.id, patch)} + onLeave={() => setLeft((s) => (s.has(r.id) ? s : new Set(s).add(r.id)))} + onRemove={() => onChange(rows.filter((x) => x.id !== r.id))} + /> + ); + })} + + )} +
    + + {rows.length === 0 && {t.none}} +
    + {costLimited && unpriced.length > 0 && } +
    + ); +} + +/** 一条上限。下面一行是用量,填错时换成错在哪 */ +function LimitLine({ + row: r, + n, + className, + autoFocus, + problem, + use, + rowDays, + now, + onChange, + onLeave, + onRemove, +}: { + row: LimitRow; + /** 第几条,读屏用 */ + n: number; + className?: string; + autoFocus: boolean; + problem: LimitProblem | undefined; + use: KeyLimitView | undefined; + rowDays: number | null; + now: number; + onChange: (patch: Partial) => void; + /** 离开了数值那一格 */ + onLeave: () => void; + onRemove: () => void; +}) { + const t = useText(limitsEditorText); + const w = useText(limitsText); + const said = + problem === "required" + ? t.required + : problem === "notPositive" + ? t.notPositive(r.measure === "cost") + : problem === "costTooSmall" + ? t.costTooSmall + : problem === "duplicate" + ? t.duplicate + : problem === "retention" + ? t.retention(w.per[r.per], ROW_DAYS_NEEDED[r.per] ?? 0, rowDays ?? 0) + : null; + return ( +
    +
    + {t.every} + onChange({ per: e.target.value as LimitRow["per"] })} + > + {PERS.map((p) => ( + + {w.per[p]} + + ))} + + {t.atMost} + onChange({ max: cleanMax(r.measure, e.target.value) })} + onBlur={onLeave} + onKeyDown={(e) => { + // 对话框会把回车当成提交 + if (e.key === "Enter") e.preventDefault(); + }} + /> + onChange({ measure: e.target.value as LimitRow["measure"], max: "", cacheReads: false })} + > + {MEASURES.map((m) => ( + + {w.measure[m]} + + ))} + + {r.measure === "tokens" && ( + + )} + +
    + {said ? ( +

    {said}

    + ) : ( + use && + )} +
    + ); +} + +/** + * 「今天 $1.23 / $5.00 · 00:00 重置」「最近一分钟 12 / 30」。 + * + * 用量是 core 数的;**上限按输入框里的**:改大改小的时候,这一行说的就是改完之后的样子。 + * 到了的写出来,数字换成琥珀色 + */ +function Usage({ row, use, now }: { row: LimitRow; use: KeyLimitView; now: number }) { + const w = useText(limitsText); + const max = parseMax(row.measure, row.max) ?? use.max; + const reached = use.used >= max; + const resets = use.resets_at_ms != null && use.resets_at_ms > now ? resetText(use.resets_at_ms, now) : null; + return ( +
    + + {w.period[row.per]}{" "} + + {amount(row.measure, use.used)} / {amount(row.measure, max)} + + {resets && ` · ${resets}`} + + {reached && ( + + {w.reached} + + )} +
    + ); +} + +/** + * 没有价格的模型:费用记 0,费用上限管不住它们。**多了先收起**,只列前几个 + */ +function Unpriced({ models }: { models: readonly string[] }) { + const t = useText(limitsEditorText); + const [all, setAll] = useState(false); + const long = models.length > UNPRICED_SHOWN; + const listed = all || !long ? models : models.slice(0, UNPRICED_SHOWN); + return ( + + {t.unpriced(models.length)} {listed.join(", ")} + {long && ( + <> + {" "} + + + )} + + ); +} diff --git a/src/keys/data.ts b/src/keys/data.ts index f065c3bb..f902a674 100644 --- a/src/keys/data.ts +++ b/src/keys/data.ts @@ -39,14 +39,16 @@ export function useKeyUsage(): Resource & { byKey: Map /** * 全部网关密钥。配置换了一版就重取(密钥页拿着概览里的版本号,直接按它;别的页 - * 听 `config_reloaded`);请求落地时也重取,「最近使用」跟着它走。 + * 听 `config_reloaded`);请求落地时也重取,「最近使用」和用量上限跟着它走。某条上限 + * 到了八成、到了顶(`key_limit_alert`)也重取:那一刻请求可能还在跑,等它落地「已达上限」 + * 就晚了。 */ export function useKeys(configVersion?: string): Resource { return useResource("keys", api.listKeys, { events: configVersion === undefined - ? ["config_reloaded", "request_finished", "request_failed", "request_cancelled"] - : ["request_finished", "request_failed", "request_cancelled"], + ? ["config_reloaded", "request_finished", "request_failed", "request_cancelled", "key_limit_alert"] + : ["request_finished", "request_failed", "request_cancelled", "key_limit_alert"], deps: configVersion === undefined ? undefined : [configVersion], }); } diff --git a/src/keys/limits.i18n.ts b/src/keys/limits.i18n.ts new file mode 100644 index 00000000..6f0ba177 --- /dev/null +++ b/src/keys/limits.i18n.ts @@ -0,0 +1,55 @@ +import { messages } from "@/i18n"; +import type { LimitMeasure, LimitPer } from "@/types"; + +/** + * 用量上限的几样说法:对话框里的一行、密钥表里「已达上限」的悬停说明共用。 + */ +export const limitsText = messages( + { + per: { minute: "分钟", hour: "小时", day: "天", week: "周", month: "月" } satisfies Record, + measure: { requests: "次请求", tokens: "token", cost: "费用 (USD)" } satisfies Record, + /** 用量那一行开头:天、周、月是这一期,分钟、小时是最近这一段 */ + period: { + minute: "最近一分钟", + hour: "最近一小时", + day: "今天", + week: "本周", + month: "本月", + } satisfies Record, + resets: (at: string) => `${at} 重置`, + /** 重置的时刻:一天之内只写钟点,再远带上日期(日期按 `locale` 写) */ + at: (hm: string) => hm, + on: (date: string, hm: string) => `${date} ${hm}`, + locale: "zh-CN", + reached: "已达上限", + /** 一条上限说成一句:「每天 $5.00 费用」「每分钟 30 次请求」 */ + phrase: (per: string, amount: string, measure: LimitMeasure, cacheReads: boolean) => + measure === "requests" + ? `每${per} ${amount} 次请求` + : measure === "tokens" + ? `每${per} ${amount} token${cacheReads ? "(含缓存读取)" : ""}` + : `每${per} ${amount} 费用`, + }, + { + per: { minute: "minute", hour: "hour", day: "day", week: "week", month: "month" }, + measure: { requests: "requests", tokens: "tokens", cost: "USD" }, + period: { + minute: "Last minute", + hour: "Last hour", + day: "Today", + week: "This week", + month: "This month", + }, + resets: (at: string) => `resets ${at}`, + at: (hm: string) => `at ${hm}`, + on: (date: string, hm: string) => `${date} at ${hm}`, + locale: "en-US", + reached: "Limit reached", + phrase: (per: string, amount: string, measure: LimitMeasure, cacheReads: boolean) => + measure === "requests" + ? `${amount} requests per ${per}` + : measure === "tokens" + ? `${amount} tokens per ${per}${cacheReads ? " (cache reads included)" : ""}` + : `${amount} per ${per}`, + }, +); diff --git a/src/keys/limits.test.ts b/src/keys/limits.test.ts new file mode 100644 index 00000000..51c25fcd --- /dev/null +++ b/src/keys/limits.test.ts @@ -0,0 +1,171 @@ +import { afterEach, describe, expect, it } from "vitest"; +import { setLang } from "@/i18n"; +import type { KeyLimitView } from "@/types"; +import { + cleanMax, + inputOfView, + inputsOf, + limitPhrase, + limitProblems, + limitRow, + maxText, + newRow, + nextReset, + parseMax, + resetText, + rowsOf, + usageOf, + type LimitRow, +} from "./limits"; + +const view = (x: Partial & Pick): KeyLimitView => ({ + cache_reads: false, + used: 0, + resets_at_ms: null, + reached: false, + ...x, +}); + +const row = (x: Partial>): LimitRow => + limitRow({ per: "day", measure: "cost", max: "5", cacheReads: false, ...x }); + +afterEach(() => setLang("zh")); + +describe("用量上限的一行", () => { + it("费用在 core 那边是微分,输入框里是美元,来回不走样", () => { + expect(maxText("cost", 5_000_000)).toBe("5.00"); + expect(maxText("cost", 5_500_000)).toBe("5.50"); + expect(maxText("cost", 125_000)).toBe("0.125"); + expect(maxText("cost", 50_000_000)).toBe("50.00"); + expect(maxText("cost", 1)).toBe("0.000001"); + expect(maxText("tokens", 1_000_000)).toBe("1000000"); + expect(parseMax("cost", "5.5")).toBe(5_500_000); + expect(parseMax("cost", "0.125")).toBe(125_000); + expect(parseMax("cost", ".5")).toBe(500_000); + expect(parseMax("cost", "5.")).toBe(5_000_000); + expect(parseMax("requests", "30")).toBe(30); + }); + + it("不是正数的上限不收", () => { + for (const bad of ["", " ", "0", "0.0", ".", "0.0000001"]) expect(parseMax("cost", bad), bad).toBeNull(); + for (const bad of ["", "0", "1.5", "-3"]) expect(parseMax("requests", bad), bad).toBeNull(); + }); + + it("输入框只留得下数字,费用多一个小数点、到微分为止", () => { + expect(cleanMax("requests", "1,000 次")).toBe("1000"); + expect(cleanMax("cost", "$5.5.0")).toBe("5.50"); + expect(cleanMax("cost", "0.12345678")).toBe("0.123456"); + }); + + it("打开时照 core 给的,保存时原样交回去", () => { + const views = [ + view({ per: "minute", measure: "requests", max: 30 }), + view({ per: "day", measure: "cost", max: 5_000_000 }), + view({ per: "week", measure: "tokens", max: 900_000, cache_reads: true }), + ]; + const rows = rowsOf(views); + expect(rows.map((r) => r.max)).toEqual(["30", "5.00", "900000"]); + expect(inputsOf(rows)).toEqual(views.map(inputOfView)); + }); + + it("缓存读取只跟着 token 上限走", () => { + expect(inputsOf([row({ measure: "requests", max: "3", cacheReads: true })])).toEqual([ + { per: "day", measure: "requests", max: 3, cache_reads: false }, + ]); + }); +}); + +describe("当场校验", () => { + it("和 core 一样:要填、要大于 0、同一种不能有两条", () => { + const rows = [ + row({ max: "" }), + row({ per: "hour", max: "0" }), + row({ per: "week", max: "3" }), + row({ per: "week", max: "8" }), + ]; + const p = limitProblems(rows, 90); + expect(rows.map((r) => p.get(r.id) ?? null)).toEqual(["required", "notPositive", null, "duplicate"]); + }); + + it("算不算缓存读取不一样,就是两条", () => { + const rows = [ + row({ measure: "tokens", max: "100" }), + row({ measure: "tokens", max: "900", cacheReads: true }), + row({ measure: "tokens", max: "900", cacheReads: true }), + ]; + const p = limitProblems(rows, 90); + expect(rows.map((r) => p.get(r.id) ?? null)).toEqual([null, null, "duplicate"]); + }); + + it("天、周、月的上限要求记录留够 1、7、31 天;分钟、小时不要;不知道留几天就不拦", () => { + for (const [per, need] of [["day", 1], ["week", 7], ["month", 31]] as const) { + const rows = [row({ per })]; + expect(limitProblems(rows, need - 1).get(rows[0]!.id)).toBe("retention"); + expect(limitProblems(rows, need).size).toBe(0); + expect(limitProblems(rows, null).size).toBe(0); + } + expect(limitProblems([row({ per: "minute" }), row({ per: "hour" })], 0).size).toBe(0); + }); + + it("费用上限至少 $0.01,和 core 一样排在重复之前", () => { + const rows = [row({ max: "0.009999" }), row({ per: "week", max: "0.01" }), row({ per: "week", max: "0.001" })]; + const p = limitProblems(rows, 90); + expect(rows.map((r) => p.get(r.id) ?? null)).toEqual(["costTooSmall", null, "costTooSmall"]); + expect(limitProblems([row({ measure: "requests", max: "1" })], 90).size).toBe(0); + }); + + it("加一行先给还没有的那一种,不一加上就重复", () => { + const first = newRow([]); + expect([first.per, first.measure]).toEqual(["day", "cost"]); + const second = newRow([first]); + expect([second.per, second.measure]).toEqual(["month", "cost"]); + }); +}); + +describe("用量", () => { + it("按周期、量和缓存读取认 core 给的那一条", () => { + const views = [ + view({ per: "day", measure: "tokens", max: 100, used: 7 }), + view({ per: "day", measure: "tokens", max: 900, used: 70, cache_reads: true }), + ]; + expect(usageOf(row({ measure: "tokens", cacheReads: true }), views)?.used).toBe(70); + expect(usageOf(row({ measure: "tokens" }), views)?.used).toBe(7); + // 新加的、改成了别的周期的,core 还没数过 + expect(usageOf(row({ per: "week", measure: "tokens" }), views)).toBeUndefined(); + }); + + it("一条上限说成一句", () => { + expect(limitPhrase(view({ per: "day", measure: "cost", max: 5_000_000 }))).toBe("每天 $5.00 费用"); + expect(limitPhrase(view({ per: "minute", measure: "requests", max: 30 }))).toBe("每分钟 30 次请求"); + expect(limitPhrase(view({ per: "week", measure: "tokens", max: 1_000_000, cache_reads: true }))).toBe( + "每周 1,000,000 token(含缓存读取)", + ); + setLang("en"); + expect(limitPhrase(view({ per: "day", measure: "cost", max: 5_000_000 }))).toBe("$5.00 per day"); + expect(limitPhrase(view({ per: "week", measure: "tokens", max: 1_000_000, cache_reads: true }))).toBe( + "1,000,000 tokens per week (cache reads included)", + ); + }); + + it("一天之内的重置只写钟点,再远带上日期", () => { + const now = new Date(2026, 9, 5, 17, 30).getTime(); + const midnight = new Date(2026, 9, 6, 0, 0).getTime(); + const monday = new Date(2026, 9, 12, 0, 0).getTime(); + expect(resetText(midnight, now)).toBe("00:00 重置"); + expect(resetText(monday, now)).toBe("10月12日 00:00 重置"); + setLang("en"); + expect(resetText(midnight, now)).toBe("resets at 00:00"); + expect(resetText(monday, now)).toBe("resets Oct 12 at 00:00"); + }); + + it("最早要重新算的那一刻:只有天、周、月有", () => { + expect(nextReset(undefined)).toBeNull(); + expect( + nextReset([ + { limits: [view({ per: "minute", measure: "requests", max: 3 })] }, + { limits: [view({ per: "month", measure: "cost", max: 9, resets_at_ms: 300 })] }, + { limits: [view({ per: "day", measure: "cost", max: 9, resets_at_ms: 200 })] }, + ]), + ).toBe(200); + }); +}); diff --git a/src/keys/limits.ts b/src/keys/limits.ts new file mode 100644 index 00000000..2eca35d4 --- /dev/null +++ b/src/keys/limits.ts @@ -0,0 +1,199 @@ +/** + * 一把密钥的用量上限:对话框里一行一条,和 core 的 `KeyLimitView` / `KeyLimitInput` 来回换。 + * + * **写得对不对由 core 的配置校验说**(`config.key_limit_*`)。这里当场查的是同一套规则里 + * 填的时候就看得出来的几条,顺序也和 core 一样:要填、要大于 0、费用至少 $0.01、同一个周期 + * 同一种量(token 再分算不算缓存读取)只能有一条、天 / 周 / 月的上限要求请求记录至少留 + * 1 / 7 / 31 天 —— 免得填完整张对话框,保存时才被拒。 + * + * 费用在 core 那边是微分(`max`、`used`),输入框里是美元。 + */ +import { compact } from "@/format"; +import { textOf } from "@/i18n"; +import { usd, type KeyLimitInput, type KeyLimitView, type LimitMeasure, type LimitPer } from "@/types"; +import { limitsText } from "./limits.i18n"; + +export const PERS: readonly LimitPer[] = ["minute", "hour", "day", "week", "month"]; +export const MEASURES: readonly LimitMeasure[] = ["requests", "tokens", "cost"]; + +/** + * 天、周、月的上限要求请求记录至少留几天:重启之后这一期的用量从记录里加回来,留得比一期 + * 短就加不全(core 的 `row_days_needed`)。分钟、小时的从空的开始,不要记录 + */ +export const ROW_DAYS_NEEDED: Readonly>> = { day: 1, week: 7, month: 31 }; + +/** 费用上限最少多少,微分($0.01,core 的 `COST_MIN`) */ +export const COST_MIN_MICROS = 10_000; + +/** 对话框里的一行 */ +export interface LimitRow { + /** 只在对话框里用:行的 key,当场校验按它认行 */ + id: number; + per: LimitPer; + measure: LimitMeasure; + /** 输入框里的字。费用是美元(`5`、`0.5`),别的是整数 */ + max: string; + /** 计入缓存读取。只有 token 上限有,别的量上一直是 false */ + cacheReads: boolean; +} + +let seq = 0; + +export function limitRow(r: Omit): LimitRow { + seq += 1; + return { id: seq, ...r }; +} + +/** 打开对话框时的那几行:照 core 给的,按配置里的顺序 */ +export function rowsOf(views: readonly KeyLimitView[] | undefined): LimitRow[] { + return (views ?? []).map((v) => + limitRow({ per: v.per, measure: v.measure, max: maxText(v.measure, v.max), cacheReads: v.cache_reads }), + ); +} + +/** 上限写进输入框的样子。费用是微分,写成美元、到分,再往下的照实写:5_500_000 → `5.50`,125_000 → `0.125` */ +export function maxText(measure: LimitMeasure, max: number): string { + if (measure !== "cost") return String(max); + return (max / 1e6).toFixed(6).replace(/(\.\d\d\d*?)0+$/, "$1"); +} + +/** 输入框里只留得下数字(费用再加一个小数点、最多到微分那一位) */ +export function cleanMax(measure: LimitMeasure, raw: string): string { + if (measure !== "cost") return raw.replace(/[^0-9]/g, ""); + const s = raw.replace(/[^0-9.]/g, ""); + const dot = s.indexOf("."); + if (dot < 0) return s; + return s.slice(0, dot + 1) + s.slice(dot + 1).replace(/\./g, "").slice(0, 6); +} + +/** 输入框里的字 → 上限(费用是微分)。空的、不是正数的是 null */ +export function parseMax(measure: LimitMeasure, text: string): number | null { + const s = text.trim(); + if (measure === "cost") { + if (!/^(\d+\.?\d*|\.\d+)$/.test(s)) return null; + const micros = Math.round(Number(s) * 1e6); + return micros > 0 && Number.isSafeInteger(micros) ? micros : null; + } + if (!/^\d+$/.test(s)) return null; + const n = Number(s); + return n > 0 && Number.isSafeInteger(n) ? n : null; +} + +/** 算不算缓存读取,只对 token 上限有意义 */ +function cacheReadsOf(measure: LimitMeasure, cacheReads: boolean): boolean { + return measure === "tokens" && cacheReads; +} + +/** core 认作同一条的:同一个周期、同一种量、缓存读取算法相同 */ +function identity(per: LimitPer, measure: LimitMeasure, cacheReads: boolean): string { + return `${per}:${measure}:${cacheReadsOf(measure, cacheReads)}`; +} + +/** + * 加一行时先给什么:**还没有的那一种**,免得一加上就和已有的重复。先按天算费用 —— + * 管住一把密钥最常见的就是每天花多少;都有了就还是它,由当场校验说重复 + */ +export function newRow(rows: readonly LimitRow[]): LimitRow { + const taken = new Set(rows.map((r) => identity(r.per, r.measure, r.cacheReads))); + const order: LimitPer[] = ["day", "month", "week", "hour", "minute"]; + for (const measure of ["cost", "requests", "tokens"] as const) { + for (const per of order) { + if (!taken.has(identity(per, measure, false))) return limitRow({ per, measure, max: "", cacheReads: false }); + } + } + return limitRow({ per: "day", measure: "cost", max: "", cacheReads: false }); +} + +export type LimitProblem = "required" | "notPositive" | "costTooSmall" | "duplicate" | "retention"; + +/** + * 每一行有什么不对,按行的 id。**一行只说一件**,和 core 查的先后一样:先说要填,再说要 + * 大于 0,再说费用不到 $0.01,再说重复(和前面哪一行一样,就标在后面那一行上),最后说 + * 天、周、月的上限要记录留够天数(`ROW_DAYS_NEEDED`)。 + * + * `rowDays`:此刻请求记录留几天(概览里的 `retention.row_days`)。不知道就不查这一条, + * 保存时由 core 说 + */ +export function limitProblems(rows: readonly LimitRow[], rowDays: number | null): Map { + const out = new Map(); + const seen = new Set(); + for (const r of rows) { + const id = identity(r.per, r.measure, r.cacheReads); + const max = parseMax(r.measure, r.max); + const need = ROW_DAYS_NEEDED[r.per]; + if (r.max.trim() === "") out.set(r.id, "required"); + else if (max == null) out.set(r.id, "notPositive"); + else if (r.measure === "cost" && max < COST_MIN_MICROS) out.set(r.id, "costTooSmall"); + else if (seen.has(id)) out.set(r.id, "duplicate"); + else if (need != null && rowDays != null && rowDays < need) out.set(r.id, "retention"); + seen.add(id); + } + return out; +} + +/** 保存时交给 core 的。**只在没有问题时调用**:填得不对的行在这里会被略过 */ +export function inputsOf(rows: readonly LimitRow[]): KeyLimitInput[] { + return rows.flatMap((r) => { + const max = parseMax(r.measure, r.max); + return max == null ? [] : [{ per: r.per, measure: r.measure, max, cache_reads: cacheReadsOf(r.measure, r.cacheReads) }]; + }); +} + +/** core 给的一条原样写回去(停用、启用这类只改别的字段的保存) */ +export function inputOfView(v: KeyLimitView): KeyLimitInput { + return { per: v.per, measure: v.measure, max: v.max, cache_reads: v.cache_reads }; +} + +/** + * 这一行此刻用了多少:core 给的那几条里,周期、量、缓存读取算法都和这一行一样的那条。 + * 新加的、改成了另一种的没有 —— 保存之前 core 没数过它 + */ +export function usageOf(row: LimitRow, views: readonly KeyLimitView[]): KeyLimitView | undefined { + const id = identity(row.per, row.measure, row.cacheReads); + return views.find((v) => identity(v.per, v.measure, v.cache_reads) === id); +} + +/** 一个用量或上限写成字:费用写美元,token 收成 k / M,请求数带千分位 */ +export function amount(measure: LimitMeasure, n: number): string { + if (measure === "cost") return usd(n); + if (measure === "tokens") return compact(n); + return n.toLocaleString("en-US"); +} + +/** 一条上限说成一句:「每天 $5.00 费用」「1,000,000 tokens per day」。数字写全 */ +export function limitPhrase(v: Pick): string { + const t = textOf(limitsText); + const n = v.measure === "cost" ? usd(v.max) : v.max.toLocaleString("en-US"); + return t.phrase(t.per[v.per], n, v.measure, v.cache_reads); +} + +const DAY_MS = 24 * 3_600_000; + +/** + * 什么时候重置:一天之内只写钟点(「00:00 重置」),再远带上日期(「10月12日 00:00 重置」)。 + * 按这台机器的时区写 —— core 在别的时区时,钟点照样是同一个时刻 + */ +export function resetText(atMs: number, nowMs: number): string { + const t = textOf(limitsText); + const d = new Date(atMs); + const hm = `${String(d.getHours()).padStart(2, "0")}:${String(d.getMinutes()).padStart(2, "0")}`; + if (atMs - nowMs <= DAY_MS) return t.resets(t.at(hm)); + const date = new Intl.DateTimeFormat(t.locale, { month: "short", day: "numeric" }).format(d); + return t.resets(t.on(date, hm)); +} + +/** 一把密钥有没有哪一条已经到了 */ +export function anyReached(views: readonly KeyLimitView[] | undefined): boolean { + return (views ?? []).some((v) => v.reached); +} + +/** 这几把密钥里最早要重新算的那一刻(天、周、月的上限才有)。没有就是 null */ +export function nextReset(keys: readonly { limits: readonly KeyLimitView[] }[] | undefined): number | null { + let soonest: number | null = null; + for (const k of keys ?? []) { + for (const l of k.limits) { + if (l.resets_at_ms != null && (soonest == null || l.resets_at_ms < soonest)) soonest = l.resets_at_ms; + } + } + return soonest; +} diff --git a/src/labels.i18n.ts b/src/labels.i18n.ts index a5a97fba..4411d03e 100644 --- a/src/labels.i18n.ts +++ b/src/labels.i18n.ts @@ -11,6 +11,13 @@ export const labelsText = messages( "url-test": "延迟最低", cheapest: "费用最低", }, + /** 轮询组按什么分请求(`balance_by`) */ + balanceBy: { + weights: "按比例", + latency: "按速度", + health: "按稳定性", + "latency-health": "按速度和稳定性", + }, allUpstreams: "全部上游", probes: { health_check: { @@ -66,12 +73,17 @@ export const labelsText = messages( noResponse: "未收到响应", estimated: "本地估算", estimatedAfter: (status: number) => `${status} · 本地估算`, + /** 开头等过了时限还没有内容,换了下一个上游 */ + slowStart: "开头超时", /** 选定上游之后被规则拒绝的那一跳:没有发给这个上游 */ deniedHop: (rule: string) => `未发送 · 被规则「${rule}」拒绝`, - // 没有发往任何上游的请求,在「上游」的位置上写的那一句 + // 没有上游接下的请求,在「上游」的位置上写的那一句:规则拒绝、没有可用的上游、 + // 密钥的用量上限拒绝、上游都满着 notSent: { denied: "规则拒绝", unavailable: "无可用上游", + limited: "用量上限", + busy: "并发已满", }, // ------------------------------------------------------------ 请求与费用 @@ -146,6 +158,12 @@ export const labelsText = messages( "url-test": "Lowest latency", cheapest: "Lowest cost", }, + balanceBy: { + weights: "By ratio", + latency: "By speed", + health: "By reliability", + "latency-health": "By speed and reliability", + }, allUpstreams: "All upstreams", probes: { health_check: { @@ -203,11 +221,14 @@ export const labelsText = messages( noResponse: "No response received", estimated: "Estimated locally", estimatedAfter: (status: number) => `${status} · Estimated locally`, + slowStart: "Start timed out", deniedHop: (rule: string) => `Not sent · denied by rule “${rule}”`, // 流量表「上游」那一列放得下的长度:再长就折成两行 notSent: { denied: "Denied", unavailable: "No upstream", + limited: "Usage limit", + busy: "At capacity", }, quote: { diff --git a/src/labels.test.ts b/src/labels.test.ts index fd8644e8..26f60d46 100644 --- a/src/labels.test.ts +++ b/src/labels.test.ts @@ -3,6 +3,7 @@ import { setLang } from "./i18n"; import { GROUP_KINDS, PROBES, + attemptText, conditionName, conditionText, mismatchText, @@ -65,3 +66,43 @@ describe("规则的条件与改写", () => { expect(setText({ field: "max_tokens", value: "4096" })).toBe("max_tokens set to 4096"); }); }); + +describe("尝试链里的一跳", () => { + const slow = { + provider: "anthropic", + outcome: "slow_start" as const, + error: { code: "gw.slow_start", args: { upstream: "anthropic", secs: "30" }, text: "" }, + ms: 30_004, + }; + const busy = { + provider: "anthropic", + outcome: "error" as const, + error: { code: "gw.busy_upstream", args: { upstream: "anthropic", limit: "2" }, text: "" }, + ms: 0, + skipped: "busy" as const, + }; + + it("开头超时、满着跳过:短名,悬停是 core 的原话", () => { + expect(attemptText(slow)).toEqual({ + text: "开头超时", + ok: false, + tip: "上游「anthropic」在 30 秒内没有返回内容,请求已转到下一个上游。", + }); + expect(attemptText(busy)).toEqual({ + text: "并发已满", + ok: false, + tip: "上游「anthropic」已有 2 个请求在进行,达到其并发上限(max_concurrent)。", + }); + setLang("en"); + expect(attemptText(slow).text).toBe("Start timed out"); + expect(attemptText(busy).text).toBe("At its concurrency limit"); + }); + + it("别的结果照旧:字就是那一句,没有悬停", () => { + expect(attemptText({ provider: "openrouter", outcome: "served", status: 200, ms: 900 })).toEqual({ + text: "成功 · 200", + ok: true, + tip: null, + }); + }); +}); diff --git a/src/labels.ts b/src/labels.ts index ad588271..701879f4 100644 --- a/src/labels.ts +++ b/src/labels.ts @@ -11,6 +11,7 @@ import { coreText } from "@/i18n/core.i18n"; import { usd, type AttemptView, + type BalanceBy, type ConditionView, type ConfigOrigin, type ConfigStage, @@ -21,7 +22,7 @@ import { type TakesEffect, type TranslatedView, } from "./types"; -import { PROTOCOLS } from "./upstreams/labels"; +import { PROTOCOLS, skipLabel } from "./upstreams/labels"; import { labelsText } from "./labels.i18n"; import type { NotSent } from "./requestRouting"; @@ -47,6 +48,13 @@ export function groupKindLabel(kind: GroupKind): string { return textOf(labelsText).groupKinds[kind]; } +/** 轮询组按什么分请求,按界面上的先后:先是只看比例,再是自动的几种 */ +export const BALANCE_BY: readonly BalanceBy[] = ["weights", "latency", "health", "latency-health"]; + +export function balanceByLabel(by: BalanceBy): string { + return textOf(labelsText).balanceBy[by]; +} + /** 内置策略组在配置里的名字。**界面上不出现它**,显示为「全部上游」 */ export const ALL_UPSTREAMS = "__all__"; @@ -148,22 +156,32 @@ export function setText(s: SetView): string { } } -/** 尝试链里的一跳。`ok` 决定颜色 */ -export function attemptText(a: AttemptView): { text: string; ok: boolean } { +/** + * 尝试链里的一跳。`ok` 决定颜色。`tip`:短名后面悬停说的那一句(core 说的原话); + * 字本身就是那一句的没有 + */ +export function attemptText(a: AttemptView): { text: string; ok: boolean; tip: string | null } { const t = textOf(labelsText); + const said = a.error ? coreText(a.error) : null; + // 没有发出去的一跳:这家满着,换了下一家。短名说原因,悬停是 core 那一句(几个请求在 + // 进行,也就是它的上限) + if (a.skipped) return { text: skipLabel(a.skipped), ok: false, tip: said }; switch (a.outcome) { case "served": if (a.status == null || a.status < 400) { - return { text: a.status == null ? t.served : t.servedStatus(a.status), ok: true }; + return { text: a.status == null ? t.served : t.servedStatus(a.status), ok: true, tip: null }; } - return { text: t.rejected(a.status), ok: false }; + return { text: t.rejected(a.status), ok: false, tip: null }; case "status": - return { text: a.status === 429 ? t.rateLimited : t.upstreamError(a.status ?? "—"), ok: false }; + return { text: a.status === 429 ? t.rateLimited : t.upstreamError(a.status ?? "—"), ok: false, tip: null }; // 数 token 由网关自己估:上游不是这种格式(没问过它),或者问过、它没实现这个接口 case "estimated": - return { text: a.status == null ? t.estimated : t.estimatedAfter(a.status), ok: true }; + return { text: a.status == null ? t.estimated : t.estimatedAfter(a.status), ok: true, tip: null }; + // 等过了开头的时限还没有内容,放弃了这一家、换了下一家。悬停说等了多久 + case "slow_start": + return { text: t.slowStart, ok: false, tip: said }; default: - return { text: a.error ? coreText(a.error) : t.noResponse, ok: false }; + return { text: said ?? t.noResponse, ok: false, tip: null }; } } @@ -172,7 +190,10 @@ export function deniedHopText(rule: string): string { return textOf(labelsText).deniedHop(rule); } -/** 没有发往任何上游的请求,在「上游」的位置上写什么:被规则拒绝,或者没有可用的上游 */ +/** + * 没有上游接下的请求,在「上游」的位置上写什么:被规则拒绝、没有可用的上游、密钥的用量 + * 上限拒绝了它,或者上游都满着(见 `NotSent`) + */ export function notSentText(kind: NotSent): string { return textOf(labelsText).notSent[kind]; } diff --git a/src/palette/items.tsx b/src/palette/items.tsx index f01f7499..b3a6db11 100644 --- a/src/palette/items.tsx +++ b/src/palette/items.tsx @@ -35,10 +35,11 @@ import { } from "@/ui/icons"; import { ClientLogo, UpstreamLogo } from "@/ui/logos"; import { StatusDot, type StatusTone } from "@/ui/status-dot"; -import { groupKindLabel, notSentText, probeLabel, targetLabel, ALL_UPSTREAMS } from "@/labels"; +import { notSentText, probeLabel, targetLabel, ALL_UPSTREAMS } from "@/labels"; import { when } from "@/format"; import { notSent } from "@/requestRouting"; import { upstreamText } from "@/requestTable"; +import { strategyText } from "@/routing/model"; import { NotSentIcon } from "@/traffic/cells"; import type { ConnView } from "@/connection/api"; import { connText } from "@/connection/connection.i18n"; @@ -348,7 +349,7 @@ export function buildItems(s: Sources): Item[] { id: `group:${g.name}`, group: "groups", title: targetLabel(g.name), - detail: `${groupKindLabel(g.kind)} · ${t.upstreamCount(g.providers.length)}`, + detail: `${strategyText(g)} · ${t.upstreamCount(g.providers.length)}`, // 显示的是译名(内置组),原名也能搜。里面的上游不算:打 `deep` 要的是 deepseek // 这个上游,不是每个含有它的组 keywords: [g.name], diff --git a/src/requestRouting.test.ts b/src/requestRouting.test.ts index cb91df88..7da7dd3e 100644 --- a/src/requestRouting.test.ts +++ b/src/requestRouting.test.ts @@ -15,6 +15,35 @@ const unavailable: Msg = { }; const served = (provider: string, ms = 900): AttemptView => ({ provider, outcome: "served", status: 200, ms }); const overloaded = (provider: string): AttemptView => ({ provider, outcome: "status", status: 529, ms: 1_870 }); +/** 网关密钥的用量上限拒绝时的那一句(`gw.key_limit.*`) */ +const limited: Msg = { + code: "gw.key_limit.cost_per_period", + args: { key: "cursor", max: "$5.00", per: "day", used: "$5.03", resets: "2026-09-26 00:00 +08:00" }, + text: "Gateway key `cursor` has reached its limit of $5.00 per day: $5.03 spent so far. It resets at 2026-09-26 00:00 +08:00.", +}; +/** 能服务的上游都满着、等过了也没空出来(`gw.busy_all`) */ +const busyAll: Msg = { + code: "gw.busy_all", + args: { upstreams: "`anthropic`, `openrouter`" }, + text: "Every upstream that can serve this request is at its concurrency limit (max_concurrent): `anthropic`, `openrouter`. None had a free slot in time; try again shortly.", +}; +/** 满着、没发出去的一跳(core 的 `hop_busy`)。`queued`:它是这段对话留着的那一家,等过空位 */ +const busy = (provider: string, queued?: number): AttemptView => ({ + provider, + outcome: "error", + error: { code: "gw.busy_upstream", args: { upstream: provider, limit: "2" }, text: "" }, + ms: 0, + skipped: "busy", + queued_ms: queued ?? null, +}); +/** 开头超时、放弃了的一跳 */ +const slow = (provider: string): AttemptView => ({ + provider, + outcome: "slow_start", + error: { code: "gw.slow_start", args: { upstream: provider, secs: "30" }, text: "" }, + ms: 30_004, + usage: { input: 48_210, cache_read: 0, cache_write: 0, estimated: true }, +}); function row(routing: Partial, over: Partial> = {}) { return { @@ -33,6 +62,16 @@ describe("没有发往任何上游的请求", () => { expect(notSent({ provider: "", error: { code: "gw.route.all_selected_disabled", args: {}, text: "" } })).toBe("unavailable"); }); + it("密钥的用量上限拒绝的:上游是空的,失败的那一句是 gw.key_limit.*", () => { + expect(notSent({ provider: "", error: limited })).toBe("limited"); + expect(notSent({ provider: "", error: { ...limited, code: "gw.key_limit.requests_rolling" } })).toBe("limited"); + }); + + it("上游都满着:那一行归在最后看过的那一家,也不算它的失败", () => { + expect(notSent({ provider: "openrouter", error: busyAll })).toBe("busy"); + expect(notSent({ provider: "", error: busyAll })).toBe("busy"); + }); + it("发往了上游的、本地应答的、还没有结局的都不算", () => { // 选定上游之后才被拒绝:记在要去的那个上游上 expect(notSent({ provider: "openrouter", error: denied("no-images", "no images") })).toBeNull(); @@ -79,6 +118,47 @@ describe("路由那一页", () => { expect(f.hops.map((h) => h.denied)).toEqual([false, false]); }); + it("开头超时、满着跳过的不说成失败:换过上游,说前几次尝试没有接下", () => { + const f = routingFacts(row({ attempts: [slow("anthropic"), served("openrouter")] }, { provider: "openrouter" }), false)!; + expect(f.note).toEqual({ kind: "switched", count: 1 }); + const g = routingFacts( + row({ attempts: [busy("anthropic"), overloaded("openrouter"), served("deepseek")] }, { provider: "deepseek" }), + false, + )!; + expect(g.note).toEqual({ kind: "switched", count: 2 }); + }); + + it("等到了空位:只有一跳,排队的时间在那一跳上,不加说明", () => { + const f = routingFacts(row({ attempts: [{ ...served("anthropic"), queued_ms: 1_240 }] }), false)!; + expect(f.note).toBeNull(); + expect(f.hops[0]!.attempt.queued_ms).toBe(1_240); + }); + + it("上游都满着:没有一跳发出去,原因是 core 说的那一句", () => { + const f = routingFacts( + row({ attempts: [busy("anthropic", 30_000), busy("openrouter")] }, { provider: "openrouter", error: busyAll }), + false, + )!; + expect(f.note).toEqual({ kind: "busy", tried: 0 }); + expect(f.reason).toEqual({ msg: busyAll }); + expect(f.hops.map((h) => h.denied)).toEqual([false, false]); + }); + + it("上游都满着:之前发出去、没成的几跳另说", () => { + const f = routingFacts( + row({ attempts: [overloaded("anthropic"), busy("openrouter")] }, { provider: "openrouter", error: busyAll }), + false, + )!; + expect(f.note).toEqual({ kind: "busy", tried: 1 }); + }); + + it("密钥的用量上限拒绝了它:没有尝试,原因是 core 说的那一句", () => { + const f = routingFacts(row({ attempts: [] }, { provider: "", error: limited }), false)!; + expect(f.ruleDenied).toBe(false); + expect(f.note).toEqual({ kind: "limited" }); + expect(f.reason).toEqual({ msg: limited }); + }); + it("选定上游之前被拒绝:没有尝试,原因是规则里写的那句", () => { const f = routingFacts( row({ route: "codex", rule: "no-opus", group: null, attempts: [] }, { provider: "", error: denied("no-opus", "Opus is not offered") }), diff --git a/src/requestRouting.ts b/src/requestRouting.ts index d6a0c9b1..7de5fa39 100644 --- a/src/requestRouting.ts +++ b/src/requestRouting.ts @@ -7,24 +7,44 @@ import type { AttemptView, HistoryRow, Msg, Stay } from "./types"; /** 规则拒绝时 core 说的那一句:`gw.route.denied {rule, reason}`。码和参数名是契约 */ const DENIED = "gw.route.denied"; +/** 网关密钥的用量上限拒绝时的那几句:`gw.key_limit.*`,一种量、一种周期一句 */ +const KEY_LIMIT = "gw.key_limit."; +/** 能服务的上游都满着、等过了也没空出位置:`gw.busy_all {upstreams}` */ +const BUSY_ALL = "gw.busy_all"; /** - * 一条请求为什么没有发往任何上游: + * 一条请求为什么没有上游接下: * * · `denied`:规则拒绝了它(选定上游之前) * · `unavailable`:规则选中的上游一个都接不了(停用、不在范围内、不提供这个模型) + * · `limited`:这把网关密钥的用量上限拒绝了它(选定上游之后、发出之前) + * · `busy`:能服务它的上游都满着(各自的并发上限),等过了也没空出位置 * * 发往了上游的、本地应答的、还没有结局的是 `null`。 * - * **上游是空的就是没有发往任何上游**:core 只在规则做了决定、请求却一个上游都不会去时 - * 把上游记成空的(本地应答另有 `local`)。是哪一种看失败的那一句:规则拒绝的是 - * `gw.route.denied`,其余是选中的上游接不了。 + * **上游是空的就是没有发往任何上游**:core 只在请求一个上游都不会去时把上游记成空的 + * (本地应答另有 `local`)。是哪一种看失败的那一句:规则拒绝的是 `gw.route.denied`, + * 用量上限的是 `gw.key_limit.*`,其余是选中的上游接不了。 + * + * **`busy` 的上游不是空的**:那一行归在尝试链的最后一跳,也就是最后看过、满着的那一家。 + * 可那一家一个字节都没收到,结局也不是它的失败 —— 这一格写「并发已满」,不写它的名字 */ -export type NotSent = "denied" | "unavailable"; +export type NotSent = "denied" | "unavailable" | "limited" | "busy"; export function notSent(r: { local?: boolean; provider: string; error?: Msg | null }): NotSent | null { - if (r.local || r.provider !== "" || !r.error) return null; - return r.error.code === DENIED ? "denied" : "unavailable"; + if (r.local || !r.error) return null; + if (r.error.code === BUSY_ALL) return "busy"; + if (r.provider !== "") return null; + if (r.error.code === DENIED) return "denied"; + return r.error.code.startsWith(KEY_LIMIT) ? "limited" : "unavailable"; +} + +/** + * 尝试链里这家满着、没有发出去就换了下一家的一跳(`skipped`)。另一种没发出去的一跳 —— + * 选定上游之后被规则拒绝 —— 由 `Hop.denied` 认 + */ +export function skippedHop(a: AttemptView): boolean { + return a.skipped != null; } /** 尝试链里的一跳。`denied`:选定上游之后的规则在这一跳拒绝了它,**没有发给这个上游** */ @@ -37,6 +57,11 @@ export interface Hop { * 尝试链下面那一句。只说字面上成立的事: * * · `failover`:前 `failed` 个上游失败,换到了下一个(最后一跳的结果在它自己那一行) + * · `switched`:前 `count` 跳没有接下它,换到了下一跳。其中有满着跳过的、或者开头超时 + * 放弃的 —— 那两种不是上游的失败,不说「失败」(每一跳为什么没接下在它自己那一行) + * · `limited`:这把网关密钥的用量上限拒绝了它:没有发往任何上游 + * · `busy`:剩下的上游都满着,等过了也没空出位置。`tried` 是在那之前真的发出去、没成的 + * 几跳;0 就是没有发往任何上游 * · `failover_denied`:前 `failed` 个上游失败,换到下一个之后被规则 `rule` 拒绝,没有发给它 * · `denied_after_pick`:唯一的那一跳被规则 `rule` 拒绝:没有发往任何上游 * · `denied_before_pick`:选定上游之前规则 `rule` 就拒绝了它:没有发往任何上游 @@ -48,6 +73,9 @@ export interface Hop { */ export type RoutingNote = | { kind: "failover"; failed: number } + | { kind: "switched"; count: number } + | { kind: "limited" } + | { kind: "busy"; tried: number } | { kind: "failover_denied"; failed: number; rule: string } | { kind: "denied_after_pick"; rule: string } | { kind: "denied_before_pick"; rule: string } @@ -110,14 +138,19 @@ export function routingFacts( if (ruleDenied || deniedBy) { const text = denyReason(deniedHop?.error) ?? denyReason(r.error); reason = text ? { text } : null; - } else if (why === "unavailable" && r.error) { + } else if ((why === "unavailable" || why === "limited" || why === "busy") && r.error) { + // 用量上限的那一句说清是哪一条、用了多少、什么时候重置;满着的那一句列出是哪几家 reason = { msg: r.error }; } let note: RoutingNote | null = null; - if (n === 0) { + if (why === "busy") { + // 满着跳过的几跳都没发出去。之前真的发出去、没成的那几跳另算 + note = { kind: "busy", tried: attempts.filter((a) => !skippedHop(a)).length }; + } else if (n === 0) { if (ruleDenied) note = { kind: "denied_before_pick", rule: routing.rule }; else if (why === "unavailable") note = { kind: "unavailable" }; + else if (why === "limited") note = { kind: "limited" }; else note = { kind: running ? "pending" : "none" }; } else if (deniedBy) { // 被拒的那一跳不算切换成功:前面几跳失败、换过来,才被拒绝 @@ -126,7 +159,9 @@ export function routingFacts( ? { kind: "failover_denied", failed: n - 1, rule: deniedBy } : { kind: "denied_after_pick", rule: deniedBy }; } else if (n > 1) { - note = { kind: "failover", failed: n - 1 }; + // 满着跳过、开头超时放弃都不是上游的失败 + const plain = attempts.slice(0, -1).every((a) => !skippedHop(a) && a.outcome !== "slow_start"); + note = plain ? { kind: "failover", failed: n - 1 } : { kind: "switched", count: n - 1 }; } return { diff --git a/src/routing/ChainMap.tsx b/src/routing/ChainMap.tsx index 59a30ebb..8a6420c4 100644 --- a/src/routing/ChainMap.tsx +++ b/src/routing/ChainMap.tsx @@ -5,7 +5,7 @@ import { StatusDot } from "@/ui/status-dot"; import { Tip } from "@/ui/tip"; import { cn } from "@/lib/utils"; import { textOf, useText } from "@/i18n"; -import { groupKindLabel, targetLabel } from "@/labels"; +import { targetLabel } from "@/labels"; import type { Overview } from "@/types"; import { buildChain, @@ -24,7 +24,7 @@ import { } from "./chain"; import { chainMapText } from "./ChainMap.i18n"; import { activityOf, type Flight } from "./flights"; -import { usersOf } from "./model"; +import { strategyText, usersOf } from "./model"; import { KeyIcon, TargetIcon, upstreamState } from "./parts"; import { partsText } from "./parts.i18n"; import { routingText } from "./routing.i18n"; @@ -421,7 +421,7 @@ function describe( ov.providers.map((p) => p.name), ); const ordered = g.kind === "fallback" || g.kind === "select"; - const line = node.idle && !g.builtin ? t.unreferenced : (ordered ? t.members : t.membersUnordered)(groupKindLabel(g.kind), ms); + const line = node.idle && !g.builtin ? t.unreferenced : (ordered ? t.members : t.membersUnordered)(strategyText(g), ms); return { body: ( <> diff --git a/src/routing/DryRunDialog.i18n.ts b/src/routing/DryRunDialog.i18n.ts index 937b5dd6..ec158413 100644 --- a/src/routing/DryRunDialog.i18n.ts +++ b/src/routing/DryRunDialog.i18n.ts @@ -25,6 +25,13 @@ export const dryRunText = messages( position: (n: number) => ` · 第 ${n} 条`, attempts: "尝试顺序", circuitOpen: "熔断中,将跳过", + /** 轮询组的候选:权重,和自动分配看的那几个数。快慢是首 token(从发出到回答的第一段内容),和请求详情同一个词 */ + weight: (n: number) => `权重 ${n}`, + ttft: (ms: string) => `首 token ${ms}`, + ttftNone: "首 token 暂无数据", + success: (pct: string) => `成功率 ${pct}`, + successNone: "成功率暂无数据", + share: (pct: string) => `占比 ${pct}`, converted: (formats: string) => `需转换格式:${formats}`, skipped: "已跳过", skipReason: (reason: string) => `(${reason})`, @@ -75,6 +82,12 @@ export const dryRunText = messages( position: (n: number) => ` · rule ${n}`, attempts: "Attempts", circuitOpen: "Circuit open, will be skipped", + weight: (n: number) => `Weight ${n}`, + ttft: (ms: string) => `First token ${ms}`, + ttftNone: "No first-token data yet", + success: (pct: string) => `Success rate ${pct}`, + successNone: "No success rate yet", + share: (pct: string) => `Share ${pct}`, converted: (formats: string) => `Format conversion: ${formats}`, skipped: "Skipped", skipReason: (reason: string) => ` (${reason})`, diff --git a/src/routing/DryRunDialog.tsx b/src/routing/DryRunDialog.tsx index db43f1c6..226178ec 100644 --- a/src/routing/DryRunDialog.tsx +++ b/src/routing/DryRunDialog.tsx @@ -25,6 +25,7 @@ import { commonText } from "@/i18n/common.i18n"; import { coreText, errorText } from "@/i18n/core.i18n"; import { PROBES, + balanceByLabel, formatLabel, groupKindLabel, mismatchText, @@ -32,13 +33,22 @@ import { targetLabel, translatedText, } from "@/labels"; -import type { Dialect, DryRunResult, KnownModel, Overview, RouteInput, RuleTrace } from "@/types"; +import type { + BalanceBy, + Dialect, + DryRunCandidate, + DryRunResult, + KnownModel, + Overview, + RouteInput, + RuleTrace, +} from "@/types"; import { skipLabel } from "@/upstreams/labels"; import { FormItem } from "@/upstreams/parts"; import { api } from "./api"; import { dryRunText } from "./DryRunDialog.i18n"; import { ModelInput, onOpenFocus } from "./fields"; -import { DIALECTS, usersOf } from "./model"; +import { DIALECTS, balanceShares, usersOf } from "./model"; import { KeyIcon, TargetIcon } from "./parts"; import { routingText } from "./routing.i18n"; import { modelViaOf, type ModelVia } from "./target"; @@ -58,7 +68,8 @@ export type DryRunTarget = * 旧结果淡一档留着,不闪成空白。 * * 尝试顺序里每个上游写出发给它的模型名和来历(别名、规则改写、指定模型):客户端写的 - * 名称和发出的不同,正是要在这里看清的事。 + * 名称和发出的不同,正是要在这里看清的事。经过轮询组时再写它的权重;按速度、稳定性 + * 分配时还写首 token 时间、成功率和算下来的占比 —— 「为什么轮到它」要从这里看得出来。 */ export function DryRunDialog({ target, @@ -366,6 +377,7 @@ function Result({ */ const short = r.outcome === "intercepted"; const candidates = r.candidate_models; + const shares = balanceShares(r); const target = r.outcome === "route" ? (r.via_group ?? candidates[0]?.provider ?? null) : null; // 这一趟经过的路:密钥 → 路由 → 规则 → 去向。和路由图同一套标志 @@ -409,7 +421,12 @@ function Result({
    {headline(r)} - {r.outcome === "route" && r.strategy && {groupKindLabel(r.strategy)}} + {r.outcome === "route" && r.strategy && ( + + {groupKindLabel(r.strategy)} + {r.balance_by && r.balance_by !== "weights" && ` · ${balanceByLabel(r.balance_by)}`} + + )}
    {steps.length > 1 && (
    @@ -464,6 +481,9 @@ function Result({ {c} {cv.sent_model && via && } + {cv.weight != null && ( + + )} {open && ( @@ -564,6 +584,29 @@ function SentModel({ model, via }: { model: string; via: ModelVia }) { ); } +/** + * 轮询组里一个候选的权重。按速度、稳定性分配时再写它看的数(没有样本的写明暂无数据) + * 和这一轮分到的占比;熔断着的这一轮不参加,不写占比 + */ +function BalanceFacts({ c, by, share }: { c: DryRunCandidate; by: BalanceBy; share: number | null }) { + const t = useText(dryRunText); + const parts = [t.weight(c.weight ?? 1)]; + if (by === "latency" || by === "latency-health") { + parts.push(c.ttfb_ms != null ? t.ttft(`${Math.round(c.ttfb_ms).toLocaleString()}ms`) : t.ttftNone); + } + if (by === "health" || by === "latency-health") { + parts.push(c.success_rate != null ? t.success(percent(c.success_rate)) : t.successNone); + } + if (by !== "weights" && share != null) parts.push(t.share(percent(share))); + return {parts.join(" · ")}; +} + +/** 0 到 1 写成百分数。不是 0、却四舍五入成 0 的写「<1%」—— 它还在轮里 */ +function percent(v: number): string { + if (v > 0 && v < 0.005) return "<1%"; + return `${Math.round(v * 100)}%`; +} + /** 路上的一站:标志加名字 */ function Step({ children, strong, className }: { children: ReactNode; strong?: boolean; className?: string }) { return ( diff --git a/src/routing/GroupDialog.i18n.ts b/src/routing/GroupDialog.i18n.ts index 741fbcd6..452acc5b 100644 --- a/src/routing/GroupDialog.i18n.ts +++ b/src/routing/GroupDialog.i18n.ts @@ -1,4 +1,5 @@ import { messages } from "@/i18n"; +import type { BalanceBy } from "@/types"; import { andList } from "./routing.i18n"; export const groupDialogText = messages( @@ -20,6 +21,17 @@ export const groupDialogText = messages( orderSelect: "拖动调整顺序。选定的上游不可用时,按顺序使用其余成员。", orderFallback: "拖动调整顺序:依次使用,前一个不可用时使用下一个。", orderOther: "拖动调整顺序。排序依据相同时按此顺序。", + orderBalance: "权重为 1 到 100 的整数。拖动调整顺序。排序依据相同时按此顺序。", + weight: "权重", + weightOf: (name: string) => `${name} 的权重`, + weightInvalid: (name: string) => `${name} 的权重须为 1 到 100 的整数`, + balanceBy: "分配依据", + balanceDesc: { + weights: "请求按成员权重的比例分配。", + latency: "以权重为基础,速度快的上游分到更多请求。", + health: "以权重为基础,失败少的上游分到更多请求。", + "latency-health": "以权重为基础,速度快、失败少的上游分到更多请求。", + } satisfies Record, }, { noMembers: "Select at least one upstream", @@ -39,5 +51,16 @@ export const groupDialogText = messages( orderSelect: "Drag to reorder. When the selected upstream is unavailable, the other members are used in order.", orderFallback: "Drag to reorder: members are used in turn, moving to the next when one is unavailable.", orderOther: "Drag to reorder. Ties are broken by this order.", + orderBalance: "Weights are whole numbers from 1 to 100. Drag to reorder. Ties are broken by this order.", + weight: "Weight", + weightOf: (name: string) => `Weight of ${name}`, + weightInvalid: (name: string) => `The weight of ${name} must be a whole number from 1 to 100`, + balanceBy: "Distribute by", + balanceDesc: { + weights: "Requests are split by the members' weights.", + latency: "Starting from the weights, faster upstreams get more requests.", + health: "Starting from the weights, upstreams that fail less get more requests.", + "latency-health": "Starting from the weights, faster upstreams that fail less get more requests.", + } satisfies Record, }, ); diff --git a/src/routing/GroupDialog.tsx b/src/routing/GroupDialog.tsx index 2f489ee8..22c3aac8 100644 --- a/src/routing/GroupDialog.tsx +++ b/src/routing/GroupDialog.tsx @@ -22,15 +22,15 @@ import { cn } from "@/lib/utils"; import { useText } from "@/i18n"; import { commonText } from "@/i18n/common.i18n"; import { errorText } from "@/i18n/core.i18n"; -import { groupKindLabel } from "@/labels"; -import type { GroupKind, Overview } from "@/types"; +import { BALANCE_BY, balanceByLabel, groupKindLabel } from "@/labels"; +import type { BalanceBy, GroupKind, Overview } from "@/types"; import { billingLabel, protocolLabel } from "@/upstreams/labels"; import { FormItem, Note } from "@/upstreams/parts"; import { api } from "./api"; import { onOpenFocus } from "./fields"; import { groupDialogText } from "./GroupDialog.i18n"; import { groupRefs } from "./GroupTable"; -import { move, strategies } from "./model"; +import { move, parseWeight, strategies } from "./model"; import { TargetIcon, upstreamState } from "./parts"; import { routingText } from "./routing.i18n"; import { useReorder } from "./useReorder"; @@ -46,6 +46,9 @@ export type GroupDialogMode = * 从上往下:名称、策略(几个里选一个,下面一句说它怎么选)、成员。成员的先后在 * 「按顺序」「手动选择」里就是优先级,所以可以拖动;手动选择还要在已选成员里定 * 一个优先使用的(行尾的「设为优先」)。 + * + * 轮询多两样:分配依据(只看比例,或者再看速度、稳定性),和每个成员的权重(行尾, + * 1 到 100,没改过的明写 1)。别的策略不用它们,切走再切回来时还在,只是不交。 */ export function GroupDialog({ mode, @@ -83,11 +86,19 @@ export function GroupDialog({ }); const [members, setMembers] = useState(source?.providers ?? []); const [selected, setSelected] = useState(source?.selected ?? null); + const [balanceBy, setBalanceBy] = useState(source?.balance_by ?? "weights"); + // 填的是文字:清空、打错的那一下也要留着让人改,交的时候才换成数 + const [weights, setWeights] = useState>(() => + Object.fromEntries(Object.entries(source?.weights ?? {}).map(([n, w]) => [n, String(w)])), + ); const [saving, setSaving] = useState(false); const [error, setError] = useState(null); const reorder = useReorder((from, to) => setOrder((o) => move(o, from, to))); const chosen = order.filter((n) => members.includes(n)); + const balanced = kind === "load-balance"; + const weightText = (n: string) => weights[n] ?? "1"; + const badWeight = balanced ? chosen.find((n) => parseWeight(weightText(n)) == null) : undefined; const preferred = kind === "select" ? (selected && chosen.includes(selected) ? selected : (chosen[0] ?? null)) : null; const refs = mode.kind === "edit" ? groupRefs(ov, mode.name) : []; @@ -103,7 +114,9 @@ export function GroupDialog({ ? rt.nameTaken(trimmed) : chosen.length === 0 ? t.noMembers - : null; + : badWeight !== undefined + ? t.weightInvalid(badWeight) + : null; const strategy = strategies().find((s) => s.id === kind); async function save() { @@ -116,6 +129,9 @@ export function GroupDialog({ kind, providers: chosen, selected: preferred, + // 不交 = 都是 1、只看比例;别的策略只能这样 + weights: balanced ? Object.fromEntries(chosen.map((n) => [n, parseWeight(weightText(n)) ?? 1])) : null, + balance_by: balanced ? balanceBy : null, }, base_version: base, }; @@ -164,6 +180,21 @@ export function GroupDialog({

    + {balanced && ( +
    + {t.balanceBy} + + label={t.balanceBy} + value={balanceBy} + onChange={setBalanceBy} + options={BALANCE_BY.map((b) => ({ id: b, label: balanceByLabel(b) }))} + /> +

    + {t.balanceDesc[balanceBy]} +

    +
    + )} +
    {rt.members} @@ -180,6 +211,7 @@ export function GroupDialog({ {t.upstream} {kind === "select" && {t.preferred}} + {balanced && {t.weight}} @@ -256,6 +288,20 @@ export function GroupDialog({ ))} )} + {balanced && ( + + {on && ( + setWeights((w) => ({ ...w, [n]: e.target.value }))} + /> + )} + + )} ); })} @@ -264,7 +310,13 @@ export function GroupDialog({
    )} - {kind === "select" ? t.orderSelect : kind === "fallback" ? t.orderFallback : t.orderOther} + {kind === "select" + ? t.orderSelect + : kind === "fallback" + ? t.orderFallback + : balanced + ? t.orderBalance + : t.orderOther}
    diff --git a/src/routing/GroupTable.tsx b/src/routing/GroupTable.tsx index 1159f395..149db7bc 100644 --- a/src/routing/GroupTable.tsx +++ b/src/routing/GroupTable.tsx @@ -13,6 +13,7 @@ import { textOf, useText } from "@/i18n"; import { groupKindLabel } from "@/labels"; import type { GroupView, Overview } from "@/types"; import { membersOf, type ChainFocus } from "./chain"; +import { balanceNotes } from "./model"; import { groupTableText } from "./GroupTable.i18n"; import { TargetIcon, upstreamState } from "./parts"; import { routingText } from "./routing.i18n"; @@ -78,6 +79,9 @@ export function GroupTable({ {shown.map(({ item: g, key, presence }) => { const items = menu(g, actions); const refs = groupRefs(ov, g.name); + // 轮询组的比例(不是平均分时)和分配依据(不是只看比例时),各占一行写在策略下面: + // 这一列窄,连成一行会从「按速度和稳定性」中间折开 + const notes = balanceNotes(g); const label = g.builtin ? t.allUpstreams : g.name; const f: ChainFocus = { kind: "group", name: g.name }; const lit = focus?.kind === "group" && focus.name === g.name; @@ -112,8 +116,13 @@ export function GroupTable({ {g.builtin && {t.builtin}} - +
    {groupKindLabel(g.kind)}
    + {notes.map((n) => ( +
    + {n} +
    + ))}
    diff --git a/src/routing/chain.test.ts b/src/routing/chain.test.ts index d47d9d23..e9542c67 100644 --- a/src/routing/chain.test.ts +++ b/src/routing/chain.test.ts @@ -19,6 +19,7 @@ const key = (name: string, x: Partial = {}): ClientView => ({ max_concurrent: null, route: null, allow: null, + limits: [], ...x, }); const rule = (name: string, x: Partial = {}): RuleView => ({ @@ -45,6 +46,8 @@ const group = (name: string, providers: string[], x: Partial = {}): G kind: "fallback", selected: null, providers, + weights: {}, + balance_by: "weights", ...x, }); const ALL = group("__all__", [], { builtin: true }); diff --git a/src/routing/flights.test.ts b/src/routing/flights.test.ts index 3b3a7386..bf8e2a47 100644 --- a/src/routing/flights.test.ts +++ b/src/routing/flights.test.ts @@ -9,6 +9,7 @@ const key = (name: string, x: Partial = {}): ClientView => ({ max_concurrent: null, route: null, allow: null, + limits: [], ...x, }); const rule = (name: string, x: Partial = {}): RuleView => ({ @@ -34,6 +35,8 @@ const group = (name: string, providers: string[], x: Partial = {}): G kind: "fallback", selected: null, providers, + weights: {}, + balance_by: "weights", ...x, }); const up = (name: string) => ({ name, disabled: false, health: "ok" }) as ProviderView; diff --git a/src/routing/model.i18n.ts b/src/routing/model.i18n.ts index 8fae7ea0..4101f6a0 100644 --- a/src/routing/model.i18n.ts +++ b/src/routing/model.i18n.ts @@ -22,6 +22,8 @@ export const modelText = messages( noCatchAll: "尚无兜底规则", builtinGroup: "内置策略组 · 按上游列表顺序", groupTarget: (kind: string, members: string) => `策略组 · ${kind}${members ? `:${members}` : ""}`, + /** 策略名后面的补充:轮询组的比例和分配依据 */ + withNotes: (kind: string, notes: string) => `${kind}(${notes})`, upstreamTarget: (protocol: string, disabled: boolean) => `上游 · ${protocol}${disabled ? " · 已停用" : ""}`, unknownTarget: "不存在的去向", setModel: (model: string) => `模型改为 ${model}`, @@ -34,7 +36,7 @@ export const modelText = messages( fallback: "依次使用成员,前一个不可用时使用下一个。", select: "使用选定的上游;它不可用时,按顺序使用其余成员。", loadBalance: "在成员之间轮流分配请求。", - urlTest: "优先使用首字节时间最短的上游。", + urlTest: "优先使用首 token 时间最短的上游。", cheapest: "优先使用输入单价最低的上游。", }, }, @@ -62,6 +64,7 @@ export const modelText = messages( noCatchAll: "No catch-all rule yet", builtinGroup: "Built-in group · In upstream list order", groupTarget: (kind: string, members: string) => `Group · ${kind}${members ? `: ${members}` : ""}`, + withNotes: (kind: string, notes: string) => `${kind} (${notes})`, upstreamTarget: (protocol: string, disabled: boolean) => `Upstream · ${protocol}${disabled ? " · Disabled" : ""}`, unknownTarget: "Destination not found", setModel: (model: string) => `Model set to ${model}`, @@ -74,7 +77,7 @@ export const modelText = messages( fallback: "Uses the members in order, moving to the next when one is unavailable.", select: "Uses the selected upstream; when it is unavailable, uses the other members in order.", loadBalance: "Distributes requests across the members in turn.", - urlTest: "Prefers the upstream with the shortest time to first byte.", + urlTest: "Prefers the upstream with the shortest time to first token.", cheapest: "Prefers the upstream with the lowest input price.", }, }, diff --git a/src/routing/model.test.ts b/src/routing/model.test.ts index 83dfafb9..2d01b305 100644 --- a/src/routing/model.test.ts +++ b/src/routing/model.test.ts @@ -1,8 +1,10 @@ import { describe, expect, it } from "vitest"; import { setLang } from "@/i18n"; -import type { ConditionView, RouteView, RuleView } from "@/types"; +import type { ConditionView, DryRunCandidate, GroupView, RouteView, RuleView } from "@/types"; import { addOnsText, + balanceNotes, + balanceShares, blankPinned, blankRule, canLift, @@ -14,9 +16,12 @@ import { insertIndex, liftShadowed, move, + parseWeight, + ratioText, routeProblems, ruleProblem, splitCompare, + strategyText, usersOf, type RuleDraft, } from "./model"; @@ -200,8 +205,8 @@ describe("路由列表", () => { it("默认路由的使用者包括没指定路由的密钥", () => { const clients = [ - { name: "claude-code", key: "tw-a", max_concurrent: null, route: null, allow: null }, - { name: "codex", key: "tw-b", max_concurrent: null, route: "codex", allow: null }, + { name: "claude-code", key: "tw-a", max_concurrent: null, route: null, allow: null, limits: [] }, + { name: "codex", key: "tw-b", max_concurrent: null, route: "codex", allow: null, limits: [] }, ]; expect(usersOf(route({ name: "默认", default: true }), clients)).toEqual(["claude-code"]); expect(usersOf(route({}), clients)).toEqual(["codex"]); @@ -256,3 +261,62 @@ describe("路由列表", () => { ]); }); }); + +describe("轮询组的比例和分配依据", () => { + const group = (p: Partial): GroupView => ({ + name: "分流", + builtin: false, + kind: "load-balance", + providers: ["anthropic", "openrouter"], + weights: { anthropic: 1, openrouter: 1 }, + balance_by: "weights", + ...p, + }); + + it("权重只收 1 到 100 的整数", () => { + expect(parseWeight("1")).toBe(1); + expect(parseWeight(" 100 ")).toBe(100); + for (const bad of ["", "0", "101", "1.5", "-3", "7k", " "]) expect(parseWeight(bad)).toBeNull(); + }); + + it("平均分时不写比例,按成员的顺序写出不平均的", () => { + expect(ratioText(group({}))).toBeNull(); + expect(ratioText(group({ weights: { anthropic: 7, openrouter: 3 } }))).toBe("7 : 3"); + // 别的类型不用权重 + expect(ratioText(group({ kind: "fallback", weights: {} }))).toBeNull(); + }); + + it("策略名后面补上比例和不是只看比例的分配依据", () => { + setLang("zh"); + expect(balanceNotes(group({}))).toEqual([]); + expect(strategyText(group({}))).toBe("轮询"); + expect(strategyText(group({ weights: { anthropic: 7, openrouter: 3 } }))).toBe("轮询(7 : 3)"); + expect(strategyText(group({ balance_by: "latency" }))).toBe("轮询(按速度)"); + expect(strategyText(group({ weights: { anthropic: 2, openrouter: 1 }, balance_by: "latency-health" }))).toBe( + "轮询(2 : 1 · 按速度和稳定性)", + ); + setLang("en"); + expect(strategyText(group({ weights: { anthropic: 7, openrouter: 3 }, balance_by: "health" }))).toBe( + "Round robin (7 : 3 · By reliability)", + ); + }); + + it("试算的占比按权重 × 系数分,熔断着的不参加", () => { + const c = (provider: string, weight: number | null, balance_factor: number | null = null): DryRunCandidate => ({ + provider, + weight, + balance_factor, + }); + expect(balanceShares({ candidate_models: [c("a", 7), c("b", 3)], circuit_open: [] })).toEqual([0.7, 0.3]); + const auto = balanceShares({ candidate_models: [c("a", 2, 2.25), c("b", 1, 0.5)], circuit_open: [] }); + expect(auto[0]).toBeCloseTo(0.9); + expect(auto[1]).toBeCloseTo(0.1); + expect(balanceShares({ candidate_models: [c("a", 1), c("b", 1), c("c", 2)], circuit_open: ["c"] })).toEqual([ + 0.5, 0.5, 0, + ]); + // 全都熔断着时都算:网关照样一家家试 + expect(balanceShares({ candidate_models: [c("a", 3), c("b", 1)], circuit_open: ["a", "b"] })).toEqual([0.75, 0.25]); + // 不是轮询组:没有权重,也就没有占比 + expect(balanceShares({ candidate_models: [c("a", null)], circuit_open: [] })).toEqual([null]); + }); +}); diff --git a/src/routing/model.ts b/src/routing/model.ts index 3d00ccf7..cb4450c5 100644 --- a/src/routing/model.ts +++ b/src/routing/model.ts @@ -5,9 +5,9 @@ * 能不能用、条件写得对不对,最后由 core 说;这里只做对话框里需要实时给出 * 的那几件事:保存按钮旁边缺什么、哪条规则被兜底挡住。 */ -import type { ClientView, ConditionField, ConditionView, Dialect, GroupKind, GroupView, KnownModel, PinnedModel, ProviderView, RouteView, RuleInput, RuleView } from "@/types"; +import type { ClientView, ConditionField, ConditionView, Dialect, DryRunResult, GroupKind, GroupView, KnownModel, PinnedModel, ProviderView, RouteView, RuleInput, RuleView } from "@/types"; import { textOf } from "@/i18n"; -import { ALL_UPSTREAMS, conditionName, groupKindLabel, targetLabel } from "@/labels"; +import { ALL_UPSTREAMS, balanceByLabel, conditionName, groupKindLabel, targetLabel } from "@/labels"; import { protocolLabel } from "@/upstreams/labels"; import { modelText } from "./model.i18n"; import { routingText } from "./routing.i18n"; @@ -386,6 +386,71 @@ export function routeProblems(route: RouteView): string[] { return out; } +// ---------------------------------------------------------------- 轮询组的比例 + +/** 权重的范围,和 core 的校验一样(`engine.group_weight_out_of_range`) */ +export const WEIGHT_MIN = 1; +export const WEIGHT_MAX = 100; + +/** 对话框里填的权重:1 到 100 的整数,别的(空、小数、超出范围)是 null */ +export function parseWeight(text: string): number | null { + const s = text.trim(); + if (!/^\d+$/.test(s)) return null; + const n = Number(s); + return n >= WEIGHT_MIN && n <= WEIGHT_MAX ? n : null; +} + +/** 成员在轮询组里的权重。core 给每个成员都列了,没列到的(别的类型)按 1 */ +export function weightOf(g: Pick, member: string): number { + return g.weights[member] ?? 1; +} + +/** + * 轮询组的比例,按成员的顺序:`7 : 3`。**权重都是 1 时没有** —— 那就是平均分, + * 写出 `1 : 1 : 1` 只是噪音。别的类型不用权重,也没有 + */ +export function ratioText(g: Pick): string | null { + if (g.kind !== "load-balance") return null; + const ws = g.providers.map((p) => weightOf(g, p)); + return ws.some((w) => w !== 1) ? ws.join(" : ") : null; +} + +/** + * 轮询组在策略名之外要说的:比例(不是平均分时)、分配依据(不是只看比例时)。 + * 别的类型、两样都是默认值时是空的 + */ +export function balanceNotes(g: Pick): string[] { + if (g.kind !== "load-balance") return []; + const out: string[] = []; + const ratio = ratioText(g); + if (ratio) out.push(ratio); + if (g.balance_by !== "weights") out.push(balanceByLabel(g.balance_by)); + return out; +} + +/** 策略名,轮询组带上比例和分配依据:`轮询(7 : 3 · 按速度)` */ +export function strategyText(g: Pick): string { + const kind = groupKindLabel(g.kind); + const notes = balanceNotes(g); + return notes.length ? textOf(modelText).withNotes(kind, notes.join(" · ")) : kind; +} + +/** + * 试算里轮询组每个候选这一轮分到请求的份额,0 到 1,和 `r.candidate_models` 一一对应; + * 不是轮询组(候选没有权重)的是 null。 + * + * 和 core 排头用的同一个数:权重 × 系数(`balance_factor`,只看比例时是 1)。**熔断着的 + * 这一轮不参加**(份额是 0,排到它的那一次本来就会被跳过),全都熔断着时都算 + */ +export function balanceShares(r: Pick): (number | null)[] { + const members = r.candidate_models.filter((c) => c.weight != null); + const sitOut = members.every((c) => r.circuit_open.includes(c.provider)) ? [] : r.circuit_open; + const eff = (c: (typeof members)[number]) => + sitOut.includes(c.provider) ? 0 : (c.weight ?? 0) * (c.balance_factor ?? 1); + const total = members.reduce((a, c) => a + eff(c), 0); + return r.candidate_models.map((c) => (c.weight == null ? null : total > 0 ? eff(c) / total : 0)); +} + /** 去向的说明:策略组的策略与成员,或上游的协议 */ export function describeTarget( name: string, @@ -397,7 +462,7 @@ export function describeTarget( if (g) { if (g.builtin) return t.builtinGroup; const members = membersText(g); - return t.groupTarget(groupKindLabel(g.kind), members); + return t.groupTarget(strategyText(g), members); } const p = providers.find((x) => x.name === name); if (p) return t.upstreamTarget(protocolLabel(p.protocol), p.disabled); diff --git a/src/settings/FailoverSection.i18n.ts b/src/settings/FailoverSection.i18n.ts index 1385452f..ec929c9d 100644 --- a/src/settings/FailoverSection.i18n.ts +++ b/src/settings/FailoverSection.i18n.ts @@ -12,6 +12,7 @@ export const failoverText = messages( quota_pause_secs: "额度用完", rate_limit_max_pause_secs: "限流", stream_start_wait_secs: "等待回答开头", + slot_wait_secs: "最多等待空位", }, what: { failures_to_pause: "服务器错误、无法连接等未说明原因的失败,连续达到此次数后暂停。", @@ -21,13 +22,18 @@ export const failoverText = messages( quota_pause_secs: "上游报告额度用完、但未给出重置时间时暂停的时长。给出重置时间的,暂停到重置为止。", rate_limit_max_pause_secs: "上游限流时按其要求的等待时间暂停,最长为此值。", stream_start_wait_secs: "流式回答在第一段内容到达前报错时,请求交给下一个上游。等待超过此时长后不再等待。", + slot_wait_secs: "上游并发已满或密钥的分钟、小时上限用满时,请求合计最多等待的时长。0 表示不等待。", }, + nextOnSlowStart: "开头超时时转到下一个上游", + nextOnSlowStartWhat: "最后一个上游照常等待。开启时,等待时长宜在 30 秒以上。", times: "次", secs: "秒", badCount: "须为 1 到 100 之间的整数。", badSecs: "须为 1 到 604800 之间的整数。", badMax: "须为整数,不小于暂停时长,不超过 604800。", badWait: "须为 1 到 120 之间的整数。", + badSlowStartWait: "开启「开头超时时转到下一个上游」时,须为 5 到 120 之间的整数。", + badSlotWait: "须为 0 到 300 之间的整数。", saveFailed: "未能保存", }, { @@ -42,6 +48,7 @@ export const failoverText = messages( quota_pause_secs: "Quota used up", rate_limit_max_pause_secs: "Rate limit", stream_start_wait_secs: "Wait for the answer to start", + slot_wait_secs: "Wait for a free slot at most", }, what: { failures_to_pause: @@ -54,13 +61,18 @@ export const failoverText = messages( rate_limit_max_pause_secs: "A rate-limited upstream is paused for the wait it asks for, at most this long.", stream_start_wait_secs: "An error before the first content of a streamed answer sends the request to the next upstream. After this long, the wait ends.", + slot_wait_secs: "Total time a request waits for a full upstream or a key's per-minute or per-hour limit. 0 means no wait.", }, + nextOnSlowStart: "Move to the next upstream when the start times out", + nextOnSlowStartWhat: "The last upstream keeps waiting. With this on, a wait of 30 s or more is advisable.", times: "times", secs: "s", badCount: "A whole number from 1 to 100.", badSecs: "A whole number from 1 to 604800.", badMax: "A whole number, not less than the pause and at most 604800.", badWait: "A whole number from 1 to 120.", + badSlowStartWait: "With “Move to the next upstream when the start times out” on, a whole number from 5 to 120.", + badSlotWait: "A whole number from 0 to 300.", saveFailed: "Not saved", }, ); diff --git a/src/settings/FailoverSection.tsx b/src/settings/FailoverSection.tsx index 80cec71a..adc8b3c1 100644 --- a/src/settings/FailoverSection.tsx +++ b/src/settings/FailoverSection.tsx @@ -1,5 +1,6 @@ -import { useState } from "react"; +import { useState, type ReactNode } from "react"; import { Banner } from "@/ui/banner"; +import { Switch } from "@/ui/switch"; import { useText } from "@/i18n"; import { errorText } from "@/i18n/core.i18n"; import { patchConfig } from "@/patch"; @@ -12,9 +13,14 @@ import { failoverText } from "./FailoverSection.i18n"; const MAX_PAUSE = 7 * 24 * 3600; /** 流开头最多等多少秒(core 的 `MAX_STREAM_START_WAIT_SECS`) */ const MAX_WAIT = 120; +/** 开着「开头超时转到下一个上游」时,流开头至少等多少秒(core 的 `MIN_SLOW_START_WAIT_SECS`) */ +const MIN_SLOW_START_WAIT = 5; +/** 等空位最多写多少秒(core 的 `MAX_SLOT_WAIT_SECS`) */ +const MAX_SLOT_WAIT = 300; -type Field = keyof FailoverView; -export type Draft = Record; +/** 这一节里的数字格子。开关(`next_on_slow_start`)另记 */ +type Field = Exclude; +export type Draft = Record & { next_on_slow_start: boolean }; /** 表单里的顺序 */ const FIELDS: Field[] = [ @@ -25,19 +31,24 @@ const FIELDS: Field[] = [ "quota_pause_secs", "rate_limit_max_pause_secs", "stream_start_wait_secs", + "slot_wait_secs", ]; /** 导出给测试用 */ -export const draftOf = (f: FailoverView): Draft => - Object.fromEntries(FIELDS.map((k) => [k, String(f[k])])) as Draft; +export const draftOf = (f: FailoverView): Draft => ({ + ...(Object.fromEntries(FIELDS.map((k) => [k, String(f[k])])) as Record), + next_on_slow_start: f.next_on_slow_start, +}); -const same = (a: Draft, b: Draft) => FIELDS.every((k) => a[k] === b[k]); +const same = (a: Draft, b: Draft) => + FIELDS.every((k) => a[k] === b[k]) && a.next_on_slow_start === b.next_on_slow_start; /** * 每一格填得对不对,范围和 core 的校验一样。 * * **没动过的格不查**(和日志保留一样):它就是配置里现在的值,保存时也不发。只有上限 - * 例外 —— 起点改大了,没动过的上限也可能跟着不对了。导出给测试用。 + * 例外 —— 起点改大了,没动过的上限也可能跟着不对了。等回答开头的秒数也一样:打开 + * 「开头超时时转到下一个上游」之后它至少要 5 秒,没动过的也要重查。导出给测试用。 */ export function checks(draft: Draft, saved: Draft): Record { const secs = (k: Field) => draft[k] === saved[k] || intIn(draft[k], 1, MAX_PAUSE); @@ -52,14 +63,20 @@ export function checks(draft: Draft, saved: Draft): Record { quota_pause_secs: secs("quota_pause_secs"), rate_limit_max_pause_secs: secs("rate_limit_max_pause_secs"), stream_start_wait_secs: - draft.stream_start_wait_secs === saved.stream_start_wait_secs || intIn(draft.stream_start_wait_secs, 1, MAX_WAIT), + (draft.stream_start_wait_secs === saved.stream_start_wait_secs && + draft.next_on_slow_start === saved.next_on_slow_start) || + intIn(draft.stream_start_wait_secs, draft.next_on_slow_start ? MIN_SLOW_START_WAIT : 1, MAX_WAIT), + slot_wait_secs: draft.slot_wait_secs === saved.slot_wait_secs || intIn(draft.slot_wait_secs, 0, MAX_SLOT_WAIT), }; } /** - * 上游失败之后停用多久、流式回答的开头最多等多久。 + * 上游失败之后停用多久、流式回答的开头最多等多久、上游满着时最多等多久。 * * 默认值显式写在格子里(概览给的就是真在用的数),不用「留空 = 默认」。 + * + * **「开头超时时转到下一个上游」和等开头的秒数是一件事**:开关挂在那一行底下,不另起 + * 一行 —— 它说的就是那个秒数到了之后怎么办。 */ export function FailoverSection({ failover, @@ -87,6 +104,8 @@ export function FailoverSection({ path: `/failover/${k}`, value: Number(draft[k]), })); + if (draft.next_on_slow_start !== saved.next_on_slow_start) + ops.push({ op: "replace", path: "/failover/next_on_slow_start", value: draft.next_on_slow_start }); setBusy(true); setError(null); try { @@ -107,7 +126,7 @@ export function FailoverSection({ }; const bad = (msg: string) => {msg}; - const row = (k: Field, unit: string, what: string, badText: string) => ( + const row = (k: Field, unit: string, what: string, badText: string, more?: ReactNode) => ( } - /> + > + {more} + ); return ( @@ -142,7 +163,33 @@ export function FailoverSection({ {row("no_balance_pause_secs", t.secs, t.what.no_balance_pause_secs, t.badSecs)} {row("quota_pause_secs", t.secs, t.what.quota_pause_secs, t.badSecs)} {row("rate_limit_max_pause_secs", t.secs, t.what.rate_limit_max_pause_secs, t.badSecs)} - {row("stream_start_wait_secs", t.secs, t.what.stream_start_wait_secs, t.badWait)} + {row( + "stream_start_wait_secs", + t.secs, + t.what.stream_start_wait_secs, + draft.next_on_slow_start ? t.badSlowStartWait : t.badWait, + // 和上面那一行排成同一个样子:说明在左、开关在右,中间不画分隔线 +
    +
    + +
    {t.nextOnSlowStartWhat}
    +
    +
    + { + setError(null); + setDraft((d) => ({ ...d, next_on_slow_start: c === true })); + }} + /> +
    +
    , + )} + {row("slot_wait_secs", t.secs, t.what.slot_wait_secs, t.badSlotWait)} ) => Object.values(c).every(Boolean); @@ -36,4 +38,25 @@ describe("故障转移", () => { expect(c.max_pause_secs).toBe(false); expect(checks({ ...saved, pause_secs: "900", max_pause_secs: "900" }, saved).max_pause_secs).toBe(true); }); + + it("等空位的秒数可以是 0(不等),最多 300", () => { + const saved = draftOf(failover); + expect(saved.slot_wait_secs).toBe("30"); + expect(checks({ ...saved, slot_wait_secs: "0" }, saved).slot_wait_secs).toBe(true); + expect(checks({ ...saved, slot_wait_secs: "300" }, saved).slot_wait_secs).toBe(true); + expect(checks({ ...saved, slot_wait_secs: "301" }, saved).slot_wait_secs).toBe(false); + expect(checks({ ...saved, slot_wait_secs: "" }, saved).slot_wait_secs).toBe(false); + }); + + it("开头超时转到下一个上游:开着时开头至少等 5 秒,没动过的秒数也重查", () => { + const saved = draftOf({ ...failover, stream_start_wait_secs: 3 }); + expect(saved.next_on_slow_start).toBe(false); + // 关着时 3 秒是合法的 + expect(checks(saved, saved).stream_start_wait_secs).toBe(true); + // 打开开关,秒数没动也不行 + const on = { ...saved, next_on_slow_start: true }; + expect(checks(on, saved).stream_start_wait_secs).toBe(false); + expect(checks({ ...on, stream_start_wait_secs: "5" }, saved).stream_start_wait_secs).toBe(true); + expect(checks({ ...on, stream_start_wait_secs: "30" }, saved).stream_start_wait_secs).toBe(true); + }); }); diff --git a/src/traffic/SessionPanel.tsx b/src/traffic/SessionPanel.tsx index 0bf180e2..6436b9bc 100644 --- a/src/traffic/SessionPanel.tsx +++ b/src/traffic/SessionPanel.tsx @@ -3,6 +3,7 @@ import { call } from "@/control"; import { useText } from "@/i18n"; import { cn } from "@/lib/utils"; import { useResource } from "@/lib/resource"; +import { notSent } from "@/requestRouting"; import type { RequestRow, SessionDetail, TurnView } from "@/types"; import { Button } from "@/ui/button"; import { UpstreamLogo } from "@/ui/logos"; @@ -131,8 +132,11 @@ export function SessionPanel({ const pending = useMemo(() => unrecordedRows(turns, rows), [turns, rows]); const n = tally(s, rows, pending); // 走过哪几个上游,按第一次出现的先后。一次任务中途换过上游,这里能看出来。 - // 没有发往任何上游的那几轮(被规则拒绝)上游是空的,不算 - const providers = [...new Set([...turns.map((x) => x.provider), ...pending.map((r) => r.provider)].filter(Boolean))]; + // 没有发往任何上游的那几轮(被规则拒绝)上游是空的,不算;上游都满着的那几轮记在最后 + // 看过的那一家上,那一家没收到它,也不算(见 `notSent`) + const providers = [ + ...new Set([...turns, ...pending].filter((x) => notSent(x) === null).map((x) => x.provider).filter(Boolean)), + ]; const client = s?.client ?? pending[0]?.client; const [tab, setTab] = useState("summary"); /** 「对话」打开过:之后切走也留着(读到哪儿、展开了哪几条都在),见 `PANE` */ diff --git a/src/traffic/SessionRow.tsx b/src/traffic/SessionRow.tsx index 51dc857a..3be2cb6a 100644 --- a/src/traffic/SessionRow.tsx +++ b/src/traffic/SessionRow.tsx @@ -2,6 +2,7 @@ import { memo } from "react"; import { ChevronRightIcon } from "lucide-react"; import { cn } from "@/lib/utils"; import { useText } from "@/i18n"; +import { notSent } from "@/requestRouting"; import { Button } from "@/ui/button"; import { UpstreamLogo } from "@/ui/logos"; import { RowMenu, RowMenuButton, type MenuItems } from "@/ui/row-menu"; @@ -66,8 +67,9 @@ export const SessionRow = memo(function SessionRow({ // 汇总加上汇总里还没有的那几轮:在跑的、刚落地的(见 `tally`) const n = tallyOf(g); // **上游从行里数,不从汇总里拿** —— `SessionView` 没有这一项, - // 而组里的每一条都知道自己走了哪个上游(没有发往任何上游的那几条是空的,不算) - const providers = [...new Set(rows.map((r) => r.provider).filter(Boolean))]; + // 而组里的每一条都知道自己走了哪个上游(没有发往任何上游的那几条是空的,不算;上游都满着 + // 的那几条记在最后看过的那一家上,可那一家没收到它,也不算 —— 见 `notSent`) + const providers = [...new Set(rows.filter((r) => notSent(r) === null).map((r) => r.provider).filter(Boolean))]; const openIt = () => { // 点组头和点请求行一样,键盘接着从这一行往下走 onCursor({ kind: "session", id }); diff --git a/src/traffic/cells.tsx b/src/traffic/cells.tsx index b72681ba..433c1ccf 100644 --- a/src/traffic/cells.tsx +++ b/src/traffic/cells.tsx @@ -6,7 +6,7 @@ import { keyText } from "@/KeyLabel"; import { cn } from "@/lib/utils"; import type { NotSent } from "@/requestRouting"; import type { RequestRow } from "@/types"; -import { IconDenied, IconNoUpstream, IconRemote } from "@/ui/icons"; +import { IconBusy, IconDenied, IconLimitReached, IconNoUpstream, IconRemote } from "@/ui/icons"; import { ClientLogo } from "@/ui/logos"; import { notify } from "@/ui/notify"; import { Tip } from "@/ui/tip"; @@ -103,22 +103,25 @@ export function RowKeyCell({ r, hints }: { r: RequestRow; hints: boolean }) { return ; } +/** 每一种在上游标志的位置上画的图形和颜色 */ +const NOT_SENT: Record = { + denied: { Icon: IconDenied, color: "text-destructive" }, + unavailable: { Icon: IconNoUpstream, color: "text-muted-foreground" }, + limited: { Icon: IconLimitReached, color: "text-warning" }, + busy: { Icon: IconBusy, color: "text-muted-foreground" }, +}; + /** - * 没有发往任何上游的请求在上游标志的位置上画什么:被规则拒绝是禁止符号(和路由图上 - * 「拒绝」那个节点同一个),没有可用的上游是划掉的上游。和上游标志一样大,名字对得齐。 + * 没有上游接下的请求在上游标志的位置上画什么:被规则拒绝是禁止符号(和路由图上 + * 「拒绝」那个节点同一个),没有可用的上游是划掉的上游,密钥的用量到了上限是顶到线的 + * 箭头,上游都满着是沙漏。和上游标志一样大,名字对得齐。 * - * 拒绝带红色,和路由图一致;`plain` 时跟着周围的字色(命令面板里的图标都是单色)。 + * 拒绝带红色,和路由图一致;用量上限是琥珀色:到了上限要留意,但不是故障。`plain` 时 + * 跟着周围的字色(命令面板里的图标都是单色)。 */ export function NotSentIcon({ kind, plain }: { kind: NotSent; plain?: boolean }) { - const Icon = kind === "denied" ? IconDenied : IconNoUpstream; - return ( - - ); + const { Icon, color } = NOT_SENT[kind]; + return ; } /** diff --git a/src/ui/icons.tsx b/src/ui/icons.tsx index aaefe203..4c5d8839 100644 --- a/src/ui/icons.tsx +++ b/src/ui/icons.tsx @@ -28,6 +28,8 @@ export { Puzzle as IconPlugin } from "lucide-react"; // 插件 —— 拼进链 export { Server as IconServer } from "lucide-react"; // 上游 —— 一摞机器 export { ServerOff as IconNoUpstream } from "lucide-react"; // 没有可用的上游 —— 那一摞机器划掉 export { Ban as IconDenied } from "lucide-react"; // 被规则拒绝 —— 禁止符号 +export { ArrowUpToLine as IconLimitReached } from "lucide-react"; // 密钥的用量到了上限 —— 顶到那条线 +export { Hourglass as IconBusy } from "lucide-react"; // 上游都满着、等不到空位 —— 沙漏 export { KeyRound as IconKey } from "lucide-react"; // 密钥 —— 钥匙,不是锁 export { SlidersHorizontal as IconSettings } from "lucide-react"; // 设置 —— 推子,齿轮留给系统设置 export { PanelLeft as IconSidebar } from "lucide-react"; // 收起/展开源列表 diff --git a/src/upstreams/ChatgptAccountSection.tsx b/src/upstreams/ChatgptAccountSection.tsx index ff0c22be..bff35f80 100644 --- a/src/upstreams/ChatgptAccountSection.tsx +++ b/src/upstreams/ChatgptAccountSection.tsx @@ -26,6 +26,7 @@ import { useText } from "@/i18n"; import { commonText } from "@/i18n/common.i18n"; import { api } from "./api"; import { chatgptAccountText } from "./ChatgptAccountSection.i18n"; +import { ConcurrencyField } from "./ConnectionSection"; import { coreText, errorText, planLabel, proxyKindLabel, quotaWindowBefore } from "./labels"; import { DialogError, FormItem } from "./parts"; import { QuotaBar } from "./QuotaBar"; @@ -142,6 +143,8 @@ export function ChatgptAccountSection({ ))} + {/* 账号一样限制同时进行的请求 */} + diff --git a/src/upstreams/ConnectionSection.i18n.ts b/src/upstreams/ConnectionSection.i18n.ts index 99d81bbc..59c592e3 100644 --- a/src/upstreams/ConnectionSection.i18n.ts +++ b/src/upstreams/ConnectionSection.i18n.ts @@ -34,6 +34,10 @@ export const connectionSectionText = messages( onProxyFail: "代理不可用时", failWithError: "返回错误", fallBackDirect: "改为直连", + concurrency: "并发上限", + noLimit: "不限", + concurrencyDesc: "同时发往此上游的请求数上限,用于限制并发的中转站或账号。", + badConcurrency: "须为 1 到 1000 之间的整数。", check: "检测连接", checking: "检测中", checkNote: "验证地址与凭据,并获取模型列表。不产生费用。", @@ -84,6 +88,10 @@ export const connectionSectionText = messages( onProxyFail: "When the proxy is unavailable", failWithError: "Return an error", fallBackDirect: "Connect directly", + concurrency: "Concurrency limit", + noLimit: "No limit", + concurrencyDesc: "The most requests sent to this upstream at once, for relays and accounts that limit concurrency.", + badConcurrency: "A whole number from 1 to 1000.", check: "Check connection", checking: "Checking", checkNote: "Verifies the URL and credentials and fetches the model list. No cost is incurred.", diff --git a/src/upstreams/ConnectionSection.tsx b/src/upstreams/ConnectionSection.tsx index 397cdfa8..d3b7dd5e 100644 --- a/src/upstreams/ConnectionSection.tsx +++ b/src/upstreams/ConnectionSection.tsx @@ -32,6 +32,7 @@ import { CHATGPT, ZAI, nameFromUrl, presetById } from "./presets"; import { ServicePicker } from "./ServicePicker"; import { authModeOf, + concurrencyOf, describeModelList, freeName, isBedrock, @@ -281,6 +282,10 @@ export function ConnectionSection({ +
    + +
    +
    + + + + ); +} + +/** + * 空着的格子写什么:价目表给的数(留空就用它)。此刻是手写的,价目表给多少这里不知道, + * 只说留空用价目表;价目表也没有的,说没有 —— 留空就是不知道。 + */ +function placeholderOf( + value: number | null | undefined, + source: SpecSource | null | undefined, + t: { fromTable: (n: string) => string; useTable: string; notInTable: string }, +): string { + if (source === "price_table" && value != null) return t.fromTable(value.toLocaleString()); + if (source === "manual") return t.useTable; + return t.notInTable; +} + +function TokensField({ + id, + label, + value, + onChange, + placeholder, + bad, +}: { + id: string; + label: string; + value: string; + onChange: (v: string) => void; + placeholder: string; + bad: boolean; +}) { + const t = useText(modelSpecDialogText); + return ( + {t.bad} : undefined}> + + onChange(e.target.value)} + /> + + tokens + + + + ); +} diff --git a/src/upstreams/ModelsPanel.i18n.ts b/src/upstreams/ModelsPanel.i18n.ts index d2da5035..117c6dc7 100644 --- a/src/upstreams/ModelsPanel.i18n.ts +++ b/src/upstreams/ModelsPanel.i18n.ts @@ -30,6 +30,12 @@ export const modelsPanelText = messages( aliasMark: (alias: string) => `别名 ${alias}`, aliasTitle: (alias: string) => `别名 ${alias} 列着这个模型:客户端用 ${alias} 请求时可以发往这个模型`, makeAlias: "起别名…", + specs: "规格…", + manualSpecs: "手动规格", + manualContext: (n: string) => `上下文窗口 ${n}`, + manualOutput: (n: string) => `输出上限 ${n}`, + listSep: ",", + manualTitle: (what: string) => `手动设置:${what}`, }, { title: "Models", @@ -64,5 +70,11 @@ export const modelsPanelText = messages( aliasMark: (alias: string) => `alias ${alias}`, aliasTitle: (alias: string) => `Alias ${alias} lists this model: requests for ${alias} can go to it`, makeAlias: "Add alias…", + specs: "Specs…", + manualSpecs: "Manual specs", + manualContext: (n: string) => `context window ${n}`, + manualOutput: (n: string) => `max output ${n}`, + listSep: ", ", + manualTitle: (what: string) => `Set by hand: ${what}`, }, ); diff --git a/src/upstreams/ModelsPanel.tsx b/src/upstreams/ModelsPanel.tsx index e63059bf..b12c7f4d 100644 --- a/src/upstreams/ModelsPanel.tsx +++ b/src/upstreams/ModelsPanel.tsx @@ -3,6 +3,7 @@ import { ChevronRightIcon, CircleAlertIcon, RefreshCwIcon, SearchIcon } from "lu import { AliasMark } from "@/aliases/AliasMark"; import { cn } from "@/lib/utils"; import { useResource } from "@/lib/resource"; +import { Badge } from "@/ui/badge"; import { Button } from "@/ui/button"; import { InputGroup, InputGroupAddon, InputGroupInput } from "@/ui/input-group"; import { Skeleton } from "@/ui/skeleton"; @@ -13,6 +14,7 @@ import { commonText } from "@/i18n/common.i18n"; import type { ModelRow, ProviderModelsView, ProviderView } from "@/types"; import { api } from "./api"; import { contextWindow, coreText, errorText, perMillion } from "./labels"; +import { hasManual } from "./modelSpec"; import { modelsPanelText } from "./ModelsPanel.i18n"; /** 列表长过这个数才给筛选框。十来个一眼就扫完了 */ @@ -31,12 +33,16 @@ const FILTER_FROM = 10; * * 列进了别名的模型,名字后面标出别名;悬停一行给「起别名…」(`onAlias`),打开新建别名的 * 对话框,这个模型已经列为上游模型。 + * + * 右边的数是上下文窗口:在这一家手写过规格(上下文窗口或输出上限)的,名字后面标「手动 + * 规格」,悬停说是哪几项。悬停一行还给「规格…」(`onSpec`),打开手写规格的对话框。 */ export function ModelsPanel({ p, perToken, onEdit, onAlias, + onSpec, }: { p: ProviderView; /** 按量计费:列出单价。别的计费方式不按单价算费用,列了也没意义 */ @@ -45,6 +51,8 @@ export function ModelsPanel({ onEdit: () => void; /** 给这个模型起别名。不给就不出「起别名…」 */ onAlias?: (model: string) => void; + /** 手写这个模型的规格。不给就不出「规格…」 */ + onSpec?: (model: ModelRow) => void; }) { const t = useText(modelsPanelText); const c = useText(commonText); @@ -173,7 +181,7 @@ export function ModelsPanel({ )}
    {shownOn.map((m) => ( - + ))} {off.length > 0 && ( <> @@ -193,7 +201,7 @@ export function ModelsPanel({ {t.notEnabled(off.length)} {(showOff || q !== "") && - shownOff.map((m) => )} + shownOff.map((m) => )} )} {q !== "" && shownOn.length + shownOff.length === 0 && ( @@ -229,21 +237,29 @@ function Row({ m, perToken, onAlias, + onSpec, }: { m: ModelRow; perToken: boolean; onAlias?: (model: string) => void; + onSpec?: (model: ModelRow) => void; }) { const t = useText(modelsPanelText); const price = perToken && m.price ? `$${perMillion(m.price.input)} / $${perMillion(m.price.output)}${m.estimated ? t.estimated : ""}` : null; + const actions = onAlias || onSpec; + /** 手写了哪几项:「上下文窗口 128K」「输出上限 16K」 */ + const manual = [ + m.context_window_source === "manual" ? t.manualContext(contextWindow(m.context_window)) : null, + m.max_output_tokens_source === "manual" ? t.manualOutput(contextWindow(m.max_output_tokens)) : null, + ].filter((x): x is string => x !== null); return (
    @@ -260,11 +276,21 @@ function Row({ {t.aliasMark(a)} ))} - {/* 悬停时这一格让给「起别名…」:两样叠在同一个位置,行高不跳 */} + {/* 放在名字这一边,不放在数旁边:悬停时右边让给按钮,这个标记和它的说明还看得见 */} + {hasManual(m) && ( + + {t.manualSpecs} + + )} + {/* 悬停时这一格让给「规格…」「起别名…」:叠在同一个位置,行高不跳 */} {price && ( @@ -276,15 +302,19 @@ function Row({ {m.context_window ? contextWindow(m.context_window) : ""} - {onAlias && ( - + {actions && ( + + {onSpec && ( + + )} + {onAlias && ( + + )} + )}
    ); diff --git a/src/upstreams/ModelsSection.tsx b/src/upstreams/ModelsSection.tsx index e186a053..0d7e901c 100644 --- a/src/upstreams/ModelsSection.tsx +++ b/src/upstreams/ModelsSection.tsx @@ -38,6 +38,11 @@ export interface ModelCatalog { status?: ModelListStatus; /** core 正在向上游问 */ fetching?: boolean; + /** + * 这一家手写了上下文窗口的模型(`model_specs`)和那个数。手写的优先于价目表,这一节的 + * 上下文窗口一列照它写,和模型弹窗里是同一个数 + */ + manualContext?: Record; } /** core 记下的那一份 */ @@ -49,6 +54,11 @@ export function catalogOf(v: ProviderModelsView): ModelCatalog { error: v.error, status: v.status, fetching: v.fetching, + manualContext: Object.fromEntries( + v.models.flatMap((m) => + m.context_window_source === "manual" && m.context_window != null ? [[m.id, m.context_window]] : [], + ), + ), }; } @@ -250,7 +260,7 @@ export function ModelsSection({ {m} - {contextWindow(price?.max_input_tokens)} + {contextWindow(catalog?.manualContext?.[m] ?? price?.max_input_tokens)} {perToken && ( diff --git a/src/upstreams/UpstreamTable.i18n.ts b/src/upstreams/UpstreamTable.i18n.ts index 83566877..b32e8485 100644 --- a/src/upstreams/UpstreamTable.i18n.ts +++ b/src/upstreams/UpstreamTable.i18n.ts @@ -45,6 +45,7 @@ export const upstreamTableText = messages( cacheLine: (model: string, here: string, hereN: number, k: number, others: string, othersN: number) => `${model}:本上游 ${here}(${hereN.toLocaleString()} 轮)· 其他 ${k} 个上游 ${others}(${othersN.toLocaleString()} 轮)`, inFlight: (n: number) => `${n} 个请求进行中`, + inFlightOf: (n: number, max: number) => `${n} 个请求进行中,并发上限 ${max} 个`, actions: (name: string) => `${name} 的操作`, check: "检测连接", linkTest: "链路测速", @@ -111,6 +112,8 @@ export const upstreamTableText = messages( cacheLine: (model: string, here: string, hereN: number, k: number, others: string, othersN: number) => `${model}: this upstream ${here} (${count(hereN, "turn", "turns")}) · ${count(k, "other upstream", "other upstreams")} ${others} (${count(othersN, "turn", "turns")})`, inFlight: (n: number) => (n === 1 ? "1 request in progress" : `${n} requests in progress`), + inFlightOf: (n: number, max: number) => + `${n === 1 ? "1 request" : `${n} requests`} in progress, concurrency limit ${max}`, actions: (name: string) => `Actions for ${name}`, check: "Check connection", linkTest: "Connection test", diff --git a/src/upstreams/UpstreamTable.tsx b/src/upstreams/UpstreamTable.tsx index 41451532..6d43d93b 100644 --- a/src/upstreams/UpstreamTable.tsx +++ b/src/upstreams/UpstreamTable.tsx @@ -14,7 +14,7 @@ import { resetAt } from "@/format"; import { useNow } from "@/useNow"; import { textOf, useText } from "@/i18n"; import { commonText } from "@/i18n/common.i18n"; -import { usd, type ProviderView, type QuotaWindow, type UpstreamHealth } from "@/types"; +import { usd, type ModelRow, type ProviderView, type QuotaWindow, type UpstreamHealth } from "@/types"; import type { UpstreamStats } from "./api"; import { discrepancies, pct, signedPct, type Discrepancies } from "./checkup"; import { slotsByUpstream, type Slot } from "./data"; @@ -31,6 +31,7 @@ import { } from "./labels"; import { labelsText } from "./labels.i18n"; import { ModelsPanel } from "./ModelsPanel"; +import { ModelSpecDialog } from "./ModelSpecDialog"; import { AliasDialog } from "@/aliases/AliasDialog"; import { ProviderTile, keepInRow, openRow } from "./parts"; import { QUOTA_FULL, QuotaBar } from "./QuotaBar"; @@ -76,6 +77,8 @@ export function UpstreamTable({ inFlight, refreshing, focus, + configVersion, + onChanged, actions, }: { providers: ProviderView[]; @@ -90,6 +93,10 @@ export function UpstreamTable({ refreshing: ReadonlySet; /** 从别的页定位到的那一行:滚进视野、亮一下。`at` 让同一个名字再定位一次也生效 */ focus: { name: string; at: number } | null; + /** 概览里的配置版本:模型弹窗里手写规格时带它 */ + configVersion: string; + /** 在这张表里写了配置(手写模型规格):外面重读概览 */ + onChanged: () => void; actions: UpstreamActions; }) { const t = useText(upstreamTableText); @@ -147,6 +154,8 @@ export function UpstreamTable({ actions.editModels(p.name)} /> @@ -304,7 +313,8 @@ function NameCell({
    {live ? ( - + // 设了并发上限的,连上限一起说:满没满一眼看得出 + {tile} ) : ( @@ -387,12 +397,26 @@ function Where({ p }: { p: ProviderView }) { * **整格是一个按钮,什么状态都能点开** —— 数目单独回答不了「要的那个模型在不在 * 里面」,而没拿到清单时,点开要能看到原因和下一步。停用的上游不提供模型。 */ -function ModelsCell({ p, busy, onEdit }: { p: ProviderView; busy: boolean; onEdit: () => void }) { +function ModelsCell({ + p, + busy, + configVersion, + onChanged, + onEdit, +}: { + p: ProviderView; + busy: boolean; + configVersion: string; + onChanged: () => void; + onEdit: () => void; +}) { const t = useText(upstreamTableText); const l = useText(labelsText); const [open, setOpen] = useState(false); /** 「起别名…」点的那个模型。对话框挂在弹窗外面:弹窗一收起,里面的东西就卸掉了 */ const [aliasFor, setAliasFor] = useState(null); + /** 「规格…」点的那一行,同上 */ + const [specFor, setSpecFor] = useState(null); if (p.disabled) { return —; } @@ -431,9 +455,21 @@ function ModelsCell({ p, busy, onEdit }: { p: ProviderView; busy: boolean; onEdi setOpen(false); setAliasFor(model); }} + onSpec={(row) => { + setOpen(false); + setSpecFor(row); + }} /> + !o && setSpecFor(null)} + provider={p.name} + row={specFor} + configVersion={configVersion} + onSaved={onChanged} + /> !o && setAliasFor(null)} diff --git a/src/upstreams/UpstreamsPage.tsx b/src/upstreams/UpstreamsPage.tsx index e1055eda..63d13c29 100644 --- a/src/upstreams/UpstreamsPage.tsx +++ b/src/upstreams/UpstreamsPage.tsx @@ -424,6 +424,8 @@ export default function UpstreamsPage({ inFlight={inFlight} refreshing={refreshing} focus={focus} + configVersion={configVersion} + onChanged={changed} actions={{ edit: (name) => setDialog({ kind: "upstream", mode: { kind: "edit", name } }), test: (name) => setDialog({ kind: "test", name }), diff --git a/src/upstreams/api.ts b/src/upstreams/api.ts index d34a6d82..c605ef62 100644 --- a/src/upstreams/api.ts +++ b/src/upstreams/api.ts @@ -14,6 +14,7 @@ import type { CostBucketGroup, CostGroup, LatencyView, + ModelSpecSave, TokenRateView, PriceQuery, PriceSheetSave, @@ -55,6 +56,8 @@ export const api = { call("PreviewProvider", { base_url: baseUrl, protocol }), providerModels: (name: string) => call("ProviderModels", null, name), refreshProviderModels: (name: string) => call("RefreshProviderModels", null, name), + /** 手写一个模型的上下文窗口、输出上限。两项都空 = 删掉手写的,回到价目表 */ + setModelSpec: (save: ModelSpecSave) => call("SetModelSpec", save), /** 补问缺失、失败、过期的清单。**立刻回**,答案随 `models_changed` 到 */ refreshStaleModels: () => call("RefreshStaleModels", null), /** 起点和格宽都由界面给:格子对齐到本地整点(见 `bucketStart`) */ diff --git a/src/upstreams/labels.i18n.ts b/src/upstreams/labels.i18n.ts index a61d09ef..ff0df44b 100644 --- a/src/upstreams/labels.i18n.ts +++ b/src/upstreams/labels.i18n.ts @@ -62,6 +62,7 @@ export const labelsText = messages( out_of_scope: "不在启用范围内", not_offered: "未提供此模型", not_allowed: "密钥不允许使用此模型", + busy: "并发已满", }, defaultSheet: "默认价目表", unpriced: "无法计价", @@ -142,6 +143,7 @@ export const labelsText = messages( out_of_scope: "Model not enabled", not_offered: "Model not offered", not_allowed: "Model not allowed for this key", + busy: "At its concurrency limit", }, defaultSheet: "Default price sheet", unpriced: "Unpriced", diff --git a/src/upstreams/labels.ts b/src/upstreams/labels.ts index 460884fe..0efda67d 100644 --- a/src/upstreams/labels.ts +++ b/src/upstreams/labels.ts @@ -356,6 +356,8 @@ export function skipLabel(reason: ServeSkip): string { return t.not_offered; case "not_allowed": return t.not_allowed; + case "busy": + return t.busy; } } diff --git a/src/upstreams/modelSpec.test.ts b/src/upstreams/modelSpec.test.ts new file mode 100644 index 00000000..0b3840e6 --- /dev/null +++ b/src/upstreams/modelSpec.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, it } from "vitest"; +import type { ModelRow } from "@/types"; +import { catalogOf } from "./ModelsSection"; +import { MAX_SPEC_TOKENS, hasManual, manualOf, tokensOf } from "./modelSpec"; + +function row(patch: Partial = {}): ModelRow { + return { id: "glm-5-air", enabled: true, estimated: false, aliases: [], ...patch }; +} + +/** 手写的模型规格:格子里的数怎么认,哪几项是手写的 */ +describe("模型规格", () => { + it("空是不写(用价目表);千分位照常认;0、小数、超出 u32 的不收", () => { + expect(tokensOf("")).toBeNull(); + expect(tokensOf(" ")).toBeNull(); + expect(tokensOf("128000")).toBe(128_000); + expect(tokensOf("128,000")).toBe(128_000); + expect(tokensOf(" 1 000 000 ")).toBe(1_000_000); + expect(tokensOf(String(MAX_SPEC_TOKENS))).toBe(MAX_SPEC_TOKENS); + for (const bad of ["0", "1.5", "-3", "128k", String(MAX_SPEC_TOKENS + 1)]) expect(tokensOf(bad)).toBeUndefined(); + }); + + it("只回填手写的那几项:来自价目表的数不进格子", () => { + const both = row({ + context_window: 1_000_000, + context_window_source: "manual", + max_output_tokens: 64_000, + max_output_tokens_source: "price_table", + }); + expect(manualOf(both)).toEqual({ context: "1000000", output: "" }); + expect(hasManual(both)).toBe(true); + const table = row({ context_window: 200_000, context_window_source: "price_table" }); + expect(manualOf(table)).toEqual({ context: "", output: "" }); + expect(hasManual(table)).toBe(false); + expect(hasManual(row({ max_output_tokens: 16_384, max_output_tokens_source: "manual" }))).toBe(true); + expect(hasManual(row())).toBe(false); + }); + + it("编辑对话框的模型一节:手写的上下文窗口和弹窗里是同一个数", () => { + const c = catalogOf({ + provider: "relay", + source: "discovered", + status: "listed", + fetching: false, + models: [ + row({ id: "a", context_window: 128_000, context_window_source: "manual" }), + row({ id: "b", context_window: 200_000, context_window_source: "price_table" }), + row({ id: "c", max_output_tokens: 8_000, max_output_tokens_source: "manual" }), + ], + }); + expect(c.manualContext).toEqual({ a: 128_000 }); + }); +}); diff --git a/src/upstreams/modelSpec.ts b/src/upstreams/modelSpec.ts new file mode 100644 index 00000000..7e6a433c --- /dev/null +++ b/src/upstreams/modelSpec.ts @@ -0,0 +1,34 @@ +/** + * 手写的模型规格(上下文窗口、输出上限):从 `ModelRow` 读出手写的那几项、格子里的数 + * 写得对不对。**先后只在 core 定**(`tw_config::model_specs::resolve`):这里只认 core + * 标的来源,不自己比大小。 + */ +import type { ModelRow } from "@/types"; + +/** 一项最多写多少:core 存的是 `u32` */ +export const MAX_SPEC_TOKENS = 4_294_967_295; + +/** + * 格子里的 token 数。空是不写(`null`,用价目表的);写的不是 1 到 `u32` 上限的整数是 + * `undefined`。千分位的逗号、空格、下划线照常认:`128,000` 和 `128000` 是同一个数。 + */ +export function tokensOf(v: string): number | null | undefined { + const s = v.replace(/[\s,_]/g, ""); + if (s === "") return null; + if (!/^\d+$/.test(s)) return undefined; + const n = Number(s); + return n >= 1 && n <= MAX_SPEC_TOKENS ? n : undefined; +} + +/** 这个模型此刻手写的两项,格子里的写法。没手写的那一项是空的 */ +export function manualOf(m: ModelRow): { context: string; output: string } { + return { + context: m.context_window_source === "manual" && m.context_window != null ? String(m.context_window) : "", + output: m.max_output_tokens_source === "manual" && m.max_output_tokens != null ? String(m.max_output_tokens) : "", + }; +} + +/** 这个模型在这一家有手写的规格 */ +export function hasManual(m: ModelRow): boolean { + return m.context_window_source === "manual" || m.max_output_tokens_source === "manual"; +} diff --git a/src/upstreams/upstreamForm.i18n.ts b/src/upstreams/upstreamForm.i18n.ts index d98c2382..e2ddc15f 100644 --- a/src/upstreams/upstreamForm.i18n.ts +++ b/src/upstreams/upstreamForm.i18n.ts @@ -10,6 +10,7 @@ export const upstreamFormText = messages( profile: "填写 AWS profile 的名称", headerName: "填写请求头名称", headerValue: (header: string) => `填写请求头「${header}」的值`, + concurrency: "并发上限须为 1 到 1000 之间的整数", pickModel: "至少选择一个模型", found: (n: number) => `发现 ${n} 个模型`, notImplemented: (status: number) => `上游未提供模型列表接口(HTTP ${status})`, @@ -25,6 +26,7 @@ export const upstreamFormText = messages( profile: "Enter the name of the AWS profile", headerName: "Enter a header name", headerValue: (header: string) => `Enter a value for header “${header}”`, + concurrency: "The concurrency limit must be a whole number from 1 to 1000", pickModel: "Select at least one model", found: (n: number) => (n === 1 ? "1 model found" : `${n} models found`), notImplemented: (status: number) => `No model list endpoint (HTTP ${status})`, diff --git a/src/upstreams/upstreamForm.test.ts b/src/upstreams/upstreamForm.test.ts index eb115bc4..c78f7569 100644 --- a/src/upstreams/upstreamForm.test.ts +++ b/src/upstreams/upstreamForm.test.ts @@ -88,6 +88,28 @@ describe("编辑时回填原样", () => { expect(toInput({ ...f, forwardClientIdentity: false }).forward_client_identity).toBe(false); }); + it("并发上限:回填、原样交回;清空就是不限。停用、启用走同一份,不会把它丢掉", () => { + const f = formFromView(view({ max_concurrent: 4 })); + expect(f.maxConcurrent).toBe("4"); + expect(toInput(f).max_concurrent).toBe(4); + expect(toInput({ ...f, maxConcurrent: "" }).max_concurrent).toBeUndefined(); + expect(formFromView(view()).maxConcurrent).toBe(""); + expect(toInput(blankForm()).max_concurrent).toBeUndefined(); + }); + + it("并发上限只收 1 到 1000 的整数,写错了保存不了", () => { + const f = formFromView(view()); + for (const ok of ["1", "1000", " 12 "]) { + expect(connectionMissing({ ...f, maxConcurrent: ok }, "relay", ["relay"])).toBeNull(); + } + for (const bad of ["0", "1001", "2.5", "-1", "abc"]) { + expect(connectionMissing({ ...f, maxConcurrent: bad }, "relay", ["relay"])).toBe( + "并发上限须为 1 到 1000 之间的整数", + ); + expect(toInput({ ...f, maxConcurrent: bad }).max_concurrent).toBeUndefined(); + } + }); + it("清空密钥就是不要密钥;清空请求头的值要补上", () => { const f = formFromView(view()); expect(toInput({ ...f, key: " " }).key).toBeUndefined(); diff --git a/src/upstreams/upstreamForm.ts b/src/upstreams/upstreamForm.ts index 340a17d2..9c503ac2 100644 --- a/src/upstreams/upstreamForm.ts +++ b/src/upstreams/upstreamForm.ts @@ -89,6 +89,8 @@ export interface UpstreamForm { billing: Billing; /** 空 = 默认价目表 */ pricing: string; + /** 同时最多发给这家几个请求,1 到 1000。空 = 不限 */ + maxConcurrent: string; disabled: boolean; } @@ -133,6 +135,7 @@ export function blankForm(): UpstreamForm { scopeList: [], billing: "per-token", pricing: "", + maxConcurrent: "", disabled: false, }; } @@ -198,6 +201,7 @@ export function formFromView(p: ProviderView): UpstreamForm { scopeList: p.models_only ?? [], billing: p.billing === "free" ? "free" : "per-token", pricing: p.pricing ?? "", + maxConcurrent: p.max_concurrent != null ? String(p.max_concurrent) : "", disabled: p.disabled, }; } @@ -283,10 +287,23 @@ export function toInput(f: UpstreamForm): ProviderInput { models_only: f.scope === "some" ? f.scopeList : undefined, billing: f.billing, pricing: f.pricing || undefined, + max_concurrent: concurrencyOf(f) ?? undefined, disabled: f.disabled, }; } +/** 并发上限最多写多少(core 的 `MAX_PROVIDER_CONCURRENCY`) */ +export const MAX_CONCURRENCY = 1000; + +/** 格子里的并发上限:空是不限(`null`),写的不是 1 到 1000 的整数是 `undefined` */ +export function concurrencyOf(f: UpstreamForm): number | null | undefined { + const v = f.maxConcurrent.trim(); + if (v === "") return null; + if (!/^\d+$/.test(v)) return undefined; + const n = Number(v); + return n >= 1 && n <= MAX_CONCURRENCY ? n : undefined; +} + /** * 连接信息和已保存的那一家比改过没有:地址、协议、凭据、请求头、出站代理。 * 改过的话,模型列表要按表单里的新值去问。 @@ -327,6 +344,7 @@ export function connectionMissing( if (header === "") return t.headerName; if (r.value.trim() === "") return t.headerValue(header); } + if (concurrencyOf(f) === undefined) return t.concurrency; return null; }