From 73735849f0615c1a1f47b4002fb5cd9f19d03a3b Mon Sep 17 00:00:00 2001 From: fylorn <249551762+fylorn@users.noreply.github.com> Date: Mon, 5 Oct 2026 17:20:57 +0800 Subject: [PATCH 01/10] chore(core): build on core's unreleased routing round (rev 3e7841e) The next round of Lite work (load-balance weights and balance_by, switching on a slow stream start, per-upstream concurrency caps, hand-set model specs, per-key usage limits) needs the protocol from core's integ/routing-2026-10, which has no tag yet. The six core crates are pinned to that commit by `rev` until core is released; the lock file changes only in their source lines, since the crates' versions and dependencies did not move. Types: tw-api.ts is regenerated. The new required fields (ClientView.limits, GroupView.weights/balance_by, FailoverView.next_on_slow_start/slot_wait_secs) are filled in test fixtures and in the notices snapshot with the values core now reports for an unchanged configuration (no limits, weight 1, balance by weights, switching off, 30 s slot wait), so every test still describes today's behaviour. The failover settings form keeps its seven fields: the two new ones are excluded from its field type until that section is designed. The new `busy` skip reason gets a label so the exhaustive switch keeps compiling. Control plane: `SetModelSpec` (PUT /provider-model-spec) is added to the endpoints the interface may call, in call.rs, control.ts and the screenshot mock, so the upcoming model-spec UI does not have to touch the allowlist. Messages: core adds 30 codes; each has a Chinese sentence now, so the Chinese interface does not fall back to English for slow-start switches, busy upstreams, key limit refusals or the new validation errors. Core fills `per` and `measure` with English words (day, tokens) and expects the interface to look them up, so three small word tables are added; new template cases run the table lookups through both the TypeScript and the Rust implementation. Screenshot data: the stored core answers are regenerated with oracle.sh, which now also reads a `rev` pin, so the shots mock still type-checks against the core types. Co-Authored-By: Claude Opus 5.5 --- scripts/shots/core/en/keys.json | 4 + scripts/shots/core/en/overview.json | 18 +- scripts/shots/core/en/status.json | 2 +- scripts/shots/core/oracle.sh | 9 +- scripts/shots/core/zh/keys.json | 4 + scripts/shots/core/zh/overview.json | 18 +- scripts/shots/core/zh/status.json | 2 +- scripts/shots/mock/core.ts | 1 + src-tauri/Cargo.lock | 14 +- src-tauri/Cargo.toml | 12 +- src-tauri/src/call.rs | 2 + src-tauri/src/clients/desktop_rule.rs | 7 +- src-tauri/src/clients/ops.rs | 2 + src-tauri/src/notices/fixtures/snapshot.json | 11 +- src-tauri/src/notices/tests.rs | 3 + src-tauri/tests/control_plane.rs | 1 + src/control.ts | 1 + src/generated/tw-api.ts | 249 +++++++++++++++++-- src/i18n/core.zh.cases.json | 63 +++++ src/i18n/core.zh.json | 48 ++++ src/routing/chain.test.ts | 3 + src/routing/flights.test.ts | 3 + src/routing/model.test.ts | 4 +- src/settings/FailoverSection.tsx | 3 +- src/settings/failover.test.ts | 2 + src/upstreams/labels.i18n.ts | 2 + src/upstreams/labels.ts | 2 + 27 files changed, 442 insertions(+), 48 deletions(-) diff --git a/scripts/shots/core/en/keys.json b/scripts/shots/core/en/keys.json index d53cbaee..bbaf468f 100644 --- a/scripts/shots/core/en/keys.json +++ b/scripts/shots/core/en/keys.json @@ -3,6 +3,7 @@ "allow": null, "default": true, "key": "tw-m9EXAMPLEj7ak", + "limits": [], "max_concurrent": null, "name": "default", "route": null @@ -11,6 +12,7 @@ "allow": null, "client": "claude-code", "key": "tw-5aEXAMPLEqmg8", + "limits": [], "max_concurrent": null, "name": "claude-code", "route": null @@ -19,6 +21,7 @@ "allow": null, "client": "codex", "key": "tw-h5EXAMPLEyrqn", + "limits": [], "max_concurrent": 4, "name": "codex", "route": "codex" @@ -30,6 +33,7 @@ ], "client": "cursor", "key": "tw-e6EXAMPLE6td6", + "limits": [], "max_concurrent": null, "name": "cursor", "route": "cursor" diff --git a/scripts/shots/core/en/overview.json b/scripts/shots/core/en/overview.json index 158330cf..f80c2019 100644 --- a/scripts/shots/core/en/overview.json +++ b/scripts/shots/core/en/overview.json @@ -26,6 +26,7 @@ "allow": null, "default": true, "key": "tw-m9…j7ak", + "limits": [], "max_concurrent": null, "name": "default", "route": null @@ -34,6 +35,7 @@ "allow": null, "client": "claude-code", "key": "tw-5a…qmg8", + "limits": [], "max_concurrent": null, "name": "claude-code", "route": null @@ -42,6 +44,7 @@ "allow": null, "client": "codex", "key": "tw-h5…yrqn", + "limits": [], "max_concurrent": 4, "name": "codex", "route": "codex" @@ -53,6 +56,7 @@ ], "client": "cursor", "key": "tw-e6…6td6", + "limits": [], "max_concurrent": null, "name": "cursor", "route": "cursor" @@ -63,32 +67,39 @@ "failover": { "failures_to_pause": 3, "max_pause_secs": 600, + "next_on_slow_start": false, "no_balance_pause_secs": 1800, "pause_secs": 60, "quota_pause_secs": 3600, "rate_limit_max_pause_secs": 3600, + "slot_wait_secs": 30, "stream_start_wait_secs": 15 }, "groups": [ { + "balance_by": "weights", "builtin": false, "kind": "fallback", "name": "main", "providers": [ "anthropic", "relay" - ] + ], + "weights": {} }, { + "balance_by": "weights", "builtin": false, "kind": "cheapest", "name": "budget", "providers": [ "anthropic", "relay" - ] + ], + "weights": {} }, { + "balance_by": "weights", "builtin": true, "kind": "fallback", "name": "__all__", @@ -100,7 +111,8 @@ "deepseek", "gemini", "ollama" - ] + ], + "weights": {} } ], "listen": { diff --git a/scripts/shots/core/en/status.json b/scripts/shots/core/en/status.json index 6d6548f8..a5b7f59b 100644 --- a/scripts/shots/core/en/status.json +++ b/scripts/shots/core/en/status.json @@ -14,5 +14,5 @@ "reachable": [] }, "uptime_secs": 0, - "version": "0.61.0" + "version": "0.62.0" } diff --git a/scripts/shots/core/oracle.sh b/scripts/shots/core/oracle.sh index 9173ad43..255f4192 100755 --- a/scripts/shots/core/oracle.sh +++ b/scripts/shots/core/oracle.sh @@ -8,15 +8,16 @@ # 钉住的 core 之后。截图页的配置类数据(上游、密钥、路由、安全规则、试算) # 直接用这几份文件,所以它们必须是 core 真的答出来的,不是照着样子手写的。 # -# 用的是 src-tauri/Cargo.toml 钉住的那个 tag:从检出里 `git archive` 一份到临时 -# 目录,放进 oracle.rs 跑一次。**不改那个检出。**要先在那边 `git fetch --tags`。 +# 用的是 src-tauri/Cargo.toml 钉住的那个 tag(core 还没发版时临时钉的 rev 也认):从检出里 +# `git archive` 一份到临时目录,放进 oracle.rs 跑一次。**不改那个检出。**要先在那边 +# `git fetch --tags`。 set -euo pipefail core=${1:?用法:oracle.sh } here=$(cd "$(dirname "$0")" && pwd) root=$(cd "$here/../../.." && pwd) -tag=$(sed -n 's/^tw-api = .*tag = "\([^"]*\)".*/\1/p' "$root/src-tauri/Cargo.toml") -[ -n "$tag" ] || { echo "src-tauri/Cargo.toml 里找不到 tw-api 的 tag" >&2; exit 1; } +tag=$(sed -nE 's/^tw-api = .*(tag|rev) = "([^"]*)".*/\2/p' "$root/src-tauri/Cargo.toml") +[ -n "$tag" ] || { echo "src-tauri/Cargo.toml 里找不到 tw-api 的 tag 或 rev" >&2; exit 1; } src=$(mktemp -d) trap 'rm -rf "$src"' EXIT diff --git a/scripts/shots/core/zh/keys.json b/scripts/shots/core/zh/keys.json index d53cbaee..bbaf468f 100644 --- a/scripts/shots/core/zh/keys.json +++ b/scripts/shots/core/zh/keys.json @@ -3,6 +3,7 @@ "allow": null, "default": true, "key": "tw-m9EXAMPLEj7ak", + "limits": [], "max_concurrent": null, "name": "default", "route": null @@ -11,6 +12,7 @@ "allow": null, "client": "claude-code", "key": "tw-5aEXAMPLEqmg8", + "limits": [], "max_concurrent": null, "name": "claude-code", "route": null @@ -19,6 +21,7 @@ "allow": null, "client": "codex", "key": "tw-h5EXAMPLEyrqn", + "limits": [], "max_concurrent": 4, "name": "codex", "route": "codex" @@ -30,6 +33,7 @@ ], "client": "cursor", "key": "tw-e6EXAMPLE6td6", + "limits": [], "max_concurrent": null, "name": "cursor", "route": "cursor" diff --git a/scripts/shots/core/zh/overview.json b/scripts/shots/core/zh/overview.json index 8665b2eb..f64795d9 100644 --- a/scripts/shots/core/zh/overview.json +++ b/scripts/shots/core/zh/overview.json @@ -26,6 +26,7 @@ "allow": null, "default": true, "key": "tw-m9…j7ak", + "limits": [], "max_concurrent": null, "name": "default", "route": null @@ -34,6 +35,7 @@ "allow": null, "client": "claude-code", "key": "tw-5a…qmg8", + "limits": [], "max_concurrent": null, "name": "claude-code", "route": null @@ -42,6 +44,7 @@ "allow": null, "client": "codex", "key": "tw-h5…yrqn", + "limits": [], "max_concurrent": 4, "name": "codex", "route": "codex" @@ -53,6 +56,7 @@ ], "client": "cursor", "key": "tw-e6…6td6", + "limits": [], "max_concurrent": null, "name": "cursor", "route": "cursor" @@ -63,32 +67,39 @@ "failover": { "failures_to_pause": 3, "max_pause_secs": 600, + "next_on_slow_start": false, "no_balance_pause_secs": 1800, "pause_secs": 60, "quota_pause_secs": 3600, "rate_limit_max_pause_secs": 3600, + "slot_wait_secs": 30, "stream_start_wait_secs": 15 }, "groups": [ { + "balance_by": "weights", "builtin": false, "kind": "fallback", "name": "主力", "providers": [ "anthropic", "relay" - ] + ], + "weights": {} }, { + "balance_by": "weights", "builtin": false, "kind": "cheapest", "name": "低价", "providers": [ "anthropic", "relay" - ] + ], + "weights": {} }, { + "balance_by": "weights", "builtin": true, "kind": "fallback", "name": "__all__", @@ -100,7 +111,8 @@ "deepseek", "gemini", "ollama" - ] + ], + "weights": {} } ], "listen": { diff --git a/scripts/shots/core/zh/status.json b/scripts/shots/core/zh/status.json index 6d6548f8..a5b7f59b 100644 --- a/scripts/shots/core/zh/status.json +++ b/scripts/shots/core/zh/status.json @@ -14,5 +14,5 @@ "reachable": [] }, "uptime_secs": 0, - "version": "0.61.0" + "version": "0.62.0" } diff --git a/scripts/shots/mock/core.ts b/scripts/shots/mock/core.ts index 31105ea0..ac542204 100644 --- a/scripts/shots/mock/core.ts +++ b/scripts/shots/mock/core.ts @@ -121,6 +121,7 @@ export const CORE: { [N in WebviewEndpoint]: Handler } = { UpdateProvider: refuse, DeleteProvider: refuse, ProviderModels: (_req, [name]) => providerModels(name!) ?? notFound(`Upstream ${name}`), + SetModelSpec: refuse, RefreshProviderModels: (_req, [name]) => providerModels(name!) ?? notFound(`Upstream ${name}`), RefreshStaleModels: () => ({ providers: [] }), diff --git a/src-tauri/Cargo.lock b/src-tauri/Cargo.lock index dffa0c2c..c4bf1cdd 100644 --- a/src-tauri/Cargo.lock +++ b/src-tauri/Cargo.lock @@ -5269,7 +5269,7 @@ dependencies = [ [[package]] name = "tw-api" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" dependencies = [ "serde", "serde_json", @@ -5281,7 +5281,7 @@ dependencies = [ [[package]] name = "tw-dialect" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" dependencies = [ "serde", "serde_json", @@ -5290,7 +5290,7 @@ dependencies = [ [[package]] name = "tw-guard" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" dependencies = [ "base64 0.22.1", "bytes", @@ -5306,7 +5306,7 @@ dependencies = [ [[package]] name = "tw-link" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" dependencies = [ "serde", "serde_json", @@ -5333,7 +5333,7 @@ dependencies = [ [[package]] name = "tw-types" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" dependencies = [ "serde", "ts-rs", @@ -5342,7 +5342,7 @@ dependencies = [ [[package]] name = "tw-watch" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" dependencies = [ "notify", "thiserror 2.0.21", @@ -5352,7 +5352,7 @@ dependencies = [ [[package]] name = "tw-yaml" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.62.0#51ef803bee644e72668cb8725d7bc27f6bb77a3f" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" dependencies = [ "saphyr-parser", "thiserror 2.0.21", diff --git a/src-tauri/Cargo.toml b/src-tauri/Cargo.toml index 05be5b2e..e8e17b5d 100644 --- a/src-tauri/Cargo.toml +++ b/src-tauri/Cargo.toml @@ -26,12 +26,12 @@ license = "MIT" # twcore 二进制必须和这里编译进去的协议镜像来自同一个 core 版本 —— 打包脚本正是 # 从 tw-api 锁到的 tag 去取二进制的(见 scripts/fetch-core.sh)。 [workspace.dependencies] -tw-api = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } -tw-types = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } -tw-yaml = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } -tw-guard = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } -tw-watch = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } -tw-link = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.62.0" } +tw-api = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } +tw-types = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } +tw-yaml = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } +tw-guard = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } +tw-watch = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } +tw-link = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } [lib] name = "thinkwatch_lite_lib" diff --git a/src-tauri/src/call.rs b/src-tauri/src/call.rs index afa29d7f..7befdab4 100644 --- a/src-tauri/src/call.rs +++ b/src-tauri/src/call.rs @@ -97,6 +97,8 @@ webview_endpoints![ UpdateProvider, DeleteProvider, ProviderModels, + // 手写一个模型的上下文窗口、输出上限 + SetModelSpec, RefreshProviderModels, RefreshStaleModels, CreateProxy, diff --git a/src-tauri/src/clients/desktop_rule.rs b/src-tauri/src/clients/desktop_rule.rs index a781a23e..3a1fd3be 100644 --- a/src-tauri/src/clients/desktop_rule.rs +++ b/src-tauri/src/clients/desktop_rule.rs @@ -477,7 +477,8 @@ mod tests { "security": {"redact": "observe", "inspect_tools": "observe", "content": "observe"}, "retention": {"body_days": 7, "row_days": 90, "body_max_bytes": 0, "body_bytes_now": 0}, "failover": {"failures_to_pause": 3, "pause_secs": 60, "max_pause_secs": 600, "no_balance_pause_secs": 1800, - "quota_pause_secs": 3600, "rate_limit_max_pause_secs": 3600, "stream_start_wait_secs": 15} + "quota_pause_secs": 3600, "rate_limit_max_pause_secs": 3600, "stream_start_wait_secs": 15, + "next_on_slow_start": false, "slot_wait_secs": 30} }) } @@ -506,8 +507,8 @@ mod tests { "rules": [{"name": "rest", "conditions": [], "to": "chatgpt", "catch_all": true, "phase_two": false, "shadowed": false}]} ]), serde_json::json!([ - {"name": "claude-desktop", "key": "tw-x", "client": "claude-desktop", "disabled": false, "default": false}, - {"name": "cd-2", "key": "tw-y", "client": "claude-desktop@Ubuntu", "route": "codex", "disabled": false, "default": false} + {"name": "claude-desktop", "key": "tw-x", "client": "claude-desktop", "disabled": false, "default": false, "limits": []}, + {"name": "cd-2", "key": "tw-y", "client": "claude-desktop@Ubuntu", "route": "codex", "disabled": false, "default": false, "limits": []} ]), ); Snapshot { diff --git a/src-tauri/src/clients/ops.rs b/src-tauri/src/clients/ops.rs index 6a6a4259..71af8053 100644 --- a/src-tauri/src/clients/ops.rs +++ b/src-tauri/src/clients/ops.rs @@ -981,6 +981,8 @@ pub(crate) mod tests { disabled: false, default, last_seen_ms: None, + limits: vec![], + unpriced_models: vec![], } } diff --git a/src-tauri/src/notices/fixtures/snapshot.json b/src-tauri/src/notices/fixtures/snapshot.json index 1b5e5d94..eddfc802 100644 --- a/src-tauri/src/notices/fixtures/snapshot.json +++ b/src-tauri/src/notices/fixtures/snapshot.json @@ -101,7 +101,9 @@ "kind": "fallback", "providers": [ "官方" - ] + ], + "weights": {}, + "balance_by": "weights" } ], "clients": [ @@ -111,7 +113,8 @@ "max_concurrent": null, "route": null, "allow": null, - "default": true + "default": true, + "limits": [] } ], "listen": { @@ -172,7 +175,9 @@ "no_balance_pause_secs": 1800, "quota_pause_secs": 3600, "rate_limit_max_pause_secs": 3600, - "stream_start_wait_secs": 15 + "stream_start_wait_secs": 15, + "next_on_slow_start": false, + "slot_wait_secs": 30 }, "price_sheets": [] }, diff --git a/src-tauri/src/notices/tests.rs b/src-tauri/src/notices/tests.rs index 3a93f012..71dc8e3a 100644 --- a/src-tauri/src/notices/tests.rs +++ b/src-tauri/src/notices/tests.rs @@ -646,6 +646,9 @@ async fn an_unreachable_upstream_is_only_listed_and_needs_real_evidence_to_clear status: Some(200), error: None, ms: 800, + usage: None, + queued_ms: None, + skipped: None, }], billing: tw_api::Billing::PerToken, }); diff --git a/src-tauri/tests/control_plane.rs b/src-tauri/tests/control_plane.rs index e08a626b..f4d35907 100644 --- a/src-tauri/tests/control_plane.rs +++ b/src-tauri/tests/control_plane.rs @@ -390,6 +390,7 @@ async fn a_bedrock_upstream_is_saved_the_way_the_dialog_sends_it() { models_only: None, billing: None, pricing: None, + max_concurrent: None, disabled: false, }, base_version: None, diff --git a/src/control.ts b/src/control.ts index ef4084f7..935cb219 100644 --- a/src/control.ts +++ b/src/control.ts @@ -59,6 +59,7 @@ export const WEBVIEW_ENDPOINTS = [ "UpdateProvider", "DeleteProvider", "ProviderModels", + "SetModelSpec", "RefreshProviderModels", "RefreshStaleModels", "CreateProxy", diff --git a/src/generated/tw-api.ts b/src/generated/tw-api.ts index 9dbdd002..6be60a99 100644 --- a/src/generated/tw-api.ts +++ b/src/generated/tw-api.ts @@ -174,7 +174,7 @@ served_by: Array, */ shadows: Array, /** - * 上下文窗口,来自默认价目表:第一家能服务它的上游发出的那个模型的 + * 上下文窗口:第一家能服务它的上游发出的那个模型的,这一家手写的优先于价目表 */ context_window?: number | null, /** @@ -219,7 +219,26 @@ suggestions: Array, }; /** * 尝试链里一跳的结果。 */ -export type AttemptOutcome = "served" | "status" | "error" | "estimated"; +export type AttemptOutcome = "served" | "status" | "error" | "estimated" | "slow_start"; + +/** + * 放弃了的一跳([`AttemptOutcome::SlowStart`])上游可能已经收了钱的输入。 + * + * 上游在流开头报了的(Anthropic 的 `message_start`)是它报的数;没报的只有 `input`,是网关 + * 估的(`estimated`,和 [`Event::RequestStarted`] 的 `input_estimate` 同一个数)。**输出不知道**: + * 先想好再输出的模型,放弃之前可能已经想了一阵,上游不说就看不到。 + * + * **不算进这个请求的费用**:上游收没收、收了多少,网关看不到 + */ +export type AttemptUsage = { +/** + * 输入 token,不含缓存读写 + */ +input: number, cache_read: number, cache_write: number, +/** + * `input` 是网关估的,上游什么都没报 + */ +estimated: boolean, }; /** * 尝试链里的一跳。 @@ -246,9 +265,24 @@ outcome: AttemptOutcome, */ status?: number | null, /** - * `error` 时的说明。和这一跳报给客户端的那条错误是同一句 + * `error` 时的说明。和这一跳报给客户端的那条错误是同一句。`slow_start` 时说等了多久 + */ +error?: Msg | null, ms: number, +/** + * 放弃了的这一跳(`slow_start`)上游可能已经收了钱的输入(见 [`AttemptUsage`])。估不 + * 出来的(请求解不开)没有。别的结果都没有:接下请求的那一跳的用量在结局里 + */ +usage?: AttemptUsage | null, +/** + * 这一跳等了多少毫秒才轮到一个空位:这家设了 `max_concurrent` 而它满着。不算在 `ms` + * 里。没等的没有 + */ +queued_ms?: number | null, +/** + * 这一跳为什么没发出去:`busy`(这家满着,换了下一家;等过它的话 `queued_ms` 是等了 + * 多久)。这时 `outcome` 是 `error`,`error` 是同一件事的那句话。发出去了的没有 */ -error?: Msg | null, ms: number, }; +skipped?: ServeSkip | null, }; /** * 上游接不接受凭据。 @@ -290,6 +324,15 @@ profile?: string | null, */ region?: string | null, }; +/** + * `load-balance` 组按什么分新对话:配置里 `balance_by` 写的那个词。 + * + * 成员的权重永远是底数,快慢、成败算出的系数乘在上面 + * ([`DryRunCandidate::balance_factor`]);进行中的对话照旧留在回答它的那一家。 + * 没有测到的上游算中等。 + */ +export type BalanceBy = "weights" | "latency" | "health" | "latency-health"; + /** * 删除时带上的版本。 */ @@ -550,7 +593,16 @@ default?: boolean, * 最后一次被用在什么时候。**按密钥算,不是按客户端自报的标识** —— * 那个可以伪造。从来没被用过时没有 */ -last_seen_ms?: number | null, }; +last_seen_ms?: number | null, +/** + * 用量上限,按配置里的顺序,各带此刻用了多少。没设的是空的 + */ +limits: Array, +/** + * 这把密钥用得到、却没有价格的模型。**只有设了费用上限的密钥才算**:这些模型的 + * 请求费用记 0,费用上限管不住它们,对话框里要提醒一句。没有就不带 + */ +unpriced_models?: Array, }; /** * 路由规则 `when` 里的键。 @@ -894,7 +946,26 @@ sent_model?: string | null, * 发出的名字为什么和请求里写的不一样:`alias`(别名对到这一家的名称)、`rule`(规则 * 改写了模型)、`pinned`(规则指定了这一家发什么模型)。一样时没有 */ -model_via?: string | null, }; +model_via?: string | null, +/** + * 经过的是 `load-balance` 组时,它在组里的权重(没写权重的是 1)。别的时候没有 + */ +weight?: number | null, +/** + * 它的典型首字节时间(最近样本的中位数),毫秒。只在顺序看它时有:`url-test`, + * 按快慢分的 `load-balance`。样本不够时没有 + */ +ttfb_ms?: number | null, +/** + * 它最近的成功率,0 到 1(最近 50 次、30 分钟以内)。只在按成败分的 `load-balance` + * 里有;不到 5 次时没有 + */ +success_rate?: number | null, +/** + * 按快慢、成败算出的系数,乘在权重([`Self::weight`])上:大于 1 分得多,小于 1 + * 分得少,没有样本的那一项算 1。只在 `balance_by` 不是 `weights` 的 `load-balance` 里有 + */ +balance_factor?: number | null, }; /** * 一次试算的结论。 @@ -944,6 +1015,10 @@ route: string, * 写在第一个」。直指 provider 时是 None。 */ strategy?: GroupKind | null, +/** + * 经过的是 `load-balance` 组时,它按什么分新对话。别的时候没有 + */ +balance_by?: BalanceBy | null, /** * `route` | `deny` | `no_match` | `unavailable`(选中的上游都服务不了, * 见 `skipped`)| `intercepted` @@ -1333,7 +1408,23 @@ window: string, /** * 什么时候重置(见 `QuotaWindow::resets_at_ms`)。上游没说就没有 */ -resets_at_ms?: number | null, at_ms: number, } | { "kind": "listen_changed", id: number, +resets_at_ms?: number | null, at_ms: number, } | { "kind": "key_limit_alert", id: number, +/** + * 密钥的名字 + */ +key: string, per: LimitPer, measure: LimitMeasure, +/** + * 上限,单位同 `KeyLimitView::max` + */ +max: number, +/** + * 报的时候用了多少,同上 + */ +used: number, cache_reads?: boolean, +/** + * `true` = 到了上限,之后的请求被拒到 `resets_at_ms`;`false` = 到了八成 + */ +reached: boolean, resets_at_ms: number, at_ms: number, } | { "kind": "listen_changed", id: number, /** * 此刻在听的那个地址,同 `Status::gateway_addr` */ @@ -1431,7 +1522,15 @@ rate_limit_max_pause_secs: number, /** * 流式回答的开头最多等多少秒 */ -stream_start_wait_secs: number, }; +stream_start_wait_secs: number, +/** + * 等过 `stream_start_wait_secs` 还没有内容就换下一家(最后一家照常等) + */ +next_on_slow_start: boolean, +/** + * 上游满着(`max_concurrent`)时,一个请求合计最多等多少秒空位。0 是不等 + */ +slot_wait_secs: number, }; /** * 一个请求失败在哪一方。和 HTTP 响应里的 `x-thinkwatch-error` 同一个词表, @@ -1450,7 +1549,16 @@ providers: Array, /** * `select` 组优先使用的成员 */ -selected?: string | null, }; +selected?: string | null, +/** + * `load-balance` 组成员的权重,1 到 100。不给 = 都是 1;给了的话没写到的成员是 1。 + * 别的类型只能不给、或者都是 1 + */ +weights?: { [key in string]: number } | null, +/** + * `load-balance` 组按什么分新对话。不给 = `weights`;别的类型只能是 `weights` + */ +balance_by?: BalanceBy | null, }; /** * 策略组按什么排候选:配置里 `type` 写的那个词。 @@ -1484,7 +1592,16 @@ kind: GroupKind, * **界面要能切它** —— 这个策略本身就是「UI 上点选或托盘里切」, * 而切不了的话它等于一个只能改 YAML 才能用的功能。 */ -selected?: string | null, providers: Array, }; +selected?: string | null, providers: Array, +/** + * `load-balance` 组每个成员的权重,**每个成员都在**,没写权重的是 1:新对话按这个 + * 比例分。别的类型不用权重,是空的 + */ +weights: { [key in string]: number }, +/** + * `load-balance` 按什么分新对话。别的类型永远是 `weights` + */ +balance_by: BalanceBy, }; /** * 哪一项防护。配置里 `security` 下的那个键,也是管理接口路径里的那一段。 @@ -1818,7 +1935,51 @@ route?: string | null, /** * 三态:不写 / 写非空 / 写 `[]`(一个都不给) */ -allow?: Array | null, disabled?: boolean, }; +allow?: Array | null, disabled?: boolean, +/** + * 用量上限,整份替换。不带 = 一条都没有 + */ +limits?: Array, }; + +/** + * 新建、保存密钥时的一条用量上限。 + */ +export type KeyLimitInput = { per: LimitPer, measure: LimitMeasure, +/** + * 上限:请求数、token 数,费用是微分。要大于 0 + */ +max: number, +/** + * 只有 token 上限能开 + */ +cache_reads?: boolean, }; + +/** + * 一条用量上限,和它此刻用了多少。 + */ +export type KeyLimitView = { per: LimitPer, measure: LimitMeasure, +/** + * 上限:请求数、token 数,费用是微分 + */ +max: number, +/** + * token 上限把从缓存读的也算进去 + */ +cache_reads: boolean, +/** + * 用了多少,单位同 `max`。**在跑的请求也算**:按它们的输入估算占着,结束时换成 + * 记下的实数 —— 准入看的就是这个数。滚动的是最近那一段时间里的,重启之后从空的 + * 开始;自然的是这一期的,重启之后从请求记录里加回来 + */ +used: number, +/** + * 这一期什么时候结束、重新算。只有天、周、月有 + */ +resets_at_ms?: number | null, +/** + * 到了:`used` 不小于 `max`,新的请求此刻会被拒(滚动的会先等一会儿) + */ +reached: boolean, }; /** * 换哪把密钥(`POST /keys/{name}/rotate`)。 @@ -1949,6 +2110,17 @@ export type LatencyView = { model: string, p50: number, p95: number, */ samples: number, }; +/** + * 一条用量上限数的是什么。 + */ +export type LimitMeasure = "requests" | "tokens" | "cost"; + +/** + * 用量上限按多长一段时间算。分钟、小时是**滚动的**(最近 60 秒、最近 60 分钟); + * 天、周、月是**自然的**,按 core 所在机器的本地时区:零点、周一零点、一号零点重新算。 + */ +export type LimitPer = "minute" | "hour" | "day" | "week" | "month"; + /** * 一张列表要的两样:看哪一段,最多几条(`GET /history`、`/sessions`)。 * @@ -2138,9 +2310,21 @@ export type ModelRow = { id: string, */ enabled: boolean, /** - * 上下文窗口,来自默认价目表 + * 上下文窗口:这一家手写的(`model_specs`),没写时来自价目表 */ context_window?: number | null, +/** + * `context_window` 从哪儿来。不知道上下文窗口时没有 + */ +context_window_source?: SpecSource | null, +/** + * 一次最多输出多少 token:这一家手写的,没写时来自价目表 + */ +max_output_tokens?: number | null, +/** + * `max_output_tokens` 从哪儿来。不知道输出上限时没有 + */ +max_output_tokens_source?: SpecSource | null, /** * 按这个上游选的价目表查到的价格。空 = 无法计价 */ @@ -2160,6 +2344,28 @@ aliases: Array, }; */ export type ModelSource = "discovered" | "manual" | "none"; +/** + * 设一家上游的一个模型的规格(`PUT /provider-model-spec`):价目表不认识这个模型、 + * 或者写错了时手写。**两项都空就是删掉这一项**,回到价目表。 + */ +export type ModelSpecSave = { provider: string, +/** + * 模型 ID,和这家的清单里写的完全相等。去掉首尾空白 + */ +model: string, +/** + * 上下文窗口(token)。空 = 用价目表的 + */ +context_window?: number | null, +/** + * 输出上限(token)。空 = 用价目表的 + */ +max_output_tokens?: number | null, +/** + * 你基于哪一版。**对不上就是 409** + */ +base_version?: string | null, }; + /** * 页面打开时补问模型清单:开始问的是哪几家。答案随 `models_changed` 到。 */ @@ -2843,6 +3049,10 @@ billing?: Billing | null, * 按哪张价目表计价。不给就是默认价目表 */ pricing?: string | null, +/** + * 同时最多发给这家几个请求,1 到 1000。不给就是不限 + */ +max_concurrent?: number | null, /** * 停用 */ @@ -3057,7 +3267,11 @@ references: Array, /** * 选的价目表。空 = 默认价目表 */ -pricing: string | null, }; +pricing: string | null, +/** + * 同时最多发给这家几个请求。不限是空 + */ +max_concurrent?: number | null, }; /** * 代理的用户名和密码。 @@ -4014,7 +4228,7 @@ export type SecurityView = { redact: GuardMode, inspect_tools: GuardMode, conten /** * 一个上游为什么服务不了这个模型。 */ -export type ServeSkip = "disabled" | "out_of_scope" | "not_offered" | "not_allowed"; +export type ServeSkip = "disabled" | "out_of_scope" | "not_offered" | "not_allowed" | "busy"; export type SessionDetail = { session: SessionView, turns: Array, }; @@ -4106,6 +4320,11 @@ export type SkippedView = { provider: string, */ reason: ServeSkip, }; +/** + * 上下文窗口、输出上限这样的模型规格从哪儿来。 + */ +export type SpecSource = "price_table" | "manual"; + /** * L3 测速要花多少。 * @@ -4648,6 +4867,7 @@ export const ENDPOINTS = { UpdateProvider: { method: "PUT", path: "/providers/{name}", params: ["name"], format: "json" }, DeleteProvider: { method: "DELETE", path: "/providers/{name}", params: ["name"], format: "json" }, ProviderModels: { method: "GET", path: "/providers/{name}/models", params: ["name"], format: "json" }, + SetModelSpec: { method: "PUT", path: "/provider-model-spec", params: [], format: "json" }, RefreshProviderModels: { method: "POST", path: "/providers/{name}/models/refresh", params: ["name"], format: "json" }, RefreshStaleModels: { method: "POST", path: "/models/refresh", params: [], format: "json" }, CreateProxy: { method: "POST", path: "/proxies", params: [], format: "json" }, @@ -4769,6 +4989,7 @@ export type Endpoints = { UpdateProvider: { req: ProviderSave; res: ConfigWritten }; DeleteProvider: { req: BaseVersion; res: ConfigWritten }; ProviderModels: { req: null; res: ProviderModelsView }; + SetModelSpec: { req: ModelSpecSave; res: ConfigWritten }; RefreshProviderModels: { req: null; res: ProviderModelsView }; RefreshStaleModels: { req: null; res: ModelsRefreshing }; CreateProxy: { req: ProxySave; res: ConfigWritten }; diff --git a/src/i18n/core.zh.cases.json b/src/i18n/core.zh.cases.json index 4cca53d3..0a7750e9 100644 --- a/src/i18n/core.zh.cases.json +++ b/src/i18n/core.zh.cases.json @@ -301,5 +301,68 @@ "text": "There is no plugin `wsl-paths`." }, "zh": "插件「wsl-paths」不存在。" + }, + { + "msg": { + "code": "gw.key_limit.cost_per_period", + "args": { + "key": "claude-code", + "max": "$5.00", + "per": "day", + "used": "$5.02", + "resets": "2026-10-06 00:00 +08:00", + "resets_at_ms": "1791216000000" + }, + "text": "Gateway key `claude-code` has reached its limit of $5.00 per day: $5.02 spent so far. It resets at 2026-10-06 00:00 +08:00." + }, + "zh": "网关密钥「claude-code」已达到每天 $5.00 的费用上限,已产生费用 $5.02,将于 2026-10-06 00:00 +08:00 重置。" + }, + { + "msg": { + "code": "gw.key_limit.requests_rolling", + "args": { + "key": "codex", + "max": "30", + "per": "minute", + "used": "30", + "retry": "12" + }, + "text": "Gateway key `codex` has reached its limit of 30 requests per minute: 30 in the last minute. Try again in 12 s." + }, + "zh": "网关密钥「codex」已达到每分钟 30 次请求的上限,最近一分钟内已有 30 次,请于 12 秒后重试。" + }, + { + "msg": { + "code": "config.key_limit_not_positive", + "args": { + "key": "codex", + "per": "week", + "measure": "tokens", + "value": "0" + }, + "text": "the tokens limit per week of gateway key `codex` is 0; it has to be more than 0" + }, + "zh": "网关密钥「codex」的一条用量上限(token 数,每周)为 0,须大于 0。" + }, + { + "msg": { + "code": "config.key_limit_two_measures", + "args": { + "key": "codex", + "measures": "tokens and cost" + }, + "text": "a limit of gateway key `codex` names tokens and cost together. Each entry takes one of requests, tokens or cost; write one entry for each" + }, + "zh": "网关密钥「codex」的一条用量上限同时写了 tokens 和 cost。每一条只能写 requests、tokens、cost 之一,请分成多条。" + }, + { + "msg": { + "code": "gw.busy_all", + "args": { + "upstreams": "`官方`, `中转`" + }, + "text": "Every upstream that can serve this request is at its concurrency limit (max_concurrent): `官方`, `中转`. None had a free slot in time; try again shortly." + }, + "zh": "可处理此请求的上游均已达到并发上限(max_concurrent):「官方」、「中转」。等待期间没有空出位置,请稍后重试。" } ] diff --git a/src/i18n/core.zh.json b/src/i18n/core.zh.json index f252d9ec..6263568f 100644 --- a/src/i18n/core.zh.json +++ b/src/i18n/core.zh.json @@ -193,6 +193,24 @@ "list_keys": "Listing the account's API keys", "create_key": "Creating an API key", "read_key": "Reading the API key" + }, + "limit_per": { + "minute": "分钟", + "hour": "小时", + "day": "天", + "week": "周", + "month": "月" + }, + "limit_measure": { + "requests": "请求数", + "tokens": "token 数", + "cost": "费用" + }, + "limit_measures": { + "requests and tokens": "requests 和 tokens", + "requests and cost": "requests 和 cost", + "tokens and cost": "tokens 和 cost", + "requests and tokens and cost": "requests、tokens 和 cost" } }, "contexts": [ @@ -304,6 +322,12 @@ "gw.auth.key_disabled": "网关密钥「{key}」已停用。在应用的密钥页启用它即可恢复。", "gw.auth.source_not_allowed": "{peer} 不在允许的来源地址中。", "gw.auth.source_not_allowed_hint": "{peer} 不在允许的来源地址中。请修改 listen.gateway.allow_from,或将 bind 改为 loopback。", + "gw.key_limit.requests_per_period": "网关密钥「{key}」已达到每{per:limit_per} {max} 次请求的上限,已使用 {used} 次,将于 {resets} 重置。", + "gw.key_limit.tokens_per_period": "网关密钥「{key}」已达到每{per:limit_per} {max} 个 token 的上限,已使用 {used} 个 token,将于 {resets} 重置。", + "gw.key_limit.cost_per_period": "网关密钥「{key}」已达到每{per:limit_per} {max} 的费用上限,已产生费用 {used},将于 {resets} 重置。", + "gw.key_limit.requests_rolling": "网关密钥「{key}」已达到每{per:limit_per} {max} 次请求的上限,最近一{per:limit_per}内已有 {used} 次,请于 {retry} 秒后重试。", + "gw.key_limit.tokens_rolling": "网关密钥「{key}」已达到每{per:limit_per} {max} 个 token 的上限,最近一{per:limit_per}内已使用 {used} 个 token,请于 {retry} 秒后重试。", + "gw.key_limit.cost_rolling": "网关密钥「{key}」已达到每{per:limit_per} {max} 的费用上限,最近一{per:limit_per}内已产生费用 {used},请于 {retry} 秒后重试。", "gw.config.no_upstreams": "尚未配置任何上游。请在 ThinkWatch Lite 中添加上游,或在 config.yaml 的 providers 中添加。", "gw.config.proxy_undefined": "上游「{upstream}」使用的代理「{proxy}」未在 proxies 中定义,内置选项只有 direct 和 system。", "gw.config.proxy_unusable": "上游「{upstream}」的代理「{proxy}」不可用:{detail}", @@ -388,6 +412,8 @@ "gw.route.rule_failed": "规则求值失败:{detail}", "gw.route.denied": "规则「{rule}」拒绝了此请求:{reason}", "gw.route.no_upstream_alive": "没有可用的上游。", + "gw.busy_upstream": "上游「{upstream}」已有 {limit} 个请求在进行,达到其并发上限(max_concurrent)。", + "gw.busy_all": "可处理此请求的上游均已达到并发上限(max_concurrent):{upstreams!names}。等待期间没有空出位置,请稍后重试。", "gw.route.upstream_missing": "配置中不存在「{upstream}」。", "gw.route.selected_upstream_missing": "规则「{rule}」选中的上游「{upstream}」在配置中不存在。", "gw.route.all_selected_disabled": "路由选中的上游均已停用:{detail}", @@ -406,6 +432,7 @@ "gw.upstream.status": "上游「{upstream}」返回 {status}。", "gw.upstream.status_message": "上游「{upstream}」返回 {status}:{message}", "gw.upstream.stream_opening_error": "上游「{upstream}」开始回答后、给出任何内容之前报错({kind}):{message}", + "gw.slow_start": "上游「{upstream}」在 {secs} 秒内没有返回内容,请求已转到下一个上游。", "gw.upstream.stream_error": "上游「{upstream}」在回答过程中报错:{message}", "gw.upstream.stream_exception": "上游「{upstream}」以 {kind} 结束了响应流:{message}", "gw.upstream.eventstream_broken": "上游「{upstream}」发来了损坏的 AWS eventstream 帧:{detail}", @@ -555,6 +582,10 @@ "control.group.upstream_twice": "上游「{upstream}」重复。", "control.group.empty": "策略组至少需要一个上游。", "control.group.preferred_not_member": "优先使用的上游「{upstream}」不在该策略组中。", + "control.group.weight_not_member": "为「{upstream}」设置了权重,但它不是该策略组的成员。", + "control.group.weight_not_load_balance": "上游「{upstream}」的权重为 {weight},只有轮询策略组使用权重。", + "control.group.weight_out_of_range": "上游「{upstream}」的权重为 {weight},权重须为 1 到 100 之间的整数。", + "control.group.balance_not_load_balance": "balance_by「{balance_by}」只适用于轮询策略组。", "// ── control.plugin:装插件、改插件、批准文件、试运行。{plugin} 在 ID 上是 id,在确认上是插件的名字 ──": "", "control.plugin.not_found": "插件「{plugin}」不存在。", "control.plugin.bad_id": "「{plugin}」不是有效的插件 ID:只能使用小写字母、数字和连字符,1 到 {max} 个字符。", @@ -597,6 +628,13 @@ "config.bad_base_url": "上游「{upstream}」的接口地址既不是 http 也不是 https:{url}", "config.empty_key": "网关密钥「{key}」的值为空。", "config.zero_concurrency": "网关密钥「{key}」的 max_concurrent 为 0,使用它的请求会一直等待。不限制并发时请删除 max_concurrent。", + "config.provider_concurrency_range": "上游「{upstream}」的 max_concurrent 为 {value},须在 1 到 {max} 之间。不限制并发时请删除 max_concurrent。", + "config.key_limit_empty": "网关密钥「{key}」的一条用量上限没有写明计量。limits 中的每一条须写 requests、tokens、cost 之一。", + "config.key_limit_two_measures": "网关密钥「{key}」的一条用量上限同时写了 {measures:limit_measures}。每一条只能写 requests、tokens、cost 之一,请分成多条。", + "config.key_limit_not_positive": "网关密钥「{key}」的一条用量上限({measure:limit_measure},每{per:limit_per})为 {value},须大于 0。", + "config.key_limit_duplicate": "网关密钥「{key}」有两条相同的用量上限({measure:limit_measure},每{per:limit_per}),请只保留一条。", + "config.key_limit_cache_reads": "网关密钥「{key}」的一条用量上限({measure:limit_measure})设置了 cache_reads,只有 token 数上限可以设置该项。", + "config.key_limit_month_retention": "网关密钥「{key}」设置了每月的用量上限,而 retention.row_days 为 {days}。重启后当月用量要从请求记录中重新累计,因此 row_days 须至少为 31。", "config.name_collision": "「{name}」同时是上游和策略组的名称,规则的 to 无法区分指的是哪一个。请重命名其中一个。", "config.bad_allow_from": "listen.gateway.allow_from 中的 {entry} 不是有效的 IP 地址或 CIDR,应写成 192.168.0.0/16 的形式。", "config.unknown_price_sheet": "上游「{upstream}」使用的价目表「{sheet}」不存在。", @@ -610,6 +648,10 @@ "config.alias_blank_model": "别名「{alias}」列出的模型中有空白项。", "config.alias_chained": "别名「{alias}」列出的 {model} 本身也是别名。别名列出的应是上游使用的模型名称,不能是其他别名。", "config.alias_only_itself": "别名「{alias}」只列出了它自己,不起任何作用。请列出各上游使用的名称,或删除该别名。", + "config.model_spec_blank_model": "上游「{upstream}」的模型规格(model_specs)中有一项的模型 ID 为空。", + "config.model_spec_wildcard": "上游「{upstream}」的模型规格「{model}」含有 * 或 ?。模型规格只能对应一个确切的模型 ID。", + "config.model_spec_empty": "上游「{upstream}」的模型规格「{model}」既未设置 context_window,也未设置 max_output_tokens。请至少设置一项,或删除该规格。", + "config.model_spec_zero": "上游「{upstream}」的模型规格「{model}」中 {field} 为 0,须为大于 0 的 token 数。要使用价目表中的值,请删除该项。", "config.reserved_name": "{what:kind}名称「{name}」以 __ 开头,该前缀保留给内置项,请使用其他名称。", "config.rule_name_empty": "有一条自定义{what:rule_line}规则没有名称。", "config.rule_name_taken": "自定义{what:rule_line}规则名称「{name}」重复。", @@ -619,6 +661,7 @@ "config.rule_label_bad": "自定义出站脱敏规则「{name}」的占位符名称为「{label}」,须以大写字母开头,由 1 到 24 个大写字母、数字或下划线组成。", "config.unknown_rule": "security.{guard} 中的「{rule}」不是内置规则。", "config.failover_range": "failover.{field} 为 {value},须在 {min} 到 {max} 之间。", + "config.slow_start_too_short": "已开启 failover.next_on_slow_start,而 failover.stream_start_wait_secs 为 {secs}。该值须至少为 {min},否则正常的回答会在开始之前被切断。", "config.plugin.bad_id": "插件 ID「{plugin}」写法有误:只能使用小写字母、数字和连字符,1 到 {max} 个字符。", "config.plugin.duplicate": "插件 ID「{plugin}」重复。", "config.plugin.file": "插件「{plugin}」的文件为 {file},应为 plugins/{plugin}.js。", @@ -692,6 +735,11 @@ "engine.unknown_target": "规则「{rule}」指向的「{target}」既不是上游也不是策略组。", "engine.empty_group": "策略组「{group}」中没有上游。", "engine.duplicate_group": "策略组名称「{group}」重复。规则按名称引用策略组,名称必须唯一。", + "engine.group_upstream_twice": "策略组「{group}」中上游「{upstream}」出现了不止一次。每个上游在策略组中只能出现一次。", + "engine.group_unknown_upstream": "策略组「{group}」列出的「{upstream}」不是上游。策略组的成员须为上游的名称。", + "engine.group_weight_not_load_balance": "策略组「{group}」为上游「{upstream}」设置了权重 {weight},而只有轮询(load-balance)策略组使用权重。请删除权重,或将该策略组改为轮询。", + "engine.group_weight_out_of_range": "策略组「{group}」为上游「{upstream}」设置的权重为 {weight}。权重须为 1 到 100 之间的整数。", + "engine.group_balance_not_load_balance": "策略组「{group}」设置了 balance_by: {balance_by},而只有轮询(load-balance)策略组使用该项。请删除该项,或将该策略组改为轮询。", "engine.duplicate_route": "路由名称「{route}」重复。网关密钥按名称绑定路由,名称必须唯一。", "engine.unknown_default_route": "default_route 指向的路由「{route}」不存在,未绑定路由的网关密钥将无法命中任何规则。", "engine.unknown_route": "网关密钥「{key}」绑定的路由「{route}」不存在。", diff --git a/src/routing/chain.test.ts b/src/routing/chain.test.ts index d47d9d23..e9542c67 100644 --- a/src/routing/chain.test.ts +++ b/src/routing/chain.test.ts @@ -19,6 +19,7 @@ const key = (name: string, x: Partial = {}): ClientView => ({ max_concurrent: null, route: null, allow: null, + limits: [], ...x, }); const rule = (name: string, x: Partial = {}): RuleView => ({ @@ -45,6 +46,8 @@ const group = (name: string, providers: string[], x: Partial = {}): G kind: "fallback", selected: null, providers, + weights: {}, + balance_by: "weights", ...x, }); const ALL = group("__all__", [], { builtin: true }); diff --git a/src/routing/flights.test.ts b/src/routing/flights.test.ts index 3b3a7386..bf8e2a47 100644 --- a/src/routing/flights.test.ts +++ b/src/routing/flights.test.ts @@ -9,6 +9,7 @@ const key = (name: string, x: Partial = {}): ClientView => ({ max_concurrent: null, route: null, allow: null, + limits: [], ...x, }); const rule = (name: string, x: Partial = {}): RuleView => ({ @@ -34,6 +35,8 @@ const group = (name: string, providers: string[], x: Partial = {}): G kind: "fallback", selected: null, providers, + weights: {}, + balance_by: "weights", ...x, }); const up = (name: string) => ({ name, disabled: false, health: "ok" }) as ProviderView; diff --git a/src/routing/model.test.ts b/src/routing/model.test.ts index 83dfafb9..0e635382 100644 --- a/src/routing/model.test.ts +++ b/src/routing/model.test.ts @@ -200,8 +200,8 @@ describe("路由列表", () => { it("默认路由的使用者包括没指定路由的密钥", () => { const clients = [ - { name: "claude-code", key: "tw-a", max_concurrent: null, route: null, allow: null }, - { name: "codex", key: "tw-b", max_concurrent: null, route: "codex", allow: null }, + { name: "claude-code", key: "tw-a", max_concurrent: null, route: null, allow: null, limits: [] }, + { name: "codex", key: "tw-b", max_concurrent: null, route: "codex", allow: null, limits: [] }, ]; expect(usersOf(route({ name: "默认", default: true }), clients)).toEqual(["claude-code"]); expect(usersOf(route({}), clients)).toEqual(["codex"]); diff --git a/src/settings/FailoverSection.tsx b/src/settings/FailoverSection.tsx index 80cec71a..94f059ac 100644 --- a/src/settings/FailoverSection.tsx +++ b/src/settings/FailoverSection.tsx @@ -13,7 +13,8 @@ const MAX_PAUSE = 7 * 24 * 3600; /** 流开头最多等多少秒(core 的 `MAX_STREAM_START_WAIT_SECS`) */ const MAX_WAIT = 120; -type Field = keyof FailoverView; +/** 这一节里的格子。慢启动换下一家、等空位的秒数还没有放进来 */ +type Field = Exclude; export type Draft = Record; /** 表单里的顺序 */ diff --git a/src/settings/failover.test.ts b/src/settings/failover.test.ts index 618b2ca6..d554451a 100644 --- a/src/settings/failover.test.ts +++ b/src/settings/failover.test.ts @@ -10,6 +10,8 @@ const failover: FailoverView = { quota_pause_secs: 3600, rate_limit_max_pause_secs: 3600, stream_start_wait_secs: 15, + next_on_slow_start: false, + slot_wait_secs: 30, }; const allOk = (c: Record) => Object.values(c).every(Boolean); diff --git a/src/upstreams/labels.i18n.ts b/src/upstreams/labels.i18n.ts index a61d09ef..ff0df44b 100644 --- a/src/upstreams/labels.i18n.ts +++ b/src/upstreams/labels.i18n.ts @@ -62,6 +62,7 @@ export const labelsText = messages( out_of_scope: "不在启用范围内", not_offered: "未提供此模型", not_allowed: "密钥不允许使用此模型", + busy: "并发已满", }, defaultSheet: "默认价目表", unpriced: "无法计价", @@ -142,6 +143,7 @@ export const labelsText = messages( out_of_scope: "Model not enabled", not_offered: "Model not offered", not_allowed: "Model not allowed for this key", + busy: "At its concurrency limit", }, defaultSheet: "Default price sheet", unpriced: "Unpriced", diff --git a/src/upstreams/labels.ts b/src/upstreams/labels.ts index 460884fe..0efda67d 100644 --- a/src/upstreams/labels.ts +++ b/src/upstreams/labels.ts @@ -356,6 +356,8 @@ export function skipLabel(reason: ServeSkip): string { return t.not_offered; case "not_allowed": return t.not_allowed; + case "busy": + return t.busy; } } From a0b1c58b26d5db1c9faa54242fc6f7cf8d92ddf4 Mon Sep 17 00:00:00 2001 From: fylorn <249551762+fylorn@users.noreply.github.com> Date: Mon, 5 Oct 2026 17:54:29 +0800 Subject: [PATCH 02/10] Routing: weights and distribute-by for load-balance groups MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Core's routing round lets a load-balance group split new conversations by member weights (1 to 100) and, on top of that ratio, by speed, reliability or both. Without UI the only way to use either was editing config.yaml, and nothing on the routing page told a 7 : 3 group from an even one. Group dialog: a load-balance group gets a "Distribute by" segmented control (by ratio / by speed / by reliability / by speed and reliability) with one line of what it means, and a Weight column for the chosen members, like the Preferred column of a select group. Weights default to an explicit 1, are checked on the spot (whole numbers 1 to 100, the same range core enforces) and are sent with balance_by through the existing group save; other strategies send neither, so their configs stay as they are. The load-balance description now says new conversations, which is what the strategy distributes. Summaries: the group table shows the ratio under the strategy when the weights are not all 1, and the distribute-by mode when it is not by ratio, each on its own line so the narrow column does not break a phrase. The routing map's tooltip, the rule target descriptions and the command palette use the same wording, e.g. "Round robin (7 : 3 · By speed)". Dry run: each candidate of a load-balance group shows its weight; in an automatic mode also its TTFB, success rate ("no data yet" when core has too few samples) and its share of new conversations, computed as core does (weight × factor, members with an open circuit sit out unless all do). The strategy badge names the mode. Co-Authored-By: Claude Opus 5.5 --- src/labels.i18n.ts | 13 ++++++ src/labels.ts | 8 ++++ src/palette/items.tsx | 5 ++- src/routing/ChainMap.tsx | 6 +-- src/routing/DryRunDialog.i18n.ts | 13 ++++++ src/routing/DryRunDialog.tsx | 51 +++++++++++++++++++++-- src/routing/GroupDialog.i18n.ts | 23 +++++++++++ src/routing/GroupDialog.tsx | 62 +++++++++++++++++++++++++--- src/routing/GroupTable.tsx | 11 ++++- src/routing/model.i18n.ts | 7 +++- src/routing/model.test.ts | 66 ++++++++++++++++++++++++++++- src/routing/model.ts | 71 ++++++++++++++++++++++++++++++-- 12 files changed, 315 insertions(+), 21 deletions(-) diff --git a/src/labels.i18n.ts b/src/labels.i18n.ts index a5a97fba..f25db5b3 100644 --- a/src/labels.i18n.ts +++ b/src/labels.i18n.ts @@ -11,6 +11,13 @@ export const labelsText = messages( "url-test": "延迟最低", cheapest: "费用最低", }, + /** 轮询组按什么分新对话(`balance_by`) */ + balanceBy: { + weights: "按比例", + latency: "按速度", + health: "按稳定性", + "latency-health": "按速度和稳定性", + }, allUpstreams: "全部上游", probes: { health_check: { @@ -146,6 +153,12 @@ export const labelsText = messages( "url-test": "Lowest latency", cheapest: "Lowest cost", }, + balanceBy: { + weights: "By ratio", + latency: "By speed", + health: "By reliability", + "latency-health": "By speed and reliability", + }, allUpstreams: "All upstreams", probes: { health_check: { diff --git a/src/labels.ts b/src/labels.ts index ad588271..4059e5ec 100644 --- a/src/labels.ts +++ b/src/labels.ts @@ -11,6 +11,7 @@ import { coreText } from "@/i18n/core.i18n"; import { usd, type AttemptView, + type BalanceBy, type ConditionView, type ConfigOrigin, type ConfigStage, @@ -47,6 +48,13 @@ export function groupKindLabel(kind: GroupKind): string { return textOf(labelsText).groupKinds[kind]; } +/** 轮询组按什么分新对话,按界面上的先后:先是只看比例,再是自动的几种 */ +export const BALANCE_BY: readonly BalanceBy[] = ["weights", "latency", "health", "latency-health"]; + +export function balanceByLabel(by: BalanceBy): string { + return textOf(labelsText).balanceBy[by]; +} + /** 内置策略组在配置里的名字。**界面上不出现它**,显示为「全部上游」 */ export const ALL_UPSTREAMS = "__all__"; diff --git a/src/palette/items.tsx b/src/palette/items.tsx index f01f7499..b3a6db11 100644 --- a/src/palette/items.tsx +++ b/src/palette/items.tsx @@ -35,10 +35,11 @@ import { } from "@/ui/icons"; import { ClientLogo, UpstreamLogo } from "@/ui/logos"; import { StatusDot, type StatusTone } from "@/ui/status-dot"; -import { groupKindLabel, notSentText, probeLabel, targetLabel, ALL_UPSTREAMS } from "@/labels"; +import { notSentText, probeLabel, targetLabel, ALL_UPSTREAMS } from "@/labels"; import { when } from "@/format"; import { notSent } from "@/requestRouting"; import { upstreamText } from "@/requestTable"; +import { strategyText } from "@/routing/model"; import { NotSentIcon } from "@/traffic/cells"; import type { ConnView } from "@/connection/api"; import { connText } from "@/connection/connection.i18n"; @@ -348,7 +349,7 @@ export function buildItems(s: Sources): Item[] { id: `group:${g.name}`, group: "groups", title: targetLabel(g.name), - detail: `${groupKindLabel(g.kind)} · ${t.upstreamCount(g.providers.length)}`, + detail: `${strategyText(g)} · ${t.upstreamCount(g.providers.length)}`, // 显示的是译名(内置组),原名也能搜。里面的上游不算:打 `deep` 要的是 deepseek // 这个上游,不是每个含有它的组 keywords: [g.name], diff --git a/src/routing/ChainMap.tsx b/src/routing/ChainMap.tsx index 59a30ebb..8a6420c4 100644 --- a/src/routing/ChainMap.tsx +++ b/src/routing/ChainMap.tsx @@ -5,7 +5,7 @@ import { StatusDot } from "@/ui/status-dot"; import { Tip } from "@/ui/tip"; import { cn } from "@/lib/utils"; import { textOf, useText } from "@/i18n"; -import { groupKindLabel, targetLabel } from "@/labels"; +import { targetLabel } from "@/labels"; import type { Overview } from "@/types"; import { buildChain, @@ -24,7 +24,7 @@ import { } from "./chain"; import { chainMapText } from "./ChainMap.i18n"; import { activityOf, type Flight } from "./flights"; -import { usersOf } from "./model"; +import { strategyText, usersOf } from "./model"; import { KeyIcon, TargetIcon, upstreamState } from "./parts"; import { partsText } from "./parts.i18n"; import { routingText } from "./routing.i18n"; @@ -421,7 +421,7 @@ function describe( ov.providers.map((p) => p.name), ); const ordered = g.kind === "fallback" || g.kind === "select"; - const line = node.idle && !g.builtin ? t.unreferenced : (ordered ? t.members : t.membersUnordered)(groupKindLabel(g.kind), ms); + const line = node.idle && !g.builtin ? t.unreferenced : (ordered ? t.members : t.membersUnordered)(strategyText(g), ms); return { body: ( <> diff --git a/src/routing/DryRunDialog.i18n.ts b/src/routing/DryRunDialog.i18n.ts index 937b5dd6..df2df60e 100644 --- a/src/routing/DryRunDialog.i18n.ts +++ b/src/routing/DryRunDialog.i18n.ts @@ -25,6 +25,13 @@ export const dryRunText = messages( position: (n: number) => ` · 第 ${n} 条`, attempts: "尝试顺序", circuitOpen: "熔断中,将跳过", + /** 轮询组的候选:权重,和自动分配看的那几个数 */ + weight: (n: number) => `权重 ${n}`, + ttfb: (ms: string) => `首字节 ${ms}`, + ttfbNone: "首字节暂无数据", + success: (pct: string) => `成功率 ${pct}`, + successNone: "成功率暂无数据", + share: (pct: string) => `占比 ${pct}`, converted: (formats: string) => `需转换格式:${formats}`, skipped: "已跳过", skipReason: (reason: string) => `(${reason})`, @@ -75,6 +82,12 @@ export const dryRunText = messages( position: (n: number) => ` · rule ${n}`, attempts: "Attempts", circuitOpen: "Circuit open, will be skipped", + weight: (n: number) => `Weight ${n}`, + ttfb: (ms: string) => `TTFB ${ms}`, + ttfbNone: "No TTFB data yet", + success: (pct: string) => `Success rate ${pct}`, + successNone: "No success rate yet", + share: (pct: string) => `Share ${pct}`, converted: (formats: string) => `Format conversion: ${formats}`, skipped: "Skipped", skipReason: (reason: string) => ` (${reason})`, diff --git a/src/routing/DryRunDialog.tsx b/src/routing/DryRunDialog.tsx index db43f1c6..a0938394 100644 --- a/src/routing/DryRunDialog.tsx +++ b/src/routing/DryRunDialog.tsx @@ -25,6 +25,7 @@ import { commonText } from "@/i18n/common.i18n"; import { coreText, errorText } from "@/i18n/core.i18n"; import { PROBES, + balanceByLabel, formatLabel, groupKindLabel, mismatchText, @@ -32,13 +33,22 @@ import { targetLabel, translatedText, } from "@/labels"; -import type { Dialect, DryRunResult, KnownModel, Overview, RouteInput, RuleTrace } from "@/types"; +import type { + BalanceBy, + Dialect, + DryRunCandidate, + DryRunResult, + KnownModel, + Overview, + RouteInput, + RuleTrace, +} from "@/types"; import { skipLabel } from "@/upstreams/labels"; import { FormItem } from "@/upstreams/parts"; import { api } from "./api"; import { dryRunText } from "./DryRunDialog.i18n"; import { ModelInput, onOpenFocus } from "./fields"; -import { DIALECTS, usersOf } from "./model"; +import { DIALECTS, balanceShares, usersOf } from "./model"; import { KeyIcon, TargetIcon } from "./parts"; import { routingText } from "./routing.i18n"; import { modelViaOf, type ModelVia } from "./target"; @@ -58,7 +68,8 @@ export type DryRunTarget = * 旧结果淡一档留着,不闪成空白。 * * 尝试顺序里每个上游写出发给它的模型名和来历(别名、规则改写、指定模型):客户端写的 - * 名称和发出的不同,正是要在这里看清的事。 + * 名称和发出的不同,正是要在这里看清的事。经过轮询组时再写它的权重;按速度、稳定性 + * 分配时还写首字节时间、成功率和算下来的占比 —— 「为什么轮到它」要从这里看得出来。 */ export function DryRunDialog({ target, @@ -366,6 +377,7 @@ function Result({ */ const short = r.outcome === "intercepted"; const candidates = r.candidate_models; + const shares = balanceShares(r); const target = r.outcome === "route" ? (r.via_group ?? candidates[0]?.provider ?? null) : null; // 这一趟经过的路:密钥 → 路由 → 规则 → 去向。和路由图同一套标志 @@ -409,7 +421,12 @@ function Result({
{headline(r)} - {r.outcome === "route" && r.strategy && {groupKindLabel(r.strategy)}} + {r.outcome === "route" && r.strategy && ( + + {groupKindLabel(r.strategy)} + {r.balance_by && r.balance_by !== "weights" && ` · ${balanceByLabel(r.balance_by)}`} + + )}
{steps.length > 1 && (
@@ -464,6 +481,9 @@ function Result({ {c} {cv.sent_model && via && } + {cv.weight != null && ( + + )} {open && ( @@ -564,6 +584,29 @@ function SentModel({ model, via }: { model: string; via: ModelVia }) { ); } +/** + * 轮询组里一个候选的权重。按速度、稳定性分配时再写它看的数(没有样本的写明暂无数据) + * 和这一轮分到的占比;熔断着的这一轮不参加,不写占比 + */ +function BalanceFacts({ c, by, share }: { c: DryRunCandidate; by: BalanceBy; share: number | null }) { + const t = useText(dryRunText); + const parts = [t.weight(c.weight ?? 1)]; + if (by === "latency" || by === "latency-health") { + parts.push(c.ttfb_ms != null ? t.ttfb(`${Math.round(c.ttfb_ms).toLocaleString()}ms`) : t.ttfbNone); + } + if (by === "health" || by === "latency-health") { + parts.push(c.success_rate != null ? t.success(percent(c.success_rate)) : t.successNone); + } + if (by !== "weights" && share != null) parts.push(t.share(percent(share))); + return {parts.join(" · ")}; +} + +/** 0 到 1 写成百分数。不是 0、却四舍五入成 0 的写「<1%」—— 它还在轮里 */ +function percent(v: number): string { + if (v > 0 && v < 0.005) return "<1%"; + return `${Math.round(v * 100)}%`; +} + /** 路上的一站:标志加名字 */ function Step({ children, strong, className }: { children: ReactNode; strong?: boolean; className?: string }) { return ( diff --git a/src/routing/GroupDialog.i18n.ts b/src/routing/GroupDialog.i18n.ts index 741fbcd6..8b874829 100644 --- a/src/routing/GroupDialog.i18n.ts +++ b/src/routing/GroupDialog.i18n.ts @@ -1,4 +1,5 @@ import { messages } from "@/i18n"; +import type { BalanceBy } from "@/types"; import { andList } from "./routing.i18n"; export const groupDialogText = messages( @@ -20,6 +21,17 @@ export const groupDialogText = messages( orderSelect: "拖动调整顺序。选定的上游不可用时,按顺序使用其余成员。", orderFallback: "拖动调整顺序:依次使用,前一个不可用时使用下一个。", orderOther: "拖动调整顺序。排序依据相同时按此顺序。", + orderBalance: "权重为 1 到 100 的整数。拖动调整顺序。排序依据相同时按此顺序。", + weight: "权重", + weightOf: (name: string) => `${name} 的权重`, + weightInvalid: (name: string) => `${name} 的权重须为 1 到 100 的整数`, + balanceBy: "分配依据", + balanceDesc: { + weights: "新对话按成员权重的比例分配。", + latency: "以权重为基础,速度快的上游分到更多新对话。", + health: "以权重为基础,失败少的上游分到更多新对话。", + "latency-health": "以权重为基础,速度快、失败少的上游分到更多新对话。", + } satisfies Record, }, { noMembers: "Select at least one upstream", @@ -39,5 +51,16 @@ export const groupDialogText = messages( orderSelect: "Drag to reorder. When the selected upstream is unavailable, the other members are used in order.", orderFallback: "Drag to reorder: members are used in turn, moving to the next when one is unavailable.", orderOther: "Drag to reorder. Ties are broken by this order.", + orderBalance: "Weights are whole numbers from 1 to 100. Drag to reorder. Ties are broken by this order.", + weight: "Weight", + weightOf: (name: string) => `Weight of ${name}`, + weightInvalid: (name: string) => `The weight of ${name} must be a whole number from 1 to 100`, + balanceBy: "Distribute by", + balanceDesc: { + weights: "New conversations are split by the members' weights.", + latency: "Starting from the weights, faster upstreams get more new conversations.", + health: "Starting from the weights, upstreams that fail less get more new conversations.", + "latency-health": "Starting from the weights, faster upstreams that fail less get more new conversations.", + } satisfies Record, }, ); diff --git a/src/routing/GroupDialog.tsx b/src/routing/GroupDialog.tsx index 2f489ee8..22c3aac8 100644 --- a/src/routing/GroupDialog.tsx +++ b/src/routing/GroupDialog.tsx @@ -22,15 +22,15 @@ import { cn } from "@/lib/utils"; import { useText } from "@/i18n"; import { commonText } from "@/i18n/common.i18n"; import { errorText } from "@/i18n/core.i18n"; -import { groupKindLabel } from "@/labels"; -import type { GroupKind, Overview } from "@/types"; +import { BALANCE_BY, balanceByLabel, groupKindLabel } from "@/labels"; +import type { BalanceBy, GroupKind, Overview } from "@/types"; import { billingLabel, protocolLabel } from "@/upstreams/labels"; import { FormItem, Note } from "@/upstreams/parts"; import { api } from "./api"; import { onOpenFocus } from "./fields"; import { groupDialogText } from "./GroupDialog.i18n"; import { groupRefs } from "./GroupTable"; -import { move, strategies } from "./model"; +import { move, parseWeight, strategies } from "./model"; import { TargetIcon, upstreamState } from "./parts"; import { routingText } from "./routing.i18n"; import { useReorder } from "./useReorder"; @@ -46,6 +46,9 @@ export type GroupDialogMode = * 从上往下:名称、策略(几个里选一个,下面一句说它怎么选)、成员。成员的先后在 * 「按顺序」「手动选择」里就是优先级,所以可以拖动;手动选择还要在已选成员里定 * 一个优先使用的(行尾的「设为优先」)。 + * + * 轮询多两样:分配依据(只看比例,或者再看速度、稳定性),和每个成员的权重(行尾, + * 1 到 100,没改过的明写 1)。别的策略不用它们,切走再切回来时还在,只是不交。 */ export function GroupDialog({ mode, @@ -83,11 +86,19 @@ export function GroupDialog({ }); const [members, setMembers] = useState(source?.providers ?? []); const [selected, setSelected] = useState(source?.selected ?? null); + const [balanceBy, setBalanceBy] = useState(source?.balance_by ?? "weights"); + // 填的是文字:清空、打错的那一下也要留着让人改,交的时候才换成数 + const [weights, setWeights] = useState>(() => + Object.fromEntries(Object.entries(source?.weights ?? {}).map(([n, w]) => [n, String(w)])), + ); const [saving, setSaving] = useState(false); const [error, setError] = useState(null); const reorder = useReorder((from, to) => setOrder((o) => move(o, from, to))); const chosen = order.filter((n) => members.includes(n)); + const balanced = kind === "load-balance"; + const weightText = (n: string) => weights[n] ?? "1"; + const badWeight = balanced ? chosen.find((n) => parseWeight(weightText(n)) == null) : undefined; const preferred = kind === "select" ? (selected && chosen.includes(selected) ? selected : (chosen[0] ?? null)) : null; const refs = mode.kind === "edit" ? groupRefs(ov, mode.name) : []; @@ -103,7 +114,9 @@ export function GroupDialog({ ? rt.nameTaken(trimmed) : chosen.length === 0 ? t.noMembers - : null; + : badWeight !== undefined + ? t.weightInvalid(badWeight) + : null; const strategy = strategies().find((s) => s.id === kind); async function save() { @@ -116,6 +129,9 @@ export function GroupDialog({ kind, providers: chosen, selected: preferred, + // 不交 = 都是 1、只看比例;别的策略只能这样 + weights: balanced ? Object.fromEntries(chosen.map((n) => [n, parseWeight(weightText(n)) ?? 1])) : null, + balance_by: balanced ? balanceBy : null, }, base_version: base, }; @@ -164,6 +180,21 @@ export function GroupDialog({

+ {balanced && ( +
+ {t.balanceBy} + + label={t.balanceBy} + value={balanceBy} + onChange={setBalanceBy} + options={BALANCE_BY.map((b) => ({ id: b, label: balanceByLabel(b) }))} + /> +

+ {t.balanceDesc[balanceBy]} +

+
+ )} +
{rt.members} @@ -180,6 +211,7 @@ export function GroupDialog({ {t.upstream} {kind === "select" && {t.preferred}} + {balanced && {t.weight}} @@ -256,6 +288,20 @@ export function GroupDialog({ ))} )} + {balanced && ( + + {on && ( + setWeights((w) => ({ ...w, [n]: e.target.value }))} + /> + )} + + )} ); })} @@ -264,7 +310,13 @@ export function GroupDialog({
)} - {kind === "select" ? t.orderSelect : kind === "fallback" ? t.orderFallback : t.orderOther} + {kind === "select" + ? t.orderSelect + : kind === "fallback" + ? t.orderFallback + : balanced + ? t.orderBalance + : t.orderOther}
diff --git a/src/routing/GroupTable.tsx b/src/routing/GroupTable.tsx index 1159f395..149db7bc 100644 --- a/src/routing/GroupTable.tsx +++ b/src/routing/GroupTable.tsx @@ -13,6 +13,7 @@ import { textOf, useText } from "@/i18n"; import { groupKindLabel } from "@/labels"; import type { GroupView, Overview } from "@/types"; import { membersOf, type ChainFocus } from "./chain"; +import { balanceNotes } from "./model"; import { groupTableText } from "./GroupTable.i18n"; import { TargetIcon, upstreamState } from "./parts"; import { routingText } from "./routing.i18n"; @@ -78,6 +79,9 @@ export function GroupTable({ {shown.map(({ item: g, key, presence }) => { const items = menu(g, actions); const refs = groupRefs(ov, g.name); + // 轮询组的比例(不是平均分时)和分配依据(不是只看比例时),各占一行写在策略下面: + // 这一列窄,连成一行会从「按速度和稳定性」中间折开 + const notes = balanceNotes(g); const label = g.builtin ? t.allUpstreams : g.name; const f: ChainFocus = { kind: "group", name: g.name }; const lit = focus?.kind === "group" && focus.name === g.name; @@ -112,8 +116,13 @@ export function GroupTable({ {g.builtin && {t.builtin}} - +
{groupKindLabel(g.kind)}
+ {notes.map((n) => ( +
+ {n} +
+ ))}
diff --git a/src/routing/model.i18n.ts b/src/routing/model.i18n.ts index 8fae7ea0..feee0655 100644 --- a/src/routing/model.i18n.ts +++ b/src/routing/model.i18n.ts @@ -22,6 +22,8 @@ export const modelText = messages( noCatchAll: "尚无兜底规则", builtinGroup: "内置策略组 · 按上游列表顺序", groupTarget: (kind: string, members: string) => `策略组 · ${kind}${members ? `:${members}` : ""}`, + /** 策略名后面的补充:轮询组的比例和分配依据 */ + withNotes: (kind: string, notes: string) => `${kind}(${notes})`, upstreamTarget: (protocol: string, disabled: boolean) => `上游 · ${protocol}${disabled ? " · 已停用" : ""}`, unknownTarget: "不存在的去向", setModel: (model: string) => `模型改为 ${model}`, @@ -33,7 +35,7 @@ export const modelText = messages( strategies: { fallback: "依次使用成员,前一个不可用时使用下一个。", select: "使用选定的上游;它不可用时,按顺序使用其余成员。", - loadBalance: "在成员之间轮流分配请求。", + loadBalance: "在成员之间轮流分配新对话。", urlTest: "优先使用首字节时间最短的上游。", cheapest: "优先使用输入单价最低的上游。", }, @@ -62,6 +64,7 @@ export const modelText = messages( noCatchAll: "No catch-all rule yet", builtinGroup: "Built-in group · In upstream list order", groupTarget: (kind: string, members: string) => `Group · ${kind}${members ? `: ${members}` : ""}`, + withNotes: (kind: string, notes: string) => `${kind} (${notes})`, upstreamTarget: (protocol: string, disabled: boolean) => `Upstream · ${protocol}${disabled ? " · Disabled" : ""}`, unknownTarget: "Destination not found", setModel: (model: string) => `Model set to ${model}`, @@ -73,7 +76,7 @@ export const modelText = messages( strategies: { fallback: "Uses the members in order, moving to the next when one is unavailable.", select: "Uses the selected upstream; when it is unavailable, uses the other members in order.", - loadBalance: "Distributes requests across the members in turn.", + loadBalance: "Distributes new conversations across the members in turn.", urlTest: "Prefers the upstream with the shortest time to first byte.", cheapest: "Prefers the upstream with the lowest input price.", }, diff --git a/src/routing/model.test.ts b/src/routing/model.test.ts index 0e635382..2d01b305 100644 --- a/src/routing/model.test.ts +++ b/src/routing/model.test.ts @@ -1,8 +1,10 @@ import { describe, expect, it } from "vitest"; import { setLang } from "@/i18n"; -import type { ConditionView, RouteView, RuleView } from "@/types"; +import type { ConditionView, DryRunCandidate, GroupView, RouteView, RuleView } from "@/types"; import { addOnsText, + balanceNotes, + balanceShares, blankPinned, blankRule, canLift, @@ -14,9 +16,12 @@ import { insertIndex, liftShadowed, move, + parseWeight, + ratioText, routeProblems, ruleProblem, splitCompare, + strategyText, usersOf, type RuleDraft, } from "./model"; @@ -256,3 +261,62 @@ describe("路由列表", () => { ]); }); }); + +describe("轮询组的比例和分配依据", () => { + const group = (p: Partial): GroupView => ({ + name: "分流", + builtin: false, + kind: "load-balance", + providers: ["anthropic", "openrouter"], + weights: { anthropic: 1, openrouter: 1 }, + balance_by: "weights", + ...p, + }); + + it("权重只收 1 到 100 的整数", () => { + expect(parseWeight("1")).toBe(1); + expect(parseWeight(" 100 ")).toBe(100); + for (const bad of ["", "0", "101", "1.5", "-3", "7k", " "]) expect(parseWeight(bad)).toBeNull(); + }); + + it("平均分时不写比例,按成员的顺序写出不平均的", () => { + expect(ratioText(group({}))).toBeNull(); + expect(ratioText(group({ weights: { anthropic: 7, openrouter: 3 } }))).toBe("7 : 3"); + // 别的类型不用权重 + expect(ratioText(group({ kind: "fallback", weights: {} }))).toBeNull(); + }); + + it("策略名后面补上比例和不是只看比例的分配依据", () => { + setLang("zh"); + expect(balanceNotes(group({}))).toEqual([]); + expect(strategyText(group({}))).toBe("轮询"); + expect(strategyText(group({ weights: { anthropic: 7, openrouter: 3 } }))).toBe("轮询(7 : 3)"); + expect(strategyText(group({ balance_by: "latency" }))).toBe("轮询(按速度)"); + expect(strategyText(group({ weights: { anthropic: 2, openrouter: 1 }, balance_by: "latency-health" }))).toBe( + "轮询(2 : 1 · 按速度和稳定性)", + ); + setLang("en"); + expect(strategyText(group({ weights: { anthropic: 7, openrouter: 3 }, balance_by: "health" }))).toBe( + "Round robin (7 : 3 · By reliability)", + ); + }); + + it("试算的占比按权重 × 系数分,熔断着的不参加", () => { + const c = (provider: string, weight: number | null, balance_factor: number | null = null): DryRunCandidate => ({ + provider, + weight, + balance_factor, + }); + expect(balanceShares({ candidate_models: [c("a", 7), c("b", 3)], circuit_open: [] })).toEqual([0.7, 0.3]); + const auto = balanceShares({ candidate_models: [c("a", 2, 2.25), c("b", 1, 0.5)], circuit_open: [] }); + expect(auto[0]).toBeCloseTo(0.9); + expect(auto[1]).toBeCloseTo(0.1); + expect(balanceShares({ candidate_models: [c("a", 1), c("b", 1), c("c", 2)], circuit_open: ["c"] })).toEqual([ + 0.5, 0.5, 0, + ]); + // 全都熔断着时都算:网关照样一家家试 + expect(balanceShares({ candidate_models: [c("a", 3), c("b", 1)], circuit_open: ["a", "b"] })).toEqual([0.75, 0.25]); + // 不是轮询组:没有权重,也就没有占比 + expect(balanceShares({ candidate_models: [c("a", null)], circuit_open: [] })).toEqual([null]); + }); +}); diff --git a/src/routing/model.ts b/src/routing/model.ts index 3d00ccf7..84f7a6c6 100644 --- a/src/routing/model.ts +++ b/src/routing/model.ts @@ -5,9 +5,9 @@ * 能不能用、条件写得对不对,最后由 core 说;这里只做对话框里需要实时给出 * 的那几件事:保存按钮旁边缺什么、哪条规则被兜底挡住。 */ -import type { ClientView, ConditionField, ConditionView, Dialect, GroupKind, GroupView, KnownModel, PinnedModel, ProviderView, RouteView, RuleInput, RuleView } from "@/types"; +import type { ClientView, ConditionField, ConditionView, Dialect, DryRunResult, GroupKind, GroupView, KnownModel, PinnedModel, ProviderView, RouteView, RuleInput, RuleView } from "@/types"; import { textOf } from "@/i18n"; -import { ALL_UPSTREAMS, conditionName, groupKindLabel, targetLabel } from "@/labels"; +import { ALL_UPSTREAMS, balanceByLabel, conditionName, groupKindLabel, targetLabel } from "@/labels"; import { protocolLabel } from "@/upstreams/labels"; import { modelText } from "./model.i18n"; import { routingText } from "./routing.i18n"; @@ -386,6 +386,71 @@ export function routeProblems(route: RouteView): string[] { return out; } +// ---------------------------------------------------------------- 轮询组的比例 + +/** 权重的范围,和 core 的校验一样(`engine.group_weight_out_of_range`) */ +export const WEIGHT_MIN = 1; +export const WEIGHT_MAX = 100; + +/** 对话框里填的权重:1 到 100 的整数,别的(空、小数、超出范围)是 null */ +export function parseWeight(text: string): number | null { + const s = text.trim(); + if (!/^\d+$/.test(s)) return null; + const n = Number(s); + return n >= WEIGHT_MIN && n <= WEIGHT_MAX ? n : null; +} + +/** 成员在轮询组里的权重。core 给每个成员都列了,没列到的(别的类型)按 1 */ +export function weightOf(g: Pick, member: string): number { + return g.weights[member] ?? 1; +} + +/** + * 轮询组的比例,按成员的顺序:`7 : 3`。**权重都是 1 时没有** —— 那就是平均分, + * 写出 `1 : 1 : 1` 只是噪音。别的类型不用权重,也没有 + */ +export function ratioText(g: Pick): string | null { + if (g.kind !== "load-balance") return null; + const ws = g.providers.map((p) => weightOf(g, p)); + return ws.some((w) => w !== 1) ? ws.join(" : ") : null; +} + +/** + * 轮询组在策略名之外要说的:比例(不是平均分时)、分配依据(不是只看比例时)。 + * 别的类型、两样都是默认值时是空的 + */ +export function balanceNotes(g: Pick): string[] { + if (g.kind !== "load-balance") return []; + const out: string[] = []; + const ratio = ratioText(g); + if (ratio) out.push(ratio); + if (g.balance_by !== "weights") out.push(balanceByLabel(g.balance_by)); + return out; +} + +/** 策略名,轮询组带上比例和分配依据:`轮询(7 : 3 · 按速度)` */ +export function strategyText(g: Pick): string { + const kind = groupKindLabel(g.kind); + const notes = balanceNotes(g); + return notes.length ? textOf(modelText).withNotes(kind, notes.join(" · ")) : kind; +} + +/** + * 试算里轮询组每个候选这一轮分到新对话的份额,0 到 1,和 `r.candidate_models` 一一对应; + * 不是轮询组(候选没有权重)的是 null。 + * + * 和 core 排头用的同一个数:权重 × 系数(`balance_factor`,只看比例时是 1)。**熔断着的 + * 这一轮不参加**(份额是 0,排到它的那一次本来就会被跳过),全都熔断着时都算 + */ +export function balanceShares(r: Pick): (number | null)[] { + const members = r.candidate_models.filter((c) => c.weight != null); + const sitOut = members.every((c) => r.circuit_open.includes(c.provider)) ? [] : r.circuit_open; + const eff = (c: (typeof members)[number]) => + sitOut.includes(c.provider) ? 0 : (c.weight ?? 0) * (c.balance_factor ?? 1); + const total = members.reduce((a, c) => a + eff(c), 0); + return r.candidate_models.map((c) => (c.weight == null ? null : total > 0 ? eff(c) / total : 0)); +} + /** 去向的说明:策略组的策略与成员,或上游的协议 */ export function describeTarget( name: string, @@ -397,7 +462,7 @@ export function describeTarget( if (g) { if (g.builtin) return t.builtinGroup; const members = membersText(g); - return t.groupTarget(groupKindLabel(g.kind), members); + return t.groupTarget(strategyText(g), members); } const p = providers.find((x) => x.name === name); if (p) return t.upstreamTarget(protocolLabel(p.protocol), p.disabled); From e7336e0ae2a0bb413c8ba6d4110e4eed54fe80fb Mon Sep 17 00:00:00 2001 From: fylorn <249551762+fylorn@users.noreply.github.com> Date: Mon, 5 Oct 2026 18:04:53 +0800 Subject: [PATCH 03/10] Traffic: slow-start switches, busy upstreams and key usage limits Core now records three new things in a request's routing: an attempt abandoned because no content arrived in time (outcome slow_start, with the input the upstream may have billed), a candidate skipped because it was at its concurrency limit (skipped: busy), and the time an attempt waited for a free slot (queued_ms). It also refuses requests with two new kinds of error: a gateway key's usage limit (gw.key_limit.*) and every upstream being full (gw.busy_all). Attempt chain: a slow start reads "Start timed out" and a busy skip "At its concurrency limit", each with core's own sentence on hover (how long it waited, how many requests the upstream already had). A busy skip was never sent, so it shows no duration, like a hop a rule denied. A wait for a slot shows as "Queued 1.2 s" next to the hop, since core keeps it out of the hop's duration. Under an abandoned hop one line gives the input tokens (the upstream's own numbers when it reported them, otherwise "about N" from the gateway's estimate) and says the upstream may have billed them; the hover adds that they are not part of this request's cost. The note under the chain no longer says "the first N upstreams failed" when some of those hops were busy skips or slow starts: neither is the upstream's failure, so it says the first N attempts did not take the request. Refusals: until now a request with an empty upstream and an error was either a rule denial or "No upstream". A key-limit refusal also has an empty upstream and would have been shown as "No upstream", which sends the user to the upstreams page; it now reads "Usage limit" with its own icon. A busy refusal is recorded on the last upstream looked at, which received nothing, so the upstream column, the drawer and the session upstream lists would have blamed that upstream; it now reads "At capacity" there and is left out of a session's upstreams. Both keep core's translated sentence as the reason and stay under "Failed only". Co-Authored-By: Claude Opus 5.5 --- src/RequestDrawer.i18n.tsx | 25 ++++++++ src/RequestDrawer.tsx | 113 ++++++++++++++++++++++++++--------- src/labels.i18n.ts | 10 +++- src/labels.test.ts | 41 +++++++++++++ src/labels.ts | 31 +++++++--- src/requestRouting.test.ts | 80 +++++++++++++++++++++++++ src/requestRouting.ts | 55 +++++++++++++---- src/traffic/SessionPanel.tsx | 8 ++- src/traffic/SessionRow.tsx | 6 +- src/traffic/cells.tsx | 29 +++++---- src/ui/icons.tsx | 2 + 11 files changed, 334 insertions(+), 66 deletions(-) diff --git a/src/RequestDrawer.i18n.tsx b/src/RequestDrawer.i18n.tsx index 948e22a6..4f31e508 100644 --- a/src/RequestDrawer.i18n.tsx +++ b/src/RequestDrawer.i18n.tsx @@ -98,6 +98,12 @@ export const requestDrawerText = messages( /** 规则写的拒绝理由,或者选中的上游为何都无法服务 */ reason: "原因", attempts: "尝试链", + /** 这一跳等空位等了多久(上游满着)。秒数已经按一位小数写好 */ + queued: (s: string) => `排队 ${s} 秒`, + /** 放弃了的一跳(开头超时)下面那一行:上游没报用量时,网关估的输入 */ + abandonedEstimate: (n: string) => `输入约 ${n} token`, + mayBeBilled: "上游可能已计费", + mayBeBilledTip: "上游是否收取这部分费用无法得知,此请求的费用不含这部分。", // 尝试链里一跳发出的模型名和客户端写的不同:悬停按原因说 sentModel: (model: string) => `规则改写了模型名:这一跳发给上游的是 ${model},费用按它计算`, sentByAlias: (upstream: string, model: string) => `别名:这一跳发给 ${upstream} 的是 ${model},费用按它计算`, @@ -112,6 +118,11 @@ export const requestDrawerText = messages( deniedAfterPick: (rule: string) => `选定上游后,规则「${rule}」拒绝了此请求,未发往任何上游。`, deniedBeforePick: (rule: string) => `选定上游之前,规则「${rule}」已拒绝此请求,未发往任何上游。`, unavailable: "规则选中的上游均无法服务此请求,未发往任何上游。", + /** 前面几跳里有满着跳过的、开头超时放弃的:它们不是上游的失败 */ + switched: (n: number) => `已自动切换上游:前 ${n} 次尝试未接下此请求。`, + limited: "网关密钥已达到用量上限,此请求未发往任何上游。", + busy: "上游均已达到并发上限,等待期间没有空出位置,此请求未发往任何上游。", + busyAfterTries: (n: number) => `发出的 ${n} 次尝试未成功,其余上游均已达到并发上限,等待期间没有空出位置。`, noRouting: "此请求由网关本地应答,未经过路由。", routingPending: "路由尚未完成", noAttempts: "此请求没有上游尝试记录。", @@ -245,6 +256,10 @@ export const requestDrawerText = messages( deniedBy: "Denied by", reason: "Reason", attempts: "Attempts", + queued: (s: string) => `Queued ${s} s`, + abandonedEstimate: (n: string) => `About ${n} input tokens`, + mayBeBilled: "may have been billed by the upstream", + mayBeBilledTip: "Whether the upstream charged for these tokens is unknown; they are not included in this request's cost.", sentModel: (model: string) => `A rule rewrote the model: this attempt sent ${model}, and the cost is priced by it`, sentByAlias: (upstream: string, model: string) => `Alias: this attempt sent ${model} to ${upstream}, and the cost is priced by it`, @@ -265,6 +280,16 @@ export const requestDrawerText = messages( deniedBeforePick: (rule: string) => `Rule “${rule}” denied this request before an upstream was chosen; it was not sent to any upstream.`, unavailable: "No upstream the rule selected can serve this request; it was not sent to any upstream.", + switched: (n: number) => + n === 1 + ? "Switched upstreams automatically: the first attempt did not take this request." + : `Switched upstreams automatically: the first ${n} attempts did not take this request.`, + limited: "The gateway key had reached a usage limit; this request was not sent to any upstream.", + busy: "Every upstream was at its concurrency limit and none freed up in time; this request was not sent to any upstream.", + busyAfterTries: (n: number) => + n === 1 + ? "The attempt sent did not succeed, and the other upstreams were at their concurrency limits with none freeing up in time." + : `The ${n} attempts sent did not succeed, and the other upstreams were at their concurrency limits with none freeing up in time.`, noRouting: "The gateway answered this request locally; it did not go through routing.", routingPending: "Routing has not finished yet", noAttempts: "No upstream attempts were recorded for this request.", diff --git a/src/RequestDrawer.tsx b/src/RequestDrawer.tsx index 888439a8..dfd5b648 100644 --- a/src/RequestDrawer.tsx +++ b/src/RequestDrawer.tsx @@ -39,10 +39,11 @@ import { } from "./labels"; import { prettyJson } from "./prettyJson"; import { requestDrawerText } from "./RequestDrawer.i18n"; -import { notSent, routingFacts, type RoutingNote } from "./requestRouting"; +import { notSent, routingFacts, skippedHop, type RoutingNote } from "./requestRouting"; import { ActionBadge, byCodepoints, EventDetail, ruleName, whereOf } from "./security/labels"; import { usd, + type AttemptUsage, type AttemptView, type BodyView, type CoreEvent, @@ -682,35 +683,51 @@ function Routing({ r, plugins, running }: { r: HistoryRow; plugins: PluginRunVie
    {f.hops.map(({ attempt: a, denied }, i) => { const outcome = attemptText(a); + // 没有发给这个上游的一跳:被规则拒绝,或者它满着、换了下一家 + const unsent = denied || skippedHop(a); return ( -
  1. - {i + 1} - - - {a.provider} - - {/* 这一跳发出的模型名:有一跳改了名、或者用了别名、指定模型时每一跳都写, - 费用也按它算 */} - {m.show && m.hops[i] && } - {denied ? ( - // 选定上游之后的规则在这一跳拒绝了它:没有发给这个上游,不是上游的失败 - - {deniedHopText(f.deniedBy ?? "")} - - ) : ( - /* **失败的原因要留着** —— 一条说「试过 A → B → C」的链和一条还说清 - 每一跳为什么失败的链,排查价值差得远 */ - - {outcome.text} - - )} - {/* 没有发出的那一跳没有耗时可言 */} - {denied ? "—" : ms(a.ms)} +
  2. +
    + {i + 1} + + + {a.provider} + + {/* 这一跳发出的模型名:有一跳改了名、或者用了别名、指定模型时每一跳都写, + 费用也按它算 */} + {m.show && m.hops[i] && } + {denied ? ( + // 选定上游之后的规则在这一跳拒绝了它:没有发给这个上游,不是上游的失败 + + {deniedHopText(f.deniedBy ?? "")} + + ) : ( + /* **失败的原因要留着** —— 一条说「试过 A → B → C」的链和一条还说清 + 每一跳为什么失败的链,排查价值差得远。短名(开头超时、并发已满)悬停 + 是 core 的原话 */ + + {outcome.tip ? ( + + {outcome.text} + + ) : ( + outcome.text + )} + + )} + {/* 等空位的时间不算在这一跳的耗时里,另写一项 */} + {a.queued_ms != null && a.queued_ms > 0 && ( + + {t.queued(seconds(a.queued_ms))} + + )} + {/* 没有发出的那一跳没有耗时可言 */} + {unsent ? "—" : ms(a.ms)} +
    + {/* 放弃了的这一跳(开头超时)上游可能已经按输入收了钱:不在这个请求的费用里 */} + {a.usage && }
  3. ); })} @@ -726,6 +743,38 @@ function Routing({ r, plugins, running }: { r: HistoryRow; plugins: PluginRunVie ); } +/** + * 放弃了的一跳(开头超时)上游可能已经收了钱的输入,写在那一跳下面一行。上游在流开头 + * 报了的写它报的几种 token;没报的是网关估的输入,写「约」。**输出不知道**,不写。 + * + * 这部分不进这个请求的费用:上游收没收、收了多少,网关看不到。悬停说这一点 + */ +function AbandonedUsage({ usage: u }: { usage: AttemptUsage }) { + const t = useText(requestDrawerText); + const parts = u.estimated + ? [t.abandonedEstimate(u.input.toLocaleString())] + : [ + `${t.input} ${u.input.toLocaleString()}`, + ...(u.cache_read > 0 ? [`${t.cacheReads} ${u.cache_read.toLocaleString()}`] : []), + ...(u.cache_write > 0 ? [`${t.cacheWrites} ${u.cache_write.toLocaleString()}`] : []), + ]; + return ( + // 和上游名对齐:序号那一格 16px 加间距 12px +

    + {parts.join(" · ")} ·{" "} + + {t.mayBeBilled} + +

    + ); +} + +/** 排队等了多久:不到 10 秒的留一位小数,最少写 0.1 */ +function seconds(ms: number): string { + const s = ms / 1000; + return String(s < 10 ? Math.max(0.1, Math.round(s * 10) / 10) : Math.round(s)); +} + /** * 尝试链里一跳发出的模型名。和客户端写的不同时带虚线下划线,悬停按原因说(别名、规则改名、 * 指定模型、插件);原因对不上现在的配置时只说发出的是什么。 @@ -766,6 +815,12 @@ function noteText(n: RoutingNote, t: (typeof requestDrawerText)["zh"]): string { switch (n.kind) { case "failover": return t.failover(n.failed); + case "switched": + return t.switched(n.count); + case "limited": + return t.limited; + case "busy": + return n.tried > 0 ? t.busyAfterTries(n.tried) : t.busy; case "failover_denied": return t.failoverDenied(n.failed, n.rule); case "denied_after_pick": diff --git a/src/labels.i18n.ts b/src/labels.i18n.ts index a5a97fba..2bf7b078 100644 --- a/src/labels.i18n.ts +++ b/src/labels.i18n.ts @@ -66,12 +66,17 @@ export const labelsText = messages( noResponse: "未收到响应", estimated: "本地估算", estimatedAfter: (status: number) => `${status} · 本地估算`, + /** 开头等过了时限还没有内容,换了下一个上游 */ + slowStart: "开头超时", /** 选定上游之后被规则拒绝的那一跳:没有发给这个上游 */ deniedHop: (rule: string) => `未发送 · 被规则「${rule}」拒绝`, - // 没有发往任何上游的请求,在「上游」的位置上写的那一句 + // 没有上游接下的请求,在「上游」的位置上写的那一句:规则拒绝、没有可用的上游、 + // 密钥的用量上限拒绝、上游都满着 notSent: { denied: "规则拒绝", unavailable: "无可用上游", + limited: "用量上限", + busy: "并发已满", }, // ------------------------------------------------------------ 请求与费用 @@ -203,11 +208,14 @@ export const labelsText = messages( noResponse: "No response received", estimated: "Estimated locally", estimatedAfter: (status: number) => `${status} · Estimated locally`, + slowStart: "Start timed out", deniedHop: (rule: string) => `Not sent · denied by rule “${rule}”`, // 流量表「上游」那一列放得下的长度:再长就折成两行 notSent: { denied: "Denied", unavailable: "No upstream", + limited: "Usage limit", + busy: "At capacity", }, quote: { diff --git a/src/labels.test.ts b/src/labels.test.ts index fd8644e8..26f60d46 100644 --- a/src/labels.test.ts +++ b/src/labels.test.ts @@ -3,6 +3,7 @@ import { setLang } from "./i18n"; import { GROUP_KINDS, PROBES, + attemptText, conditionName, conditionText, mismatchText, @@ -65,3 +66,43 @@ describe("规则的条件与改写", () => { expect(setText({ field: "max_tokens", value: "4096" })).toBe("max_tokens set to 4096"); }); }); + +describe("尝试链里的一跳", () => { + const slow = { + provider: "anthropic", + outcome: "slow_start" as const, + error: { code: "gw.slow_start", args: { upstream: "anthropic", secs: "30" }, text: "" }, + ms: 30_004, + }; + const busy = { + provider: "anthropic", + outcome: "error" as const, + error: { code: "gw.busy_upstream", args: { upstream: "anthropic", limit: "2" }, text: "" }, + ms: 0, + skipped: "busy" as const, + }; + + it("开头超时、满着跳过:短名,悬停是 core 的原话", () => { + expect(attemptText(slow)).toEqual({ + text: "开头超时", + ok: false, + tip: "上游「anthropic」在 30 秒内没有返回内容,请求已转到下一个上游。", + }); + expect(attemptText(busy)).toEqual({ + text: "并发已满", + ok: false, + tip: "上游「anthropic」已有 2 个请求在进行,达到其并发上限(max_concurrent)。", + }); + setLang("en"); + expect(attemptText(slow).text).toBe("Start timed out"); + expect(attemptText(busy).text).toBe("At its concurrency limit"); + }); + + it("别的结果照旧:字就是那一句,没有悬停", () => { + expect(attemptText({ provider: "openrouter", outcome: "served", status: 200, ms: 900 })).toEqual({ + text: "成功 · 200", + ok: true, + tip: null, + }); + }); +}); diff --git a/src/labels.ts b/src/labels.ts index ad588271..4024db89 100644 --- a/src/labels.ts +++ b/src/labels.ts @@ -21,7 +21,7 @@ import { type TakesEffect, type TranslatedView, } from "./types"; -import { PROTOCOLS } from "./upstreams/labels"; +import { PROTOCOLS, skipLabel } from "./upstreams/labels"; import { labelsText } from "./labels.i18n"; import type { NotSent } from "./requestRouting"; @@ -148,22 +148,32 @@ export function setText(s: SetView): string { } } -/** 尝试链里的一跳。`ok` 决定颜色 */ -export function attemptText(a: AttemptView): { text: string; ok: boolean } { +/** + * 尝试链里的一跳。`ok` 决定颜色。`tip`:短名后面悬停说的那一句(core 说的原话); + * 字本身就是那一句的没有 + */ +export function attemptText(a: AttemptView): { text: string; ok: boolean; tip: string | null } { const t = textOf(labelsText); + const said = a.error ? coreText(a.error) : null; + // 没有发出去的一跳:这家满着,换了下一家。短名说原因,悬停是 core 那一句(几个请求在 + // 进行,也就是它的上限) + if (a.skipped) return { text: skipLabel(a.skipped), ok: false, tip: said }; switch (a.outcome) { case "served": if (a.status == null || a.status < 400) { - return { text: a.status == null ? t.served : t.servedStatus(a.status), ok: true }; + return { text: a.status == null ? t.served : t.servedStatus(a.status), ok: true, tip: null }; } - return { text: t.rejected(a.status), ok: false }; + return { text: t.rejected(a.status), ok: false, tip: null }; case "status": - return { text: a.status === 429 ? t.rateLimited : t.upstreamError(a.status ?? "—"), ok: false }; + return { text: a.status === 429 ? t.rateLimited : t.upstreamError(a.status ?? "—"), ok: false, tip: null }; // 数 token 由网关自己估:上游不是这种格式(没问过它),或者问过、它没实现这个接口 case "estimated": - return { text: a.status == null ? t.estimated : t.estimatedAfter(a.status), ok: true }; + return { text: a.status == null ? t.estimated : t.estimatedAfter(a.status), ok: true, tip: null }; + // 等过了开头的时限还没有内容,放弃了这一家、换了下一家。悬停说等了多久 + case "slow_start": + return { text: t.slowStart, ok: false, tip: said }; default: - return { text: a.error ? coreText(a.error) : t.noResponse, ok: false }; + return { text: said ?? t.noResponse, ok: false, tip: null }; } } @@ -172,7 +182,10 @@ export function deniedHopText(rule: string): string { return textOf(labelsText).deniedHop(rule); } -/** 没有发往任何上游的请求,在「上游」的位置上写什么:被规则拒绝,或者没有可用的上游 */ +/** + * 没有上游接下的请求,在「上游」的位置上写什么:被规则拒绝、没有可用的上游、密钥的用量 + * 上限拒绝了它,或者上游都满着(见 `NotSent`) + */ export function notSentText(kind: NotSent): string { return textOf(labelsText).notSent[kind]; } diff --git a/src/requestRouting.test.ts b/src/requestRouting.test.ts index cb91df88..7da7dd3e 100644 --- a/src/requestRouting.test.ts +++ b/src/requestRouting.test.ts @@ -15,6 +15,35 @@ const unavailable: Msg = { }; const served = (provider: string, ms = 900): AttemptView => ({ provider, outcome: "served", status: 200, ms }); const overloaded = (provider: string): AttemptView => ({ provider, outcome: "status", status: 529, ms: 1_870 }); +/** 网关密钥的用量上限拒绝时的那一句(`gw.key_limit.*`) */ +const limited: Msg = { + code: "gw.key_limit.cost_per_period", + args: { key: "cursor", max: "$5.00", per: "day", used: "$5.03", resets: "2026-09-26 00:00 +08:00" }, + text: "Gateway key `cursor` has reached its limit of $5.00 per day: $5.03 spent so far. It resets at 2026-09-26 00:00 +08:00.", +}; +/** 能服务的上游都满着、等过了也没空出来(`gw.busy_all`) */ +const busyAll: Msg = { + code: "gw.busy_all", + args: { upstreams: "`anthropic`, `openrouter`" }, + text: "Every upstream that can serve this request is at its concurrency limit (max_concurrent): `anthropic`, `openrouter`. None had a free slot in time; try again shortly.", +}; +/** 满着、没发出去的一跳(core 的 `hop_busy`)。`queued`:它是这段对话留着的那一家,等过空位 */ +const busy = (provider: string, queued?: number): AttemptView => ({ + provider, + outcome: "error", + error: { code: "gw.busy_upstream", args: { upstream: provider, limit: "2" }, text: "" }, + ms: 0, + skipped: "busy", + queued_ms: queued ?? null, +}); +/** 开头超时、放弃了的一跳 */ +const slow = (provider: string): AttemptView => ({ + provider, + outcome: "slow_start", + error: { code: "gw.slow_start", args: { upstream: provider, secs: "30" }, text: "" }, + ms: 30_004, + usage: { input: 48_210, cache_read: 0, cache_write: 0, estimated: true }, +}); function row(routing: Partial, over: Partial> = {}) { return { @@ -33,6 +62,16 @@ describe("没有发往任何上游的请求", () => { expect(notSent({ provider: "", error: { code: "gw.route.all_selected_disabled", args: {}, text: "" } })).toBe("unavailable"); }); + it("密钥的用量上限拒绝的:上游是空的,失败的那一句是 gw.key_limit.*", () => { + expect(notSent({ provider: "", error: limited })).toBe("limited"); + expect(notSent({ provider: "", error: { ...limited, code: "gw.key_limit.requests_rolling" } })).toBe("limited"); + }); + + it("上游都满着:那一行归在最后看过的那一家,也不算它的失败", () => { + expect(notSent({ provider: "openrouter", error: busyAll })).toBe("busy"); + expect(notSent({ provider: "", error: busyAll })).toBe("busy"); + }); + it("发往了上游的、本地应答的、还没有结局的都不算", () => { // 选定上游之后才被拒绝:记在要去的那个上游上 expect(notSent({ provider: "openrouter", error: denied("no-images", "no images") })).toBeNull(); @@ -79,6 +118,47 @@ describe("路由那一页", () => { expect(f.hops.map((h) => h.denied)).toEqual([false, false]); }); + it("开头超时、满着跳过的不说成失败:换过上游,说前几次尝试没有接下", () => { + const f = routingFacts(row({ attempts: [slow("anthropic"), served("openrouter")] }, { provider: "openrouter" }), false)!; + expect(f.note).toEqual({ kind: "switched", count: 1 }); + const g = routingFacts( + row({ attempts: [busy("anthropic"), overloaded("openrouter"), served("deepseek")] }, { provider: "deepseek" }), + false, + )!; + expect(g.note).toEqual({ kind: "switched", count: 2 }); + }); + + it("等到了空位:只有一跳,排队的时间在那一跳上,不加说明", () => { + const f = routingFacts(row({ attempts: [{ ...served("anthropic"), queued_ms: 1_240 }] }), false)!; + expect(f.note).toBeNull(); + expect(f.hops[0]!.attempt.queued_ms).toBe(1_240); + }); + + it("上游都满着:没有一跳发出去,原因是 core 说的那一句", () => { + const f = routingFacts( + row({ attempts: [busy("anthropic", 30_000), busy("openrouter")] }, { provider: "openrouter", error: busyAll }), + false, + )!; + expect(f.note).toEqual({ kind: "busy", tried: 0 }); + expect(f.reason).toEqual({ msg: busyAll }); + expect(f.hops.map((h) => h.denied)).toEqual([false, false]); + }); + + it("上游都满着:之前发出去、没成的几跳另说", () => { + const f = routingFacts( + row({ attempts: [overloaded("anthropic"), busy("openrouter")] }, { provider: "openrouter", error: busyAll }), + false, + )!; + expect(f.note).toEqual({ kind: "busy", tried: 1 }); + }); + + it("密钥的用量上限拒绝了它:没有尝试,原因是 core 说的那一句", () => { + const f = routingFacts(row({ attempts: [] }, { provider: "", error: limited }), false)!; + expect(f.ruleDenied).toBe(false); + expect(f.note).toEqual({ kind: "limited" }); + expect(f.reason).toEqual({ msg: limited }); + }); + it("选定上游之前被拒绝:没有尝试,原因是规则里写的那句", () => { const f = routingFacts( row({ route: "codex", rule: "no-opus", group: null, attempts: [] }, { provider: "", error: denied("no-opus", "Opus is not offered") }), diff --git a/src/requestRouting.ts b/src/requestRouting.ts index d6a0c9b1..7de5fa39 100644 --- a/src/requestRouting.ts +++ b/src/requestRouting.ts @@ -7,24 +7,44 @@ import type { AttemptView, HistoryRow, Msg, Stay } from "./types"; /** 规则拒绝时 core 说的那一句:`gw.route.denied {rule, reason}`。码和参数名是契约 */ const DENIED = "gw.route.denied"; +/** 网关密钥的用量上限拒绝时的那几句:`gw.key_limit.*`,一种量、一种周期一句 */ +const KEY_LIMIT = "gw.key_limit."; +/** 能服务的上游都满着、等过了也没空出位置:`gw.busy_all {upstreams}` */ +const BUSY_ALL = "gw.busy_all"; /** - * 一条请求为什么没有发往任何上游: + * 一条请求为什么没有上游接下: * * · `denied`:规则拒绝了它(选定上游之前) * · `unavailable`:规则选中的上游一个都接不了(停用、不在范围内、不提供这个模型) + * · `limited`:这把网关密钥的用量上限拒绝了它(选定上游之后、发出之前) + * · `busy`:能服务它的上游都满着(各自的并发上限),等过了也没空出位置 * * 发往了上游的、本地应答的、还没有结局的是 `null`。 * - * **上游是空的就是没有发往任何上游**:core 只在规则做了决定、请求却一个上游都不会去时 - * 把上游记成空的(本地应答另有 `local`)。是哪一种看失败的那一句:规则拒绝的是 - * `gw.route.denied`,其余是选中的上游接不了。 + * **上游是空的就是没有发往任何上游**:core 只在请求一个上游都不会去时把上游记成空的 + * (本地应答另有 `local`)。是哪一种看失败的那一句:规则拒绝的是 `gw.route.denied`, + * 用量上限的是 `gw.key_limit.*`,其余是选中的上游接不了。 + * + * **`busy` 的上游不是空的**:那一行归在尝试链的最后一跳,也就是最后看过、满着的那一家。 + * 可那一家一个字节都没收到,结局也不是它的失败 —— 这一格写「并发已满」,不写它的名字 */ -export type NotSent = "denied" | "unavailable"; +export type NotSent = "denied" | "unavailable" | "limited" | "busy"; export function notSent(r: { local?: boolean; provider: string; error?: Msg | null }): NotSent | null { - if (r.local || r.provider !== "" || !r.error) return null; - return r.error.code === DENIED ? "denied" : "unavailable"; + if (r.local || !r.error) return null; + if (r.error.code === BUSY_ALL) return "busy"; + if (r.provider !== "") return null; + if (r.error.code === DENIED) return "denied"; + return r.error.code.startsWith(KEY_LIMIT) ? "limited" : "unavailable"; +} + +/** + * 尝试链里这家满着、没有发出去就换了下一家的一跳(`skipped`)。另一种没发出去的一跳 —— + * 选定上游之后被规则拒绝 —— 由 `Hop.denied` 认 + */ +export function skippedHop(a: AttemptView): boolean { + return a.skipped != null; } /** 尝试链里的一跳。`denied`:选定上游之后的规则在这一跳拒绝了它,**没有发给这个上游** */ @@ -37,6 +57,11 @@ export interface Hop { * 尝试链下面那一句。只说字面上成立的事: * * · `failover`:前 `failed` 个上游失败,换到了下一个(最后一跳的结果在它自己那一行) + * · `switched`:前 `count` 跳没有接下它,换到了下一跳。其中有满着跳过的、或者开头超时 + * 放弃的 —— 那两种不是上游的失败,不说「失败」(每一跳为什么没接下在它自己那一行) + * · `limited`:这把网关密钥的用量上限拒绝了它:没有发往任何上游 + * · `busy`:剩下的上游都满着,等过了也没空出位置。`tried` 是在那之前真的发出去、没成的 + * 几跳;0 就是没有发往任何上游 * · `failover_denied`:前 `failed` 个上游失败,换到下一个之后被规则 `rule` 拒绝,没有发给它 * · `denied_after_pick`:唯一的那一跳被规则 `rule` 拒绝:没有发往任何上游 * · `denied_before_pick`:选定上游之前规则 `rule` 就拒绝了它:没有发往任何上游 @@ -48,6 +73,9 @@ export interface Hop { */ export type RoutingNote = | { kind: "failover"; failed: number } + | { kind: "switched"; count: number } + | { kind: "limited" } + | { kind: "busy"; tried: number } | { kind: "failover_denied"; failed: number; rule: string } | { kind: "denied_after_pick"; rule: string } | { kind: "denied_before_pick"; rule: string } @@ -110,14 +138,19 @@ export function routingFacts( if (ruleDenied || deniedBy) { const text = denyReason(deniedHop?.error) ?? denyReason(r.error); reason = text ? { text } : null; - } else if (why === "unavailable" && r.error) { + } else if ((why === "unavailable" || why === "limited" || why === "busy") && r.error) { + // 用量上限的那一句说清是哪一条、用了多少、什么时候重置;满着的那一句列出是哪几家 reason = { msg: r.error }; } let note: RoutingNote | null = null; - if (n === 0) { + if (why === "busy") { + // 满着跳过的几跳都没发出去。之前真的发出去、没成的那几跳另算 + note = { kind: "busy", tried: attempts.filter((a) => !skippedHop(a)).length }; + } else if (n === 0) { if (ruleDenied) note = { kind: "denied_before_pick", rule: routing.rule }; else if (why === "unavailable") note = { kind: "unavailable" }; + else if (why === "limited") note = { kind: "limited" }; else note = { kind: running ? "pending" : "none" }; } else if (deniedBy) { // 被拒的那一跳不算切换成功:前面几跳失败、换过来,才被拒绝 @@ -126,7 +159,9 @@ export function routingFacts( ? { kind: "failover_denied", failed: n - 1, rule: deniedBy } : { kind: "denied_after_pick", rule: deniedBy }; } else if (n > 1) { - note = { kind: "failover", failed: n - 1 }; + // 满着跳过、开头超时放弃都不是上游的失败 + const plain = attempts.slice(0, -1).every((a) => !skippedHop(a) && a.outcome !== "slow_start"); + note = plain ? { kind: "failover", failed: n - 1 } : { kind: "switched", count: n - 1 }; } return { diff --git a/src/traffic/SessionPanel.tsx b/src/traffic/SessionPanel.tsx index 0bf180e2..6436b9bc 100644 --- a/src/traffic/SessionPanel.tsx +++ b/src/traffic/SessionPanel.tsx @@ -3,6 +3,7 @@ import { call } from "@/control"; import { useText } from "@/i18n"; import { cn } from "@/lib/utils"; import { useResource } from "@/lib/resource"; +import { notSent } from "@/requestRouting"; import type { RequestRow, SessionDetail, TurnView } from "@/types"; import { Button } from "@/ui/button"; import { UpstreamLogo } from "@/ui/logos"; @@ -131,8 +132,11 @@ export function SessionPanel({ const pending = useMemo(() => unrecordedRows(turns, rows), [turns, rows]); const n = tally(s, rows, pending); // 走过哪几个上游,按第一次出现的先后。一次任务中途换过上游,这里能看出来。 - // 没有发往任何上游的那几轮(被规则拒绝)上游是空的,不算 - const providers = [...new Set([...turns.map((x) => x.provider), ...pending.map((r) => r.provider)].filter(Boolean))]; + // 没有发往任何上游的那几轮(被规则拒绝)上游是空的,不算;上游都满着的那几轮记在最后 + // 看过的那一家上,那一家没收到它,也不算(见 `notSent`) + const providers = [ + ...new Set([...turns, ...pending].filter((x) => notSent(x) === null).map((x) => x.provider).filter(Boolean)), + ]; const client = s?.client ?? pending[0]?.client; const [tab, setTab] = useState("summary"); /** 「对话」打开过:之后切走也留着(读到哪儿、展开了哪几条都在),见 `PANE` */ diff --git a/src/traffic/SessionRow.tsx b/src/traffic/SessionRow.tsx index 51dc857a..3be2cb6a 100644 --- a/src/traffic/SessionRow.tsx +++ b/src/traffic/SessionRow.tsx @@ -2,6 +2,7 @@ import { memo } from "react"; import { ChevronRightIcon } from "lucide-react"; import { cn } from "@/lib/utils"; import { useText } from "@/i18n"; +import { notSent } from "@/requestRouting"; import { Button } from "@/ui/button"; import { UpstreamLogo } from "@/ui/logos"; import { RowMenu, RowMenuButton, type MenuItems } from "@/ui/row-menu"; @@ -66,8 +67,9 @@ export const SessionRow = memo(function SessionRow({ // 汇总加上汇总里还没有的那几轮:在跑的、刚落地的(见 `tally`) const n = tallyOf(g); // **上游从行里数,不从汇总里拿** —— `SessionView` 没有这一项, - // 而组里的每一条都知道自己走了哪个上游(没有发往任何上游的那几条是空的,不算) - const providers = [...new Set(rows.map((r) => r.provider).filter(Boolean))]; + // 而组里的每一条都知道自己走了哪个上游(没有发往任何上游的那几条是空的,不算;上游都满着 + // 的那几条记在最后看过的那一家上,可那一家没收到它,也不算 —— 见 `notSent`) + const providers = [...new Set(rows.filter((r) => notSent(r) === null).map((r) => r.provider).filter(Boolean))]; const openIt = () => { // 点组头和点请求行一样,键盘接着从这一行往下走 onCursor({ kind: "session", id }); diff --git a/src/traffic/cells.tsx b/src/traffic/cells.tsx index b72681ba..433c1ccf 100644 --- a/src/traffic/cells.tsx +++ b/src/traffic/cells.tsx @@ -6,7 +6,7 @@ import { keyText } from "@/KeyLabel"; import { cn } from "@/lib/utils"; import type { NotSent } from "@/requestRouting"; import type { RequestRow } from "@/types"; -import { IconDenied, IconNoUpstream, IconRemote } from "@/ui/icons"; +import { IconBusy, IconDenied, IconLimitReached, IconNoUpstream, IconRemote } from "@/ui/icons"; import { ClientLogo } from "@/ui/logos"; import { notify } from "@/ui/notify"; import { Tip } from "@/ui/tip"; @@ -103,22 +103,25 @@ export function RowKeyCell({ r, hints }: { r: RequestRow; hints: boolean }) { return ; } +/** 每一种在上游标志的位置上画的图形和颜色 */ +const NOT_SENT: Record = { + denied: { Icon: IconDenied, color: "text-destructive" }, + unavailable: { Icon: IconNoUpstream, color: "text-muted-foreground" }, + limited: { Icon: IconLimitReached, color: "text-warning" }, + busy: { Icon: IconBusy, color: "text-muted-foreground" }, +}; + /** - * 没有发往任何上游的请求在上游标志的位置上画什么:被规则拒绝是禁止符号(和路由图上 - * 「拒绝」那个节点同一个),没有可用的上游是划掉的上游。和上游标志一样大,名字对得齐。 + * 没有上游接下的请求在上游标志的位置上画什么:被规则拒绝是禁止符号(和路由图上 + * 「拒绝」那个节点同一个),没有可用的上游是划掉的上游,密钥的用量到了上限是顶到线的 + * 箭头,上游都满着是沙漏。和上游标志一样大,名字对得齐。 * - * 拒绝带红色,和路由图一致;`plain` 时跟着周围的字色(命令面板里的图标都是单色)。 + * 拒绝带红色,和路由图一致;用量上限是琥珀色:到了上限要留意,但不是故障。`plain` 时 + * 跟着周围的字色(命令面板里的图标都是单色)。 */ export function NotSentIcon({ kind, plain }: { kind: NotSent; plain?: boolean }) { - const Icon = kind === "denied" ? IconDenied : IconNoUpstream; - return ( - - ); + const { Icon, color } = NOT_SENT[kind]; + return ; } /** diff --git a/src/ui/icons.tsx b/src/ui/icons.tsx index aaefe203..4c5d8839 100644 --- a/src/ui/icons.tsx +++ b/src/ui/icons.tsx @@ -28,6 +28,8 @@ export { Puzzle as IconPlugin } from "lucide-react"; // 插件 —— 拼进链 export { Server as IconServer } from "lucide-react"; // 上游 —— 一摞机器 export { ServerOff as IconNoUpstream } from "lucide-react"; // 没有可用的上游 —— 那一摞机器划掉 export { Ban as IconDenied } from "lucide-react"; // 被规则拒绝 —— 禁止符号 +export { ArrowUpToLine as IconLimitReached } from "lucide-react"; // 密钥的用量到了上限 —— 顶到那条线 +export { Hourglass as IconBusy } from "lucide-react"; // 上游都满着、等不到空位 —— 沙漏 export { KeyRound as IconKey } from "lucide-react"; // 密钥 —— 钥匙,不是锁 export { SlidersHorizontal as IconSettings } from "lucide-react"; // 设置 —— 推子,齿轮留给系统设置 export { PanelLeft as IconSidebar } from "lucide-react"; // 收起/展开源列表 From 53fc6cef73017f3e4f25155e06dca9375da9783b Mon Sep 17 00:00:00 2001 From: fylorn <249551762+fylorn@users.noreply.github.com> Date: Mon, 5 Oct 2026 18:09:31 +0800 Subject: [PATCH 04/10] Keys: usage limits per key in the key dialog, a "Limit reached" chip, notifications at 80% and at the limit MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Core can now cap a gateway key by requests, tokens or cost per minute, hour, day, week or month (clients[].limits), and reports each limit with what it has used. This gives that a place in Lite. - Key dialog: a "Usage limits" section. Each limit is one row that reads as a sentence (Per [day] at most [5.00] [USD]); token limits add "Count cache reads". Rows are checked on the spot with core's rules (a positive amount, no two limits with the same period, measure and cache-read setting, and a monthly limit needs records kept for 31 days), so a wrong limit is caught before Save. A blank amount is only flagged once the field is left, so a newly added row is not red. Cost is typed in dollars and sent as micros. - For an existing key each row shows what is used ("Today $1.23 / $5.00 · resets at 00:00", "Last minute 12 / 30") and marks a reached limit. With a cost limit, the models this key can use that have no price are listed (folded after six): they count as $0, so the limit does not hold them back. - Keys table: a "Limit reached" chip next to the disabled state; hover names the limit and when it resets. The list is fetched again on key_limit_alert, at the next period reset and after the clock jumps, so the chip does not linger after midnight on an idle key. - Disabling or enabling a key from the table now sends its limits back: the key input replaces the whole list, so leaving them out would have deleted every limit. - Notifications: KeyLimitAlert becomes a system notification through the notice bus, naming the key, the limit and when it resets. One notice per limit: reaching the limit replaces the 80% notice, and the next period's 80% notice replaces yesterday's "reached". Each period counts as a new event, so a notice read yesterday does not silence today's. Clicking it opens the Keys page. Co-Authored-By: Claude Opus 5.5 --- src-tauri/src/notices/rules.rs | 131 +++++++++++++++ src-tauri/src/notices/tests.rs | 143 +++++++++++++++++ src/keys/KeyDialog.i18n.ts | 4 + src/keys/KeyDialog.tsx | 27 +++- src/keys/KeysPage.tsx | 21 +++ src/keys/KeysTable.i18n.ts | 4 + src/keys/KeysTable.tsx | 33 ++++ src/keys/LimitsEditor.i18n.ts | 50 ++++++ src/keys/LimitsEditor.tsx | 286 +++++++++++++++++++++++++++++++++ src/keys/data.ts | 8 +- src/keys/limits.i18n.ts | 55 +++++++ src/keys/limits.test.ts | 161 +++++++++++++++++++ src/keys/limits.ts | 188 ++++++++++++++++++++++ 13 files changed, 1105 insertions(+), 6 deletions(-) create mode 100644 src/keys/LimitsEditor.i18n.ts create mode 100644 src/keys/LimitsEditor.tsx create mode 100644 src/keys/limits.i18n.ts create mode 100644 src/keys/limits.test.ts create mode 100644 src/keys/limits.ts diff --git a/src-tauri/src/notices/rules.rs b/src-tauri/src/notices/rules.rs index 00350099..c3b094c4 100644 --- a/src-tauri/src/notices/rules.rs +++ b/src-tauri/src/notices/rules.rs @@ -48,6 +48,7 @@ const UPSTREAMS: &str = "upstreams"; const SECURITY: &str = "security"; const MCP: &str = "mcp"; const PLUGINS: &str = "plugins"; +const KEYS: &str = "keys"; const SETTINGS: &str = "settings"; /// 设置页的「网关监听」一节(`settings:<节>`,界面滚到那一节) const LISTEN_SETTINGS: &str = "settings:listen"; @@ -60,6 +61,7 @@ pub fn default_view(key: &str) -> &'static str { // 客户端配置里的可疑内容在 MCP 页:服务器、技能、钩子和扫描发现都在那儿 "scan" => MCP, "plugin" => PLUGINS, + "keylimit" => KEYS, // 网关、配置文件、监听,以及认不出来的:设置页至少能看到网关在不在跑 _ => SETTINGS, } @@ -396,6 +398,7 @@ pub fn from_event(ev: &Event) -> Vec { .event(), ] } + Event::KeyLimitAlert { .. } => key_limit(ev), Event::PluginFailed { plugin_id, plugin_name, @@ -407,6 +410,134 @@ pub fn from_event(ev: &Event) -> Vec { } } +/// 一把网关密钥这一期(天、周、月)的用量到了一条上限的八成,或者到了上限。 +/// +/// **每一期都是新的一件**(`event`):core 每一期、每一档只报一次,昨天看过的那一条不该 +/// 压住今天的。**同一条上限只留一条**:到顶时收起这一期八成的那条;新的一期到了八成, +/// 上一期到顶的那条也不再是现状,一并收起。正文写密钥的名字(不写它的值)、哪一条、 +/// 什么时候重置 +fn key_limit(ev: &Event) -> Vec { + let &Event::KeyLimitAlert { + ref key, + per, + measure, + max, + used, + cache_reads, + reached, + resets_at_ms, + .. + } = ev + else { + return Vec::new(); + }; + // 同一条上限:同一个周期、同一种量、缓存读取算法相同(和 core 认重复的规矩一样) + let limit = format!( + "keylimit:{key}:{per}-{measure}{}", + if cache_reads { "-cache" } else { "" } + ); + let near = format!("{limit}:near"); + let phrase = limit_phrase(per, measure, max, cache_reads); + let secs = resets_at_ms.saturating_sub(super::now_ms()) / 1000; + let reset = after(secs); + if reached { + return vec![ + Signal::cleared(near), + Signal::raised( + limit, + Level::Warning, + tr!( + format!("密钥「{key}」已达用量上限"), + format!("Key “{key}” Reached Its Usage Limit") + ), + ) + .body(tr!( + format!("上限「{phrase}」已用满{reset}。使用此密钥的请求会被拒绝。"), + format!( + "The limit of {phrase} has been reached{reset}. Requests with this key will be rejected." + ) + )) + .view(KEYS) + .event(), + ]; + } + let percent = (used.saturating_mul(100) / max.max(1)).min(99); + let n = limit_amount(measure, used); + let spent = match measure { + tw_api::LimitMeasure::Requests => tr!(format!("{n} 次"), format!("{n} requests")), + tw_api::LimitMeasure::Tokens => tr!(format!("{n} token"), format!("{n} tokens")), + tw_api::LimitMeasure::Cost => n, + }; + vec![ + Signal::cleared(limit.clone()), + Signal::raised( + near, + Level::Warning, + tr!( + format!("密钥「{key}」的用量接近上限"), + format!("Key “{key}” Is Nearing Its Usage Limit") + ), + ) + .body(tr!( + format!( + "上限「{phrase}」已使用 {percent}%({spent}){reset}。达到上限后,使用此密钥的请求会被拒绝。" + ), + format!( + "The limit of {phrase} is {percent}% used ({spent}){reset}. Once it is reached, requests with this key will be rejected." + ) + )) + .view(KEYS) + .event(), + ] +} + +/// 一条上限说成一句:「每天 $5.00 费用」「每天 1,000,000 token(含缓存读取)」;英文是 +/// 「$5.00 per day」「1,000,000 tokens per day (cache reads included)」。和密钥对话框里的同一种说法 +fn limit_phrase( + per: tw_api::LimitPer, + measure: tw_api::LimitMeasure, + max: u64, + cache_reads: bool, +) -> String { + use tw_api::{LimitMeasure, LimitPer}; + let (zh, en) = match per { + LimitPer::Minute => ("分钟", "minute"), + LimitPer::Hour => ("小时", "hour"), + LimitPer::Day => ("天", "day"), + LimitPer::Week => ("周", "week"), + LimitPer::Month => ("月", "month"), + }; + let n = limit_amount(measure, max); + match measure { + LimitMeasure::Requests => tr!( + format!("每{zh} {n} 次请求"), + format!("{n} requests per {en}") + ), + LimitMeasure::Tokens => { + let cache = if cache_reads { + tr!("(含缓存读取)", " (cache reads included)") + } else { + "" + }; + tr!( + format!("每{zh} {n} token{cache}"), + format!("{n} tokens per {en}{cache}") + ) + } + LimitMeasure::Cost => tr!(format!("每{zh} {n} 费用"), format!("{n} per {en}")), + } +} + +/// 用量或上限写成字:费用是微分,写到分;请求数、token 数带千分位 +fn limit_amount(measure: tw_api::LimitMeasure, n: u64) -> String { + use crate::menubar::model::{cost_long, grouped}; + let n = i64::try_from(n).unwrap_or(i64::MAX); + match measure { + tw_api::LimitMeasure::Cost => cost_long(n), + _ => grouped(n), + } +} + /// 一个插件没能把事情做成(core 的 `plugin_failed`):在一个请求上运行出错(`request` 有), /// 或者文件变了、加载不了,从此不再运行(没有)。 /// diff --git a/src-tauri/src/notices/tests.rs b/src-tauri/src/notices/tests.rs index 71dc8e3a..a0faa09b 100644 --- a/src-tauri/src/notices/tests.rs +++ b/src-tauri/src/notices/tests.rs @@ -608,6 +608,7 @@ fn every_key_lands_on_the_page_that_handles_it() { ("toolwall:relay", "security"), ("scan", "mcp"), ("plugin:add-date", "plugins"), + ("keylimit:codex:day-cost", "keys"), ] { assert_eq!(rules::default_view(key), view, "{key}"); } @@ -1531,3 +1532,145 @@ async fn reconciling_fills_in_what_was_missed_without_saying_anything_twice() { "{shown:?}" ); } + +/// 一把网关密钥的一条上限到了八成(`reached` 为假)或者到顶,`secs` 秒之后重置 +fn key_limit_alert( + measure: tw_api::LimitMeasure, + max: u64, + used: u64, + cache_reads: bool, + reached: bool, + secs: u64, +) -> tw_api::Event { + tw_api::Event::KeyLimitAlert { + id: 1, + key: "codex".into(), + per: tw_api::LimitPer::Day, + measure, + max, + used, + cache_reads, + reached, + resets_at_ms: resets_in(secs).unwrap(), + at_ms: T0, + } +} + +/// 通知写明是哪把密钥(名字,不是值)、哪一条上限、什么时候重置 +#[test] +fn a_key_limit_names_the_key_the_limit_and_the_reset() { + use tw_api::LimitMeasure::{Cost, Requests, Tokens}; + let said = |ev: &tw_api::Event| { + let s = rules::from_event(ev) + .into_iter() + .find(|s| s.change == Change::Raised) + .expect("要说"); + (s.title, s.body) + }; + with_lang(Lang::Zh, || { + assert_eq!( + said(&key_limit_alert( + Cost, + 5_000_000, + 5_020_000, + false, + true, + 5 * 3600 + )), + ( + "密钥「codex」已达用量上限".to_string(), + "上限「每天 $5.00 费用」已用满,约 5 小时后重置。使用此密钥的请求会被拒绝。" + .to_string() + ) + ); + assert_eq!( + said(&key_limit_alert( + Tokens, 1_000_000, 812_345, true, false, 600 + )) + .1, + "上限「每天 1,000,000 token(含缓存读取)」已使用 81%(812,345 token),约 10 分钟后重置。\ + 达到上限后,使用此密钥的请求会被拒绝。" + ); + assert_eq!( + said(&key_limit_alert( + Requests, + 500, + 400, + false, + false, + 3 * 86_400 + )) + .1, + "上限「每天 500 次请求」已使用 80%(400 次),约 3 天后重置。达到上限后,使用此密钥的请求会被拒绝。" + ); + }); + with_lang(Lang::En, || { + assert_eq!( + said(&key_limit_alert( + Cost, + 5_000_000, + 5_020_000, + false, + true, + 5 * 3600 + )), + ( + "Key “codex” Reached Its Usage Limit".to_string(), + "The limit of $5.00 per day has been reached and resets in about 5 hours. \ + Requests with this key will be rejected." + .to_string() + ) + ); + let (title, body) = said(&key_limit_alert( + Tokens, 1_000_000, 812_345, true, false, 600, + )); + assert_eq!(title, "Key “codex” Is Nearing Its Usage Limit"); + assert_eq!( + body, + "The limit of 1,000,000 tokens per day (cache reads included) is 81% used \ + (812,345 tokens) and resets in about 10 minutes. Once it is reached, requests with \ + this key will be rejected." + ); + }); +} + +/// 八成说一次、到顶再说一次;**同一条上限只留一条**。新的一期到八成时,上一期到顶的那条 +/// 收起、八成那条重新说(每一期都是新的一件,昨天看过的不压住今天的) +#[tokio::test] +async fn a_key_limit_is_told_at_80_percent_and_at_the_limit_once_per_period() { + use tw_api::LimitMeasure::Cost; + let b = bed(); + let near = key_limit_alert(Cost, 5_000_000, 4_100_000, false, false, 3600); + let reached = key_limit_alert(Cost, 5_000_000, 5_000_000, false, true, 3600); + b.bus.on_event(&near); + b.bus.on_event(&reached); + assert_eq!(b.titles().len(), 2, "两档都弹系统通知:{:?}", b.titles()); + let keys = |b: &Bed| b.bus.list().into_iter().map(|n| n.key).collect::>(); + assert_eq!(keys(&b), ["keylimit:codex:day-cost"], "到顶时收起八成那条"); + assert_eq!( + *b.withdrawn.lock().unwrap(), + ["keylimit:codex:day-cost:near"] + ); + + b.bus.mark_all_read(); + // 第二天 + b.bus.on_event(&near); + assert_eq!( + keys(&b), + ["keylimit:codex:day-cost:near"], + "上一期到顶的那条收起" + ); + assert!(!b.bus.list()[0].read, "新的一期是新的一件"); + assert_eq!(b.titles().len(), 3); + + // 算缓存读取的 token 上限和不算的是两条 + b.bus.on_event(&key_limit_alert( + tw_api::LimitMeasure::Tokens, + 100, + 100, + true, + true, + 3600, + )); + assert!(keys(&b).contains(&"keylimit:codex:day-tokens-cache".to_string())); +} diff --git a/src/keys/KeyDialog.i18n.ts b/src/keys/KeyDialog.i18n.ts index d2152e83..36db35bf 100644 --- a/src/keys/KeyDialog.i18n.ts +++ b/src/keys/KeyDialog.i18n.ts @@ -5,6 +5,8 @@ export const keyDialogText = messages( nameRequired: "请填写名称", nameTaken: "这个名称已被占用", patternsRequired: "请至少添加一条规则或选中一个模型", + limitRequired: "请填写用量上限的数值", + limitsInvalid: "请修正用量上限", editTitle: "编辑密钥", newTitle: "新建密钥", newDescription: "新密钥立即可用。客户端把它填进请求头即可连接网关。", @@ -28,6 +30,8 @@ export const keyDialogText = messages( nameRequired: "A name is required", nameTaken: "This name is already in use", patternsRequired: "At least one pattern or one model is required", + limitRequired: "Enter an amount for each usage limit", + limitsInvalid: "Fix the usage limits", editTitle: "Edit key", newTitle: "New key", newDescription: diff --git a/src/keys/KeyDialog.tsx b/src/keys/KeyDialog.tsx index 38f3c99c..275759e0 100644 --- a/src/keys/KeyDialog.tsx +++ b/src/keys/KeyDialog.tsx @@ -21,6 +21,8 @@ import { api } from "./api"; import type { KeyUse } from "./data"; import { keyDialogText } from "./KeyDialog.i18n"; import { errorText, routeLabel, takeoverOf } from "./labels"; +import { inputsOf, limitProblems, rowsOf, type LimitRow } from "./limits"; +import { LimitsEditor } from "./LimitsEditor"; import { TakeoverBadge } from "./KeysTable"; import { ModelScope } from "./ModelScope"; import { CopyButton, focusSelf, useDialogFocus } from "./parts"; @@ -42,6 +44,7 @@ export function KeyDialog({ routes, defaultRoute, catalog, + rowDays, version, onClose, onSaved, @@ -59,6 +62,8 @@ export function KeyDialog({ defaultRoute: string; /** 网关知道的全部模型,用来勾选可见范围。取不到时为空 */ catalog: KnownModel[]; + /** 请求记录留几天(概览里的)。每月的用量上限要求至少 31 天 */ + rowDays: number | null; /** 这一页最后知道的配置版本。**打开时读一次**,保存带的是那一个(见下面的 `base`) */ version: { get: () => string }; onClose: () => void; @@ -74,6 +79,7 @@ export function KeyDialog({ const [scope, setScope] = useState(scopeOf(editing?.allow)); const [entries, setEntries] = useState(editing?.allow ?? []); const [limit, setLimit] = useState(editing?.max_concurrent != null ? String(editing.max_concurrent) : ""); + const [limits, setLimits] = useState(() => rowsOf(editing?.limits)); const [enabled, setEnabled] = useState(!editing?.disabled); const [saving, setSaving] = useState(false); const [error, setError] = useState(null); @@ -86,14 +92,19 @@ export function KeyDialog({ const owner = editing ? takeoverOf(editing, clients, manual) : null; const taken = keys.some((k) => k.name === name.trim() && k.name !== editing?.name); + const problems = limitProblems(limits, rowDays); const missing = name.trim().length === 0 ? t.nameRequired : taken ? t.nameTaken - : scope === "some" && entries.length === 0 - ? t.patternsRequired - : null; + : problems.size > 0 + ? [...problems.values()].every((p) => p === "required") + ? t.limitRequired + : t.limitsInvalid + : scope === "some" && entries.length === 0 + ? t.patternsRequired + : null; async function save() { setSaving(true); @@ -105,6 +116,7 @@ export function KeyDialog({ allow: allowOf(scope, entries), max_concurrent: limit.trim() ? Number(limit.trim()) : null, disabled: !enabled, + limits: inputsOf(limits), }, base_version: base, }; @@ -214,6 +226,15 @@ export function KeyDialog({ /> + +
    diff --git a/src/keys/KeysPage.tsx b/src/keys/KeysPage.tsx index 59f4651c..935ad9db 100644 --- a/src/keys/KeysPage.tsx +++ b/src/keys/KeysPage.tsx @@ -14,6 +14,7 @@ import type { ClientView, KeyInput, Overview } from "@/types"; import { CostFigure } from "@/CostFigure"; import { useText } from "@/i18n"; import { useClients } from "@/clients/data"; +import { useCoreEvent } from "@/useCoreEvent"; import { writeQueue } from "@/lib/writeQueue"; import { api } from "./api"; import { CreatedDialog } from "./CreatedDialog"; @@ -23,9 +24,13 @@ import { KeyDialog } from "./KeyDialog"; import { KeysTable } from "./KeysTable"; import { keysPageText } from "./KeysPage.i18n"; import { takeoverOf } from "./labels"; +import { inputOfView, nextReset } from "./limits"; import { RowsSkeleton } from "./parts"; import { RotateDialog } from "./RotateDialog"; +/** setTimeout 能等的最长时间(2^31 − 1 毫秒):再长会当成 0,立刻就响 */ +const MAX_TIMER_MS = 2_147_483_647; + type DialogState = | null | { kind: "edit"; name: string | null } @@ -102,6 +107,19 @@ export default function KeysPage({ return () => clearTimeout(h); }, [highlight]); + /* + 用量上限按天、周、月重新算:到了那一刻重取一次,「已达上限」和对话框里的用量跟着 + 换(平时请求落地就会重取,这一次是给一直没有请求的时候)。定时器量的钟睡着时不走: + 睡醒、改了时钟(`clock_changed`)也重取。setTimeout 最多等 24.8 天,再远的到时候再排 + */ + const reset = nextReset(list); + useEffect(() => { + if (reset == null) return; + const h = setTimeout(() => void keys.reload(), Math.min(MAX_TIMER_MS, Math.max(1_000, reset - Date.now() + 1_000))); + return () => clearTimeout(h); + }, [reset]); + useCoreEvent(["clock_changed"], () => void keys.reload()); + /** 写完一次:记下新版本,重读列表,告诉外壳(概览跟着重读) */ function wrote(v: string) { version.set(v); @@ -252,6 +270,7 @@ export default function KeysPage({ routes={ov.routes} defaultRoute={ov.default_route} catalog={catalog.data ?? []} + rowDays={ov.retention.row_days} version={version} onClose={() => setDialog(null)} onSaved={(name, v) => { @@ -372,6 +391,8 @@ function inputOf(k: ClientView, patch: Partial): KeyInput { allow: k.allow ?? null, max_concurrent: k.max_concurrent, disabled: k.disabled ?? false, + // 上限是整份替换的:不带就是一条都不要了 + limits: k.limits.map(inputOfView), ...patch, }; } diff --git a/src/keys/KeysTable.i18n.ts b/src/keys/KeysTable.i18n.ts index 7af85b3c..152f8cfd 100644 --- a/src/keys/KeysTable.i18n.ts +++ b/src/keys/KeysTable.i18n.ts @@ -10,6 +10,8 @@ export const keysTableText = messages( actions: "操作", default: "默认", disabled: "已停用", + limitReached: "已达上限", + reachedUntil: (limit: string, resets: string) => `${limit},${resets}`, actionsFor: (name: string) => `${name} 的操作`, copyKey: "复制密钥", edit: "编辑…", @@ -31,6 +33,8 @@ export const keysTableText = messages( actions: "Actions", default: "Default", disabled: "Disabled", + limitReached: "Limit reached", + reachedUntil: (limit: string, resets: string) => `${limit}, ${resets}`, actionsFor: (name: string) => `Actions for ${name}`, copyKey: "Copy key", edit: "Edit…", diff --git a/src/keys/KeysTable.tsx b/src/keys/KeysTable.tsx index fa3a47b9..f5bc935b 100644 --- a/src/keys/KeysTable.tsx +++ b/src/keys/KeysTable.tsx @@ -13,6 +13,7 @@ import type { KeyUse } from "./data"; import { keysTableText } from "./KeysTable.i18n"; import { labelsText } from "./labels.i18n"; import { routeLabel, scopeLabel, takeoverOf, type KeyOwner } from "./labels"; +import { limitPhrase, resetText } from "./limits"; import { ClientMark, CopyIconButton, CostCell, OPENABLE_ROW, Tile, UsageCell, openable, stop } from "./parts"; export interface KeyActions { @@ -123,6 +124,7 @@ export function KeysTable({ {t.disabled} )} + {owner && }
    actions.copy(k.name, true)} /> @@ -153,6 +155,37 @@ export function KeysTable({ ); } +/** + * 用到上限的那几把:「已达上限」,悬停写是哪一条、什么时候重置。**没到就不出现** —— + * 状态只在异常时出现 + */ +function LimitReached({ limits }: { limits: ClientView["limits"] }) { + const t = useText(keysTableText); + const reached = limits.filter((l) => l.reached); + if (reached.length === 0) return null; + const now = Date.now(); + return ( + + {reached.map((l) => ( +

    + {l.resets_at_ms != null && l.resets_at_ms > now + ? t.reachedUntil(limitPhrase(l), resetText(l.resets_at_ms, now)) + : limitPhrase(l)} +

    + ))} + + } + > + + + {t.limitReached} + +
    + ); +} + /** * 为某个客户端生成的那几把,**单独一个标记**,写出是为谁生成的 —— 接管时生成的写 * 「接管 · Claude Code」,手动配置时生成的写「手动配置 · Cursor」。 diff --git a/src/keys/LimitsEditor.i18n.ts b/src/keys/LimitsEditor.i18n.ts new file mode 100644 index 00000000..71b01798 --- /dev/null +++ b/src/keys/LimitsEditor.i18n.ts @@ -0,0 +1,50 @@ +import { messages } from "@/i18n"; + +export const limitsEditorText = messages( + { + title: "用量上限", + hint: "任一上限用满后,此密钥的请求会被拒绝", + none: "未设上限,用量不限", + add: "添加上限", + every: "每", + atMost: "最多", + perLabel: (n: number) => `第 ${n} 条上限的周期`, + maxLabel: (n: number) => `第 ${n} 条上限的数值`, + measureLabel: (n: number) => `第 ${n} 条上限的计量`, + cacheReads: "计入缓存读取", + remove: "删除", + removeLabel: (n: number) => `删除第 ${n} 条上限`, + required: "请填写上限", + notPositive: (cost: boolean): string => (cost ? "须为大于 0 的金额" : "须为大于 0 的整数"), + duplicate: "与前面的一条上限重复", + monthRetention: (days: number) => `每月上限要求请求记录至少保留 31 天,当前为 ${days} 天`, + unpriced: (n: number) => `${n} 个可用模型没有价格,其费用按 0 计入上限:`, + more: (n: number) => `另有 ${n} 个`, + fewer: "收起", + }, + { + title: "Usage limits", + hint: "Once any limit is used up, requests with this key are rejected", + none: "No limits set: usage is unlimited", + add: "Add limit", + every: "Per", + atMost: "at most", + perLabel: (n: number) => `Period of limit ${n}`, + maxLabel: (n: number) => `Amount of limit ${n}`, + measureLabel: (n: number) => `Measure of limit ${n}`, + cacheReads: "Count cache reads", + remove: "Remove", + removeLabel: (n: number) => `Remove limit ${n}`, + required: "Enter a limit", + notPositive: (cost: boolean): string => (cost ? "Must be an amount above 0" : "Must be a whole number above 0"), + duplicate: "Same as a limit above", + monthRetention: (days: number) => + `A monthly limit needs request records kept for at least 31 days; they are kept for ${days}`, + unpriced: (n: number) => + n === 1 + ? "1 model this key can use has no price and counts as $0 toward the limit:" + : `${n} models this key can use have no price and count as $0 toward the limit:`, + more: (n: number) => `${n} more`, + fewer: "Show less", + }, +); diff --git a/src/keys/LimitsEditor.tsx b/src/keys/LimitsEditor.tsx new file mode 100644 index 00000000..f9ba664f --- /dev/null +++ b/src/keys/LimitsEditor.tsx @@ -0,0 +1,286 @@ +import { useState } from "react"; +import { PlusIcon } from "lucide-react"; +import { Button } from "@/ui/button"; +import { Checkbox } from "@/ui/checkbox"; +import { Input } from "@/ui/input"; +import { rowMotion, usePresentList } from "@/ui/motion"; +import { NativeSelect, NativeSelectOption } from "@/ui/native-select"; +import { StatusLabel } from "@/ui/status-dot"; +import { cn } from "@/lib/utils"; +import { useText } from "@/i18n"; +import { useNow } from "@/useNow"; +import { Boxed, Note } from "@/upstreams/parts"; +import type { KeyLimitView } from "@/types"; +import { + MEASURES, + PERS, + amount, + cleanMax, + newRow, + parseMax, + resetText, + usageOf, + type LimitProblem, + type LimitRow, +} from "./limits"; +import { limitsText } from "./limits.i18n"; +import { limitsEditorText } from "./LimitsEditor.i18n"; + +/** 没有价格的模型多于这么多个时先收起,只列前面几个 */ +const UNPRICED_SHOWN = 6; + +/** + * 对话框里的「用量上限」:一条一行,读起来是一句话 ——「每 [天] 最多 [5] [费用 (USD)]」, + * token 上限再加一个「计入缓存读取」。 + * + * 编辑一把已有的密钥时,每一行下面写着此刻用了多少(core 的 `KeyLimitView`):天、周、月 + * 是这一期的,带重置的时刻;分钟、小时是最近这一段的。到了的那一行标出来。 + * + * 填错的当场说,写在那一行下面。**空着的那一格等离开它才说**:刚加的一行一出现就标红, + * 说的是一件用户正要去做的事。删一条写成字,× 在这个应用里只表示关闭。 + */ +export function LimitsEditor({ + rows, + views, + unpriced, + problems, + rowDays, + onChange, +}: { + rows: LimitRow[]; + /** core 给的这把密钥的上限和用量。新建时是空的 */ + views: readonly KeyLimitView[]; + /** 这把密钥用得到、却没有价格的模型。core 只在已经存了费用上限时才算 */ + unpriced: readonly string[]; + problems: ReadonlyMap; + /** 此刻请求记录留几天,「每月」那一条的提示用 */ + rowDays: number | null; + onChange: (rows: LimitRow[]) => void; +}) { + const t = useText(limitsEditorText); + // 「今天」「00:00 重置」随时间走:开着对话框过了零点,这一行要跟着换 + const now = useNow(60_000); + const [focus, setFocus] = useState(null); + /** 离开过数值那一格的行:空着的从这时起才说「请填写」 */ + const [left, setLeft] = useState>(() => new Set()); + const shown = usePresentList(rows, (r) => r.id); + + function update(id: number, patch: Partial) { + onChange(rows.map((r) => (r.id === id ? { ...r, ...patch } : r))); + } + + function add() { + const r = newRow(rows); + setFocus(r.id); + onChange([...rows, r]); + } + + const costLimited = rows.some((r) => r.measure === "cost"); + + return ( +
    +
    + {t.title} + {t.hint} +
    + {rows.length > 0 && ( + + {shown.map(({ item: r, key, presence }) => { + const n = rows.indexOf(r) + 1; + const problem = problems.get(r.id); + return ( + update(r.id, patch)} + onLeave={() => setLeft((s) => (s.has(r.id) ? s : new Set(s).add(r.id)))} + onRemove={() => onChange(rows.filter((x) => x.id !== r.id))} + /> + ); + })} + + )} +
    + + {rows.length === 0 && {t.none}} +
    + {costLimited && unpriced.length > 0 && } +
    + ); +} + +/** 一条上限。下面一行是用量,填错时换成错在哪 */ +function LimitLine({ + row: r, + n, + className, + autoFocus, + problem, + use, + rowDays, + now, + onChange, + onLeave, + onRemove, +}: { + row: LimitRow; + /** 第几条,读屏用 */ + n: number; + className?: string; + autoFocus: boolean; + problem: LimitProblem | undefined; + use: KeyLimitView | undefined; + rowDays: number | null; + now: number; + onChange: (patch: Partial) => void; + /** 离开了数值那一格 */ + onLeave: () => void; + onRemove: () => void; +}) { + const t = useText(limitsEditorText); + const w = useText(limitsText); + const said = + problem === "required" + ? t.required + : problem === "notPositive" + ? t.notPositive(r.measure === "cost") + : problem === "duplicate" + ? t.duplicate + : problem === "monthRetention" + ? t.monthRetention(rowDays ?? 0) + : null; + return ( +
    +
    + {t.every} + onChange({ per: e.target.value as LimitRow["per"] })} + > + {PERS.map((p) => ( + + {w.per[p]} + + ))} + + {t.atMost} + onChange({ max: cleanMax(r.measure, e.target.value) })} + onBlur={onLeave} + onKeyDown={(e) => { + // 对话框会把回车当成提交 + if (e.key === "Enter") e.preventDefault(); + }} + /> + onChange({ measure: e.target.value as LimitRow["measure"], max: "", cacheReads: false })} + > + {MEASURES.map((m) => ( + + {w.measure[m]} + + ))} + + {r.measure === "tokens" && ( + + )} + +
    + {said ? ( +

    {said}

    + ) : ( + use && + )} +
    + ); +} + +/** + * 「今天 $1.23 / $5.00 · 00:00 重置」「最近一分钟 12 / 30」。 + * + * 用量是 core 数的;**上限按输入框里的**:改大改小的时候,这一行说的就是改完之后的样子。 + * 到了的写出来,数字换成琥珀色 + */ +function Usage({ row, use, now }: { row: LimitRow; use: KeyLimitView; now: number }) { + const w = useText(limitsText); + const max = parseMax(row.measure, row.max) ?? use.max; + const reached = use.used >= max; + const resets = use.resets_at_ms != null && use.resets_at_ms > now ? resetText(use.resets_at_ms, now) : null; + return ( +
    + + {w.period[row.per]}{" "} + + {amount(row.measure, use.used)} / {amount(row.measure, max)} + + {resets && ` · ${resets}`} + + {reached && ( + + {w.reached} + + )} +
    + ); +} + +/** + * 没有价格的模型:费用记 0,费用上限管不住它们。**多了先收起**,只列前几个 + */ +function Unpriced({ models }: { models: readonly string[] }) { + const t = useText(limitsEditorText); + const [all, setAll] = useState(false); + const long = models.length > UNPRICED_SHOWN; + const listed = all || !long ? models : models.slice(0, UNPRICED_SHOWN); + return ( + + {t.unpriced(models.length)} {listed.join(", ")} + {long && ( + <> + {" "} + + + )} + + ); +} diff --git a/src/keys/data.ts b/src/keys/data.ts index f065c3bb..f902a674 100644 --- a/src/keys/data.ts +++ b/src/keys/data.ts @@ -39,14 +39,16 @@ export function useKeyUsage(): Resource & { byKey: Map /** * 全部网关密钥。配置换了一版就重取(密钥页拿着概览里的版本号,直接按它;别的页 - * 听 `config_reloaded`);请求落地时也重取,「最近使用」跟着它走。 + * 听 `config_reloaded`);请求落地时也重取,「最近使用」和用量上限跟着它走。某条上限 + * 到了八成、到了顶(`key_limit_alert`)也重取:那一刻请求可能还在跑,等它落地「已达上限」 + * 就晚了。 */ export function useKeys(configVersion?: string): Resource { return useResource("keys", api.listKeys, { events: configVersion === undefined - ? ["config_reloaded", "request_finished", "request_failed", "request_cancelled"] - : ["request_finished", "request_failed", "request_cancelled"], + ? ["config_reloaded", "request_finished", "request_failed", "request_cancelled", "key_limit_alert"] + : ["request_finished", "request_failed", "request_cancelled", "key_limit_alert"], deps: configVersion === undefined ? undefined : [configVersion], }); } diff --git a/src/keys/limits.i18n.ts b/src/keys/limits.i18n.ts new file mode 100644 index 00000000..6f0ba177 --- /dev/null +++ b/src/keys/limits.i18n.ts @@ -0,0 +1,55 @@ +import { messages } from "@/i18n"; +import type { LimitMeasure, LimitPer } from "@/types"; + +/** + * 用量上限的几样说法:对话框里的一行、密钥表里「已达上限」的悬停说明共用。 + */ +export const limitsText = messages( + { + per: { minute: "分钟", hour: "小时", day: "天", week: "周", month: "月" } satisfies Record, + measure: { requests: "次请求", tokens: "token", cost: "费用 (USD)" } satisfies Record, + /** 用量那一行开头:天、周、月是这一期,分钟、小时是最近这一段 */ + period: { + minute: "最近一分钟", + hour: "最近一小时", + day: "今天", + week: "本周", + month: "本月", + } satisfies Record, + resets: (at: string) => `${at} 重置`, + /** 重置的时刻:一天之内只写钟点,再远带上日期(日期按 `locale` 写) */ + at: (hm: string) => hm, + on: (date: string, hm: string) => `${date} ${hm}`, + locale: "zh-CN", + reached: "已达上限", + /** 一条上限说成一句:「每天 $5.00 费用」「每分钟 30 次请求」 */ + phrase: (per: string, amount: string, measure: LimitMeasure, cacheReads: boolean) => + measure === "requests" + ? `每${per} ${amount} 次请求` + : measure === "tokens" + ? `每${per} ${amount} token${cacheReads ? "(含缓存读取)" : ""}` + : `每${per} ${amount} 费用`, + }, + { + per: { minute: "minute", hour: "hour", day: "day", week: "week", month: "month" }, + measure: { requests: "requests", tokens: "tokens", cost: "USD" }, + period: { + minute: "Last minute", + hour: "Last hour", + day: "Today", + week: "This week", + month: "This month", + }, + resets: (at: string) => `resets ${at}`, + at: (hm: string) => `at ${hm}`, + on: (date: string, hm: string) => `${date} at ${hm}`, + locale: "en-US", + reached: "Limit reached", + phrase: (per: string, amount: string, measure: LimitMeasure, cacheReads: boolean) => + measure === "requests" + ? `${amount} requests per ${per}` + : measure === "tokens" + ? `${amount} tokens per ${per}${cacheReads ? " (cache reads included)" : ""}` + : `${amount} per ${per}`, + }, +); diff --git a/src/keys/limits.test.ts b/src/keys/limits.test.ts new file mode 100644 index 00000000..9756f056 --- /dev/null +++ b/src/keys/limits.test.ts @@ -0,0 +1,161 @@ +import { afterEach, describe, expect, it } from "vitest"; +import { setLang } from "@/i18n"; +import type { KeyLimitView } from "@/types"; +import { + cleanMax, + inputOfView, + inputsOf, + limitPhrase, + limitProblems, + limitRow, + maxText, + newRow, + nextReset, + parseMax, + resetText, + rowsOf, + usageOf, + type LimitRow, +} from "./limits"; + +const view = (x: Partial & Pick): KeyLimitView => ({ + cache_reads: false, + used: 0, + resets_at_ms: null, + reached: false, + ...x, +}); + +const row = (x: Partial>): LimitRow => + limitRow({ per: "day", measure: "cost", max: "5", cacheReads: false, ...x }); + +afterEach(() => setLang("zh")); + +describe("用量上限的一行", () => { + it("费用在 core 那边是微分,输入框里是美元,来回不走样", () => { + expect(maxText("cost", 5_000_000)).toBe("5.00"); + expect(maxText("cost", 5_500_000)).toBe("5.50"); + expect(maxText("cost", 125_000)).toBe("0.125"); + expect(maxText("cost", 50_000_000)).toBe("50.00"); + expect(maxText("cost", 1)).toBe("0.000001"); + expect(maxText("tokens", 1_000_000)).toBe("1000000"); + expect(parseMax("cost", "5.5")).toBe(5_500_000); + expect(parseMax("cost", "0.125")).toBe(125_000); + expect(parseMax("cost", ".5")).toBe(500_000); + expect(parseMax("cost", "5.")).toBe(5_000_000); + expect(parseMax("requests", "30")).toBe(30); + }); + + it("不是正数的上限不收", () => { + for (const bad of ["", " ", "0", "0.0", ".", "0.0000001"]) expect(parseMax("cost", bad), bad).toBeNull(); + for (const bad of ["", "0", "1.5", "-3"]) expect(parseMax("requests", bad), bad).toBeNull(); + }); + + it("输入框只留得下数字,费用多一个小数点、到微分为止", () => { + expect(cleanMax("requests", "1,000 次")).toBe("1000"); + expect(cleanMax("cost", "$5.5.0")).toBe("5.50"); + expect(cleanMax("cost", "0.12345678")).toBe("0.123456"); + }); + + it("打开时照 core 给的,保存时原样交回去", () => { + const views = [ + view({ per: "minute", measure: "requests", max: 30 }), + view({ per: "day", measure: "cost", max: 5_000_000 }), + view({ per: "week", measure: "tokens", max: 900_000, cache_reads: true }), + ]; + const rows = rowsOf(views); + expect(rows.map((r) => r.max)).toEqual(["30", "5.00", "900000"]); + expect(inputsOf(rows)).toEqual(views.map(inputOfView)); + }); + + it("缓存读取只跟着 token 上限走", () => { + expect(inputsOf([row({ measure: "requests", max: "3", cacheReads: true })])).toEqual([ + { per: "day", measure: "requests", max: 3, cache_reads: false }, + ]); + }); +}); + +describe("当场校验", () => { + it("和 core 一样:要填、要大于 0、同一种不能有两条", () => { + const rows = [ + row({ max: "" }), + row({ per: "hour", max: "0" }), + row({ per: "week", max: "3" }), + row({ per: "week", max: "8" }), + ]; + const p = limitProblems(rows, 90); + expect(rows.map((r) => p.get(r.id) ?? null)).toEqual(["required", "notPositive", null, "duplicate"]); + }); + + it("算不算缓存读取不一样,就是两条", () => { + const rows = [ + row({ measure: "tokens", max: "100" }), + row({ measure: "tokens", max: "900", cacheReads: true }), + row({ measure: "tokens", max: "900", cacheReads: true }), + ]; + const p = limitProblems(rows, 90); + expect(rows.map((r) => p.get(r.id) ?? null)).toEqual([null, null, "duplicate"]); + }); + + it("每月的上限要求记录留够 31 天;不知道留几天就不拦", () => { + const rows = [row({ per: "month" })]; + expect(limitProblems(rows, 30).get(rows[0]!.id)).toBe("monthRetention"); + expect(limitProblems(rows, 31).size).toBe(0); + expect(limitProblems(rows, null).size).toBe(0); + }); + + it("加一行先给还没有的那一种,不一加上就重复", () => { + const first = newRow([]); + expect([first.per, first.measure]).toEqual(["day", "cost"]); + const second = newRow([first]); + expect([second.per, second.measure]).toEqual(["month", "cost"]); + }); +}); + +describe("用量", () => { + it("按周期、量和缓存读取认 core 给的那一条", () => { + const views = [ + view({ per: "day", measure: "tokens", max: 100, used: 7 }), + view({ per: "day", measure: "tokens", max: 900, used: 70, cache_reads: true }), + ]; + expect(usageOf(row({ measure: "tokens", cacheReads: true }), views)?.used).toBe(70); + expect(usageOf(row({ measure: "tokens" }), views)?.used).toBe(7); + // 新加的、改成了别的周期的,core 还没数过 + expect(usageOf(row({ per: "week", measure: "tokens" }), views)).toBeUndefined(); + }); + + it("一条上限说成一句", () => { + expect(limitPhrase(view({ per: "day", measure: "cost", max: 5_000_000 }))).toBe("每天 $5.00 费用"); + expect(limitPhrase(view({ per: "minute", measure: "requests", max: 30 }))).toBe("每分钟 30 次请求"); + expect(limitPhrase(view({ per: "week", measure: "tokens", max: 1_000_000, cache_reads: true }))).toBe( + "每周 1,000,000 token(含缓存读取)", + ); + setLang("en"); + expect(limitPhrase(view({ per: "day", measure: "cost", max: 5_000_000 }))).toBe("$5.00 per day"); + expect(limitPhrase(view({ per: "week", measure: "tokens", max: 1_000_000, cache_reads: true }))).toBe( + "1,000,000 tokens per week (cache reads included)", + ); + }); + + it("一天之内的重置只写钟点,再远带上日期", () => { + const now = new Date(2026, 9, 5, 17, 30).getTime(); + const midnight = new Date(2026, 9, 6, 0, 0).getTime(); + const monday = new Date(2026, 9, 12, 0, 0).getTime(); + expect(resetText(midnight, now)).toBe("00:00 重置"); + expect(resetText(monday, now)).toBe("10月12日 00:00 重置"); + setLang("en"); + expect(resetText(midnight, now)).toBe("resets at 00:00"); + expect(resetText(monday, now)).toBe("resets Oct 12 at 00:00"); + }); + + it("最早要重新算的那一刻:只有天、周、月有", () => { + expect(nextReset(undefined)).toBeNull(); + expect( + nextReset([ + { limits: [view({ per: "minute", measure: "requests", max: 3 })] }, + { limits: [view({ per: "month", measure: "cost", max: 9, resets_at_ms: 300 })] }, + { limits: [view({ per: "day", measure: "cost", max: 9, resets_at_ms: 200 })] }, + ]), + ).toBe(200); + }); +}); diff --git a/src/keys/limits.ts b/src/keys/limits.ts new file mode 100644 index 00000000..835431dc --- /dev/null +++ b/src/keys/limits.ts @@ -0,0 +1,188 @@ +/** + * 一把密钥的用量上限:对话框里一行一条,和 core 的 `KeyLimitView` / `KeyLimitInput` 来回换。 + * + * **写得对不对由 core 的配置校验说**(`config.key_limit_*`)。这里当场查的是同一套规则里 + * 填的时候就看得出来的几条:要填、要大于 0、同一个周期同一种量(token 再分算不算缓存 + * 读取)只能有一条、每月的要求请求记录至少留 31 天 —— 免得填完整张对话框,保存时才被拒。 + * + * 费用在 core 那边是微分(`max`、`used`),输入框里是美元。 + */ +import { compact } from "@/format"; +import { textOf } from "@/i18n"; +import { usd, type KeyLimitInput, type KeyLimitView, type LimitMeasure, type LimitPer } from "@/types"; +import { limitsText } from "./limits.i18n"; + +export const PERS: readonly LimitPer[] = ["minute", "hour", "day", "week", "month"]; +export const MEASURES: readonly LimitMeasure[] = ["requests", "tokens", "cost"]; + +/** 每月上限要求请求记录至少留这么多天:重启之后当月的用量从记录里加回来 */ +export const MONTH_ROW_DAYS = 31; + +/** 对话框里的一行 */ +export interface LimitRow { + /** 只在对话框里用:行的 key,当场校验按它认行 */ + id: number; + per: LimitPer; + measure: LimitMeasure; + /** 输入框里的字。费用是美元(`5`、`0.5`),别的是整数 */ + max: string; + /** 计入缓存读取。只有 token 上限有,别的量上一直是 false */ + cacheReads: boolean; +} + +let seq = 0; + +export function limitRow(r: Omit): LimitRow { + seq += 1; + return { id: seq, ...r }; +} + +/** 打开对话框时的那几行:照 core 给的,按配置里的顺序 */ +export function rowsOf(views: readonly KeyLimitView[] | undefined): LimitRow[] { + return (views ?? []).map((v) => + limitRow({ per: v.per, measure: v.measure, max: maxText(v.measure, v.max), cacheReads: v.cache_reads }), + ); +} + +/** 上限写进输入框的样子。费用是微分,写成美元、到分,再往下的照实写:5_500_000 → `5.50`,125_000 → `0.125` */ +export function maxText(measure: LimitMeasure, max: number): string { + if (measure !== "cost") return String(max); + return (max / 1e6).toFixed(6).replace(/(\.\d\d\d*?)0+$/, "$1"); +} + +/** 输入框里只留得下数字(费用再加一个小数点、最多到微分那一位) */ +export function cleanMax(measure: LimitMeasure, raw: string): string { + if (measure !== "cost") return raw.replace(/[^0-9]/g, ""); + const s = raw.replace(/[^0-9.]/g, ""); + const dot = s.indexOf("."); + if (dot < 0) return s; + return s.slice(0, dot + 1) + s.slice(dot + 1).replace(/\./g, "").slice(0, 6); +} + +/** 输入框里的字 → 上限(费用是微分)。空的、不是正数的是 null */ +export function parseMax(measure: LimitMeasure, text: string): number | null { + const s = text.trim(); + if (measure === "cost") { + if (!/^(\d+\.?\d*|\.\d+)$/.test(s)) return null; + const micros = Math.round(Number(s) * 1e6); + return micros > 0 && Number.isSafeInteger(micros) ? micros : null; + } + if (!/^\d+$/.test(s)) return null; + const n = Number(s); + return n > 0 && Number.isSafeInteger(n) ? n : null; +} + +/** 算不算缓存读取,只对 token 上限有意义 */ +function cacheReadsOf(measure: LimitMeasure, cacheReads: boolean): boolean { + return measure === "tokens" && cacheReads; +} + +/** core 认作同一条的:同一个周期、同一种量、缓存读取算法相同 */ +function identity(per: LimitPer, measure: LimitMeasure, cacheReads: boolean): string { + return `${per}:${measure}:${cacheReadsOf(measure, cacheReads)}`; +} + +/** + * 加一行时先给什么:**还没有的那一种**,免得一加上就和已有的重复。先按天算费用 —— + * 管住一把密钥最常见的就是每天花多少;都有了就还是它,由当场校验说重复 + */ +export function newRow(rows: readonly LimitRow[]): LimitRow { + const taken = new Set(rows.map((r) => identity(r.per, r.measure, r.cacheReads))); + const order: LimitPer[] = ["day", "month", "week", "hour", "minute"]; + for (const measure of ["cost", "requests", "tokens"] as const) { + for (const per of order) { + if (!taken.has(identity(per, measure, false))) return limitRow({ per, measure, max: "", cacheReads: false }); + } + } + return limitRow({ per: "day", measure: "cost", max: "", cacheReads: false }); +} + +export type LimitProblem = "required" | "notPositive" | "duplicate" | "monthRetention"; + +/** + * 每一行有什么不对,按行的 id。**一行只说一件**:先说要填,再说要大于 0,再说重复(和 + * 前面哪一行一样,就标在后面那一行上),最后说每月的上限要记录留够天数。 + * + * `rowDays`:此刻请求记录留几天(概览里的 `retention.row_days`)。不知道就不查这一条, + * 保存时由 core 说 + */ +export function limitProblems(rows: readonly LimitRow[], rowDays: number | null): Map { + const out = new Map(); + const seen = new Set(); + for (const r of rows) { + const id = identity(r.per, r.measure, r.cacheReads); + if (r.max.trim() === "") out.set(r.id, "required"); + else if (parseMax(r.measure, r.max) == null) out.set(r.id, "notPositive"); + else if (seen.has(id)) out.set(r.id, "duplicate"); + else if (r.per === "month" && rowDays != null && rowDays < MONTH_ROW_DAYS) out.set(r.id, "monthRetention"); + seen.add(id); + } + return out; +} + +/** 保存时交给 core 的。**只在没有问题时调用**:填得不对的行在这里会被略过 */ +export function inputsOf(rows: readonly LimitRow[]): KeyLimitInput[] { + return rows.flatMap((r) => { + const max = parseMax(r.measure, r.max); + return max == null ? [] : [{ per: r.per, measure: r.measure, max, cache_reads: cacheReadsOf(r.measure, r.cacheReads) }]; + }); +} + +/** core 给的一条原样写回去(停用、启用这类只改别的字段的保存) */ +export function inputOfView(v: KeyLimitView): KeyLimitInput { + return { per: v.per, measure: v.measure, max: v.max, cache_reads: v.cache_reads }; +} + +/** + * 这一行此刻用了多少:core 给的那几条里,周期、量、缓存读取算法都和这一行一样的那条。 + * 新加的、改成了另一种的没有 —— 保存之前 core 没数过它 + */ +export function usageOf(row: LimitRow, views: readonly KeyLimitView[]): KeyLimitView | undefined { + const id = identity(row.per, row.measure, row.cacheReads); + return views.find((v) => identity(v.per, v.measure, v.cache_reads) === id); +} + +/** 一个用量或上限写成字:费用写美元,token 收成 k / M,请求数带千分位 */ +export function amount(measure: LimitMeasure, n: number): string { + if (measure === "cost") return usd(n); + if (measure === "tokens") return compact(n); + return n.toLocaleString("en-US"); +} + +/** 一条上限说成一句:「每天 $5.00 费用」「1,000,000 tokens per day」。数字写全 */ +export function limitPhrase(v: Pick): string { + const t = textOf(limitsText); + const n = v.measure === "cost" ? usd(v.max) : v.max.toLocaleString("en-US"); + return t.phrase(t.per[v.per], n, v.measure, v.cache_reads); +} + +const DAY_MS = 24 * 3_600_000; + +/** + * 什么时候重置:一天之内只写钟点(「00:00 重置」),再远带上日期(「10月12日 00:00 重置」)。 + * 按这台机器的时区写 —— core 在别的时区时,钟点照样是同一个时刻 + */ +export function resetText(atMs: number, nowMs: number): string { + const t = textOf(limitsText); + const d = new Date(atMs); + const hm = `${String(d.getHours()).padStart(2, "0")}:${String(d.getMinutes()).padStart(2, "0")}`; + if (atMs - nowMs <= DAY_MS) return t.resets(t.at(hm)); + const date = new Intl.DateTimeFormat(t.locale, { month: "short", day: "numeric" }).format(d); + return t.resets(t.on(date, hm)); +} + +/** 一把密钥有没有哪一条已经到了 */ +export function anyReached(views: readonly KeyLimitView[] | undefined): boolean { + return (views ?? []).some((v) => v.reached); +} + +/** 这几把密钥里最早要重新算的那一刻(天、周、月的上限才有)。没有就是 null */ +export function nextReset(keys: readonly { limits: readonly KeyLimitView[] }[] | undefined): number | null { + let soonest: number | null = null; + for (const k of keys ?? []) { + for (const l of k.limits) { + if (l.resets_at_ms != null && (soonest == null || l.resets_at_ms < soonest)) soonest = l.resets_at_ms; + } + } + return soonest; +} From 8abaf20f08cb98cd8c90373c80b8c72d6c543c94 Mon Sep 17 00:00:00 2001 From: fylorn <249551762+fylorn@users.noreply.github.com> Date: Mon, 5 Oct 2026 19:14:08 +0800 Subject: [PATCH 05/10] Upstreams: concurrency limit, hand-set model specs; failover: slow-start switch and slot wait Core's routing round adds three things an upstream owner sets by hand and two failover settings; this makes all of them reachable without editing config.yaml. Upstream dialog: an optional "Concurrency limit" (max_concurrent, 1-1000, blank shows "No limit"). It sits under the proxy fields in the Connection section and in the Account section of ChatGPT accounts, since accounts limit concurrent requests too. It is checked on the spot and blocks saving like the other required fields. The form now carries max_concurrent both ways, so the table's enable/disable toggle, which saves the upstream from its view, keeps the limit instead of dropping it. The row's in-flight tooltip names the limit when one is set. Models popover: hovering a model offers "Specs..." next to "Add alias..."; it opens a small dialog for the context window and max output of that model on that upstream, saved through PUT /provider-model-spec with the version the dialog opened on. A blank field shows the price table's value (or that the table has none) as its placeholder, so blank never means something hidden; clearing both removes the hand-set entry, and the footer says so. Models with a hand-set value carry a "Manual specs" mark next to their name (kept on the name side so it stays visible while the row's right side gives way to the buttons); its tooltip lists which values are manual. The context window column of the edit dialog's Models section now prefers the hand-set value too, so the two places never show different numbers for one model. Settings > Failover: "Move to the next upstream when the start times out" is a switch hung under the stream-start wait, since it says what happens when that wait runs out; "Wait for a free slot at most" (slot_wait_secs, 0-300, default 30 shown) follows. With the switch on the wait must be 5 to 120 seconds, checked on the spot like the section's other ranges, including when only the switch changed; core's config.slow_start_too_short still surfaces in the section's error banner if the file changed underneath. Co-Authored-By: Claude Opus 5.5 --- src/settings/FailoverSection.i18n.ts | 13 ++ src/settings/FailoverSection.tsx | 72 ++++++-- src/settings/failover.test.ts | 21 +++ src/upstreams/ChatgptAccountSection.tsx | 3 + src/upstreams/ConnectionSection.i18n.ts | 8 + src/upstreams/ConnectionSection.tsx | 41 +++++ src/upstreams/ModelSpecDialog.i18n.ts | 26 +++ src/upstreams/ModelSpecDialog.tsx | 234 ++++++++++++++++++++++++ src/upstreams/ModelsPanel.i18n.ts | 12 ++ src/upstreams/ModelsPanel.tsx | 58 ++++-- src/upstreams/ModelsSection.tsx | 12 +- src/upstreams/UpstreamTable.i18n.ts | 3 + src/upstreams/UpstreamTable.tsx | 42 ++++- src/upstreams/UpstreamsPage.tsx | 2 + src/upstreams/api.ts | 3 + src/upstreams/modelSpec.test.ts | 52 ++++++ src/upstreams/modelSpec.ts | 34 ++++ src/upstreams/upstreamForm.i18n.ts | 2 + src/upstreams/upstreamForm.test.ts | 22 +++ src/upstreams/upstreamForm.ts | 18 ++ 20 files changed, 647 insertions(+), 31 deletions(-) create mode 100644 src/upstreams/ModelSpecDialog.i18n.ts create mode 100644 src/upstreams/ModelSpecDialog.tsx create mode 100644 src/upstreams/modelSpec.test.ts create mode 100644 src/upstreams/modelSpec.ts diff --git a/src/settings/FailoverSection.i18n.ts b/src/settings/FailoverSection.i18n.ts index 1385452f..05f333f6 100644 --- a/src/settings/FailoverSection.i18n.ts +++ b/src/settings/FailoverSection.i18n.ts @@ -12,6 +12,7 @@ export const failoverText = messages( quota_pause_secs: "额度用完", rate_limit_max_pause_secs: "限流", stream_start_wait_secs: "等待回答开头", + slot_wait_secs: "并发已满时最多等待", }, what: { failures_to_pause: "服务器错误、无法连接等未说明原因的失败,连续达到此次数后暂停。", @@ -21,13 +22,18 @@ export const failoverText = messages( quota_pause_secs: "上游报告额度用完、但未给出重置时间时暂停的时长。给出重置时间的,暂停到重置为止。", rate_limit_max_pause_secs: "上游限流时按其要求的等待时间暂停,最长为此值。", stream_start_wait_secs: "流式回答在第一段内容到达前报错时,请求交给下一个上游。等待超过此时长后不再等待。", + slot_wait_secs: "上游达到并发上限时,请求等待空位的最长时间。0 表示不等待。", }, + nextOnSlowStart: "开头超时时转到下一个上游", + nextOnSlowStartWhat: "最后一个上游照常等待。开启时,等待时长宜在 30 秒以上。", times: "次", secs: "秒", badCount: "须为 1 到 100 之间的整数。", badSecs: "须为 1 到 604800 之间的整数。", badMax: "须为整数,不小于暂停时长,不超过 604800。", badWait: "须为 1 到 120 之间的整数。", + badSlowStartWait: "开启「开头超时时转到下一个上游」时,须为 5 到 120 之间的整数。", + badSlotWait: "须为 0 到 300 之间的整数。", saveFailed: "未能保存", }, { @@ -42,6 +48,7 @@ export const failoverText = messages( quota_pause_secs: "Quota used up", rate_limit_max_pause_secs: "Rate limit", stream_start_wait_secs: "Wait for the answer to start", + slot_wait_secs: "Wait for a free slot at most", }, what: { failures_to_pause: @@ -54,13 +61,19 @@ export const failoverText = messages( rate_limit_max_pause_secs: "A rate-limited upstream is paused for the wait it asks for, at most this long.", stream_start_wait_secs: "An error before the first content of a streamed answer sends the request to the next upstream. After this long, the wait ends.", + slot_wait_secs: + "How long a request waits for a free slot when upstreams are at their concurrency limit. 0 means no wait.", }, + nextOnSlowStart: "Move to the next upstream when the start times out", + nextOnSlowStartWhat: "The last upstream keeps waiting. With this on, a wait of 30 s or more is advisable.", times: "times", secs: "s", badCount: "A whole number from 1 to 100.", badSecs: "A whole number from 1 to 604800.", badMax: "A whole number, not less than the pause and at most 604800.", badWait: "A whole number from 1 to 120.", + badSlowStartWait: "With “Move to the next upstream when the start times out” on, a whole number from 5 to 120.", + badSlotWait: "A whole number from 0 to 300.", saveFailed: "Not saved", }, ); diff --git a/src/settings/FailoverSection.tsx b/src/settings/FailoverSection.tsx index 94f059ac..adc8b3c1 100644 --- a/src/settings/FailoverSection.tsx +++ b/src/settings/FailoverSection.tsx @@ -1,5 +1,6 @@ -import { useState } from "react"; +import { useState, type ReactNode } from "react"; import { Banner } from "@/ui/banner"; +import { Switch } from "@/ui/switch"; import { useText } from "@/i18n"; import { errorText } from "@/i18n/core.i18n"; import { patchConfig } from "@/patch"; @@ -12,10 +13,14 @@ import { failoverText } from "./FailoverSection.i18n"; const MAX_PAUSE = 7 * 24 * 3600; /** 流开头最多等多少秒(core 的 `MAX_STREAM_START_WAIT_SECS`) */ const MAX_WAIT = 120; +/** 开着「开头超时转到下一个上游」时,流开头至少等多少秒(core 的 `MIN_SLOW_START_WAIT_SECS`) */ +const MIN_SLOW_START_WAIT = 5; +/** 等空位最多写多少秒(core 的 `MAX_SLOT_WAIT_SECS`) */ +const MAX_SLOT_WAIT = 300; -/** 这一节里的格子。慢启动换下一家、等空位的秒数还没有放进来 */ -type Field = Exclude; -export type Draft = Record; +/** 这一节里的数字格子。开关(`next_on_slow_start`)另记 */ +type Field = Exclude; +export type Draft = Record & { next_on_slow_start: boolean }; /** 表单里的顺序 */ const FIELDS: Field[] = [ @@ -26,19 +31,24 @@ const FIELDS: Field[] = [ "quota_pause_secs", "rate_limit_max_pause_secs", "stream_start_wait_secs", + "slot_wait_secs", ]; /** 导出给测试用 */ -export const draftOf = (f: FailoverView): Draft => - Object.fromEntries(FIELDS.map((k) => [k, String(f[k])])) as Draft; +export const draftOf = (f: FailoverView): Draft => ({ + ...(Object.fromEntries(FIELDS.map((k) => [k, String(f[k])])) as Record), + next_on_slow_start: f.next_on_slow_start, +}); -const same = (a: Draft, b: Draft) => FIELDS.every((k) => a[k] === b[k]); +const same = (a: Draft, b: Draft) => + FIELDS.every((k) => a[k] === b[k]) && a.next_on_slow_start === b.next_on_slow_start; /** * 每一格填得对不对,范围和 core 的校验一样。 * * **没动过的格不查**(和日志保留一样):它就是配置里现在的值,保存时也不发。只有上限 - * 例外 —— 起点改大了,没动过的上限也可能跟着不对了。导出给测试用。 + * 例外 —— 起点改大了,没动过的上限也可能跟着不对了。等回答开头的秒数也一样:打开 + * 「开头超时时转到下一个上游」之后它至少要 5 秒,没动过的也要重查。导出给测试用。 */ export function checks(draft: Draft, saved: Draft): Record { const secs = (k: Field) => draft[k] === saved[k] || intIn(draft[k], 1, MAX_PAUSE); @@ -53,14 +63,20 @@ export function checks(draft: Draft, saved: Draft): Record { quota_pause_secs: secs("quota_pause_secs"), rate_limit_max_pause_secs: secs("rate_limit_max_pause_secs"), stream_start_wait_secs: - draft.stream_start_wait_secs === saved.stream_start_wait_secs || intIn(draft.stream_start_wait_secs, 1, MAX_WAIT), + (draft.stream_start_wait_secs === saved.stream_start_wait_secs && + draft.next_on_slow_start === saved.next_on_slow_start) || + intIn(draft.stream_start_wait_secs, draft.next_on_slow_start ? MIN_SLOW_START_WAIT : 1, MAX_WAIT), + slot_wait_secs: draft.slot_wait_secs === saved.slot_wait_secs || intIn(draft.slot_wait_secs, 0, MAX_SLOT_WAIT), }; } /** - * 上游失败之后停用多久、流式回答的开头最多等多久。 + * 上游失败之后停用多久、流式回答的开头最多等多久、上游满着时最多等多久。 * * 默认值显式写在格子里(概览给的就是真在用的数),不用「留空 = 默认」。 + * + * **「开头超时时转到下一个上游」和等开头的秒数是一件事**:开关挂在那一行底下,不另起 + * 一行 —— 它说的就是那个秒数到了之后怎么办。 */ export function FailoverSection({ failover, @@ -88,6 +104,8 @@ export function FailoverSection({ path: `/failover/${k}`, value: Number(draft[k]), })); + if (draft.next_on_slow_start !== saved.next_on_slow_start) + ops.push({ op: "replace", path: "/failover/next_on_slow_start", value: draft.next_on_slow_start }); setBusy(true); setError(null); try { @@ -108,7 +126,7 @@ export function FailoverSection({ }; const bad = (msg: string) => {msg}; - const row = (k: Field, unit: string, what: string, badText: string) => ( + const row = (k: Field, unit: string, what: string, badText: string, more?: ReactNode) => ( } - /> + > + {more} + ); return ( @@ -143,7 +163,33 @@ export function FailoverSection({ {row("no_balance_pause_secs", t.secs, t.what.no_balance_pause_secs, t.badSecs)} {row("quota_pause_secs", t.secs, t.what.quota_pause_secs, t.badSecs)} {row("rate_limit_max_pause_secs", t.secs, t.what.rate_limit_max_pause_secs, t.badSecs)} - {row("stream_start_wait_secs", t.secs, t.what.stream_start_wait_secs, t.badWait)} + {row( + "stream_start_wait_secs", + t.secs, + t.what.stream_start_wait_secs, + draft.next_on_slow_start ? t.badSlowStartWait : t.badWait, + // 和上面那一行排成同一个样子:说明在左、开关在右,中间不画分隔线 +
    +
    + +
    {t.nextOnSlowStartWhat}
    +
    +
    + { + setError(null); + setDraft((d) => ({ ...d, next_on_slow_start: c === true })); + }} + /> +
    +
    , + )} + {row("slot_wait_secs", t.secs, t.what.slot_wait_secs, t.badSlotWait)} { expect(c.max_pause_secs).toBe(false); expect(checks({ ...saved, pause_secs: "900", max_pause_secs: "900" }, saved).max_pause_secs).toBe(true); }); + + it("等空位的秒数可以是 0(不等),最多 300", () => { + const saved = draftOf(failover); + expect(saved.slot_wait_secs).toBe("30"); + expect(checks({ ...saved, slot_wait_secs: "0" }, saved).slot_wait_secs).toBe(true); + expect(checks({ ...saved, slot_wait_secs: "300" }, saved).slot_wait_secs).toBe(true); + expect(checks({ ...saved, slot_wait_secs: "301" }, saved).slot_wait_secs).toBe(false); + expect(checks({ ...saved, slot_wait_secs: "" }, saved).slot_wait_secs).toBe(false); + }); + + it("开头超时转到下一个上游:开着时开头至少等 5 秒,没动过的秒数也重查", () => { + const saved = draftOf({ ...failover, stream_start_wait_secs: 3 }); + expect(saved.next_on_slow_start).toBe(false); + // 关着时 3 秒是合法的 + expect(checks(saved, saved).stream_start_wait_secs).toBe(true); + // 打开开关,秒数没动也不行 + const on = { ...saved, next_on_slow_start: true }; + expect(checks(on, saved).stream_start_wait_secs).toBe(false); + expect(checks({ ...on, stream_start_wait_secs: "5" }, saved).stream_start_wait_secs).toBe(true); + expect(checks({ ...on, stream_start_wait_secs: "30" }, saved).stream_start_wait_secs).toBe(true); + }); }); diff --git a/src/upstreams/ChatgptAccountSection.tsx b/src/upstreams/ChatgptAccountSection.tsx index ff0c22be..bff35f80 100644 --- a/src/upstreams/ChatgptAccountSection.tsx +++ b/src/upstreams/ChatgptAccountSection.tsx @@ -26,6 +26,7 @@ import { useText } from "@/i18n"; import { commonText } from "@/i18n/common.i18n"; import { api } from "./api"; import { chatgptAccountText } from "./ChatgptAccountSection.i18n"; +import { ConcurrencyField } from "./ConnectionSection"; import { coreText, errorText, planLabel, proxyKindLabel, quotaWindowBefore } from "./labels"; import { DialogError, FormItem } from "./parts"; import { QuotaBar } from "./QuotaBar"; @@ -142,6 +143,8 @@ export function ChatgptAccountSection({ ))} + {/* 账号一样限制同时进行的请求 */} + diff --git a/src/upstreams/ConnectionSection.i18n.ts b/src/upstreams/ConnectionSection.i18n.ts index 99d81bbc..59c592e3 100644 --- a/src/upstreams/ConnectionSection.i18n.ts +++ b/src/upstreams/ConnectionSection.i18n.ts @@ -34,6 +34,10 @@ export const connectionSectionText = messages( onProxyFail: "代理不可用时", failWithError: "返回错误", fallBackDirect: "改为直连", + concurrency: "并发上限", + noLimit: "不限", + concurrencyDesc: "同时发往此上游的请求数上限,用于限制并发的中转站或账号。", + badConcurrency: "须为 1 到 1000 之间的整数。", check: "检测连接", checking: "检测中", checkNote: "验证地址与凭据,并获取模型列表。不产生费用。", @@ -84,6 +88,10 @@ export const connectionSectionText = messages( onProxyFail: "When the proxy is unavailable", failWithError: "Return an error", fallBackDirect: "Connect directly", + concurrency: "Concurrency limit", + noLimit: "No limit", + concurrencyDesc: "The most requests sent to this upstream at once, for relays and accounts that limit concurrency.", + badConcurrency: "A whole number from 1 to 1000.", check: "Check connection", checking: "Checking", checkNote: "Verifies the URL and credentials and fetches the model list. No cost is incurred.", diff --git a/src/upstreams/ConnectionSection.tsx b/src/upstreams/ConnectionSection.tsx index 397cdfa8..d3b7dd5e 100644 --- a/src/upstreams/ConnectionSection.tsx +++ b/src/upstreams/ConnectionSection.tsx @@ -32,6 +32,7 @@ import { CHATGPT, ZAI, nameFromUrl, presetById } from "./presets"; import { ServicePicker } from "./ServicePicker"; import { authModeOf, + concurrencyOf, describeModelList, freeName, isBedrock, @@ -281,6 +282,10 @@ export function ConnectionSection({ +
    + +
    +
    + + + + ); +} + +/** + * 空着的格子写什么:价目表给的数(留空就用它)。此刻是手写的,价目表给多少这里不知道, + * 只说留空用价目表;价目表也没有的,说没有 —— 留空就是不知道。 + */ +function placeholderOf( + value: number | null | undefined, + source: SpecSource | null | undefined, + t: { fromTable: (n: string) => string; useTable: string; notInTable: string }, +): string { + if (source === "price_table" && value != null) return t.fromTable(value.toLocaleString()); + if (source === "manual") return t.useTable; + return t.notInTable; +} + +function TokensField({ + id, + label, + value, + onChange, + placeholder, + bad, +}: { + id: string; + label: string; + value: string; + onChange: (v: string) => void; + placeholder: string; + bad: boolean; +}) { + const t = useText(modelSpecDialogText); + return ( + {t.bad} : undefined}> + + onChange(e.target.value)} + /> + + tokens + + + + ); +} diff --git a/src/upstreams/ModelsPanel.i18n.ts b/src/upstreams/ModelsPanel.i18n.ts index d2da5035..117c6dc7 100644 --- a/src/upstreams/ModelsPanel.i18n.ts +++ b/src/upstreams/ModelsPanel.i18n.ts @@ -30,6 +30,12 @@ export const modelsPanelText = messages( aliasMark: (alias: string) => `别名 ${alias}`, aliasTitle: (alias: string) => `别名 ${alias} 列着这个模型:客户端用 ${alias} 请求时可以发往这个模型`, makeAlias: "起别名…", + specs: "规格…", + manualSpecs: "手动规格", + manualContext: (n: string) => `上下文窗口 ${n}`, + manualOutput: (n: string) => `输出上限 ${n}`, + listSep: ",", + manualTitle: (what: string) => `手动设置:${what}`, }, { title: "Models", @@ -64,5 +70,11 @@ export const modelsPanelText = messages( aliasMark: (alias: string) => `alias ${alias}`, aliasTitle: (alias: string) => `Alias ${alias} lists this model: requests for ${alias} can go to it`, makeAlias: "Add alias…", + specs: "Specs…", + manualSpecs: "Manual specs", + manualContext: (n: string) => `context window ${n}`, + manualOutput: (n: string) => `max output ${n}`, + listSep: ", ", + manualTitle: (what: string) => `Set by hand: ${what}`, }, ); diff --git a/src/upstreams/ModelsPanel.tsx b/src/upstreams/ModelsPanel.tsx index e63059bf..b12c7f4d 100644 --- a/src/upstreams/ModelsPanel.tsx +++ b/src/upstreams/ModelsPanel.tsx @@ -3,6 +3,7 @@ import { ChevronRightIcon, CircleAlertIcon, RefreshCwIcon, SearchIcon } from "lu import { AliasMark } from "@/aliases/AliasMark"; import { cn } from "@/lib/utils"; import { useResource } from "@/lib/resource"; +import { Badge } from "@/ui/badge"; import { Button } from "@/ui/button"; import { InputGroup, InputGroupAddon, InputGroupInput } from "@/ui/input-group"; import { Skeleton } from "@/ui/skeleton"; @@ -13,6 +14,7 @@ import { commonText } from "@/i18n/common.i18n"; import type { ModelRow, ProviderModelsView, ProviderView } from "@/types"; import { api } from "./api"; import { contextWindow, coreText, errorText, perMillion } from "./labels"; +import { hasManual } from "./modelSpec"; import { modelsPanelText } from "./ModelsPanel.i18n"; /** 列表长过这个数才给筛选框。十来个一眼就扫完了 */ @@ -31,12 +33,16 @@ const FILTER_FROM = 10; * * 列进了别名的模型,名字后面标出别名;悬停一行给「起别名…」(`onAlias`),打开新建别名的 * 对话框,这个模型已经列为上游模型。 + * + * 右边的数是上下文窗口:在这一家手写过规格(上下文窗口或输出上限)的,名字后面标「手动 + * 规格」,悬停说是哪几项。悬停一行还给「规格…」(`onSpec`),打开手写规格的对话框。 */ export function ModelsPanel({ p, perToken, onEdit, onAlias, + onSpec, }: { p: ProviderView; /** 按量计费:列出单价。别的计费方式不按单价算费用,列了也没意义 */ @@ -45,6 +51,8 @@ export function ModelsPanel({ onEdit: () => void; /** 给这个模型起别名。不给就不出「起别名…」 */ onAlias?: (model: string) => void; + /** 手写这个模型的规格。不给就不出「规格…」 */ + onSpec?: (model: ModelRow) => void; }) { const t = useText(modelsPanelText); const c = useText(commonText); @@ -173,7 +181,7 @@ export function ModelsPanel({ )}
    {shownOn.map((m) => ( - + ))} {off.length > 0 && ( <> @@ -193,7 +201,7 @@ export function ModelsPanel({ {t.notEnabled(off.length)} {(showOff || q !== "") && - shownOff.map((m) => )} + shownOff.map((m) => )} )} {q !== "" && shownOn.length + shownOff.length === 0 && ( @@ -229,21 +237,29 @@ function Row({ m, perToken, onAlias, + onSpec, }: { m: ModelRow; perToken: boolean; onAlias?: (model: string) => void; + onSpec?: (model: ModelRow) => void; }) { const t = useText(modelsPanelText); const price = perToken && m.price ? `$${perMillion(m.price.input)} / $${perMillion(m.price.output)}${m.estimated ? t.estimated : ""}` : null; + const actions = onAlias || onSpec; + /** 手写了哪几项:「上下文窗口 128K」「输出上限 16K」 */ + const manual = [ + m.context_window_source === "manual" ? t.manualContext(contextWindow(m.context_window)) : null, + m.max_output_tokens_source === "manual" ? t.manualOutput(contextWindow(m.max_output_tokens)) : null, + ].filter((x): x is string => x !== null); return (
    @@ -260,11 +276,21 @@ function Row({ {t.aliasMark(a)} ))} - {/* 悬停时这一格让给「起别名…」:两样叠在同一个位置,行高不跳 */} + {/* 放在名字这一边,不放在数旁边:悬停时右边让给按钮,这个标记和它的说明还看得见 */} + {hasManual(m) && ( + + {t.manualSpecs} + + )} + {/* 悬停时这一格让给「规格…」「起别名…」:叠在同一个位置,行高不跳 */} {price && ( @@ -276,15 +302,19 @@ function Row({ {m.context_window ? contextWindow(m.context_window) : ""} - {onAlias && ( - + {actions && ( + + {onSpec && ( + + )} + {onAlias && ( + + )} + )}
    ); diff --git a/src/upstreams/ModelsSection.tsx b/src/upstreams/ModelsSection.tsx index e186a053..0d7e901c 100644 --- a/src/upstreams/ModelsSection.tsx +++ b/src/upstreams/ModelsSection.tsx @@ -38,6 +38,11 @@ export interface ModelCatalog { status?: ModelListStatus; /** core 正在向上游问 */ fetching?: boolean; + /** + * 这一家手写了上下文窗口的模型(`model_specs`)和那个数。手写的优先于价目表,这一节的 + * 上下文窗口一列照它写,和模型弹窗里是同一个数 + */ + manualContext?: Record; } /** core 记下的那一份 */ @@ -49,6 +54,11 @@ export function catalogOf(v: ProviderModelsView): ModelCatalog { error: v.error, status: v.status, fetching: v.fetching, + manualContext: Object.fromEntries( + v.models.flatMap((m) => + m.context_window_source === "manual" && m.context_window != null ? [[m.id, m.context_window]] : [], + ), + ), }; } @@ -250,7 +260,7 @@ export function ModelsSection({ {m} - {contextWindow(price?.max_input_tokens)} + {contextWindow(catalog?.manualContext?.[m] ?? price?.max_input_tokens)} {perToken && ( diff --git a/src/upstreams/UpstreamTable.i18n.ts b/src/upstreams/UpstreamTable.i18n.ts index 83566877..b32e8485 100644 --- a/src/upstreams/UpstreamTable.i18n.ts +++ b/src/upstreams/UpstreamTable.i18n.ts @@ -45,6 +45,7 @@ export const upstreamTableText = messages( cacheLine: (model: string, here: string, hereN: number, k: number, others: string, othersN: number) => `${model}:本上游 ${here}(${hereN.toLocaleString()} 轮)· 其他 ${k} 个上游 ${others}(${othersN.toLocaleString()} 轮)`, inFlight: (n: number) => `${n} 个请求进行中`, + inFlightOf: (n: number, max: number) => `${n} 个请求进行中,并发上限 ${max} 个`, actions: (name: string) => `${name} 的操作`, check: "检测连接", linkTest: "链路测速", @@ -111,6 +112,8 @@ export const upstreamTableText = messages( cacheLine: (model: string, here: string, hereN: number, k: number, others: string, othersN: number) => `${model}: this upstream ${here} (${count(hereN, "turn", "turns")}) · ${count(k, "other upstream", "other upstreams")} ${others} (${count(othersN, "turn", "turns")})`, inFlight: (n: number) => (n === 1 ? "1 request in progress" : `${n} requests in progress`), + inFlightOf: (n: number, max: number) => + `${n === 1 ? "1 request" : `${n} requests`} in progress, concurrency limit ${max}`, actions: (name: string) => `Actions for ${name}`, check: "Check connection", linkTest: "Connection test", diff --git a/src/upstreams/UpstreamTable.tsx b/src/upstreams/UpstreamTable.tsx index 41451532..6d43d93b 100644 --- a/src/upstreams/UpstreamTable.tsx +++ b/src/upstreams/UpstreamTable.tsx @@ -14,7 +14,7 @@ import { resetAt } from "@/format"; import { useNow } from "@/useNow"; import { textOf, useText } from "@/i18n"; import { commonText } from "@/i18n/common.i18n"; -import { usd, type ProviderView, type QuotaWindow, type UpstreamHealth } from "@/types"; +import { usd, type ModelRow, type ProviderView, type QuotaWindow, type UpstreamHealth } from "@/types"; import type { UpstreamStats } from "./api"; import { discrepancies, pct, signedPct, type Discrepancies } from "./checkup"; import { slotsByUpstream, type Slot } from "./data"; @@ -31,6 +31,7 @@ import { } from "./labels"; import { labelsText } from "./labels.i18n"; import { ModelsPanel } from "./ModelsPanel"; +import { ModelSpecDialog } from "./ModelSpecDialog"; import { AliasDialog } from "@/aliases/AliasDialog"; import { ProviderTile, keepInRow, openRow } from "./parts"; import { QUOTA_FULL, QuotaBar } from "./QuotaBar"; @@ -76,6 +77,8 @@ export function UpstreamTable({ inFlight, refreshing, focus, + configVersion, + onChanged, actions, }: { providers: ProviderView[]; @@ -90,6 +93,10 @@ export function UpstreamTable({ refreshing: ReadonlySet; /** 从别的页定位到的那一行:滚进视野、亮一下。`at` 让同一个名字再定位一次也生效 */ focus: { name: string; at: number } | null; + /** 概览里的配置版本:模型弹窗里手写规格时带它 */ + configVersion: string; + /** 在这张表里写了配置(手写模型规格):外面重读概览 */ + onChanged: () => void; actions: UpstreamActions; }) { const t = useText(upstreamTableText); @@ -147,6 +154,8 @@ export function UpstreamTable({ actions.editModels(p.name)} /> @@ -304,7 +313,8 @@ function NameCell({
    {live ? ( - + // 设了并发上限的,连上限一起说:满没满一眼看得出 + {tile} ) : ( @@ -387,12 +397,26 @@ function Where({ p }: { p: ProviderView }) { * **整格是一个按钮,什么状态都能点开** —— 数目单独回答不了「要的那个模型在不在 * 里面」,而没拿到清单时,点开要能看到原因和下一步。停用的上游不提供模型。 */ -function ModelsCell({ p, busy, onEdit }: { p: ProviderView; busy: boolean; onEdit: () => void }) { +function ModelsCell({ + p, + busy, + configVersion, + onChanged, + onEdit, +}: { + p: ProviderView; + busy: boolean; + configVersion: string; + onChanged: () => void; + onEdit: () => void; +}) { const t = useText(upstreamTableText); const l = useText(labelsText); const [open, setOpen] = useState(false); /** 「起别名…」点的那个模型。对话框挂在弹窗外面:弹窗一收起,里面的东西就卸掉了 */ const [aliasFor, setAliasFor] = useState(null); + /** 「规格…」点的那一行,同上 */ + const [specFor, setSpecFor] = useState(null); if (p.disabled) { return —; } @@ -431,9 +455,21 @@ function ModelsCell({ p, busy, onEdit }: { p: ProviderView; busy: boolean; onEdi setOpen(false); setAliasFor(model); }} + onSpec={(row) => { + setOpen(false); + setSpecFor(row); + }} /> + !o && setSpecFor(null)} + provider={p.name} + row={specFor} + configVersion={configVersion} + onSaved={onChanged} + /> !o && setAliasFor(null)} diff --git a/src/upstreams/UpstreamsPage.tsx b/src/upstreams/UpstreamsPage.tsx index e1055eda..63d13c29 100644 --- a/src/upstreams/UpstreamsPage.tsx +++ b/src/upstreams/UpstreamsPage.tsx @@ -424,6 +424,8 @@ export default function UpstreamsPage({ inFlight={inFlight} refreshing={refreshing} focus={focus} + configVersion={configVersion} + onChanged={changed} actions={{ edit: (name) => setDialog({ kind: "upstream", mode: { kind: "edit", name } }), test: (name) => setDialog({ kind: "test", name }), diff --git a/src/upstreams/api.ts b/src/upstreams/api.ts index d34a6d82..c605ef62 100644 --- a/src/upstreams/api.ts +++ b/src/upstreams/api.ts @@ -14,6 +14,7 @@ import type { CostBucketGroup, CostGroup, LatencyView, + ModelSpecSave, TokenRateView, PriceQuery, PriceSheetSave, @@ -55,6 +56,8 @@ export const api = { call("PreviewProvider", { base_url: baseUrl, protocol }), providerModels: (name: string) => call("ProviderModels", null, name), refreshProviderModels: (name: string) => call("RefreshProviderModels", null, name), + /** 手写一个模型的上下文窗口、输出上限。两项都空 = 删掉手写的,回到价目表 */ + setModelSpec: (save: ModelSpecSave) => call("SetModelSpec", save), /** 补问缺失、失败、过期的清单。**立刻回**,答案随 `models_changed` 到 */ refreshStaleModels: () => call("RefreshStaleModels", null), /** 起点和格宽都由界面给:格子对齐到本地整点(见 `bucketStart`) */ diff --git a/src/upstreams/modelSpec.test.ts b/src/upstreams/modelSpec.test.ts new file mode 100644 index 00000000..0b3840e6 --- /dev/null +++ b/src/upstreams/modelSpec.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, it } from "vitest"; +import type { ModelRow } from "@/types"; +import { catalogOf } from "./ModelsSection"; +import { MAX_SPEC_TOKENS, hasManual, manualOf, tokensOf } from "./modelSpec"; + +function row(patch: Partial = {}): ModelRow { + return { id: "glm-5-air", enabled: true, estimated: false, aliases: [], ...patch }; +} + +/** 手写的模型规格:格子里的数怎么认,哪几项是手写的 */ +describe("模型规格", () => { + it("空是不写(用价目表);千分位照常认;0、小数、超出 u32 的不收", () => { + expect(tokensOf("")).toBeNull(); + expect(tokensOf(" ")).toBeNull(); + expect(tokensOf("128000")).toBe(128_000); + expect(tokensOf("128,000")).toBe(128_000); + expect(tokensOf(" 1 000 000 ")).toBe(1_000_000); + expect(tokensOf(String(MAX_SPEC_TOKENS))).toBe(MAX_SPEC_TOKENS); + for (const bad of ["0", "1.5", "-3", "128k", String(MAX_SPEC_TOKENS + 1)]) expect(tokensOf(bad)).toBeUndefined(); + }); + + it("只回填手写的那几项:来自价目表的数不进格子", () => { + const both = row({ + context_window: 1_000_000, + context_window_source: "manual", + max_output_tokens: 64_000, + max_output_tokens_source: "price_table", + }); + expect(manualOf(both)).toEqual({ context: "1000000", output: "" }); + expect(hasManual(both)).toBe(true); + const table = row({ context_window: 200_000, context_window_source: "price_table" }); + expect(manualOf(table)).toEqual({ context: "", output: "" }); + expect(hasManual(table)).toBe(false); + expect(hasManual(row({ max_output_tokens: 16_384, max_output_tokens_source: "manual" }))).toBe(true); + expect(hasManual(row())).toBe(false); + }); + + it("编辑对话框的模型一节:手写的上下文窗口和弹窗里是同一个数", () => { + const c = catalogOf({ + provider: "relay", + source: "discovered", + status: "listed", + fetching: false, + models: [ + row({ id: "a", context_window: 128_000, context_window_source: "manual" }), + row({ id: "b", context_window: 200_000, context_window_source: "price_table" }), + row({ id: "c", max_output_tokens: 8_000, max_output_tokens_source: "manual" }), + ], + }); + expect(c.manualContext).toEqual({ a: 128_000 }); + }); +}); diff --git a/src/upstreams/modelSpec.ts b/src/upstreams/modelSpec.ts new file mode 100644 index 00000000..7e6a433c --- /dev/null +++ b/src/upstreams/modelSpec.ts @@ -0,0 +1,34 @@ +/** + * 手写的模型规格(上下文窗口、输出上限):从 `ModelRow` 读出手写的那几项、格子里的数 + * 写得对不对。**先后只在 core 定**(`tw_config::model_specs::resolve`):这里只认 core + * 标的来源,不自己比大小。 + */ +import type { ModelRow } from "@/types"; + +/** 一项最多写多少:core 存的是 `u32` */ +export const MAX_SPEC_TOKENS = 4_294_967_295; + +/** + * 格子里的 token 数。空是不写(`null`,用价目表的);写的不是 1 到 `u32` 上限的整数是 + * `undefined`。千分位的逗号、空格、下划线照常认:`128,000` 和 `128000` 是同一个数。 + */ +export function tokensOf(v: string): number | null | undefined { + const s = v.replace(/[\s,_]/g, ""); + if (s === "") return null; + if (!/^\d+$/.test(s)) return undefined; + const n = Number(s); + return n >= 1 && n <= MAX_SPEC_TOKENS ? n : undefined; +} + +/** 这个模型此刻手写的两项,格子里的写法。没手写的那一项是空的 */ +export function manualOf(m: ModelRow): { context: string; output: string } { + return { + context: m.context_window_source === "manual" && m.context_window != null ? String(m.context_window) : "", + output: m.max_output_tokens_source === "manual" && m.max_output_tokens != null ? String(m.max_output_tokens) : "", + }; +} + +/** 这个模型在这一家有手写的规格 */ +export function hasManual(m: ModelRow): boolean { + return m.context_window_source === "manual" || m.max_output_tokens_source === "manual"; +} diff --git a/src/upstreams/upstreamForm.i18n.ts b/src/upstreams/upstreamForm.i18n.ts index d98c2382..e2ddc15f 100644 --- a/src/upstreams/upstreamForm.i18n.ts +++ b/src/upstreams/upstreamForm.i18n.ts @@ -10,6 +10,7 @@ export const upstreamFormText = messages( profile: "填写 AWS profile 的名称", headerName: "填写请求头名称", headerValue: (header: string) => `填写请求头「${header}」的值`, + concurrency: "并发上限须为 1 到 1000 之间的整数", pickModel: "至少选择一个模型", found: (n: number) => `发现 ${n} 个模型`, notImplemented: (status: number) => `上游未提供模型列表接口(HTTP ${status})`, @@ -25,6 +26,7 @@ export const upstreamFormText = messages( profile: "Enter the name of the AWS profile", headerName: "Enter a header name", headerValue: (header: string) => `Enter a value for header “${header}”`, + concurrency: "The concurrency limit must be a whole number from 1 to 1000", pickModel: "Select at least one model", found: (n: number) => (n === 1 ? "1 model found" : `${n} models found`), notImplemented: (status: number) => `No model list endpoint (HTTP ${status})`, diff --git a/src/upstreams/upstreamForm.test.ts b/src/upstreams/upstreamForm.test.ts index eb115bc4..c78f7569 100644 --- a/src/upstreams/upstreamForm.test.ts +++ b/src/upstreams/upstreamForm.test.ts @@ -88,6 +88,28 @@ describe("编辑时回填原样", () => { expect(toInput({ ...f, forwardClientIdentity: false }).forward_client_identity).toBe(false); }); + it("并发上限:回填、原样交回;清空就是不限。停用、启用走同一份,不会把它丢掉", () => { + const f = formFromView(view({ max_concurrent: 4 })); + expect(f.maxConcurrent).toBe("4"); + expect(toInput(f).max_concurrent).toBe(4); + expect(toInput({ ...f, maxConcurrent: "" }).max_concurrent).toBeUndefined(); + expect(formFromView(view()).maxConcurrent).toBe(""); + expect(toInput(blankForm()).max_concurrent).toBeUndefined(); + }); + + it("并发上限只收 1 到 1000 的整数,写错了保存不了", () => { + const f = formFromView(view()); + for (const ok of ["1", "1000", " 12 "]) { + expect(connectionMissing({ ...f, maxConcurrent: ok }, "relay", ["relay"])).toBeNull(); + } + for (const bad of ["0", "1001", "2.5", "-1", "abc"]) { + expect(connectionMissing({ ...f, maxConcurrent: bad }, "relay", ["relay"])).toBe( + "并发上限须为 1 到 1000 之间的整数", + ); + expect(toInput({ ...f, maxConcurrent: bad }).max_concurrent).toBeUndefined(); + } + }); + it("清空密钥就是不要密钥;清空请求头的值要补上", () => { const f = formFromView(view()); expect(toInput({ ...f, key: " " }).key).toBeUndefined(); diff --git a/src/upstreams/upstreamForm.ts b/src/upstreams/upstreamForm.ts index 340a17d2..9c503ac2 100644 --- a/src/upstreams/upstreamForm.ts +++ b/src/upstreams/upstreamForm.ts @@ -89,6 +89,8 @@ export interface UpstreamForm { billing: Billing; /** 空 = 默认价目表 */ pricing: string; + /** 同时最多发给这家几个请求,1 到 1000。空 = 不限 */ + maxConcurrent: string; disabled: boolean; } @@ -133,6 +135,7 @@ export function blankForm(): UpstreamForm { scopeList: [], billing: "per-token", pricing: "", + maxConcurrent: "", disabled: false, }; } @@ -198,6 +201,7 @@ export function formFromView(p: ProviderView): UpstreamForm { scopeList: p.models_only ?? [], billing: p.billing === "free" ? "free" : "per-token", pricing: p.pricing ?? "", + maxConcurrent: p.max_concurrent != null ? String(p.max_concurrent) : "", disabled: p.disabled, }; } @@ -283,10 +287,23 @@ export function toInput(f: UpstreamForm): ProviderInput { models_only: f.scope === "some" ? f.scopeList : undefined, billing: f.billing, pricing: f.pricing || undefined, + max_concurrent: concurrencyOf(f) ?? undefined, disabled: f.disabled, }; } +/** 并发上限最多写多少(core 的 `MAX_PROVIDER_CONCURRENCY`) */ +export const MAX_CONCURRENCY = 1000; + +/** 格子里的并发上限:空是不限(`null`),写的不是 1 到 1000 的整数是 `undefined` */ +export function concurrencyOf(f: UpstreamForm): number | null | undefined { + const v = f.maxConcurrent.trim(); + if (v === "") return null; + if (!/^\d+$/.test(v)) return undefined; + const n = Number(v); + return n >= 1 && n <= MAX_CONCURRENCY ? n : undefined; +} + /** * 连接信息和已保存的那一家比改过没有:地址、协议、凭据、请求头、出站代理。 * 改过的话,模型列表要按表单里的新值去问。 @@ -327,6 +344,7 @@ export function connectionMissing( if (header === "") return t.headerName; if (r.value.trim() === "") return t.headerValue(header); } + if (concurrencyOf(f) === undefined) return t.concurrency; return null; } From 9ad429c66427b8990140ad18f52a8d31a105cc62 Mon Sep 17 00:00:00 2001 From: fylorn <249551762+fylorn@users.noreply.github.com> Date: Mon, 5 Oct 2026 19:30:39 +0800 Subject: [PATCH 06/10] Routing: call the speed sample "first token", and say weights share requests Core measures the speed sample behind url-test and load-balance by speed from sending a hop to the first content of the answer, not to the response headers. The dry run and the url-test description called it TTFB / first byte, the word the request detail keeps for the headers; they now say first token, as the request detail, traffic and upstream table already do for that measurement. The terminology table records both terms. Weights share requests, not new conversations: a turn that stays on an upstream is charged to it. The load-balance and distribute-by descriptions now talk about requests. Co-Authored-By: Claude Opus 5.5 --- src/i18n/terminology.md | 3 ++- src/labels.i18n.ts | 2 +- src/labels.ts | 2 +- src/routing/DryRunDialog.i18n.ts | 10 +++++----- src/routing/DryRunDialog.tsx | 4 ++-- src/routing/GroupDialog.i18n.ts | 16 ++++++++-------- src/routing/model.i18n.ts | 8 ++++---- src/routing/model.ts | 2 +- 8 files changed, 24 insertions(+), 23 deletions(-) diff --git a/src/i18n/terminology.md b/src/i18n/terminology.md index c69625e2..d16bff36 100644 --- a/src/i18n/terminology.md +++ b/src/i18n/terminology.md @@ -93,7 +93,8 @@ known colloquialisms. | 积分 | credits | the unit of a GLM Coding Plan billed in credits: 剩余 1,976 / 2,000 积分 = 1,976 / 2,000 credits left; not the ChatGPT reset credits | | 会话日志 | session log | the whole conversation DeepSeek Harness attaches to each request | | 按量计费 / 不计费 | Per token / Free | billing: the only two modes; subscription accounts are billed per token | -| 首字节 | time to first byte (TTFB) | column headers may use "TTFB" | +| 首字节 | time to first byte (TTFB) | column headers may use "TTFB"; the response headers arriving, not the answer | +| 首 token / 首 token 时间 | first token / time to first token | from sending to the first content of the answer; group ordering by speed (url-test, load-balance by speed) uses this | | 延迟 / 总耗时 / 生成用时 | latency / total time / generation time | | | 故障转移 | failover | | | 尝试链 | attempts | | diff --git a/src/labels.i18n.ts b/src/labels.i18n.ts index 1afdec89..4411d03e 100644 --- a/src/labels.i18n.ts +++ b/src/labels.i18n.ts @@ -11,7 +11,7 @@ export const labelsText = messages( "url-test": "延迟最低", cheapest: "费用最低", }, - /** 轮询组按什么分新对话(`balance_by`) */ + /** 轮询组按什么分请求(`balance_by`) */ balanceBy: { weights: "按比例", latency: "按速度", diff --git a/src/labels.ts b/src/labels.ts index cacffd7c..701879f4 100644 --- a/src/labels.ts +++ b/src/labels.ts @@ -48,7 +48,7 @@ export function groupKindLabel(kind: GroupKind): string { return textOf(labelsText).groupKinds[kind]; } -/** 轮询组按什么分新对话,按界面上的先后:先是只看比例,再是自动的几种 */ +/** 轮询组按什么分请求,按界面上的先后:先是只看比例,再是自动的几种 */ export const BALANCE_BY: readonly BalanceBy[] = ["weights", "latency", "health", "latency-health"]; export function balanceByLabel(by: BalanceBy): string { diff --git a/src/routing/DryRunDialog.i18n.ts b/src/routing/DryRunDialog.i18n.ts index df2df60e..ec158413 100644 --- a/src/routing/DryRunDialog.i18n.ts +++ b/src/routing/DryRunDialog.i18n.ts @@ -25,10 +25,10 @@ export const dryRunText = messages( position: (n: number) => ` · 第 ${n} 条`, attempts: "尝试顺序", circuitOpen: "熔断中,将跳过", - /** 轮询组的候选:权重,和自动分配看的那几个数 */ + /** 轮询组的候选:权重,和自动分配看的那几个数。快慢是首 token(从发出到回答的第一段内容),和请求详情同一个词 */ weight: (n: number) => `权重 ${n}`, - ttfb: (ms: string) => `首字节 ${ms}`, - ttfbNone: "首字节暂无数据", + ttft: (ms: string) => `首 token ${ms}`, + ttftNone: "首 token 暂无数据", success: (pct: string) => `成功率 ${pct}`, successNone: "成功率暂无数据", share: (pct: string) => `占比 ${pct}`, @@ -83,8 +83,8 @@ export const dryRunText = messages( attempts: "Attempts", circuitOpen: "Circuit open, will be skipped", weight: (n: number) => `Weight ${n}`, - ttfb: (ms: string) => `TTFB ${ms}`, - ttfbNone: "No TTFB data yet", + ttft: (ms: string) => `First token ${ms}`, + ttftNone: "No first-token data yet", success: (pct: string) => `Success rate ${pct}`, successNone: "No success rate yet", share: (pct: string) => `Share ${pct}`, diff --git a/src/routing/DryRunDialog.tsx b/src/routing/DryRunDialog.tsx index a0938394..226178ec 100644 --- a/src/routing/DryRunDialog.tsx +++ b/src/routing/DryRunDialog.tsx @@ -69,7 +69,7 @@ export type DryRunTarget = * * 尝试顺序里每个上游写出发给它的模型名和来历(别名、规则改写、指定模型):客户端写的 * 名称和发出的不同,正是要在这里看清的事。经过轮询组时再写它的权重;按速度、稳定性 - * 分配时还写首字节时间、成功率和算下来的占比 —— 「为什么轮到它」要从这里看得出来。 + * 分配时还写首 token 时间、成功率和算下来的占比 —— 「为什么轮到它」要从这里看得出来。 */ export function DryRunDialog({ target, @@ -592,7 +592,7 @@ function BalanceFacts({ c, by, share }: { c: DryRunCandidate; by: BalanceBy; sha const t = useText(dryRunText); const parts = [t.weight(c.weight ?? 1)]; if (by === "latency" || by === "latency-health") { - parts.push(c.ttfb_ms != null ? t.ttfb(`${Math.round(c.ttfb_ms).toLocaleString()}ms`) : t.ttfbNone); + parts.push(c.ttfb_ms != null ? t.ttft(`${Math.round(c.ttfb_ms).toLocaleString()}ms`) : t.ttftNone); } if (by === "health" || by === "latency-health") { parts.push(c.success_rate != null ? t.success(percent(c.success_rate)) : t.successNone); diff --git a/src/routing/GroupDialog.i18n.ts b/src/routing/GroupDialog.i18n.ts index 8b874829..452acc5b 100644 --- a/src/routing/GroupDialog.i18n.ts +++ b/src/routing/GroupDialog.i18n.ts @@ -27,10 +27,10 @@ export const groupDialogText = messages( weightInvalid: (name: string) => `${name} 的权重须为 1 到 100 的整数`, balanceBy: "分配依据", balanceDesc: { - weights: "新对话按成员权重的比例分配。", - latency: "以权重为基础,速度快的上游分到更多新对话。", - health: "以权重为基础,失败少的上游分到更多新对话。", - "latency-health": "以权重为基础,速度快、失败少的上游分到更多新对话。", + weights: "请求按成员权重的比例分配。", + latency: "以权重为基础,速度快的上游分到更多请求。", + health: "以权重为基础,失败少的上游分到更多请求。", + "latency-health": "以权重为基础,速度快、失败少的上游分到更多请求。", } satisfies Record, }, { @@ -57,10 +57,10 @@ export const groupDialogText = messages( weightInvalid: (name: string) => `The weight of ${name} must be a whole number from 1 to 100`, balanceBy: "Distribute by", balanceDesc: { - weights: "New conversations are split by the members' weights.", - latency: "Starting from the weights, faster upstreams get more new conversations.", - health: "Starting from the weights, upstreams that fail less get more new conversations.", - "latency-health": "Starting from the weights, faster upstreams that fail less get more new conversations.", + weights: "Requests are split by the members' weights.", + latency: "Starting from the weights, faster upstreams get more requests.", + health: "Starting from the weights, upstreams that fail less get more requests.", + "latency-health": "Starting from the weights, faster upstreams that fail less get more requests.", } satisfies Record, }, ); diff --git a/src/routing/model.i18n.ts b/src/routing/model.i18n.ts index feee0655..4101f6a0 100644 --- a/src/routing/model.i18n.ts +++ b/src/routing/model.i18n.ts @@ -35,8 +35,8 @@ export const modelText = messages( strategies: { fallback: "依次使用成员,前一个不可用时使用下一个。", select: "使用选定的上游;它不可用时,按顺序使用其余成员。", - loadBalance: "在成员之间轮流分配新对话。", - urlTest: "优先使用首字节时间最短的上游。", + loadBalance: "在成员之间轮流分配请求。", + urlTest: "优先使用首 token 时间最短的上游。", cheapest: "优先使用输入单价最低的上游。", }, }, @@ -76,8 +76,8 @@ export const modelText = messages( strategies: { fallback: "Uses the members in order, moving to the next when one is unavailable.", select: "Uses the selected upstream; when it is unavailable, uses the other members in order.", - loadBalance: "Distributes new conversations across the members in turn.", - urlTest: "Prefers the upstream with the shortest time to first byte.", + loadBalance: "Distributes requests across the members in turn.", + urlTest: "Prefers the upstream with the shortest time to first token.", cheapest: "Prefers the upstream with the lowest input price.", }, }, diff --git a/src/routing/model.ts b/src/routing/model.ts index 84f7a6c6..cb4450c5 100644 --- a/src/routing/model.ts +++ b/src/routing/model.ts @@ -436,7 +436,7 @@ export function strategyText(g: Pick Date: Mon, 5 Oct 2026 20:25:57 +0800 Subject: [PATCH 07/10] chore(core): build on core's final routing round (rev adadbf5) Core's routing round is final at adadbf5 (integ/routing-2026-10, protocol 39). The six core crates move to that rev; the lock file changes only in their source lines. tw-api.ts is regenerated: the protocol constant and doc comments changed, no field did. Core's checks changed, so the UI follows them: - Key limits: a cost limit has to be at least $0.01, and day, week and month limits need request records kept for 1, 7 and 31 days (config.key_limit_cost_too_small, config.key_limit_retention, which replaces the month-only code). The key dialog checks both on the spot, in core's order, and names the period in the retention line. - New codes translated: config.key_limit_retention, config.key_limit_cost_too_small, gw.ws.upstream_closed; the retired config.key_limit_month_retention is removed. - failover.slot_wait_secs is now one total wait per request, for a full upstream and for a key's minute or hour limit alike, so the setting no longer says it is only about concurrency. Co-Authored-By: Claude Opus 5.5 --- src-tauri/Cargo.lock | 14 ++++----- src-tauri/Cargo.toml | 12 ++++---- src/generated/tw-api.ts | 43 ++++++++++++++++------------ src/i18n/core.zh.json | 4 ++- src/keys/LimitsEditor.i18n.ts | 10 +++++-- src/keys/LimitsEditor.tsx | 15 ++++++---- src/keys/limits.test.ts | 20 +++++++++---- src/keys/limits.ts | 29 +++++++++++++------ src/settings/FailoverSection.i18n.ts | 6 ++-- 9 files changed, 94 insertions(+), 59 deletions(-) diff --git a/src-tauri/Cargo.lock b/src-tauri/Cargo.lock index c4bf1cdd..2796202f 100644 --- a/src-tauri/Cargo.lock +++ b/src-tauri/Cargo.lock @@ -5269,7 +5269,7 @@ dependencies = [ [[package]] name = "tw-api" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" dependencies = [ "serde", "serde_json", @@ -5281,7 +5281,7 @@ dependencies = [ [[package]] name = "tw-dialect" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" dependencies = [ "serde", "serde_json", @@ -5290,7 +5290,7 @@ dependencies = [ [[package]] name = "tw-guard" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" dependencies = [ "base64 0.22.1", "bytes", @@ -5306,7 +5306,7 @@ dependencies = [ [[package]] name = "tw-link" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" dependencies = [ "serde", "serde_json", @@ -5333,7 +5333,7 @@ dependencies = [ [[package]] name = "tw-types" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" dependencies = [ "serde", "ts-rs", @@ -5342,7 +5342,7 @@ dependencies = [ [[package]] name = "tw-watch" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" dependencies = [ "notify", "thiserror 2.0.21", @@ -5352,7 +5352,7 @@ dependencies = [ [[package]] name = "tw-yaml" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=3e7841e2120134f883d90810f70aa9a1ad33a57e#3e7841e2120134f883d90810f70aa9a1ad33a57e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" dependencies = [ "saphyr-parser", "thiserror 2.0.21", diff --git a/src-tauri/Cargo.toml b/src-tauri/Cargo.toml index e8e17b5d..5898ec88 100644 --- a/src-tauri/Cargo.toml +++ b/src-tauri/Cargo.toml @@ -26,12 +26,12 @@ license = "MIT" # twcore 二进制必须和这里编译进去的协议镜像来自同一个 core 版本 —— 打包脚本正是 # 从 tw-api 锁到的 tag 去取二进制的(见 scripts/fetch-core.sh)。 [workspace.dependencies] -tw-api = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } -tw-types = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } -tw-yaml = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } -tw-guard = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } -tw-watch = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } -tw-link = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "3e7841e2120134f883d90810f70aa9a1ad33a57e" } +tw-api = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } +tw-types = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } +tw-yaml = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } +tw-guard = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } +tw-watch = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } +tw-link = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } [lib] name = "thinkwatch_lite_lib" diff --git a/src/generated/tw-api.ts b/src/generated/tw-api.ts index 6be60a99..50bf80bc 100644 --- a/src/generated/tw-api.ts +++ b/src/generated/tw-api.ts @@ -1,6 +1,6 @@ // Generated by tw-api (`tw_api::ts::export_all`). Do not edit by hand. -export const CONTROL_API_VERSION = 38; +export const CONTROL_API_VERSION = 39; /** * 一个账号上游登的是哪个账号。 @@ -256,8 +256,9 @@ export type AttemptView = { provider: string, */ model?: string | null, /** - * WebSocket 的那一跳是一次握手:上游同意升级(101)是 `served`,回了别的 - * 状态码是 `status`,连不上是 `error`。 + * WebSocket 连接的那一跳是一次握手:上游同意升级(101)是 `served`,回了别的 + * 状态码是 `status`,连不上是 `error`。Responses 的连接上每一轮是一个请求,那一跳是这条 + * 已经接下的连接:发出去了是 `served`,状态码记 200。 */ outcome: AttemptOutcome, /** @@ -325,11 +326,11 @@ profile?: string | null, region?: string | null, }; /** - * `load-balance` 组按什么分新对话:配置里 `balance_by` 写的那个词。 + * `load-balance` 组按什么分请求:配置里 `balance_by` 写的那个词。 * * 成员的权重永远是底数,快慢、成败算出的系数乘在上面 - * ([`DryRunCandidate::balance_factor`]);进行中的对话照旧留在回答它的那一家。 - * 没有测到的上游算中等。 + * ([`DryRunCandidate::balance_factor`]),长期看各家分到的请求是乘出来的比例;进行中的 + * 对话照旧留在回答它的那一家,记在那一家的份额里。没有测到的上游算中等。 */ export type BalanceBy = "weights" | "latency" | "health" | "latency-health"; @@ -952,8 +953,8 @@ model_via?: string | null, */ weight?: number | null, /** - * 它的典型首字节时间(最近样本的中位数),毫秒。只在顺序看它时有:`url-test`, - * 按快慢分的 `load-balance`。样本不够时没有 + * 它典型的快慢:从发出去到回答的第一段内容(最近样本的中位数),毫秒。只在顺序看它时 + * 有:`url-test`,按快慢分的 `load-balance`。样本不够时没有 */ ttfb_ms?: number | null, /** @@ -1016,7 +1017,7 @@ route: string, */ strategy?: GroupKind | null, /** - * 经过的是 `load-balance` 组时,它按什么分新对话。别的时候没有 + * 经过的是 `load-balance` 组时,它按什么分请求。别的时候没有 */ balance_by?: BalanceBy | null, /** @@ -1090,7 +1091,8 @@ client_hint?: string | null, * 的指纹)、离这段对话的上一个请求不超过半小时,就还是那一次;隔久了算 * 新的一次。**开始时就给出来**,界面才能把一个还在跑的请求放进它的会话、 * 把那次会话标成进行中。认不出会话的没有:正文里没有任何能认人的东西, - * 或者是 WebSocket 升级(升级请求没有正文) + * 或者是整条连接一行的 WebSocket(升级请求没有正文)。Responses 连接上的每一轮 + * 按那一帧认,和 HTTP 的请求一样 */ session?: string | null, /** @@ -1168,7 +1170,8 @@ session_log_bytes?: number | null, at_ms: number, } | { "kind": "request_headers * 在结局里才到的。模型名只在开始事件里的话,一个开始时没人在听、 * 结束时有人在听的请求,它的用量就不知道该记在哪个模型上。 * - * WebSocket 那条路是空串:升级请求里没有模型名(和开始事件一样)。 + * 整条连接一行的 WebSocket 和开始事件一样:Realtime 是查询串里的那个,别的连接 + * 升级时还不知道,是空串。 */ model: string, status: number, bytes: number, duration_ms: number, /** @@ -1189,7 +1192,7 @@ tokens_per_sec?: number | null, * 上游在回答里写的模型名:Anthropic 和 Chat 的 `model`、Responses 的 * `response.model`、Gemini 的 `modelVersion`。**原样,不归一。** * - * 回答里没写的没有:Bedrock 的 Converse 不写,WebSocket 那条路不看。和 + * 回答里没写的没有:Bedrock 的 Converse 不写,整条连接一行的 WebSocket 不看。和 * `model` 不是一回事 —— 那是客户端要的,这是上游说它用的 */ answered_model?: string | null, } | { "kind": "request_failed", id: number, @@ -1528,7 +1531,8 @@ stream_start_wait_secs: number, */ next_on_slow_start: boolean, /** - * 上游满着(`max_concurrent`)时,一个请求合计最多等多少秒空位。0 是不等 + * 一个请求合计最多等多少秒:等密钥的分钟、小时上限空出名额,和等满着(`max_concurrent`) + * 的上游空出位置,共用这一段。0 是不等 */ slot_wait_secs: number, }; @@ -1556,7 +1560,7 @@ selected?: string | null, */ weights?: { [key in string]: number } | null, /** - * `load-balance` 组按什么分新对话。不给 = `weights`;别的类型只能是 `weights` + * `load-balance` 组按什么分请求。不给 = `weights`;别的类型只能是 `weights` */ balance_by?: BalanceBy | null, }; @@ -1594,12 +1598,13 @@ kind: GroupKind, */ selected?: string | null, providers: Array, /** - * `load-balance` 组每个成员的权重,**每个成员都在**,没写权重的是 1:新对话按这个 - * 比例分。别的类型不用权重,是空的 + * `load-balance` 组每个成员的权重,**每个成员都在**,没写权重的是 1:长期看各家分到的 + * 请求就是这个比例。进行中的对话留在回答它的那一家,那一轮记在那一家的份额里,新对话把 + * 差的补回去。别的类型不用权重,是空的 */ weights: { [key in string]: number }, /** - * `load-balance` 按什么分新对话。别的类型永远是 `weights` + * `load-balance` 按什么分请求。别的类型永远是 `weights` */ balance_by: BalanceBy, }; @@ -1719,7 +1724,7 @@ translated?: TranslatedView | null, * 任务;看着一次很贵的任务,也回不到具体是哪一条。库里这一列一直 * 都在(`requests.session`,还建了索引),只是没有交出来。 * - * 认不出会话的请求(拼不出指纹的,比如 WebSocket、本地应答)是 `None`。 + * 认不出会话的请求(拼不出指纹的,比如整条连接一行的 WebSocket、本地应答)是 `None`。 */ session?: string | null, /** @@ -4497,7 +4502,7 @@ cost_micros_exact: number, cost_micros_estimated: number, unpriced_requests: number, /** * 有多少条请求**没有拿到用量**,所以同样算不出钱:上游没报,或者连接 - * 在它报之前就结束了(客户端取消、WebSocket 会话)。 + * 在它报之前就结束了(客户端取消、整条连接一行的 WebSocket 会话)。 * * 和 `unpriced_requests` 一样让金额合计偏低,但配价格解决不了它 —— * 界面上是两句不同的话。上游确实接下了的才算:成功的响应和客户端 diff --git a/src/i18n/core.zh.json b/src/i18n/core.zh.json index 6263568f..41888daf 100644 --- a/src/i18n/core.zh.json +++ b/src/i18n/core.zh.json @@ -451,6 +451,7 @@ "gw.ws.connect_failed": "无法连接上游的 WebSocket:{detail}", "gw.ws.send_failed": "向上游发送数据失败:{detail}", "gw.ws.upstream_broke": "上游连接中断:{detail}", + "gw.ws.upstream_closed": "上游在回答完成前关闭了连接。", "gw.ws.proxy_unsupported": "上游「{upstream}」配置了代理({proxy}),WebSocket 连接暂不支持经代理转发,仅支持直连的上游。", "gw.toolcall.connection_cut": "回答中的 {tool} 调用命中规则「{?why:{rule:scan_rule}|{name}}」{?why:({rule:rule_why})},已切断连接。", "// ── control:控制面的 HTTP 错误 ──────────────────────────────────": "", @@ -634,7 +635,8 @@ "config.key_limit_not_positive": "网关密钥「{key}」的一条用量上限({measure:limit_measure},每{per:limit_per})为 {value},须大于 0。", "config.key_limit_duplicate": "网关密钥「{key}」有两条相同的用量上限({measure:limit_measure},每{per:limit_per}),请只保留一条。", "config.key_limit_cache_reads": "网关密钥「{key}」的一条用量上限({measure:limit_measure})设置了 cache_reads,只有 token 数上限可以设置该项。", - "config.key_limit_month_retention": "网关密钥「{key}」设置了每月的用量上限,而 retention.row_days 为 {days}。重启后当月用量要从请求记录中重新累计,因此 row_days 须至少为 31。", + "config.key_limit_retention": "网关密钥「{key}」设置了每{per:limit_per}的用量上限,而 retention.row_days 为 {days}。重启后当{per:limit_per}用量要从请求记录中重新累计,因此 row_days 须至少为 {min}。", + "config.key_limit_cost_too_small": "网关密钥「{key}」每{per:limit_per}的费用上限为 {value},须至少为 0.01。", "config.name_collision": "「{name}」同时是上游和策略组的名称,规则的 to 无法区分指的是哪一个。请重命名其中一个。", "config.bad_allow_from": "listen.gateway.allow_from 中的 {entry} 不是有效的 IP 地址或 CIDR,应写成 192.168.0.0/16 的形式。", "config.unknown_price_sheet": "上游「{upstream}」使用的价目表「{sheet}」不存在。", diff --git a/src/keys/LimitsEditor.i18n.ts b/src/keys/LimitsEditor.i18n.ts index 71b01798..a244fdb1 100644 --- a/src/keys/LimitsEditor.i18n.ts +++ b/src/keys/LimitsEditor.i18n.ts @@ -16,8 +16,11 @@ export const limitsEditorText = messages( removeLabel: (n: number) => `删除第 ${n} 条上限`, required: "请填写上限", notPositive: (cost: boolean): string => (cost ? "须为大于 0 的金额" : "须为大于 0 的整数"), + costTooSmall: "金额须至少为 $0.01", duplicate: "与前面的一条上限重复", - monthRetention: (days: number) => `每月上限要求请求记录至少保留 31 天,当前为 ${days} 天`, + /** `per` 是周期的字(天、周、月),`need` 是它要求的天数 */ + retention: (per: string, need: number, days: number) => + `每${per}的上限要求请求记录至少保留 ${need} 天,当前为 ${days} 天`, unpriced: (n: number) => `${n} 个可用模型没有价格,其费用按 0 计入上限:`, more: (n: number) => `另有 ${n} 个`, fewer: "收起", @@ -37,9 +40,10 @@ export const limitsEditorText = messages( removeLabel: (n: number) => `Remove limit ${n}`, required: "Enter a limit", notPositive: (cost: boolean): string => (cost ? "Must be an amount above 0" : "Must be a whole number above 0"), + costTooSmall: "Must be at least $0.01", duplicate: "Same as a limit above", - monthRetention: (days: number) => - `A monthly limit needs request records kept for at least 31 days; they are kept for ${days}`, + retention: (per: string, need: number, days: number) => + `A limit per ${per} needs request records kept for at least ${need === 1 ? "1 day" : `${need} days`}; they are kept for ${days === 1 ? "1 day" : `${days} days`}`, unpriced: (n: number) => n === 1 ? "1 model this key can use has no price and counts as $0 toward the limit:" diff --git a/src/keys/LimitsEditor.tsx b/src/keys/LimitsEditor.tsx index f9ba664f..f266ad2c 100644 --- a/src/keys/LimitsEditor.tsx +++ b/src/keys/LimitsEditor.tsx @@ -14,6 +14,7 @@ import type { KeyLimitView } from "@/types"; import { MEASURES, PERS, + ROW_DAYS_NEEDED, amount, cleanMax, newRow, @@ -154,11 +155,13 @@ function LimitLine({ ? t.required : problem === "notPositive" ? t.notPositive(r.measure === "cost") - : problem === "duplicate" - ? t.duplicate - : problem === "monthRetention" - ? t.monthRetention(rowDays ?? 0) - : null; + : problem === "costTooSmall" + ? t.costTooSmall + : problem === "duplicate" + ? t.duplicate + : problem === "retention" + ? t.retention(w.per[r.per], ROW_DAYS_NEEDED[r.per] ?? 0, rowDays ?? 0) + : null; return (
    @@ -177,7 +180,7 @@ function LimitLine({ {t.atMost} { expect(rows.map((r) => p.get(r.id) ?? null)).toEqual([null, null, "duplicate"]); }); - it("每月的上限要求记录留够 31 天;不知道留几天就不拦", () => { - const rows = [row({ per: "month" })]; - expect(limitProblems(rows, 30).get(rows[0]!.id)).toBe("monthRetention"); - expect(limitProblems(rows, 31).size).toBe(0); - expect(limitProblems(rows, null).size).toBe(0); + it("天、周、月的上限要求记录留够 1、7、31 天;分钟、小时不要;不知道留几天就不拦", () => { + for (const [per, need] of [["day", 1], ["week", 7], ["month", 31]] as const) { + const rows = [row({ per })]; + expect(limitProblems(rows, need - 1).get(rows[0]!.id)).toBe("retention"); + expect(limitProblems(rows, need).size).toBe(0); + expect(limitProblems(rows, null).size).toBe(0); + } + expect(limitProblems([row({ per: "minute" }), row({ per: "hour" })], 0).size).toBe(0); + }); + + it("费用上限至少 $0.01,和 core 一样排在重复之前", () => { + const rows = [row({ max: "0.009999" }), row({ per: "week", max: "0.01" }), row({ per: "week", max: "0.001" })]; + const p = limitProblems(rows, 90); + expect(rows.map((r) => p.get(r.id) ?? null)).toEqual(["costTooSmall", null, "costTooSmall"]); + expect(limitProblems([row({ measure: "requests", max: "1" })], 90).size).toBe(0); }); it("加一行先给还没有的那一种,不一加上就重复", () => { diff --git a/src/keys/limits.ts b/src/keys/limits.ts index 835431dc..2eca35d4 100644 --- a/src/keys/limits.ts +++ b/src/keys/limits.ts @@ -2,8 +2,9 @@ * 一把密钥的用量上限:对话框里一行一条,和 core 的 `KeyLimitView` / `KeyLimitInput` 来回换。 * * **写得对不对由 core 的配置校验说**(`config.key_limit_*`)。这里当场查的是同一套规则里 - * 填的时候就看得出来的几条:要填、要大于 0、同一个周期同一种量(token 再分算不算缓存 - * 读取)只能有一条、每月的要求请求记录至少留 31 天 —— 免得填完整张对话框,保存时才被拒。 + * 填的时候就看得出来的几条,顺序也和 core 一样:要填、要大于 0、费用至少 $0.01、同一个周期 + * 同一种量(token 再分算不算缓存读取)只能有一条、天 / 周 / 月的上限要求请求记录至少留 + * 1 / 7 / 31 天 —— 免得填完整张对话框,保存时才被拒。 * * 费用在 core 那边是微分(`max`、`used`),输入框里是美元。 */ @@ -15,8 +16,14 @@ import { limitsText } from "./limits.i18n"; export const PERS: readonly LimitPer[] = ["minute", "hour", "day", "week", "month"]; export const MEASURES: readonly LimitMeasure[] = ["requests", "tokens", "cost"]; -/** 每月上限要求请求记录至少留这么多天:重启之后当月的用量从记录里加回来 */ -export const MONTH_ROW_DAYS = 31; +/** + * 天、周、月的上限要求请求记录至少留几天:重启之后这一期的用量从记录里加回来,留得比一期 + * 短就加不全(core 的 `row_days_needed`)。分钟、小时的从空的开始,不要记录 + */ +export const ROW_DAYS_NEEDED: Readonly>> = { day: 1, week: 7, month: 31 }; + +/** 费用上限最少多少,微分($0.01,core 的 `COST_MIN`) */ +export const COST_MIN_MICROS = 10_000; /** 对话框里的一行 */ export interface LimitRow { @@ -97,11 +104,12 @@ export function newRow(rows: readonly LimitRow[]): LimitRow { return limitRow({ per: "day", measure: "cost", max: "", cacheReads: false }); } -export type LimitProblem = "required" | "notPositive" | "duplicate" | "monthRetention"; +export type LimitProblem = "required" | "notPositive" | "costTooSmall" | "duplicate" | "retention"; /** - * 每一行有什么不对,按行的 id。**一行只说一件**:先说要填,再说要大于 0,再说重复(和 - * 前面哪一行一样,就标在后面那一行上),最后说每月的上限要记录留够天数。 + * 每一行有什么不对,按行的 id。**一行只说一件**,和 core 查的先后一样:先说要填,再说要 + * 大于 0,再说费用不到 $0.01,再说重复(和前面哪一行一样,就标在后面那一行上),最后说 + * 天、周、月的上限要记录留够天数(`ROW_DAYS_NEEDED`)。 * * `rowDays`:此刻请求记录留几天(概览里的 `retention.row_days`)。不知道就不查这一条, * 保存时由 core 说 @@ -111,10 +119,13 @@ export function limitProblems(rows: readonly LimitRow[], rowDays: number | null) const seen = new Set(); for (const r of rows) { const id = identity(r.per, r.measure, r.cacheReads); + const max = parseMax(r.measure, r.max); + const need = ROW_DAYS_NEEDED[r.per]; if (r.max.trim() === "") out.set(r.id, "required"); - else if (parseMax(r.measure, r.max) == null) out.set(r.id, "notPositive"); + else if (max == null) out.set(r.id, "notPositive"); + else if (r.measure === "cost" && max < COST_MIN_MICROS) out.set(r.id, "costTooSmall"); else if (seen.has(id)) out.set(r.id, "duplicate"); - else if (r.per === "month" && rowDays != null && rowDays < MONTH_ROW_DAYS) out.set(r.id, "monthRetention"); + else if (need != null && rowDays != null && rowDays < need) out.set(r.id, "retention"); seen.add(id); } return out; diff --git a/src/settings/FailoverSection.i18n.ts b/src/settings/FailoverSection.i18n.ts index 05f333f6..c16b4c22 100644 --- a/src/settings/FailoverSection.i18n.ts +++ b/src/settings/FailoverSection.i18n.ts @@ -12,7 +12,7 @@ export const failoverText = messages( quota_pause_secs: "额度用完", rate_limit_max_pause_secs: "限流", stream_start_wait_secs: "等待回答开头", - slot_wait_secs: "并发已满时最多等待", + slot_wait_secs: "最多等待空位", }, what: { failures_to_pause: "服务器错误、无法连接等未说明原因的失败,连续达到此次数后暂停。", @@ -22,7 +22,7 @@ export const failoverText = messages( quota_pause_secs: "上游报告额度用完、但未给出重置时间时暂停的时长。给出重置时间的,暂停到重置为止。", rate_limit_max_pause_secs: "上游限流时按其要求的等待时间暂停,最长为此值。", stream_start_wait_secs: "流式回答在第一段内容到达前报错时,请求交给下一个上游。等待超过此时长后不再等待。", - slot_wait_secs: "上游达到并发上限时,请求等待空位的最长时间。0 表示不等待。", + slot_wait_secs: "上游达到并发上限、或密钥达到每分钟或每小时上限时,一个请求合计最多等待的时长。0 表示不等待。", }, nextOnSlowStart: "开头超时时转到下一个上游", nextOnSlowStartWhat: "最后一个上游照常等待。开启时,等待时长宜在 30 秒以上。", @@ -62,7 +62,7 @@ export const failoverText = messages( stream_start_wait_secs: "An error before the first content of a streamed answer sends the request to the next upstream. After this long, the wait ends.", slot_wait_secs: - "How long a request waits for a free slot when upstreams are at their concurrency limit. 0 means no wait.", + "How long a request waits in total when upstreams are at their concurrency limit or its key is at a per-minute or per-hour limit. 0 means no wait.", }, nextOnSlowStart: "Move to the next upstream when the start times out", nextOnSlowStartWhat: "The last upstream keeps waiting. With this on, a wait of 30 s or more is advisable.", From 8c880afce4c8c9b9cb1ca934e88ea0c3ba1b9362 Mon Sep 17 00:00:00 2001 From: fylorn <249551762+fylorn@users.noreply.github.com> Date: Mon, 5 Oct 2026 20:36:47 +0800 Subject: [PATCH 08/10] Failover: keep the slot-wait description to one line The description that now covers both waits (a full upstream, a key's minute or hour limit) wrapped to two lines where every other failover setting takes one. Co-Authored-By: Claude Opus 5.5 --- src/settings/FailoverSection.i18n.ts | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/src/settings/FailoverSection.i18n.ts b/src/settings/FailoverSection.i18n.ts index c16b4c22..ec929c9d 100644 --- a/src/settings/FailoverSection.i18n.ts +++ b/src/settings/FailoverSection.i18n.ts @@ -22,7 +22,7 @@ export const failoverText = messages( quota_pause_secs: "上游报告额度用完、但未给出重置时间时暂停的时长。给出重置时间的,暂停到重置为止。", rate_limit_max_pause_secs: "上游限流时按其要求的等待时间暂停,最长为此值。", stream_start_wait_secs: "流式回答在第一段内容到达前报错时,请求交给下一个上游。等待超过此时长后不再等待。", - slot_wait_secs: "上游达到并发上限、或密钥达到每分钟或每小时上限时,一个请求合计最多等待的时长。0 表示不等待。", + slot_wait_secs: "上游并发已满或密钥的分钟、小时上限用满时,请求合计最多等待的时长。0 表示不等待。", }, nextOnSlowStart: "开头超时时转到下一个上游", nextOnSlowStartWhat: "最后一个上游照常等待。开启时,等待时长宜在 30 秒以上。", @@ -61,8 +61,7 @@ export const failoverText = messages( rate_limit_max_pause_secs: "A rate-limited upstream is paused for the wait it asks for, at most this long.", stream_start_wait_secs: "An error before the first content of a streamed answer sends the request to the next upstream. After this long, the wait ends.", - slot_wait_secs: - "How long a request waits in total when upstreams are at their concurrency limit or its key is at a per-minute or per-hour limit. 0 means no wait.", + slot_wait_secs: "Total time a request waits for a full upstream or a key's per-minute or per-hour limit. 0 means no wait.", }, nextOnSlowStart: "Move to the next upstream when the start times out", nextOnSlowStartWhat: "The last upstream keeps waiting. With this on, a wait of 30 s or more is advisable.", From fb9d0aa17e1a5b994acdab3611dfa948fe5c0d1e Mon Sep 17 00:00:00 2001 From: fylorn <249551762+fylorn@users.noreply.github.com> Date: Mon, 5 Oct 2026 20:37:54 +0800 Subject: [PATCH 09/10] Pin core to main after ThinkWatch-Core#292 (same tree as the integration commit) Co-Authored-By: Claude Opus 5.5 --- src-tauri/Cargo.lock | 14 +++++++------- src-tauri/Cargo.toml | 12 ++++++------ 2 files changed, 13 insertions(+), 13 deletions(-) diff --git a/src-tauri/Cargo.lock b/src-tauri/Cargo.lock index 2796202f..c057e94d 100644 --- a/src-tauri/Cargo.lock +++ b/src-tauri/Cargo.lock @@ -5269,7 +5269,7 @@ dependencies = [ [[package]] name = "tw-api" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" dependencies = [ "serde", "serde_json", @@ -5281,7 +5281,7 @@ dependencies = [ [[package]] name = "tw-dialect" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" dependencies = [ "serde", "serde_json", @@ -5290,7 +5290,7 @@ dependencies = [ [[package]] name = "tw-guard" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" dependencies = [ "base64 0.22.1", "bytes", @@ -5306,7 +5306,7 @@ dependencies = [ [[package]] name = "tw-link" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" dependencies = [ "serde", "serde_json", @@ -5333,7 +5333,7 @@ dependencies = [ [[package]] name = "tw-types" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" dependencies = [ "serde", "ts-rs", @@ -5342,7 +5342,7 @@ dependencies = [ [[package]] name = "tw-watch" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" dependencies = [ "notify", "thiserror 2.0.21", @@ -5352,7 +5352,7 @@ dependencies = [ [[package]] name = "tw-yaml" version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e#adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" dependencies = [ "saphyr-parser", "thiserror 2.0.21", diff --git a/src-tauri/Cargo.toml b/src-tauri/Cargo.toml index 5898ec88..16eb0593 100644 --- a/src-tauri/Cargo.toml +++ b/src-tauri/Cargo.toml @@ -26,12 +26,12 @@ license = "MIT" # twcore 二进制必须和这里编译进去的协议镜像来自同一个 core 版本 —— 打包脚本正是 # 从 tw-api 锁到的 tag 去取二进制的(见 scripts/fetch-core.sh)。 [workspace.dependencies] -tw-api = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } -tw-types = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } -tw-yaml = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } -tw-guard = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } -tw-watch = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } -tw-link = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "adadbf53cfe93c10e8e42ffdb94e16d4e5a6180e" } +tw-api = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } +tw-types = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } +tw-yaml = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } +tw-guard = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } +tw-watch = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } +tw-link = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } [lib] name = "thinkwatch_lite_lib" From b115aa2d74433383453c75bf0d56d3568500bcc8 Mon Sep 17 00:00:00 2001 From: fylorn <249551762+fylorn@users.noreply.github.com> Date: Mon, 5 Oct 2026 21:54:06 +0800 Subject: [PATCH 10/10] 2026.10.6: core v0.63.0 Co-Authored-By: Claude Opus 5.5 --- package.json | 2 +- release-notes/2026.10.6.md | 30 ++++++++++++++++++++++++++++++ src-tauri/Cargo.lock | 30 +++++++++++++++--------------- src-tauri/Cargo.toml | 14 +++++++------- src-tauri/tauri.conf.json | 2 +- 5 files changed, 54 insertions(+), 24 deletions(-) create mode 100644 release-notes/2026.10.6.md diff --git a/package.json b/package.json index fc57eb2e..9b3780d1 100644 --- a/package.json +++ b/package.json @@ -2,7 +2,7 @@ "name": "thinkwatch-lite", "private": true, "description": "Desktop app for a local AI API gateway on macOS, Windows and Linux", - "version": "2026.10.5", + "version": "2026.10.6", "type": "module", "packageManager": "pnpm@11.13.0", "scripts": { diff --git a/release-notes/2026.10.6.md b/release-notes/2026.10.6.md new file mode 100644 index 00000000..ee6096a9 --- /dev/null +++ b/release-notes/2026.10.6.md @@ -0,0 +1,30 @@ +**Upgrade notes:** +- The bundled core is now 0.63.0, which speaks control-plane protocol 39. A remote server has to run core 0.63.0 too: this version does not connect to 0.62.0, and 2026.10.5 and earlier do not connect to 0.63.0. +- Request history is kept. +- A configuration edited by hand that lists an upstream twice in a strategy group, or a name that is not an upstream, no longer loads; safe mode names the group and the member. +- A key's concurrency limit now counts streamed answers until they end, so a key at its limit waits more often than before. + +**Strategy groups:** +- Load-balance groups take a weight per member, from 1 to 100, and a choice of how to distribute: by ratio, by speed, by reliability, or both. Speed is the time to first token, reliability the recent success rate; both adjust the weights, and an upstream without enough measurements counts as average. +- Ongoing conversations stay on their upstream and count toward its share. The route test shows each member's weight, speed, success rate and share, and predicts the next upstream. + +**Upstreams:** +- An upstream can have a concurrency limit. A conversation in progress waits for a free slot; other requests go to the next upstream, and when every upstream is full the request waits, then fails with a clear message. +- A model's context window and output limit can be set by hand under "Specs…" in the upstream's model list, for relays the price table does not know or gets wrong. A manual value is marked and replaces the price table's. +- Check-up findings (a different model name in answers, input reported high or low, few cache reads) now appear on the upstream's row; the separate Check-up tab is gone. + +**Failover settings:** +- "Move to the next upstream when the start times out" (off by default): a streamed answer that has sent nothing within the wait moves to the next upstream, if one can take it. The last upstream keeps waiting. +- "Wait for a free slot at most" (30 seconds by default) is the total time a request waits for a full upstream or a key's per-minute or per-hour limit. + +**Keys:** +- A key can have usage limits: requests, tokens or cost (USD) per minute, hour, day, week or month, all of which must hold. A token limit can count cache reads. Day, week and month reset at local midnight, on Monday and on the 1st. +- The key dialog shows each limit's usage and reset time, and lists the models the key can use that have no price; they count as $0. Keys at a limit are marked in the list, and a notification comes at 80% and when a limit is reached. + +**Traffic:** +- The attempt chain shows an upstream given up for a slow start, with the input tokens it may have billed, upstreams skipped at their concurrency limit, and time spent waiting for a slot. +- Requests refused by a key's usage limit or because every upstream was full are labelled as such. +- On a Responses WebSocket connection, each turn is now a request of its own, with its usage and cost. + +**Also:** +- The menu bar's menu follows the appearance chosen in the app instead of the menu bar's tint, so it no longer opens light in Dark Mode over a bright wallpaper. diff --git a/src-tauri/Cargo.lock b/src-tauri/Cargo.lock index c057e94d..7984f4bc 100644 --- a/src-tauri/Cargo.lock +++ b/src-tauri/Cargo.lock @@ -4766,7 +4766,7 @@ dependencies = [ [[package]] name = "thinkwatch-lite" -version = "2026.10.5" +version = "2026.10.6" dependencies = [ "anyhow", "block2", @@ -5268,8 +5268,8 @@ dependencies = [ [[package]] name = "tw-api" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "serde", "serde_json", @@ -5280,8 +5280,8 @@ dependencies = [ [[package]] name = "tw-dialect" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "serde", "serde_json", @@ -5289,8 +5289,8 @@ dependencies = [ [[package]] name = "tw-guard" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "base64 0.22.1", "bytes", @@ -5305,8 +5305,8 @@ dependencies = [ [[package]] name = "tw-link" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "serde", "serde_json", @@ -5332,8 +5332,8 @@ dependencies = [ [[package]] name = "tw-types" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "serde", "ts-rs", @@ -5341,8 +5341,8 @@ dependencies = [ [[package]] name = "tw-watch" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "notify", "thiserror 2.0.21", @@ -5351,8 +5351,8 @@ dependencies = [ [[package]] name = "tw-yaml" -version = "0.62.0" -source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?rev=c2fbbc2c23d284f021157f11a9cfab3c05e9b84e#c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" +version = "0.63.0" +source = "git+https://github.com/ThinkWatchProject/ThinkWatch-Core.git?tag=v0.63.0#3ac509473a50c9cbe2f92a895a462e02ca08498d" dependencies = [ "saphyr-parser", "thiserror 2.0.21", diff --git a/src-tauri/Cargo.toml b/src-tauri/Cargo.toml index 16eb0593..1a130ac5 100644 --- a/src-tauri/Cargo.toml +++ b/src-tauri/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "thinkwatch-lite" -version = "2026.10.5" +version = "2026.10.6" edition = "2024" # **这个数决定依赖能升到哪一版。**edition 2024 的解析器只挑声明的 Rust # 版本编得动的依赖:写 1.85 的时候,`cargo update` 一直停在旧的 time 和 @@ -26,12 +26,12 @@ license = "MIT" # twcore 二进制必须和这里编译进去的协议镜像来自同一个 core 版本 —— 打包脚本正是 # 从 tw-api 锁到的 tag 去取二进制的(见 scripts/fetch-core.sh)。 [workspace.dependencies] -tw-api = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } -tw-types = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } -tw-yaml = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } -tw-guard = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } -tw-watch = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } -tw-link = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", rev = "c2fbbc2c23d284f021157f11a9cfab3c05e9b84e" } +tw-api = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } +tw-types = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } +tw-yaml = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } +tw-guard = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } +tw-watch = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } +tw-link = { git = "https://github.com/ThinkWatchProject/ThinkWatch-Core.git", tag = "v0.63.0" } [lib] name = "thinkwatch_lite_lib" diff --git a/src-tauri/tauri.conf.json b/src-tauri/tauri.conf.json index 121160ad..8cf7b534 100644 --- a/src-tauri/tauri.conf.json +++ b/src-tauri/tauri.conf.json @@ -1,7 +1,7 @@ { "$schema": "https://schema.tauri.app/config/2", "productName": "ThinkWatch Lite", - "version": "2026.10.5", + "version": "2026.10.6", "identifier": "app.thinkwatch.lite", "build": { "beforeDevCommand": "pnpm dev",