From 0d35cd06bdd41b4a9e19cdc466d06098af1e0168 Mon Sep 17 00:00:00 2001 From: xuezhizone Date: Wed, 30 Sep 2026 11:36:31 +0800 Subject: [PATCH 1/4] feat(llm): task-level cumulative budget engine (issue #569) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replace the fixed 20-round segment counter as the only consumption guard with an optional task-level cumulative budget across four dimensions: tool-loop rounds, tool-call count, fresh tokens (input + output, cached prefix reads excluded via the new cache_read/cache_write usage fields), and active wall-clock seconds. Counters accrue across continues, compactions, sub-agent folds, and session resumes — they are never reset by 'continue'; limits come from config.yaml (llm.task_budget, all dimensions optional) and live from /setting. Budget checks run at the loop top before any model request; exhaustion saves the completed tool evidence through the #572 partial-turn pipeline, emits OpEnd(reason=budget_exhausted), and returns AishError::BudgetExhausted(dimension). Raising a limit is an explicit /setting action pushed live and audit-logged with before/after values. Task budget state persists in SessionStateSnapshot (serde-default, legacy snapshots unaffected) and is re-seeded on resume so restoring cannot bypass it. Near-threshold (80%) advisory fires once per cycle and re-arms on adjustment. Sub-agent loops accrue the same counters and fold into the parent task exactly once on completion. /token shows per-dimension used/limit with cost explicitly marked unavailable. The unbounded default keeps behavior identical to before. Workspace: 2214 tests passed, fmt and clippy -D warnings clean. Verified on a real binary against a mock LLM: max_rounds=5 stops after exactly 5 rounds, the snapshot matches the mock usage, and a resumed session blocks new turns with zero model requests. --- CONFIGURATION.md | 9 + crates/aish-config/src/lib.rs | 2 +- crates/aish-config/src/model.rs | 60 +++++ crates/aish-core/src/error.rs | 6 + crates/aish-i18n/locales/de-DE.yaml | 20 ++ crates/aish-i18n/locales/en-US.yaml | 20 ++ crates/aish-i18n/locales/es-ES.yaml | 20 ++ crates/aish-i18n/locales/fr-FR.yaml | 20 ++ crates/aish-i18n/locales/ja-JP.yaml | 20 ++ crates/aish-i18n/locales/zh-CN.yaml | 20 ++ crates/aish-llm/src/agents/spawn.rs | 41 +++ crates/aish-llm/src/agents/tool_loop.rs | 8 + crates/aish-llm/src/budget.rs | 297 ++++++++++++++++++++ crates/aish-llm/src/lib.rs | 1 + crates/aish-llm/src/session.rs | 344 +++++++++++++++++++++++- crates/aish-llm/src/usage.rs | 78 +++++- crates/aish-session/src/lib.rs | 2 +- crates/aish-session/src/models.rs | 69 +++++ crates/aish-session/src/store.rs | 3 + crates/aish-shell/src/ai_handler.rs | 162 +++++++++++ crates/aish-shell/src/app.rs | 199 +++++++++++++- crates/aish-shell/src/settings_panel.rs | 38 +++ 22 files changed, 1420 insertions(+), 19 deletions(-) create mode 100644 crates/aish-llm/src/budget.rs diff --git a/CONFIGURATION.md b/CONFIGURATION.md index 2bbc840c..0457652d 100644 --- a/CONFIGURATION.md +++ b/CONFIGURATION.md @@ -70,6 +70,7 @@ aish run --config ~/work/ai-shell-config.yaml | `context_token_budget` | integer/null | `null` | 可选的上下文 token 预算限制 | 如:4000,为 null 则仅使用消息数量限制 | | `enable_token_estimation` | boolean | `true` | 启用基于 tiktoken 的 token 估算 | true/false | | `context_auto_compact` | object | 见下方 | 自动上下文压缩配置,面向 shell/tool 输出控量 | 可选 | +| `task_budget` | object | 见下方 | 任务级累计预算(issue #569),任一维度达到上限即停止循环并提示(继续不会重置;调整预算需显式确认并写入审计) | 各维度 integer/null | ### 工具输出配置 @@ -203,6 +204,14 @@ max_llm_messages: 50 max_shell_messages: 20 context_token_budget: null enable_token_estimation: true + +# 任务级累计预算(可选;缺省不限,仅计数) +task_budget: + max_rounds: null # 累计工具循环轮次 + max_tool_calls: null # 累计工具调用次数 + max_tokens: null # 累计 token(输入+输出,不含缓存命中) + max_duration_secs: null # 累计活跃时长(秒) + context_auto_compact: enabled: true full_compact_enabled: true diff --git a/crates/aish-config/src/lib.rs b/crates/aish-config/src/lib.rs index a18818b1..9d936595 100644 --- a/crates/aish-config/src/lib.rs +++ b/crates/aish-config/src/lib.rs @@ -19,5 +19,5 @@ pub mod model; pub use loader::ConfigLoader; pub use model::{ compile_remote_danger_patterns, ApiAccountConfig, ConfigModel, InlineCompletionConfig, - MemoryConfig, OutputOffloadConfig, RegistrySource, SkillsConfig, ToolArgPreviewConfig, + MemoryConfig, OutputOffloadConfig, RegistrySource, SkillsConfig, TaskBudgetConfig, }; diff --git a/crates/aish-config/src/model.rs b/crates/aish-config/src/model.rs index 788120d1..ebcf4233 100644 --- a/crates/aish-config/src/model.rs +++ b/crates/aish-config/src/model.rs @@ -194,6 +194,37 @@ impl Default for ContextAutoCompactConfig { } } +// --------------------------------------------------------------------------- +// Task budget sub-config (issue #569) +// --------------------------------------------------------------------------- + +/// Task-level cumulative budget bounds. All dimensions are optional and +/// default to unlimited; the runtime still accrues counters for display. +/// A tripped budget stops the loop with `BudgetExhausted`; raising a limit +/// is an explicit, audit-logged action — never implied by "continue". +#[derive(Debug, Clone, Serialize, Deserialize, Default, PartialEq)] +#[serde(default)] +pub struct TaskBudgetConfig { + /// Cumulative tool-loop rounds for the whole task. + pub max_rounds: Option, + /// Cumulative tool executions (parallel calls each count). + pub max_tool_calls: Option, + /// Cumulative fresh tokens (input + output; cached reads excluded). + pub max_tokens: Option, + /// Cumulative active wall-clock seconds (turn time only, no idle). + pub max_duration_secs: Option, +} + +impl TaskBudgetConfig { + /// True when any dimension is bounded. + pub fn is_bounded(&self) -> bool { + self.max_rounds.is_some() + || self.max_tool_calls.is_some() + || self.max_tokens.is_some() + || self.max_duration_secs.is_some() + } +} + // --------------------------------------------------------------------------- // Inline AI completion sub-config // --------------------------------------------------------------------------- @@ -483,6 +514,11 @@ pub struct ConfigModel { #[serde(default)] pub context_token_budget: Option, + /// Task-level cumulative budget (issue #569). Every dimension is + /// optional; absent = unlimited (counters still accrue for /token). + #[serde(default)] + pub task_budget: TaskBudgetConfig, + /// Enable tiktoken-based token estimation for context trimming #[serde(default = "default_true")] pub enable_token_estimation: bool, @@ -582,6 +618,7 @@ impl Default for ConfigModel { context_token_budget: None, enable_token_estimation: default_true(), context_auto_compact: ContextAutoCompactConfig::default(), + task_budget: TaskBudgetConfig::default(), enable_scripts: default_true(), history_size: default_history_size(), terminal_resize_mode: default_terminal_resize_mode(), @@ -1018,4 +1055,27 @@ inline_completion: assert_eq!(m.inline_completion.debounce_ms, 400); assert_eq!(m.inline_completion.max_tokens, 512); } + + #[test] + fn task_budget_config_parses_from_yaml() { + let yaml = r#" +max_llm_messages: 30 +task_budget: + max_rounds: 200 + max_tokens: 2000000 +"#; + let config: ConfigModel = serde_yaml::from_str(yaml).unwrap(); + assert!(config.task_budget.is_bounded()); + assert_eq!(config.task_budget.max_rounds, Some(200)); + assert_eq!(config.task_budget.max_tokens, Some(2_000_000)); + assert_eq!(config.task_budget.max_tool_calls, None); + assert_eq!(config.task_budget.max_duration_secs, None); + } + + #[test] + fn task_budget_config_defaults_to_unbounded() { + let config = ConfigModel::default(); + assert!(!config.task_budget.is_bounded()); + assert_eq!(config.task_budget.max_rounds, None); + } } diff --git a/crates/aish-core/src/error.rs b/crates/aish-core/src/error.rs index eaceed96..367cfbf5 100644 --- a/crates/aish-core/src/error.rs +++ b/crates/aish-core/src/error.rs @@ -45,6 +45,11 @@ pub enum AishError { #[error("tool loop stopped at the iteration limit")] IterationLimit, + /// Task-level cumulative budget exhausted (issue #569). The field names + /// the dimension that tripped (rounds / tool_calls / tokens / duration). + #[error("task budget exhausted: {0}")] + BudgetExhausted(String), + #[error("operation timed out")] Timeout, } @@ -74,6 +79,7 @@ impl AishError { AishError::Shell(_) => "shell", AishError::Cancelled => "cancelled", AishError::IterationLimit => "iteration_limit", + AishError::BudgetExhausted(_) => "budget_exhausted", AishError::Parse(_) => "parse", AishError::Timeout => "timeout", } diff --git a/crates/aish-i18n/locales/de-DE.yaml b/crates/aish-i18n/locales/de-DE.yaml index 2bda0b36..f09f976f 100644 --- a/crates/aish-i18n/locales/de-DE.yaml +++ b/crates/aish-i18n/locales/de-DE.yaml @@ -332,6 +332,15 @@ shell: no_evidence: "(nicht angegeben)" nl_detection: confirm_ask_ai: "Das sieht nach einer natürlichen Sprachfrage aus. KI fragen? (Y/n)" + budget: + exhausted_title: "⛔ Aufgabenbudget erschoepft" + exhausted_detail: "Dimension: {dimension} · Genutzt: {used} · Limit: {limit}" + exhausted_options: "Budget anpassen und fortfahren? Anpassungen werden auditiert (y=anpassen / n=stoppen)" + stopped_by_user: "Durch Budget gestoppt. Abgeschlossene Werkzeugschritte sind gespeichert; fortsetzen vom gespeicherten Punkt." + new_limit_prompt: "Neues Limit fuer diese Dimension eingeben:" + raised: "Budget angepasst: {dimension} -> {value}. Aufgabe erneut ausgeben, um fortzufahren." + invalid_limit: "Ungueltiger Wert (muss ueber der Nutzung liegen); Budget unveraendert." + error: load_script_failed: "Skript '{script}' konnte nicht geladen werden: {error}" pty_error: "PTY-Fehler: {error}" @@ -343,6 +352,7 @@ shell: llm_error_message: "Fehler: {error}" partial_turn_saved: "Die abgeschlossenen Werkzeugschritte dieses Durchgangs wurden gespeichert ({steps} Schritte). Die Arbeit vor dem Fehler geht nicht verloren; geben Sie eine neue Aufgabe ein oder ;retry, um ab dem gespeicherten Punkt fortzufahren." partial_turn_saved_limit: "Die abgeschlossenen Werkzeugschritte dieses Durchgangs wurden vor dem Stopp am Iterationslimit gespeichert ({steps} Schritte). Die Arbeit bleibt erhalten; geben Sie eine neue Aufgabe ein oder ;retry, um ab dem gespeicherten Punkt fortzufahren." + partial_turn_saved_budget: "Die abgeschlossenen Werkzeugschritte dieses Durchgangs wurden vor dem Stopp am Aufgabenbudgetlimit gespeichert ({steps} Schritte). Die Arbeit bleibt erhalten; geben Sie eine neue Aufgabe ein oder ;retry, um vom gespeicherten Punkt fortzufahren." llm_error_details_title: "Debug-Details" tool_args_json_invalid_hint: "❌ Anfrage fehlgeschlagen, bitte erneut versuchen." prompt: @@ -446,6 +456,13 @@ shell: output_tokens: "Ausgabe-Tokens" total: "Gesamt" api_calls: "API-Aufrufe" + budget_title: "Aufgabenbudget" + budget_rounds: "Runden (genutzt/Limit):" + budget_tool_calls: "Werkzeugaufrufe (genutzt/Limit):" + budget_tokens: "Token (genutzt/Limit):" + budget_duration: "Aktive Sekunden (genutzt/Limit):" + budget_cost: "Kosten:" + budget_cost_unavailable: "nicht verfuegbar (Anbieter melden keine Preise)" setup: no_config_manager: "Neukonfiguration nicht möglich: Konfigurationsmanager nicht verfügbar" cancelled: "Einrichtung abgebrochen; aktuelle Einstellungen werden beibehalten" @@ -559,6 +576,9 @@ shell: inline_enforce_json: label: "Inline Enforce JSON" desc: "Inline-Completion zwingen, striktes JSON zurückzugeben" + task_budget_max_rounds: + label: "Aufgabenbudget-Rundenlimit" + desc: "Kumulierte Werkzugschleifen-Runden pro Aufgabe; leer = unbegrenzt (Weitermachen setzt nicht zurueck)" max_llm_messages: label: "Max LLM-Messages" desc: "Maximale Konversations-Turns im LLM-Verlauf" diff --git a/crates/aish-i18n/locales/en-US.yaml b/crates/aish-i18n/locales/en-US.yaml index e8c9ec46..0432eada 100644 --- a/crates/aish-i18n/locales/en-US.yaml +++ b/crates/aish-i18n/locales/en-US.yaml @@ -768,6 +768,15 @@ shell: nl_detection: confirm_ask_ai: "This looks like a natural language question. Ask AI? (Y/n)" + budget: + exhausted_title: "⛔ Task budget exhausted" + exhausted_detail: "Dimension: {dimension} · Used: {used} · Limit: {limit}" + exhausted_options: "Adjust the budget and continue? Adjustments are audit-logged (y=adjust / n=stop)" + stopped_by_user: "Stopped by budget. Completed tool evidence is saved; resume from the saved point later." + new_limit_prompt: "Enter the new limit for this dimension:" + raised: "Budget adjusted: {dimension} -> {value}. Re-issue the task to continue from the saved point." + invalid_limit: "Invalid value (must exceed usage); budget unchanged." + error: execution_error: "❌ Error during execution: {error}" pty_decode_error: "Caught PTY output decode error: {error}" @@ -779,6 +788,7 @@ shell: llm_error_message: "Error: {error}" partial_turn_saved: "Saved the completed tool steps of this turn ({steps} steps). Work done before the failure is not lost; enter a new task or ;retry to continue from the saved point." partial_turn_saved_limit: "Saved the completed tool steps of this turn ({steps} steps) before stopping at the iteration limit. The work is kept; enter a new task or ;retry to continue from the saved point." + partial_turn_saved_budget: "Saved the completed tool steps of this turn ({steps} steps) before stopping at the task budget limit. The work is kept; enter a new task or ;retry to continue from the saved point." llm_error_details_title: "Debug Details" tool_args_json_invalid_hint: "❌ Request failed, please retry." @@ -888,6 +898,13 @@ shell: output_tokens: "Output tokens" total: "Total" api_calls: "API calls" + budget_title: "Task budget" + budget_rounds: "Rounds (used/limit):" + budget_tool_calls: "Tool calls (used/limit):" + budget_tokens: "Tokens (used/limit):" + budget_duration: "Active seconds (used/limit):" + budget_cost: "Cost:" + budget_cost_unavailable: "unavailable (providers do not report pricing)" setup: no_config_manager: "Cannot reconfigure: configuration manager unavailable" @@ -983,6 +1000,9 @@ shell: inline_completion: label: "Inline AI Completion" desc: "Gray ghost-text AI suggestions while typing an AI prompt" + task_budget_max_rounds: + label: "Task Budget Round Limit" + desc: "Cumulative tool-loop rounds per task; blank = unlimited (continues never reset it)" max_llm_messages: label: "Max AI Messages" desc: "How many AI conversation turns to keep in context" diff --git a/crates/aish-i18n/locales/es-ES.yaml b/crates/aish-i18n/locales/es-ES.yaml index 0c1df33a..1d9cabec 100644 --- a/crates/aish-i18n/locales/es-ES.yaml +++ b/crates/aish-i18n/locales/es-ES.yaml @@ -332,6 +332,15 @@ shell: no_evidence: "(no proporcionada)" nl_detection: confirm_ask_ai: "Esto parece una pregunta en lenguaje natural. ¿Preguntar a IA? (Y/n)" + budget: + exhausted_title: "⛔ Presupuesto de tarea agotado" + exhausted_detail: "Dimensión: {dimension} · Usado: {used} · Límite: {limit}" + exhausted_options: "¿Ajustar el presupuesto y continuar? Los ajustes se registran en la auditoría (y=ajustar / n=detener)" + stopped_by_user: "Detenido por presupuesto. La evidencia de herramientas completadas está guardada; reanude desde el punto guardado." + new_limit_prompt: "Introduzca el nuevo límite para esta dimensión:" + raised: "Presupuesto ajustado: {dimension} -> {value}. Emita la tarea de nuevo para continuar." + invalid_limit: "Valor inválido (debe superar el uso); presupuesto sin cambios." + error: load_script_failed: "No se pudo cargar el script '{script}': {error}" pty_error: "Error de PTY: {error}" @@ -343,6 +352,7 @@ shell: llm_error_message: "Error: {error}" partial_turn_saved: "Se guardaron los pasos de herramienta completados de este turno ({steps} pasos). El trabajo previo al error no se pierde; escriba una nueva tarea o ;retry para continuar desde el punto guardado." partial_turn_saved_limit: "Se guardaron los pasos de herramienta completados de este turno ({steps} pasos) antes de detenerse en el límite de iteraciones. El trabajo se conserva; escriba una nueva tarea o ;retry para continuar desde el punto guardado." + partial_turn_saved_budget: "Se guardaron los pasos de herramientas completados de este turno ({steps} pasos) antes de detenerse en el límite de presupuesto de la tarea. El trabajo se conserva; escriba una nueva tarea o ;retry para continuar desde el punto guardado." llm_error_details_title: "Detalles de depuración" tool_args_json_invalid_hint: "❌ La solicitud falló; inténtalo de nuevo." prompt: @@ -446,6 +456,13 @@ shell: output_tokens: "Tokens de salida" total: "Total" api_calls: "Llamadas API" + budget_title: "Presupuesto de tarea" + budget_rounds: "Rondas (usado/límite):" + budget_tool_calls: "Llamadas de herramientas (usado/límite):" + budget_tokens: "Tokens (usado/límite):" + budget_duration: "Segundos activos (usado/límite):" + budget_cost: "Costo:" + budget_cost_unavailable: "no disponible (los proveedores no informan precios)" setup: no_config_manager: "No se puede reconfigurar: el gestor de configuración no está disponible" cancelled: "Configuración cancelada; se conservan los ajustes actuales" @@ -559,6 +576,9 @@ shell: inline_enforce_json: label: "Forzar JSON en línea" desc: "Forzar que el completado en línea devuelva JSON estricto" + task_budget_max_rounds: + label: "Límite de rondas del presupuesto" + desc: "Rondas acumuladas del bucle de herramientas por tarea; vacío = ilimitado (continuar nunca lo restablece)" max_llm_messages: label: "Mensajes LLM máx." desc: "Máximo de turnos de conversación conservados en el historial del LLM" diff --git a/crates/aish-i18n/locales/fr-FR.yaml b/crates/aish-i18n/locales/fr-FR.yaml index 1344c93a..13c0c440 100644 --- a/crates/aish-i18n/locales/fr-FR.yaml +++ b/crates/aish-i18n/locales/fr-FR.yaml @@ -332,6 +332,15 @@ shell: no_evidence: "(non fournie)" nl_detection: confirm_ask_ai: "Cela ressemble à une question en langage naturel. Demander à l'IA ? (Y/n)" + budget: + exhausted_title: "⛔ Budget de tache epuise" + exhausted_detail: "Dimension : {dimension} · Utilise : {used} · Limite : {limit}" + exhausted_options: "Ajuster le budget et continuer ? Les ajustements sont journalises (y=ajuster / n=arreter)" + stopped_by_user: "Arrete par budget. Les preuves d'outils completes sont sauvegardees ; reprenez depuis le point sauvegarde." + new_limit_prompt: "Saisissez la nouvelle limite pour cette dimension :" + raised: "Budget ajuste : {dimension} -> {value}. Relancez la tache pour continuer." + invalid_limit: "Valeur invalide (doit depasser l'utilisation) ; budget inchange." + error: load_script_failed: "Échec du chargement du script '{script}' : {error}" pty_error: "Erreur PTY : {error}" @@ -343,6 +352,7 @@ shell: llm_error_message: "Erreur : {error}" partial_turn_saved: "Les etapes d'outils completees de ce tour ont ete sauvegardees ({steps} etapes). Le travail precedent l'erreur n'est pas perdu ; saisissez une nouvelle tache ou ;retry pour reprendre depuis le point sauvegarde." partial_turn_saved_limit: "Les etapes d'outils completees de ce tour ont ete sauvegardees ({steps} etapes) avant l'arret a la limite d'iterations. Le travail est conserve ; saisissez une nouvelle tache ou ;retry pour reprendre depuis le point sauvegarde." + partial_turn_saved_budget: "Les etapes d'outils completees de ce tour ont ete sauvegardees ({steps} etapes) avant l'arret a la limite de budget de la tache. Le travail est conserve ; saisissez une nouvelle tache ou ;retry pour continuer depuis le point sauvegarde." llm_error_details_title: "Details de debogage" tool_args_json_invalid_hint: "❌ La requete a echoue, veuillez reessayer." prompt: @@ -446,6 +456,13 @@ shell: output_tokens: "Tokens de sortie" total: "Total" api_calls: "Appels API" + budget_title: "Budget de tache" + budget_rounds: "Tours (utilise/limite):" + budget_tool_calls: "Appels d'outils (utilise/limite):" + budget_tokens: "Tokens (utilise/limite):" + budget_duration: "Secondes actives (utilise/limite):" + budget_cost: "Cout:" + budget_cost_unavailable: "indisponible (les fournisseurs ne rapportent pas de prix)" setup: no_config_manager: "Reconfiguration impossible : gestionnaire de configuration indisponible" cancelled: "Configuration annulee, les parametres actuels sont conserves" @@ -559,6 +576,9 @@ shell: inline_enforce_json: label: "Forcer JSON en ligne" desc: "Forcer la complétion en ligne à renvoyer du JSON strict" + task_budget_max_rounds: + label: "Limite de tours du budget" + desc: "Tours cumules de la boucle d'outils par tache ; vide = illimite (continuer ne reinitialise jamais)" max_llm_messages: label: "Messages LLM max" desc: "Tours de conversation maximum conservés dans l'historique LLM" diff --git a/crates/aish-i18n/locales/ja-JP.yaml b/crates/aish-i18n/locales/ja-JP.yaml index a302ac10..f7caf130 100644 --- a/crates/aish-i18n/locales/ja-JP.yaml +++ b/crates/aish-i18n/locales/ja-JP.yaml @@ -332,6 +332,15 @@ shell: no_evidence: "(未提供)" nl_detection: confirm_ask_ai: "自然言語の質問のようです。AIに聞きますか?(Y/n)" + budget: + exhausted_title: "⛔ タスク予算が尽きました" + exhausted_detail: "次元: {dimension} · 使用: {used} · 上限: {limit}" + exhausted_options: "予算を調整して続行しますか?調整は監査ログに記録されます(y=調整 / n=停止)" + stopped_by_user: "予算により停止。完了したツール実行記録は保存されています。保存ポイントから再開できます。" + new_limit_prompt: "この次元の新しい上限を入力してください:" + raised: "予算を調整しました: {dimension} -> {value}。タスクを再度実行して続行できます。" + invalid_limit: "無効な値(使用量を超える必要)。予算は変更されません。" + error: load_script_failed: "スクリプト '{script}' の読み込みに失敗: {error}" pty_error: "PTY エラー: {error}" @@ -343,6 +352,7 @@ shell: llm_error_message: "エラー: {error}" partial_turn_saved: "このターンで完了したツール実行ステップを保存しました({steps} ステップ)。失敗前の作業は失われません。新しいタスクを入力するか ;retry で保存ポイントから続行できます。" partial_turn_saved_limit: "イテレーション上限で停止する前に、このターンで完了したツール実行ステップを保存しました({steps} ステップ)。完了した作業は失われません。新しいタスクを入力するか ;retry で保存ポイントから続行できます。" + partial_turn_saved_budget: "タスク予算上限で停止する前に、このターンで完了したツール実行ステップを保存しました({steps} ステップ)。完了した作業は失われません。新しいタスクを入力するか ;retry で保存ポイントから続行できます。" llm_error_details_title: "デバッグ詳細" tool_args_json_invalid_hint: "❌ リクエストに失敗しました。再試行してください。" prompt: @@ -446,6 +456,13 @@ shell: output_tokens: "出力 tokens" total: "合計" api_calls: "API 呼び出し回数" + budget_title: "タスク予算" + budget_rounds: "ラウンド(使用/上限):" + budget_tool_calls: "ツール呼び出し(使用/上限):" + budget_tokens: "Token(使用/上限):" + budget_duration: "アクティブ秒数(使用/上限):" + budget_cost: "費用:" + budget_cost_unavailable: "利用不可(プロバイダは価格を報告しません)" setup: no_config_manager: "再設定できません: 設定マネージャーを利用できません" cancelled: "セットアップをキャンセルしました。現在の設定を維持します" @@ -559,6 +576,9 @@ shell: inline_enforce_json: label: "インライン JSON 強制" desc: "インライン補完で厳密な JSON を返すよう強制" + task_budget_max_rounds: + label: "タスク予算ラウンド上限" + desc: "タスクごとの累積ツールループ上限。空欄 = 無制限(続行ではリセットされません)" max_llm_messages: label: "最大 LLM メッセージ数" desc: "LLM 履歴に保持する最大会話ターン数" diff --git a/crates/aish-i18n/locales/zh-CN.yaml b/crates/aish-i18n/locales/zh-CN.yaml index c2f2341d..bd60a5b9 100644 --- a/crates/aish-i18n/locales/zh-CN.yaml +++ b/crates/aish-i18n/locales/zh-CN.yaml @@ -768,6 +768,15 @@ shell: nl_detection: confirm_ask_ai: "这看起来像自然语言问题,是否让 AI 解答?(Y/n)" + budget: + exhausted_title: "⛔ 任务预算已耗尽" + exhausted_detail: "维度: {dimension} · 已用: {used} · 上限: {limit}" + exhausted_options: "是否调整预算并继续? 调整将被记录到审计日志 (y=调整 / n=停止)" + stopped_by_user: "已按预算停止。已完成的工具执行记录已保存,可稍后从保存点继续。" + new_limit_prompt: "请输入该维度的新上限数值:" + raised: "预算已调整: {dimension} -> {value}。重新发起任务以从保存点继续。" + invalid_limit: "无效数值(必须大于已用量),预算保持不变。" + error: execution_error: "❌ 执行过程中出错: {error}" pty_decode_error: "捕获 pty 输出解码错误: {error}" @@ -776,6 +785,7 @@ shell: llm_error_message: "错误: {error}" partial_turn_saved: "已保存本轮已完成的工具执行记录({steps} 步),失败前的工作不会丢失;输入新任务或 ;重试 可从保存点继续。" partial_turn_saved_limit: "已在达到轮次上限停止前保存本轮已完成的工具执行记录({steps} 步),已完成的工作不会丢失;输入新任务或 ;重试 可从保存点继续。" + partial_turn_saved_budget: "已在达到任务预算上限停止前保存本轮已完成的工具执行记录({steps} 步),已完成的工作不会丢失;输入新任务或 ;重试 可从保存点继续。" llm_error_details_title: "调试详情" tool_args_json_invalid_hint: "❌ 请求失败,请重试。" pty_error: "PTY 错误:{error}" @@ -888,6 +898,13 @@ shell: output_tokens: "输出 tokens" total: "总计" api_calls: "API 调用次数" + budget_title: "任务预算" + budget_rounds: "轮次(已用/上限):" + budget_tool_calls: "工具调用(已用/上限):" + budget_tokens: "Token(已用/上限):" + budget_duration: "活跃时长秒(已用/上限):" + budget_cost: "费用:" + budget_cost_unavailable: "不可用(提供商未回报价格数据)" setup: no_config_manager: "无法重新配置:配置管理器不可用" @@ -983,6 +1000,9 @@ shell: inline_completion: label: "行内 AI 补全" desc: "输入 AI 提示时显示灰色幽灵文本建议" + task_budget_max_rounds: + label: "任务预算轮次上限" + desc: "单个任务累计工具循环轮次上限,留空为不限(继续不重置)" max_llm_messages: label: "AI 最大消息数" desc: "上下文中保留的 AI 对话轮数" diff --git a/crates/aish-llm/src/agents/spawn.rs b/crates/aish-llm/src/agents/spawn.rs index 7e69b2f6..3d863aa8 100644 --- a/crates/aish-llm/src/agents/spawn.rs +++ b/crates/aish-llm/src/agents/spawn.rs @@ -119,6 +119,12 @@ where // billed exactly like main-session turns). Totals only — see // TokenStats::merge_totals for why last_prompt_tokens is skipped. parent.merge_token_stats(&sub.token_stats()); + // Issue #569 acceptance 3: fold the sub-agent's budget counters + // (rounds, tool calls, wall-clock seconds, tokens) into the parent + // task exactly once on completion — no double counting on retries, + // because the merge runs in the single completion path here. + sub.flush_task_budget_wall_clock(); + parent.merge_task_budget(&sub.task_budget_state()); SpawnResult { text: outcome.text, @@ -385,6 +391,41 @@ mod tests { .any(|s| s.function.name == "read_file")); } + #[tokio::test] + async fn test_spawn_folds_subsession_budget_into_parent_task() { + // Issue #569 acceptance 3: sub-agent rounds/tool-calls/tokens fold + // into the parent's task-level counters exactly once. + let mut parent = LlmSession::new("http://localhost", "key", "model", None, None); + parent.register_tool(Box::new(MockTool::new("grep"))); + let rounds_before = parent.task_budget_state().task_rounds; + let calls_before = parent.task_budget_state().tool_calls; + + let registry = AgentRegistry::builtin(); + let _ = spawn_builtin(&parent, ®istry, "explore", "task", |sub, _specs| { + configure_spawn_test( + sub, + vec![ + Ok(mock_tool_call_response_with_usage( + &[("c1", "grep", "{}")], + 500, + 40, + )), + Ok(mock_text_response_with_usage("done", 700, 60)), + ], + ); + }) + .await + .expect("spawn_builtin should succeed"); + + let budget = parent.task_budget_state(); + // The sub-agent ran 2 loop rounds and 1 tool call; the parent's own + // loop did not run, so the totals are exactly the sub's. + assert_eq!(budget.task_rounds, rounds_before + 2); + assert_eq!(budget.tool_calls, calls_before + 1); + assert_eq!(budget.token_usage.total_input, 1200); + assert_eq!(budget.token_usage.total_output, 100); + } + #[tokio::test] async fn test_spawn_merges_subsession_usage_into_parent() { // The sub-agent's billed usage must land in the parent's cumulative diff --git a/crates/aish-llm/src/agents/tool_loop.rs b/crates/aish-llm/src/agents/tool_loop.rs index 9f1487cf..2d3b4422 100644 --- a/crates/aish-llm/src/agents/tool_loop.rs +++ b/crates/aish-llm/src/agents/tool_loop.rs @@ -127,6 +127,14 @@ pub async fn run_tool_loop_until_done( return LoopOutcome::from_spawn_outcome(outcome, loop_messages, None); } iterations += 1; + // Task budget accounting (issue #569): sub-agent loops accrue the + // same counters as the main loop so a spawn folds real consumption + // into the parent task. + { + let mut state = session.task_budget_state_mut(); + state.start_turn(); + state.task_rounds += 1; + } messages = session.prepare_messages_for_send(messages).await; diff --git a/crates/aish-llm/src/budget.rs b/crates/aish-llm/src/budget.rs new file mode 100644 index 00000000..07852b4a --- /dev/null +++ b/crates/aish-llm/src/budget.rs @@ -0,0 +1,297 @@ +use std::time::Instant; + +use crate::usage::TokenStats; + +/// Task-level cumulative budget counters (issue #569). +/// +/// Unlike the per-segment iteration counter (reset when the user chooses to +/// continue), these counters accumulate for the lifetime of the task across +/// continues, compactions, and sub-agent merges. They are never reset by +/// "continue" — only persisted state or an explicit budget adjustment can +/// change their trajectory. +#[derive(Debug, Default, Clone)] +pub struct TaskBudgetState { + /// Cumulative tool-loop rounds across the whole task (continues included). + pub task_rounds: u64, + /// Cumulative tool executions across the whole task (parallel calls + /// within one round each count). + pub tool_calls: u64, + /// Active wall-clock seconds accumulated while a turn was in flight. + /// Advanced incrementally (`last_accounted += elapsed`) so continues, + /// pauses, and resumes never double-count. Idle time is not counted. + pub active_secs: u64, + /// Monotonic anchor for wall-clock accounting: the instant from which + /// active time is being accumulated. `None` while no turn is running. + pub last_accounted: Option, + /// Cumulative token usage. Cached-prefix reads are excluded from budget + /// math via `TokenStats::total_tokens()`. + pub token_usage: TokenStats, +} + +impl TaskBudgetState { + /// Begin (or resume) wall-clock accounting. + pub fn start_turn(&mut self) { + self.last_accounted = Some(Instant::now()); + } + + /// Flush pending wall-clock time into `active_secs` using the advancing + /// anchor pattern: the anchor moves forward by exactly the flushed amount, + /// so a later flush never double-counts and a pause/continue never resets + /// the baseline. Whole seconds only — sub-second remainder stays pending. + pub fn flush_wall_clock(&mut self) { + if let Some(anchor) = self.last_accounted { + let elapsed = anchor.elapsed().as_secs(); + if elapsed > 0 { + self.active_secs += elapsed; + self.last_accounted = Some(anchor + std::time::Duration::from_secs(elapsed)); + } + } + } + + /// Stop wall-clock accounting (turn ended). Flushes any pending whole + /// seconds first. + pub fn end_turn(&mut self) { + self.flush_wall_clock(); + self.last_accounted = None; + } + + /// Merge a sub-agent's cumulative counters into this parent task. + /// Each sub-agent reports exactly once on completion, so totals do not + /// double-count. Wall-clock anchors are not merged — they are process + /// local. + pub fn merge_task(&mut self, other: &TaskBudgetState) { + self.task_rounds += other.task_rounds; + self.tool_calls += other.tool_calls; + self.active_secs += other.active_secs; + self.token_usage.merge_totals(&other.token_usage); + } +} + +/// Near-threshold advisory state (issue #569 acceptance 3 / omp +/// `budgetReportedFor` pattern): the "you are approaching the budget" hint +/// fires once per threshold cycle and is re-armed by an explicit budget +/// adjustment. Tracked on the session, not the persisted counters. +#[derive(Debug, Default, Clone)] +pub struct NearThresholdState { + /// True once the hint has been shown since the last re-arm. + pub reported: bool, +} + +impl NearThresholdState { + /// True when usage crossed `NEAR_THRESHOLD_PERCENT` and the hint has + /// not been shown yet in this cycle. Idempotent: subsequent calls with + /// usage still above the threshold return false. + pub fn should_report(&mut self, check: &BudgetCheck) -> bool { + let crossed = check + .nearest_percent + .is_some_and(|pct| pct >= NEAR_THRESHOLD_PERCENT); + if crossed && !self.reported { + self.reported = true; + return true; + } + if !crossed { + // Usage dropped below the threshold (e.g. a limit raise): + // re-arm so a future crossing reports again. + self.reported = false; + } + false + } + + /// Explicit re-arm (budget adjustment path). + pub fn rearm(&mut self) { + self.reported = false; + } +} + +/// Configurable upper bounds for one task. `None` = unlimited for that +/// dimension. Limits come from `config.yaml` (`llm.task_budget`) and can only +/// be changed through the explicit budget-adjustment flow (never by "continue"). +#[derive(Debug, Clone, Default)] +pub struct TaskBudgetLimit { + pub max_rounds: Option, + pub max_tool_calls: Option, + pub max_tokens: Option, + pub max_duration_secs: Option, +} + +impl TaskBudgetLimit { + /// True when any dimension is bounded. + pub fn is_bounded(&self) -> bool { + self.max_rounds.is_some() + || self.max_tool_calls.is_some() + || self.max_tokens.is_some() + || self.max_duration_secs.is_some() + } +} + +/// Which dimension exhausted, in check order. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BudgetDimension { + Rounds, + ToolCalls, + Tokens, + Duration, +} + +impl BudgetDimension { + pub fn key(self) -> &'static str { + match self { + BudgetDimension::Rounds => "rounds", + BudgetDimension::ToolCalls => "tool_calls", + BudgetDimension::Tokens => "tokens", + BudgetDimension::Duration => "duration", + } + } +} + +/// Result of a budget check: which dimension is over (if any) and how close +/// the task is to the nearest threshold (for near-threshold UI, 0-100). +pub struct BudgetCheck { + pub exhausted: Option, + /// Highest usage/threshold ratio across bounded dimensions, scaled to + /// percent. `None` when nothing is bounded. + pub nearest_percent: Option, +} + +pub const NEAR_THRESHOLD_PERCENT: u64 = 80; + +impl TaskBudgetLimit { + /// Evaluate the current state against all bounds. Pure computation — + /// callers decide what to do with exhaustion (stop the loop) or + /// near-threshold pressure (checkpoint + UI hint). + pub fn check(&self, state: &TaskBudgetState) -> BudgetCheck { + // First exhausted dimension in a stable check order. + let exhausted_dim = [ + (state.task_rounds, self.max_rounds, BudgetDimension::Rounds), + ( + state.tool_calls, + self.max_tool_calls, + BudgetDimension::ToolCalls, + ), + ( + state.token_usage.total_tokens(), + self.max_tokens, + BudgetDimension::Tokens, + ), + ( + state.active_secs, + self.max_duration_secs, + BudgetDimension::Duration, + ), + ] + .into_iter() + .find(|(used, limit, _)| limit.is_some_and(|max| *used >= max)) + .map(|(_, _, dim)| dim); + + // Highest usage/threshold ratio across bounded dimensions. + let mut nearest: Option = None; + for (used, limit) in [ + (state.task_rounds, self.max_rounds), + (state.tool_calls, self.max_tool_calls), + (state.token_usage.total_tokens(), self.max_tokens), + (state.active_secs, self.max_duration_secs), + ] { + if let Some(max) = limit { + if let Some(pct) = used.saturating_mul(100).checked_div(max) { + nearest = Some(nearest.map_or(pct, |p: u64| p.max(pct))); + } + } + } + + BudgetCheck { + exhausted: exhausted_dim, + nearest_percent: nearest, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn advancing_wall_clock_never_double_counts() { + let mut state = TaskBudgetState::default(); + state.start_turn(); + state.flush_wall_clock(); + let first = state.active_secs; + state.flush_wall_clock(); + let second = state.active_secs; + assert!(second >= first, "flush must advance the counter"); + state.end_turn(); + assert!(state.last_accounted.is_none()); + } + + #[test] + fn exhaustion_is_detected_per_dimension_in_order() { + let mut state = TaskBudgetState::default(); + state.task_rounds = 10; + state.tool_calls = 50; + let limit = TaskBudgetLimit { + max_rounds: Some(10), + max_tool_calls: Some(50), + max_tokens: None, + max_duration_secs: None, + }; + assert_eq!(limit.check(&state).exhausted, Some(BudgetDimension::Rounds)); + + state.task_rounds = 9; + assert_eq!( + limit.check(&state).exhausted, + Some(BudgetDimension::ToolCalls) + ); + } + + #[test] + fn near_threshold_percent_is_the_max_ratio() { + let mut state = TaskBudgetState::default(); + state.task_rounds = 10; + state.tool_calls = 45; + let limit = TaskBudgetLimit { + max_rounds: Some(100), + max_tool_calls: Some(50), + max_tokens: None, + max_duration_secs: None, + }; + let check = limit.check(&state); + assert_eq!(check.nearest_percent, Some(90)); + assert!(check.exhausted.is_none()); + } + + #[test] + fn near_threshold_reports_once_then_rearms() { + let mut nt = NearThresholdState::default(); + let state = TaskBudgetState { + task_rounds: 85, + ..Default::default() + }; + let limit = TaskBudgetLimit { + max_rounds: Some(100), + ..Default::default() + }; + let check = limit.check(&state); + assert!(nt.should_report(&check), "first crossing reports"); + assert!( + !nt.should_report(&check), + "second call in the same cycle is deduplicated" + ); + nt.rearm(); + assert!(nt.should_report(&check), "rearm allows reporting again"); + } + + #[test] + fn near_threshold_ignores_unbounded_dimensions() { + let mut nt = NearThresholdState::default(); + let check = TaskBudgetLimit::default().check(&TaskBudgetState::default()); + assert!(!nt.should_report(&check)); + } + + #[test] + fn unbounded_limit_never_exhausts() { + let mut state = TaskBudgetState::default(); + state.task_rounds = u64::MAX; + let check = TaskBudgetLimit::default().check(&state); + assert!(check.exhausted.is_none()); + assert!(check.nearest_percent.is_none()); + } +} diff --git a/crates/aish-llm/src/lib.rs b/crates/aish-llm/src/lib.rs index 354fe797..39f7ad3d 100644 --- a/crates/aish-llm/src/lib.rs +++ b/crates/aish-llm/src/lib.rs @@ -16,6 +16,7 @@ pub mod agents; pub mod api; pub mod approval_memory; +pub mod budget; pub mod client; pub mod langfuse; pub mod llm_stream; diff --git a/crates/aish-llm/src/session.rs b/crates/aish-llm/src/session.rs index 0164c197..2ea0d244 100644 --- a/crates/aish-llm/src/session.rs +++ b/crates/aish-llm/src/session.rs @@ -110,7 +110,19 @@ pub struct LlmSession { /// `take_last_partial_turn` so the shell layer can persist the /// evidence. Empty unless the previous turn failed mid-loop. last_partial_turn: parking_lot::Mutex>, - /// Scripted chat completion responses for unit/integration tests (pop in order). + /// Task-level cumulative budget counters (issue #569). Accumulates + /// across continues, compactions, and sub-agent merges; never reset by + /// "continue". Guarded by a mutex because `execute_tool` (called from + /// async contexts) updates `tool_calls` mid-loop. + task_budget_state: parking_lot::Mutex, + /// Upper bounds for the task budget. `None`-dimension = unlimited. + task_budget_limit: parking_lot::Mutex, + /// Set when the loop stopped because a budget dimension was exhausted; + /// holds the dimension key for the shell's stop-note wording. + budget_exhausted_dimension: parking_lot::Mutex>, + /// Near-threshold advisory state: fires the "approaching budget" hint + /// once per cycle; re-armed by an explicit budget adjustment. + near_threshold_state: parking_lot::Mutex, #[cfg(test)] test_chat_responses: Option>>>>, } @@ -171,11 +183,67 @@ impl LlmSession { is_sub_agent: false, rotation: None, last_partial_turn: parking_lot::Mutex::new(Vec::new()), + task_budget_state: parking_lot::Mutex::new(crate::budget::TaskBudgetState::default()), + task_budget_limit: parking_lot::Mutex::new(crate::budget::TaskBudgetLimit::default()), + budget_exhausted_dimension: parking_lot::Mutex::new(None), + near_threshold_state: parking_lot::Mutex::new( + crate::budget::NearThresholdState::default(), + ), #[cfg(test)] test_chat_responses: None, } } + /// Set the task budget bounds (from `config.yaml llm.task_budget`). + pub fn set_task_budget_limit(&self, limit: crate::budget::TaskBudgetLimit) { + *self.task_budget_limit.lock() = limit; + } + + /// Fold a completed sub-agent's cumulative budget counters into this + /// session's task (issue #569 acceptance 3). Called once per sub-agent + /// completion; wall-clock anchors stay process-local and unmerged. + pub fn merge_task_budget(&self, other: &crate::budget::TaskBudgetState) { + self.task_budget_state.lock().merge_task(other); + } + + /// Snapshot the configured task budget bounds. + pub fn task_budget_limit(&self) -> crate::budget::TaskBudgetLimit { + self.task_budget_limit.lock().clone() + } + + /// Snapshot the cumulative task budget counters. + pub fn task_budget_state(&self) -> crate::budget::TaskBudgetState { + self.task_budget_state.lock().clone() + } + + /// Flush pending wall-clock seconds into the budget counters. Public + /// so the spawn path can settle a sub-session before folding it into + /// the parent task. + pub fn flush_task_budget_wall_clock(&self) { + self.task_budget_state.lock().end_turn(); + } + + /// Internal mutable access for loop accounting (tool_loop / spawn). + pub(crate) fn task_budget_state_mut( + &self, + ) -> parking_lot::MutexGuard<'_, crate::budget::TaskBudgetState> { + self.task_budget_state.lock() + } + + /// Seed cumulative counters on resume (issue #569 acceptance 5: budget + /// state must survive session restore so it cannot be bypassed by + /// restarting the process). Wall-clock anchor is never seeded — it is + /// process-local and starts with the next turn. + pub fn seed_task_budget_state(&self, mut state: crate::budget::TaskBudgetState) { + state.end_turn(); // drop any cross-process Instant anchor + *self.task_budget_state.lock() = state; + } + + /// Take the budget dimension that stopped the last turn, if any. + pub fn take_budget_exhausted_dimension(&self) -> Option { + self.budget_exhausted_dimension.lock().take() + } + /// Drain the tool-loop messages completed before the last mid-turn /// failure. Returns an empty vec when the previous turn succeeded (or /// failed before any tool ran). The caller is expected to persist these @@ -441,6 +509,12 @@ impl LlmSession { } fn record_usage(&self, usage: crate::usage::TokenUsage) { + // Mirror into the task budget counters first (issue #569): budget + // math must see the same usage the session footer reports. + self.task_budget_state + .lock() + .token_usage + .record(usage.clone()); self.token_stats.lock().unwrap().record(usage); } @@ -772,6 +846,56 @@ impl LlmSession { return Err(AishError::IterationLimit); } } + + // Task-level cumulative budget check (issue #569): evaluated at + // the loop top, BEFORE any model request, so an exhausted budget + // never fires another call. The previous round's tools have all + // completed by now (rounds are serial), so the partial-turn + // evidence below is complete and consistent. + { + let (limit, state_snapshot) = { + let state = self.task_budget_state.lock(); + (self.task_budget_limit.lock().clone(), state.clone()) + }; + let check = limit.check(&state_snapshot); + if let Some(dim) = check.exhausted { + // Preserve the same evidence contract as the iteration + // limit / API error paths (#572/#452): user message + // prefix + the post-initial loop messages, call/result + // pairing intact. + if messages.len() > initial_len { + let mut partial: Vec = vec![user_msg.clone()]; + partial.extend(messages[initial_len..].iter().cloned()); + *self.last_partial_turn.lock() = partial; + } + let dim_key = dim.key().to_string(); + *self.budget_exhausted_dimension.lock() = Some(dim_key.clone()); + self.emit_event(LlmEvent { + event_type: LlmEventType::Error, + data: serde_json::json!({ + "error": "task budget exhausted", + "dimension": dim_key, + }), + timestamp: now_timestamp(), + metadata: None, + }); + self.emit_event(LlmEvent { + event_type: LlmEventType::OpEnd, + data: serde_json::json!({"reason": "budget_exhausted"}), + timestamp: now_timestamp(), + metadata: None, + }); + return Err(AishError::BudgetExhausted(dim_key)); + } + } + + // Start wall-clock accounting for this round, count the round, + // then proceed with the model request. + { + let mut state = self.task_budget_state.lock(); + state.start_turn(); + state.task_rounds += 1; + } iterations += 1; messages = self.prepare_messages_for_send(messages).await; @@ -790,6 +914,7 @@ impl LlmSession { event_type: LlmEventType::GenerationStart, data: serde_json::json!({ "iteration": iterations, + "task_rounds": self.task_budget_state().task_rounds, "has_tools": has_tools, }), timestamp: now_timestamp(), @@ -947,6 +1072,8 @@ impl LlmSession { // Accumulate token usage from SSE chunks let mut stream_prompt_tokens: u64 = 0; let mut stream_completion_tokens: u64 = 0; + let mut stream_cache_read_tokens: u64 = 0; + let mut stream_cache_write_tokens: u64 = 0; while !stream_done { if self.cancellation_token.is_cancelled() { @@ -968,6 +1095,8 @@ impl LlmSession { if let Some(u) = chunk_usage { stream_prompt_tokens = u.prompt_tokens; stream_completion_tokens = u.completion_tokens; + stream_cache_read_tokens = u.cache_read_tokens; + stream_cache_write_tokens = u.cache_write_tokens; } // Content is emitted before ToolCallDelta in // the same parse batch, so look ahead: a @@ -1106,6 +1235,8 @@ impl LlmSession { self.record_usage(crate::usage::TokenUsage { prompt_tokens: stream_prompt_tokens, completion_tokens: stream_completion_tokens, + cache_read_tokens: stream_cache_read_tokens, + cache_write_tokens: stream_cache_write_tokens, }); } @@ -1260,10 +1391,17 @@ impl LlmSession { ) .await { + // Round finished — flush pending whole seconds of + // active wall-clock into the task budget (#569). + let mut state = self.task_budget_state.lock(); + state.end_turn(); + let check = self.task_budget_limit.lock().check(&state); + if self.near_threshold_state.lock().should_report(&check) { + self.emit_budget_near_threshold(&check); + } return Ok(pr); } - // Smart-trim old tool outputs to prevent unbounded growth smart_trim_tool_loop(&mut messages, initial_len); } } @@ -1313,6 +1451,9 @@ impl LlmSession { /// - Retries once on execution failure (matching Python's robustness). /// - Emits structured TOOL_EXECUTION_START / TOOL_EXECUTION_END events. async fn execute_tool(&self, tool_call: &ToolCall) -> ToolResult { + // Count every tool execution against the task budget (issue #569): + // parallel calls within one round each count. + self.task_budget_state.lock().tool_calls += 1; let args: serde_json::Value = serde_json::from_str(&tool_call.arguments).unwrap_or(serde_json::Value::Null); @@ -1684,6 +1825,12 @@ impl LlmSession { context_budget_policy: self.context_budget_policy.clone(), rotation: None, last_partial_turn: parking_lot::Mutex::new(Vec::new()), + task_budget_state: parking_lot::Mutex::new(crate::budget::TaskBudgetState::default()), + task_budget_limit: parking_lot::Mutex::new(crate::budget::TaskBudgetLimit::default()), + budget_exhausted_dimension: parking_lot::Mutex::new(None), + near_threshold_state: parking_lot::Mutex::new( + crate::budget::NearThresholdState::default(), + ), turn_seq: std::sync::atomic::AtomicU32::new(0), plan_state: Arc::new(Mutex::new(PlanModeState::default())), token_stats: std::sync::Mutex::new(crate::usage::TokenStats::default()), @@ -1858,6 +2005,27 @@ impl LlmSession { }); } + /// Issue #569: emit a one-shot advisory when cumulative usage crosses + /// the near-threshold percentage of any bounded dimension. Re-armed by + /// an explicit budget adjustment (`rearm_near_threshold`). + fn emit_budget_near_threshold(&self, check: &crate::budget::BudgetCheck) { + self.emit_event(LlmEvent { + event_type: LlmEventType::Error, + data: serde_json::json!({ + "error": "task budget near threshold", + "nearest_percent": check.nearest_percent, + "warning": true, + }), + timestamp: now_timestamp(), + metadata: None, + }); + } + + /// Re-arm the near-threshold advisory after an explicit budget raise. + pub fn rearm_near_threshold(&self) { + self.near_threshold_state.lock().rearm(); + } + pub fn emit_context_compaction_start(&self, scope: &str, mode: &str) { self.emit_event(LlmEvent { event_type: LlmEventType::ContextCompactionStart, @@ -4333,4 +4501,176 @@ mod tests { assert_eq!(called_ids, ids, "all completed call ids in order"); assert_eq!(called_ids, answered_ids, "no dangling tool_call_id"); } + + /// Issue #569: a token-only task budget stops the loop BEFORE the model + /// request that would exceed it, preserves the #572 evidence contract, + /// and reports the exhausted dimension. + #[tokio::test] + async fn task_budget_token_exhaustion_stops_loop_and_saves_evidence() { + use crate::agents::mock_tool_call_response_with_usage; + + let log = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); + // Three scripted rounds; each round burns 60 prompt + 40 output + // tokens. The 150-token budget must stop before round 3's request. + let responses: Vec> = (0..3) + .map(|i| { + Ok(mock_tool_call_response_with_usage( + &[(format!("c{}", i).as_str(), "grep", "{}")], + 60, + 40, + )) + }) + .collect(); + let session = partial_turn_session(responses, log.clone()); + session.set_task_budget_limit(crate::budget::TaskBudgetLimit { + max_tokens: Some(150), + ..Default::default() + }); + + let result = session + .process_input(&ChatMessage::user("do the task"), &[], Some("core"), false) + .await; + let err = result.expect_err("budget exhaustion must fail the turn"); + assert!( + matches!(&err, AishError::BudgetExhausted(d) if d == "tokens"), + "must surface BudgetExhausted(tokens), got {err:?}" + ); + assert_eq!( + session.take_budget_exhausted_dimension().as_deref(), + Some("tokens") + ); + // Rounds 1 and 2 ran (2×100=200 ≥ 150 after round 2); round 3 never + // sent a request, so the log shows exactly two tool executions. + assert_eq!(log.lock().unwrap().len(), 2, "round 3 must not run"); + let state = session.task_budget_state(); + assert_eq!(state.task_rounds, 2); + assert_eq!(state.tool_calls, 2); + assert_eq!(state.token_usage.total_tokens(), 200); + + // Evidence contract: user prefix + both completed call/result pairs. + let partial = session.take_last_partial_turn(); + let roles: Vec<&str> = partial.iter().map(|m| m.role.as_str()).collect(); + assert_eq!( + roles, + vec!["user", "assistant", "tool", "assistant", "tool"] + ); + let called_ids: Vec = partial + .iter() + .filter(|m| m.role == "assistant") + .flat_map(|m| m.tool_calls.iter().flatten().map(|tc| tc.id.clone())) + .collect(); + assert_eq!(called_ids, vec!["c0".to_string(), "c1".to_string()]); + } + + /// Issue #569: rounds-dimension exhaustion stops before the round that + /// would exceed it and reports the rounds dimension. + #[tokio::test] + async fn task_budget_rounds_exhaustion_stops_with_dimension() { + use crate::agents::mock_tool_call_response; + + let log = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); + let responses: Vec> = (0..5) + .map(|i| { + Ok(mock_tool_call_response(&[( + format!("c{}", i).as_str(), + "grep", + "{}", + )])) + }) + .collect(); + let session = partial_turn_session(responses, log.clone()); + session.set_task_budget_limit(crate::budget::TaskBudgetLimit { + max_rounds: Some(3), + ..Default::default() + }); + + let result = session + .process_input(&ChatMessage::user("do the task"), &[], Some("core"), false) + .await; + let err = result.expect_err("rounds exhaustion must fail the turn"); + assert!( + matches!(&err, AishError::BudgetExhausted(d) if d == "rounds"), + "must surface BudgetExhausted(rounds), got {err:?}" + ); + assert_eq!(log.lock().unwrap().len(), 3, "round 4 must not run"); + assert_eq!(session.task_budget_state().task_rounds, 3); + // Evidence: user prefix + 3 completed call/result pairs. + let partial = session.take_last_partial_turn(); + let assistants = partial.iter().filter(|m| m.role == "assistant").count(); + let tools = partial.iter().filter(|m| m.role == "tool").count(); + assert_eq!(assistants, 3); + assert_eq!(tools, 3); + } + + /// Issue #569: a bounded-but-unreached budget never stops the loop and + /// a completed turn keeps the partial buffer empty. + #[tokio::test] + async fn task_budget_within_bounds_completes_normally() { + use crate::agents::mock_tool_call_response; + + let log = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); + let session = partial_turn_session( + vec![ + Ok(mock_tool_call_response(&[("c1", "grep", "{}")])), + Ok(crate::agents::mock_text_response("done")), + ], + log.clone(), + ); + session.set_task_budget_limit(crate::budget::TaskBudgetLimit { + max_rounds: Some(100), + max_tool_calls: Some(100), + max_tokens: Some(1_000_000), + max_duration_secs: Some(3_600), + }); + + let result = session + .process_input(&ChatMessage::user("do the task"), &[], Some("core"), false) + .await; + assert!(result.is_ok(), "within-bounds turn must succeed"); + assert!(session.take_last_partial_turn().is_empty()); + assert_eq!(session.task_budget_state().task_rounds, 2); + assert_eq!(session.task_budget_state().tool_calls, 1); + } + + /// Issue #569 acceptance 2: "continue" resets only the segment counter, + /// never the cumulative task rounds. + #[tokio::test] + async fn task_rounds_accumulate_across_continue() { + use crate::agents::mock_tool_call_response; + + let log = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); + // Segment 1: 20 rounds then continue (segment counter resets), segment + // 2: 20 more rounds then stop. 40 rounds execute; the 41st scripted + // response is never consumed because the stop branch fires at the top + // of round 41 before any request. + let mut responses: Vec> = Vec::new(); + for i in 0..41 { + responses.push(Ok(mock_tool_call_response(&[( + format!("c{}", i).as_str(), + "grep", + "{}", + )]))); + } + let mut session = partial_turn_session(responses, log.clone()); + let first_vote = Arc::new(std::sync::atomic::AtomicBool::new(true)); + let vote = first_vote.clone(); + session.set_iteration_limit_callback(Arc::new(move |_| { + vote.swap(false, std::sync::atomic::Ordering::SeqCst) + })); + + let result = session + .process_input(&ChatMessage::user("do the task"), &[], Some("core"), false) + .await; + assert!(matches!(result, Err(AishError::IterationLimit))); + + // The task-level counter accumulated across the continue: 40 rounds. + assert_eq!( + session.task_budget_state().task_rounds, + 40, + "task rounds must accumulate across continues" + ); + assert_eq!(log.lock().unwrap().len(), 40); + // But the budget dimension was NOT the stopper. + assert!(session.take_budget_exhausted_dimension().is_none()); + } } diff --git a/crates/aish-llm/src/usage.rs b/crates/aish-llm/src/usage.rs index 31916911..831b3022 100644 --- a/crates/aish-llm/src/usage.rs +++ b/crates/aish-llm/src/usage.rs @@ -3,13 +3,18 @@ pub struct TokenUsage { pub prompt_tokens: u64, pub completion_tokens: u64, + /// Prompt tokens served from the provider's prompt cache (reused + /// prefix, not new work). OpenAI: `prompt_tokens_details.cached_tokens`; + /// Anthropic: `cache_read_input_tokens`. + pub cache_read_tokens: u64, + /// Prompt tokens written to the provider's prompt cache (billed as new + /// work). Anthropic: `cache_creation_input_tokens`; OpenAI-compatible + /// gateways usually fold this into `prompt_tokens` and report 0 here. + pub cache_write_tokens: u64, } impl TokenUsage { /// Extract token usage from an OpenAI-compatible API response JSON. - /// - /// Looks for `usage.prompt_tokens` and `usage.completion_tokens`. - /// Returns default (zeroed) if the fields are missing. pub fn from_response_json(json: &serde_json::Value) -> Self { let usage = json.get("usage"); Self { @@ -21,6 +26,12 @@ impl TokenUsage { .and_then(|u| u.get("completion_tokens")) .and_then(|v| v.as_u64()) .unwrap_or(0), + cache_read_tokens: usage + .and_then(|u| u.get("prompt_tokens_details")) + .and_then(|d| d.get("cached_tokens")) + .and_then(|v| v.as_u64()) + .unwrap_or(0), + cache_write_tokens: 0, } } @@ -36,6 +47,14 @@ impl TokenUsage { .and_then(|u| u.get("output_tokens")) .and_then(|v| v.as_u64()) .unwrap_or(0), + cache_read_tokens: usage + .and_then(|u| u.get("cache_read_input_tokens")) + .and_then(|v| v.as_u64()) + .unwrap_or(0), + cache_write_tokens: usage + .and_then(|u| u.get("cache_creation_input_tokens")) + .and_then(|v| v.as_u64()) + .unwrap_or(0), } } } @@ -46,6 +65,14 @@ pub struct TokenStats { pub total_input: u64, pub total_output: u64, pub request_count: u64, + /// Cumulative prompt tokens served from the provider's prompt cache. + /// Reused prefix, not new work — excluded from `total_tokens()` so + /// budgets measure fresh consumption only (same accounting as omp + /// goals: input + cacheWrite + output, cacheRead excluded). + pub total_cache_read: u64, + /// Cumulative prompt tokens written to the provider's prompt cache. + /// Billed as new work — included in `total_input`. + pub total_cache_write: u64, /// Prompt tokens from the most recent API call — the actual context /// window consumption at the current conversation depth. pub last_prompt_tokens: u64, @@ -56,6 +83,8 @@ impl TokenStats { pub fn record(&mut self, usage: TokenUsage) { self.total_input += usage.prompt_tokens; self.total_output += usage.completion_tokens; + self.total_cache_read += usage.cache_read_tokens; + self.total_cache_write += usage.cache_write_tokens; self.last_prompt_tokens = usage.prompt_tokens; self.request_count += 1; } @@ -67,10 +96,13 @@ impl TokenStats { pub fn merge_totals(&mut self, other: &TokenStats) { self.total_input += other.total_input; self.total_output += other.total_output; + self.total_cache_read += other.total_cache_read; + self.total_cache_write += other.total_cache_write; self.request_count += other.request_count; } - /// Total tokens consumed (input + output). + /// Total fresh tokens consumed (input + output; cached prefix reads are + /// excluded because they are reused context, not new work). pub fn total_tokens(&self) -> u64 { self.total_input + self.total_output } @@ -120,14 +152,52 @@ mod tests { stats.record(TokenUsage { prompt_tokens: 100, completion_tokens: 50, + cache_read_tokens: 0, + cache_write_tokens: 0, }); stats.record(TokenUsage { prompt_tokens: 200, completion_tokens: 80, + cache_read_tokens: 40, + cache_write_tokens: 10, }); assert_eq!(stats.total_input, 300); assert_eq!(stats.total_output, 130); + assert_eq!(stats.total_cache_read, 40); + assert_eq!(stats.total_cache_write, 10); assert_eq!(stats.request_count, 2); assert_eq!(stats.total_tokens(), 430); } + + #[test] + fn test_token_usage_openai_cached_tokens() { + let json = serde_json::json!({ + "usage": { + "prompt_tokens": 150, + "completion_tokens": 50, + "prompt_tokens_details": { "cached_tokens": 120 } + } + }); + let usage = TokenUsage::from_response_json(&json); + assert_eq!(usage.prompt_tokens, 150); + assert_eq!(usage.cache_read_tokens, 120); + assert_eq!(usage.cache_write_tokens, 0); + } + + #[test] + fn test_token_usage_anthropic_cache_fields() { + let json = serde_json::json!({ + "usage": { + "input_tokens": 30, + "output_tokens": 40, + "cache_read_input_tokens": 120, + "cache_creation_input_tokens": 25 + } + }); + let usage = TokenUsage::from_anthropic_json(&json); + assert_eq!(usage.prompt_tokens, 30); + assert_eq!(usage.completion_tokens, 40); + assert_eq!(usage.cache_read_tokens, 120); + assert_eq!(usage.cache_write_tokens, 25); + } } diff --git a/crates/aish-session/src/lib.rs b/crates/aish-session/src/lib.rs index 70c0d349..77017e86 100644 --- a/crates/aish-session/src/lib.rs +++ b/crates/aish-session/src/lib.rs @@ -18,6 +18,6 @@ pub mod store; pub use models::{ AuditEventRecord, AuditQuery, HistoryEntry, SessionContextMessage, SessionRecord, - SessionStateSnapshot, + SessionStateSnapshot, TaskBudgetSnapshot, }; pub use store::{AuditStore, SessionStore}; diff --git a/crates/aish-session/src/models.rs b/crates/aish-session/src/models.rs index 6cb7e526..37638970 100644 --- a/crates/aish-session/src/models.rs +++ b/crates/aish-session/src/models.rs @@ -42,6 +42,31 @@ pub struct SessionStateSnapshot { #[serde(default)] pub context_messages_snapshot: Vec, pub updated_at: Option>, + /// Task-level cumulative budget counters (issue #569), persisted so a + /// resumed session keeps accruing against the same task budget instead + /// of restarting it. Pure data: the runtime `Instant` anchor is + /// process-local and is never serialized. + #[serde(default)] + pub task_budget: Option, +} + +/// Serializable snapshot of the task budget state (issue #569). +#[derive(Debug, Clone, Serialize, Deserialize, Default)] +pub struct TaskBudgetSnapshot { + pub task_rounds: u64, + pub tool_calls: u64, + pub active_secs: u64, + pub total_input: u64, + pub total_output: u64, + pub total_cache_read: u64, + pub total_cache_write: u64, + pub request_count: u64, + /// Upper bounds persisted alongside the counters so resuming cannot + /// bypass the budget with a config that no longer matches. + pub max_rounds: Option, + pub max_tool_calls: Option, + pub max_tokens: Option, + pub max_duration_secs: Option, } impl SessionRecord { @@ -125,3 +150,47 @@ impl AuditQuery { Self::default() } } + +#[cfg(test)] +mod task_budget_snapshot_tests { + use super::*; + + #[test] + fn task_budget_snapshot_round_trips_through_json() { + let snapshot = SessionStateSnapshot { + cwd: Some("/tmp".into()), + summary_preview: None, + context_messages_snapshot: Vec::new(), + updated_at: None, + task_budget: Some(TaskBudgetSnapshot { + task_rounds: 25, + tool_calls: 31, + active_secs: 95, + total_input: 12_345, + total_output: 6_789, + total_cache_read: 999, + total_cache_write: 111, + request_count: 25, + max_rounds: Some(200), + max_tool_calls: None, + max_tokens: Some(2_000_000), + max_duration_secs: Some(3_600), + }), + }; + let json = serde_json::to_value(&snapshot).expect("serialize"); + // Legacy compatibility: an old snapshot without the field must parse. + let legacy = serde_json::json!({"cwd": "/tmp"}); + let parsed_legacy: SessionStateSnapshot = + serde_json::from_value(legacy).expect("legacy snapshot must parse"); + assert!(parsed_legacy.task_budget.is_none()); + + let parsed: SessionStateSnapshot = serde_json::from_value(json).expect("deserialize"); + let budget = parsed.task_budget.expect("budget present"); + assert_eq!(budget.task_rounds, 25); + assert_eq!(budget.tool_calls, 31); + assert_eq!(budget.active_secs, 95); + assert_eq!(budget.total_cache_read, 999); + assert_eq!(budget.max_rounds, Some(200)); + assert_eq!(budget.max_duration_secs, Some(3_600)); + } +} diff --git a/crates/aish-session/src/store.rs b/crates/aish-session/src/store.rs index 09c930ac..82a574c7 100644 --- a/crates/aish-session/src/store.rs +++ b/crates/aish-session/src/store.rs @@ -930,6 +930,7 @@ mod tests { reasoning_content: None, }], updated_at: Some(Utc::now()), + task_budget: None, }; store @@ -979,6 +980,7 @@ mod tests { summary_preview: Some("Rollback: systemctl revert nginx".to_string()), context_messages_snapshot: vec![summary_msg, tail_user], updated_at: Some(Utc::now()), + task_budget: None, }; store .update_session_state(&record.session_uuid, &snapshot) @@ -1338,6 +1340,7 @@ mod tests { reasoning_content: None, }], updated_at: Some(Utc::now()), + task_budget: None, }; store .update_session_state(&parent.session_uuid, &snapshot) diff --git a/crates/aish-shell/src/ai_handler.rs b/crates/aish-shell/src/ai_handler.rs index 14c6178d..5e1e7180 100644 --- a/crates/aish-shell/src/ai_handler.rs +++ b/crates/aish-shell/src/ai_handler.rs @@ -186,6 +186,9 @@ pub struct AiHandler { /// Whether the most recent partial turn was stopped at the iteration /// limit (as opposed to a provider failure); selects the user hint. last_partial_turn_limit: bool, + /// Whether the most recent partial turn was stopped by task-budget + /// exhaustion (issue #569); selects the user hint. + last_partial_turn_budget: bool, } impl AiHandler { @@ -221,6 +224,7 @@ impl AiHandler { secret_redactor: None, last_partial_turn_steps: 0, last_partial_turn_limit: false, + last_partial_turn_budget: false, } } @@ -632,6 +636,7 @@ impl AiHandler { // never reports evidence from an already-completed turn. self.last_partial_turn_steps = 0; self.last_partial_turn_limit = false; + self.last_partial_turn_budget = false; result } Err(err) => { @@ -643,6 +648,11 @@ impl AiHandler { // already done and must not blindly redo. if matches!(err, aish_core::AishError::IterationLimit) { self.commit_partial_turn_stopped_at_limit(&question_processed); + } else if matches!(err, aish_core::AishError::BudgetExhausted(_)) { + // Issue #569: task budget exhausted — same evidence + // contract, but the stop note must say budget, not + // provider failure or iteration limit. + self.commit_partial_turn_stopped_at_budget(&question_processed); } else { self.commit_partial_turn(&question_processed); } @@ -715,6 +725,103 @@ impl AiHandler { self.commit_partial_messages(question_processed, &partial, true); } + /// Issue #569 variant: commit a partial turn that ended because the + /// task-level cumulative budget was exhausted. Same evidence contract; + /// the note must say budget, not provider failure. + fn commit_partial_turn_stopped_at_budget(&mut self, question_processed: &str) { + let partial = self.llm_session.take_last_partial_turn(); + self.commit_partial_messages(question_processed, &partial, true); + self.last_partial_turn_budget = true; + } + + /// Whether the most recent partial turn was stopped by task-budget + /// exhaustion (issue #569); used by the shell to pick the user hint. + pub fn last_partial_turn_stopped_at_budget(&self) -> bool { + self.last_partial_turn_budget + } + /// Read-only access to the task budget state for /token and the footer. + pub fn llm_budget_state(&self) -> aish_llm::budget::TaskBudgetState { + self.llm_session.task_budget_state() + } + + /// Update the task budget bounds at runtime (explicit adjustment only). + pub fn set_llm_budget_limit(&self, limit: aish_llm::budget::TaskBudgetLimit) { + self.llm_session.set_task_budget_limit(limit); + } + + /// Install limits from `config.yaml llm.task_budget` (issue #569). + /// Absent dimensions stay unlimited; counters still accrue for display. + pub fn apply_task_budget_config(&self, config: &aish_config::TaskBudgetConfig) { + self.llm_session + .set_task_budget_limit(aish_llm::budget::TaskBudgetLimit { + max_rounds: config.max_rounds, + max_tool_calls: config.max_tool_calls, + max_tokens: config.max_tokens, + max_duration_secs: config.max_duration_secs, + }); + } + + /// Re-arm the near-threshold advisory after an explicit budget raise. + pub fn rearm_budget_advisory(&self) { + self.llm_session.rearm_near_threshold(); + } + + /// Read-only access to the task budget limits for /token and the footer. + pub fn llm_budget_limit(&self) -> aish_llm::budget::TaskBudgetLimit { + self.llm_session.task_budget_limit() + } + + /// Issue #569: serializable budget counters + limits for the session + /// snapshot. `None` when the process never bounded or accrued anything + /// (fresh session, no config limits) — keeps legacy snapshots identical. + pub fn task_budget_snapshot(&self) -> Option { + let state = self.llm_session.task_budget_state(); + let limit = self.llm_session.task_budget_limit(); + if !limit.is_bounded() && state.task_rounds == 0 && state.tool_calls == 0 { + return None; + } + Some(aish_session::TaskBudgetSnapshot { + task_rounds: state.task_rounds, + tool_calls: state.tool_calls, + active_secs: state.active_secs, + total_input: state.token_usage.total_input, + total_output: state.token_usage.total_output, + total_cache_read: state.token_usage.total_cache_read, + total_cache_write: state.token_usage.total_cache_write, + request_count: state.token_usage.request_count, + max_rounds: limit.max_rounds, + max_tool_calls: limit.max_tool_calls, + max_tokens: limit.max_tokens, + max_duration_secs: limit.max_duration_secs, + }) + } + + /// Issue #569: restore budget counters + limits from a session snapshot + /// on resume so the task keeps accruing against the same budget. + pub fn seed_task_budget(&self, snapshot: aish_session::TaskBudgetSnapshot) { + use aish_llm::usage::TokenStats; + let mut state = aish_llm::budget::TaskBudgetState::default(); + state.task_rounds = snapshot.task_rounds; + state.tool_calls = snapshot.tool_calls; + state.active_secs = snapshot.active_secs; + state.token_usage = TokenStats { + total_input: snapshot.total_input, + total_output: snapshot.total_output, + total_cache_read: snapshot.total_cache_read, + total_cache_write: snapshot.total_cache_write, + request_count: snapshot.request_count, + last_prompt_tokens: 0, + }; + self.llm_session.seed_task_budget_state(state); + self.llm_session + .set_task_budget_limit(aish_llm::budget::TaskBudgetLimit { + max_rounds: snapshot.max_rounds, + max_tool_calls: snapshot.max_tool_calls, + max_tokens: snapshot.max_tokens, + max_duration_secs: snapshot.max_duration_secs, + }); + } + /// Shared commit path: append the user message, the completed /// tool-call/tool-result pairs (redacted), and a synthetic assistant /// note describing why the turn stopped. `stopped_by_limit` selects the @@ -728,6 +835,7 @@ impl AiHandler { if partial.is_empty() { self.last_partial_turn_steps = 0; self.last_partial_turn_limit = false; + self.last_partial_turn_budget = false; return; } // The session prepends the user message; the shell commits it here @@ -3723,4 +3831,58 @@ mod tests { aish_llm::ToolResult::success("probe ran") } } + + /// Issue #569 acceptance 5: seeding the budget from a snapshot restores + /// both the cumulative counters and the limits, and drops the process- + /// local wall-clock anchor so a resumed session never inherits a stale + /// Instant. + #[test] + fn seed_task_budget_restores_counters_and_limits() { + use aish_llm::budget::TaskBudgetLimit; + use aish_session::TaskBudgetSnapshot; + + let handler = test_handler(); + handler.seed_task_budget(TaskBudgetSnapshot { + task_rounds: 25, + tool_calls: 31, + active_secs: 95, + total_input: 12_345, + total_output: 6_789, + total_cache_read: 999, + total_cache_write: 111, + request_count: 25, + max_rounds: Some(200), + max_tool_calls: None, + max_tokens: Some(2_000_000), + max_duration_secs: Some(3_600), + }); + + let state = handler.llm_session.task_budget_state(); + assert_eq!(state.task_rounds, 25); + assert_eq!(state.tool_calls, 31); + assert_eq!(state.active_secs, 95); + assert_eq!(state.token_usage.total_input, 12_345); + assert_eq!(state.token_usage.total_cache_read, 999); + assert!(state.last_accounted.is_none(), "anchor must not be seeded"); + + let limit = handler.llm_session.task_budget_limit(); + assert_eq!(limit.max_rounds, Some(200)); + assert_eq!(limit.max_tokens, Some(2_000_000)); + assert_eq!(limit.max_duration_secs, Some(3_600)); + assert!(!TaskBudgetLimit::default().is_bounded()); + assert!(limit.is_bounded()); + + // The snapshot serializer round-trips the restored state. + let snapshot = handler.task_budget_snapshot().expect("bounded task"); + assert_eq!(snapshot.task_rounds, 25); + assert_eq!(snapshot.max_rounds, Some(200)); + } + + /// A fresh, unbounded handler yields no snapshot payload — legacy + /// sessions stay byte-identical. + #[test] + fn task_budget_snapshot_is_none_for_fresh_unbounded_session() { + let handler = test_handler(); + assert!(handler.task_budget_snapshot().is_none()); + } } diff --git a/crates/aish-shell/src/app.rs b/crates/aish-shell/src/app.rs index e3c80e6f..2b9c9499 100644 --- a/crates/aish-shell/src/app.rs +++ b/crates/aish-shell/src/app.rs @@ -2201,6 +2201,9 @@ impl AishShell { context_budget_policy, config.skills.auto_search, ); + // Issue #569: install the configured task budget limits so the tool + // loop enforces them from the very first round. + ai_handler.apply_task_budget_config(&config.task_budget); // Redact secrets from tool outputs before they enter persistent // LLM context (same scanner as the audit path). @@ -2927,7 +2930,13 @@ impl AishShell { // Errors are already displayed via the LlmEventType::Error // event callback — avoid printing twice. Only handle // non-LLM errors that bypass the event system. - if !matches!( + if let aish_core::AishError::BudgetExhausted(dim) = &e { + // Issue #569 acceptance 5: the stop + // panel offers an explicit budget + // adjustment (recorded to audit) — + // plain "continue" never raises it. + self.handle_budget_exhaustion(dim); + } else if !matches!( e, aish_core::AishError::Llm(_) | aish_core::AishError::IterationLimit @@ -2942,11 +2951,6 @@ impl AishShell { } } - // InputGuard pre-check for AI prompts - if !self.screen_ai_prompt(&question) { - continue; - } - // Security gate: detect secrets in AI input if !self.check_security_gate(&mut question) { continue; @@ -3178,7 +3182,9 @@ impl AishShell { Err(e) => { if !matches!( e, - aish_core::AishError::Llm(_) | aish_core::AishError::IterationLimit + aish_core::AishError::Llm(_) + | aish_core::AishError::IterationLimit + | aish_core::AishError::BudgetExhausted(_) ) { let msg = t("shell.error.llm_error_message") .replace("{error}", &e.to_string()); @@ -3377,6 +3383,7 @@ impl AishShell { e, aish_core::AishError::Llm(_) | aish_core::AishError::IterationLimit + | aish_core::AishError::BudgetExhausted(_) ) { let msg = t("shell.error.llm_error_message") .replace("{error}", &e.to_string()); @@ -3755,7 +3762,9 @@ impl AishShell { Err(e) => { if !matches!( e, - aish_core::AishError::Llm(_) | aish_core::AishError::IterationLimit + aish_core::AishError::Llm(_) + | aish_core::AishError::IterationLimit + | aish_core::AishError::BudgetExhausted(_) ) { let msg = t("shell.error.llm_error_message") .replace("{error}", &e.to_string()); @@ -3846,8 +3855,11 @@ impl AishShell { return false; } // Issue #572: a deliberate iteration-limit stop is not a failure; - // use the matching hint so the saved evidence is not misdescribed. - let key = if self.ai_handler.last_partial_turn_stopped_at_limit() { + // issue #569: neither is a task-budget stop. Use the matching hint + // so the saved evidence is not misdescribed. + let key = if self.ai_handler.last_partial_turn_stopped_at_budget() { + "shell.error.partial_turn_saved_budget" + } else if self.ai_handler.last_partial_turn_stopped_at_limit() { "shell.error.partial_turn_saved_limit" } else { "shell.error.partial_turn_saved" @@ -3858,6 +3870,113 @@ impl AishShell { true } + /// Issue #569: interactive panel shown when the task budget stops a + /// turn. Offers: stop (default) / adjust the budget explicitly / show + /// the saved-evidence hint. An adjustment is recorded to the audit + /// store with its before/after values — "continue" alone never raises + /// the limit. + fn handle_budget_exhaustion(&mut self, dimension: &str) { + let limit = self.ai_handler.llm_budget_limit(); + let state = self.ai_handler.llm_budget_state(); + let (used, bound) = match dimension { + "rounds" => (state.task_rounds, limit.max_rounds), + "tool_calls" => (state.tool_calls, limit.max_tool_calls), + "tokens" => (state.token_usage.total_tokens(), limit.max_tokens), + "duration" => (state.active_secs, limit.max_duration_secs), + _ => (0, None), + }; + println!(); + println!( + "{}", + theme::warning(&aish_i18n::t("shell.budget.exhausted_title")) + ); + println!( + " {}", + aish_i18n::t_with_args("shell.budget.exhausted_detail", &{ + let mut m = std::collections::HashMap::new(); + m.insert("dimension".to_string(), dimension.to_string()); + m.insert("used".to_string(), format_number(used)); + m.insert( + "limit".to_string(), + bound.map_or_else(|| "∞".to_string(), format_number), + ); + m + }) + ); + println!(" {}", aish_i18n::t("shell.budget.exhausted_options")); + print!(" "); + let _ = std::io::Write::flush(&mut std::io::stdout()); + + let choice = read_raw_confirmation(); + if !choice { + let msg = aish_i18n::t("shell.budget.stopped_by_user"); + println!("{}", theme::warning(&msg)); + return; + } + + // Explicit increase: prompt for the new value for the tripped + // dimension only. + println!( + " {}", + aish_i18n::t_with_args("shell.budget.new_limit_prompt", &{ + let mut m = std::collections::HashMap::new(); + m.insert("dimension".to_string(), dimension.to_string()); + m + }) + ); + print!(" "); + let _ = std::io::Write::flush(&mut std::io::stdout()); + + let mut new_limit_raw = String::new(); + if std::io::stdin().read_line(&mut new_limit_raw).is_ok() { + let parsed = new_limit_raw.trim().parse::().ok(); + let applied = parsed.filter(|v| *v > used).map(|v| (v, used)); + if let Some((new_limit, old_used)) = applied { + self.raise_budget_limit(dimension, new_limit); + // Audit the explicit adjustment with before/after values. + if let Some(ref audit) = self.audit_store { + let event = aish_core::AuditEvent::command( + chrono::Utc::now(), + Some(self.session_uuid.clone()), + self.audit_user.clone(), + self.audit_host.clone(), + format!("budget {} {} -> {}", dimension, old_used, new_limit), + "budget".to_string(), + 0, + ); + audit.record(event); + } + let msg = aish_i18n::t_with_args("shell.budget.raised", &{ + let mut m = std::collections::HashMap::new(); + m.insert("dimension".to_string(), dimension.to_string()); + m.insert("value".to_string(), format_number(new_limit)); + m + }); + println!("{}", theme::accent(&msg)); + } else { + let msg = aish_i18n::t("shell.budget.invalid_limit"); + println!("{}", theme::error(&msg)); + } + } + } + + /// Raise one budget dimension's limit in-place (also updates the LLM + /// session so the loop sees it immediately). + fn raise_budget_limit(&mut self, dimension: &str, new_limit: u64) { + let mut limit = self.ai_handler.llm_budget_limit(); + match dimension { + "rounds" => limit.max_rounds = Some(new_limit), + "tool_calls" => limit.max_tool_calls = Some(new_limit), + "tokens" => limit.max_tokens = Some(new_limit), + "duration" => limit.max_duration_secs = Some(new_limit), + _ => {} + } + self.ai_handler.set_llm_budget_limit(limit); + self.ai_handler.rearm_budget_advisory(); + } + + /// Issue #569: budget counters ride the session snapshot; limits are + /// persisted too so resuming cannot bypass the budget. fn session_state_snapshot( &self, updated_at: chrono::DateTime, @@ -3868,6 +3987,7 @@ impl AishShell { summary_preview: summary_preview_from_context(&context_messages), context_messages_snapshot: context_messages, updated_at: Some(updated_at), + task_budget: self.ai_handler.task_budget_snapshot(), } } @@ -7144,6 +7264,13 @@ impl AishShell { self.ai_handler .restore_context_messages(restored_context.clone()); + // Issue #569 acceptance 5: seed the persisted task-budget counters + // and limits so a resumed session keeps accruing against the same + // budget instead of restarting it. + if let Some(budget) = &snapshot.task_budget { + self.ai_handler.seed_task_budget(budget.clone()); + } + // Issue #530: a legacy session (snapshot but no transcript rows) // resumed here must not let its first new turn become the only // exported message — seed the snapshot into the transcript once. @@ -8611,7 +8738,16 @@ impl AishShell { self.refresh_config_dependent_tools(); } LiveEffect::ToolsRefresh => self.refresh_config_dependent_tools(), - LiveEffect::None => {} + LiveEffect::None => { + // Issue #569: the task-budget limit is read from the + // session on every loop round — push it immediately so + // /setting changes take effect without a restart. + if matches!(key, crate::settings_panel::SettingKey::TaskBudgetMaxRounds) { + self.ai_handler + .apply_task_budget_config(&self.config.task_budget); + self.ai_handler.rearm_budget_advisory(); + } + } } } @@ -8923,6 +9059,47 @@ impl AishShell { aish_i18n::t("shell.token.api_calls"), format_number(stats.request_count) ); + + // Issue #569: task budget usage alongside token totals. Cost is not + // reported by providers, so it is explicitly shown as unavailable. + let limit = self.ai_handler.llm_budget_limit(); + let state = self.ai_handler.llm_budget_state(); + if limit.is_bounded() { + println!(); + println!( + "{}", + theme::accent(&aish_i18n::t("shell.token.budget_title")) + ); + let row = |label: String, used: u64, limit: Option| { + let rest = match limit { + Some(max) => format!("{} / {}", format_number(used), format_number(max)), + None => format_number(used), + }; + println!(" {} {}", label, rest); + }; + row( + aish_i18n::t("shell.token.budget_rounds"), + state.task_rounds, + limit.max_rounds, + ); + row( + aish_i18n::t("shell.token.budget_tool_calls"), + state.tool_calls, + limit.max_tool_calls, + ); + row( + aish_i18n::t("shell.token.budget_tokens"), + state.token_usage.total_tokens(), + limit.max_tokens, + ); + row( + aish_i18n::t("shell.token.budget_duration"), + state.active_secs, + limit.max_duration_secs, + ); + let cost = aish_i18n::t("shell.token.budget_cost_unavailable"); + println!(" {} {}", aish_i18n::t("shell.token.budget_cost"), cost); + } println!(); } diff --git a/crates/aish-shell/src/settings_panel.rs b/crates/aish-shell/src/settings_panel.rs index 918c77ee..b65dd86c 100644 --- a/crates/aish-shell/src/settings_panel.rs +++ b/crates/aish-shell/src/settings_panel.rs @@ -138,6 +138,7 @@ pub enum SettingKey { InlineDisableThinking, InlineEnforceJson, MaxLlmMessages, + TaskBudgetMaxRounds, EnableTokenEstimation, // Security InputGuardEnabled, @@ -200,6 +201,7 @@ impl SettingKey { SettingKey::InlineDisableThinking => "inline_disable_thinking", SettingKey::InlineEnforceJson => "inline_enforce_json", SettingKey::MaxLlmMessages => "max_llm_messages", + SettingKey::TaskBudgetMaxRounds => "task_budget_max_rounds", SettingKey::EnableTokenEstimation => "enable_token_estimation", SettingKey::InputGuardEnabled => "input_guard_enabled", SettingKey::EnableSandbox => "enable_sandbox", @@ -345,6 +347,11 @@ pub const SETTINGS: &[SettingDef] = &[ category: SettingCategory::Ai, kind: SettingKind::Int, }, + SettingDef { + key: SettingKey::TaskBudgetMaxRounds, + category: SettingCategory::Ai, + kind: SettingKind::Int, + }, SettingDef { key: SettingKey::EnableTokenEstimation, category: SettingCategory::Ai, @@ -698,6 +705,11 @@ pub fn current_raw(cfg: &ConfigModel, key: SettingKey) -> String { SettingKey::InlineDisableThinking => bool_str(cfg.inline_completion.disable_thinking), SettingKey::InlineEnforceJson => bool_str(cfg.inline_completion.enforce_json), SettingKey::MaxLlmMessages => cfg.max_llm_messages.to_string(), + SettingKey::TaskBudgetMaxRounds => cfg + .task_budget + .max_rounds + .map(|n| n.to_string()) + .unwrap_or_default(), SettingKey::EnableTokenEstimation => bool_str(cfg.enable_token_estimation), SettingKey::InputGuardEnabled | SettingKey::EnableSandbox @@ -855,6 +867,13 @@ pub fn apply(cfg: &mut ConfigModel, key: SettingKey, value: &str) -> Result<(), cfg.inline_completion.enforce_json = parse_bool(value)?; } SettingKey::MaxLlmMessages => cfg.max_llm_messages = parse_usize(value)?, + SettingKey::TaskBudgetMaxRounds => { + cfg.task_budget.max_rounds = if value.is_empty() { + None + } else { + Some(parse_usize(value)? as u64) + }; + } SettingKey::EnableTokenEstimation => cfg.enable_token_estimation = parse_bool(value)?, SettingKey::InputGuardEnabled | SettingKey::EnableSandbox @@ -1120,6 +1139,7 @@ mod tests { SettingKey::InlineDisableThinking, SettingKey::InlineEnforceJson, SettingKey::MaxLlmMessages, + SettingKey::TaskBudgetMaxRounds, SettingKey::EnableTokenEstimation, SettingKey::InputGuardEnabled, SettingKey::EnableSandbox, @@ -1163,6 +1183,24 @@ mod tests { } } + #[test] + fn task_budget_max_rounds_round_trips() { + // Unset renders as "" and blank input clears; a number applies. + let mut cfg = default_cfg(); + assert_eq!(current_raw(&cfg, SettingKey::TaskBudgetMaxRounds), ""); + + apply(&mut cfg, SettingKey::TaskBudgetMaxRounds, "200").unwrap(); + assert_eq!(cfg.task_budget.max_rounds, Some(200)); + assert_eq!(current_raw(&cfg, SettingKey::TaskBudgetMaxRounds), "200"); + + apply(&mut cfg, SettingKey::TaskBudgetMaxRounds, "").unwrap(); + assert_eq!(cfg.task_budget.max_rounds, None); + + // Rejects non-numeric input without mutating the live value. + assert!(apply(&mut cfg, SettingKey::TaskBudgetMaxRounds, "abc").is_err()); + assert_eq!(cfg.task_budget.max_rounds, None); + } + #[test] fn choice_current_value_is_case_normalized() { // Security enums always render as canonical lowercase options. From 01f39ec01c6aba56eb3e4fb14811620e43729f7f Mon Sep 17 00:00:00 2001 From: xuezhizone Date: Wed, 30 Sep 2026 12:05:58 +0800 Subject: [PATCH 2/4] fix(shell,llm): address code review findings on task budget (issue #569) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit C2 (major): wall-clock time was only settled on the streaming tool-call exit — start_turn() reset the anchor every round, so elapsed time leaked on every other exit path and max_duration_secs could never trip. process_input now starts accounting once and installs a drop guard that ends it on every return; the loop flushes before each budget check. tool_loop (spawn) starts once at loop entry and relies on the existing pre-merge flush. C3 (major): the budget stop reused the iteration-limit flag, so the persisted note said 'iteration limit' instead of naming the budget. Both booleans are replaced by a PartialTurnStopReason enum; the note, tracing reason, and shell hint all derive from it, and the reason is recorded even when the partial is empty (no stale flags). C4 (major): only the error-correction path opened the budget panel; the main ';'-path, the natural-language path, and the popup path hid the error silently — a resumed already-exhausted session showed nothing. All three now call handle_budget_exhaustion() before report_partial_turn_saved(). C5 (minor): the audit record logged 'used -> new_limit'; it now logs the limit's before/after values ('limit unbounded|N -> M (used U)'). C1 (minor): de-DE typo Werkzugschleifen -> Werkzeugschleifen. aish-shell 498 / aish-llm 349 tests pass; new tests cover the budget note, reason-on-empty-partial, and reason round-trip. --- crates/aish-i18n/locales/de-DE.yaml | 2 +- crates/aish-llm/src/agents/tool_loop.rs | 16 +- crates/aish-llm/src/session.rs | 33 +++-- crates/aish-shell/src/ai_handler.rs | 186 +++++++++++++++++------- crates/aish-shell/src/app.rs | 58 +++++--- 5 files changed, 205 insertions(+), 90 deletions(-) diff --git a/crates/aish-i18n/locales/de-DE.yaml b/crates/aish-i18n/locales/de-DE.yaml index f09f976f..3baa1da0 100644 --- a/crates/aish-i18n/locales/de-DE.yaml +++ b/crates/aish-i18n/locales/de-DE.yaml @@ -578,7 +578,7 @@ shell: desc: "Inline-Completion zwingen, striktes JSON zurückzugeben" task_budget_max_rounds: label: "Aufgabenbudget-Rundenlimit" - desc: "Kumulierte Werkzugschleifen-Runden pro Aufgabe; leer = unbegrenzt (Weitermachen setzt nicht zurueck)" + desc: "Kumulierte Werkzeugschleifen-Runden pro Aufgabe; leer = unbegrenzt (Weitermachen setzt nicht zurueck)" max_llm_messages: label: "Max LLM-Messages" desc: "Maximale Konversations-Turns im LLM-Verlauf" diff --git a/crates/aish-llm/src/agents/tool_loop.rs b/crates/aish-llm/src/agents/tool_loop.rs index 2d3b4422..b3ac30fc 100644 --- a/crates/aish-llm/src/agents/tool_loop.rs +++ b/crates/aish-llm/src/agents/tool_loop.rs @@ -97,6 +97,12 @@ pub async fn run_tool_loop_until_done( config: &ToolLoopConfig, ) -> LoopOutcome { let base_system = config.system_message.as_deref().unwrap_or(""); + // Wall-clock accounting for the whole spawn (issue #569): start once, + // flush every round before the parent merges the counters. The spawn + // path calls flush_task_budget_wall_clock() before merging, so no drop + // guard is needed here — the loop's own exits all leave the anchor in + // place for that flush. + session.task_budget_state_mut().start_turn(); let bundle = PromptAssembly::build(session, config.prompt_context.clone(), base_system); let mut messages: Vec = Vec::new(); if config.system_message.is_some() { @@ -128,13 +134,9 @@ pub async fn run_tool_loop_until_done( } iterations += 1; // Task budget accounting (issue #569): sub-agent loops accrue the - // same counters as the main loop so a spawn folds real consumption - // into the parent task. - { - let mut state = session.task_budget_state_mut(); - state.start_turn(); - state.task_rounds += 1; - } + // same counters as the main loop. Wall-clock stays anchored from + // loop start; the spawn path flushes it before merging. + session.task_budget_state_mut().task_rounds += 1; messages = session.prepare_messages_for_send(messages).await; diff --git a/crates/aish-llm/src/session.rs b/crates/aish-llm/src/session.rs index 2ea0d244..b5194628 100644 --- a/crates/aish-llm/src/session.rs +++ b/crates/aish-llm/src/session.rs @@ -721,6 +721,21 @@ impl LlmSession { ) -> Result { self.cancellation_token.reset(); self.last_turn_compaction.lock().unwrap().take(); + + // Wall-clock settle guard (issue #569): every exit path of this + // function — success, error, cancel, iteration limit, budget stop — + // must flush pending wall-clock seconds and stop accounting, or the + // next round's start_turn() would reset the anchor and lose the + // elapsed time of the aborted round. + struct _WallClockGuard<'a>(&'a parking_lot::Mutex); + impl Drop for _WallClockGuard<'_> { + fn drop(&mut self) { + self.0.lock().end_turn(); + } + } + self.task_budget_state.lock().start_turn(); + let _wall_clock_guard = _WallClockGuard(&self.task_budget_state); + let turn_seq = self .turn_seq .fetch_add(1, std::sync::atomic::Ordering::Relaxed); @@ -854,7 +869,10 @@ impl LlmSession { // evidence below is complete and consistent. { let (limit, state_snapshot) = { - let state = self.task_budget_state.lock(); + let mut state = self.task_budget_state.lock(); + // Settle the previous round's wall-clock time BEFORE the + // check so duration exhaustion sees complete data. + state.flush_wall_clock(); (self.task_budget_limit.lock().clone(), state.clone()) }; let check = limit.check(&state_snapshot); @@ -888,14 +906,11 @@ impl LlmSession { return Err(AishError::BudgetExhausted(dim_key)); } } - - // Start wall-clock accounting for this round, count the round, - // then proceed with the model request. - { - let mut state = self.task_budget_state.lock(); - state.start_turn(); - state.task_rounds += 1; - } + // Count this round. Wall-clock accounting is already running + // (started by the entry guard and flushed below before the + // budget check), so do NOT restart the anchor here — restarting + // it would discard the previous round's elapsed time. + self.task_budget_state.lock().task_rounds += 1; iterations += 1; messages = self.prepare_messages_for_send(messages).await; diff --git a/crates/aish-shell/src/ai_handler.rs b/crates/aish-shell/src/ai_handler.rs index 5e1e7180..c7c2714a 100644 --- a/crates/aish-shell/src/ai_handler.rs +++ b/crates/aish-shell/src/ai_handler.rs @@ -183,12 +183,23 @@ pub struct AiHandler { /// Tool-message count committed by the most recent `commit_partial_turn` /// (0 when the last turn succeeded or failed before any tool ran). last_partial_turn_steps: usize, - /// Whether the most recent partial turn was stopped at the iteration - /// limit (as opposed to a provider failure); selects the user hint. - last_partial_turn_limit: bool, - /// Whether the most recent partial turn was stopped by task-budget - /// exhaustion (issue #569); selects the user hint. - last_partial_turn_budget: bool, + /// Why the most recent partial turn stopped; selects the user hint and + /// the synthetic note written into the persisted context. + last_partial_turn_reason: PartialTurnStopReason, +} + +/// Why a partial turn stopped (issue #569 review: replaces the +/// `stopped_by_limit` / `stopped_by_budget` boolean pair, whose stale-flag +/// interactions made the persisted note and user hint diverge). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum PartialTurnStopReason { + /// Provider died mid-turn (#452) — partial evidence. + #[default] + ProviderError, + /// User chose to stop at the 20-round iteration limit (#572). + IterationLimit, + /// Task-level cumulative budget exhausted (#569). + BudgetExhausted, } impl AiHandler { @@ -223,8 +234,7 @@ impl AiHandler { auto_search, secret_redactor: None, last_partial_turn_steps: 0, - last_partial_turn_limit: false, - last_partial_turn_budget: false, + last_partial_turn_reason: PartialTurnStopReason::ProviderError, } } @@ -635,8 +645,7 @@ impl AiHandler { // Clear the stale partial-turn count so a later error path // never reports evidence from an already-completed turn. self.last_partial_turn_steps = 0; - self.last_partial_turn_limit = false; - self.last_partial_turn_budget = false; + self.last_partial_turn_reason = PartialTurnStopReason::ProviderError; result } Err(err) => { @@ -712,32 +721,39 @@ impl AiHandler { /// sees the evidence instead of dangling calls. fn commit_partial_turn(&mut self, question_processed: &str) { let partial = self.llm_session.take_last_partial_turn(); - let stopped_by_limit = false; - self.commit_partial_messages(question_processed, &partial, stopped_by_limit); + self.commit_partial_messages( + question_processed, + &partial, + PartialTurnStopReason::ProviderError, + ); } /// Issue #572 variant: commit a partial turn that ended because the - /// user chose to stop at the tool-call iteration limit. The synthetic - /// note and the user hint differ from the provider-failure case so the - /// persisted record is not misdescribed as a provider error. + /// user chose to stop at the tool-call iteration limit. fn commit_partial_turn_stopped_at_limit(&mut self, question_processed: &str) { let partial = self.llm_session.take_last_partial_turn(); - self.commit_partial_messages(question_processed, &partial, true); + self.commit_partial_messages( + question_processed, + &partial, + PartialTurnStopReason::IterationLimit, + ); } /// Issue #569 variant: commit a partial turn that ended because the - /// task-level cumulative budget was exhausted. Same evidence contract; - /// the note must say budget, not provider failure. + /// task-level cumulative budget was exhausted. fn commit_partial_turn_stopped_at_budget(&mut self, question_processed: &str) { let partial = self.llm_session.take_last_partial_turn(); - self.commit_partial_messages(question_processed, &partial, true); - self.last_partial_turn_budget = true; + self.commit_partial_messages( + question_processed, + &partial, + PartialTurnStopReason::BudgetExhausted, + ); } - /// Whether the most recent partial turn was stopped by task-budget - /// exhaustion (issue #569); used by the shell to pick the user hint. - pub fn last_partial_turn_stopped_at_budget(&self) -> bool { - self.last_partial_turn_budget + /// Why the most recent partial turn stopped; used by the shell to pick + /// the user hint. + pub fn last_partial_turn_reason(&self) -> PartialTurnStopReason { + self.last_partial_turn_reason } /// Read-only access to the task budget state for /token and the footer. pub fn llm_budget_state(&self) -> aish_llm::budget::TaskBudgetState { @@ -824,18 +840,19 @@ impl AiHandler { /// Shared commit path: append the user message, the completed /// tool-call/tool-result pairs (redacted), and a synthetic assistant - /// note describing why the turn stopped. `stopped_by_limit` selects the - /// note text and the user-hint state exposed to the shell. + /// note describing why the turn stopped. `reason` selects the note text + /// and the user-hint state exposed to the shell. The reason is recorded + /// unconditionally — including the empty-partial case — so the flag can + /// never go stale. fn commit_partial_messages( &mut self, question_processed: &str, partial: &[ChatMessage], - stopped_by_limit: bool, + reason: PartialTurnStopReason, ) { + self.last_partial_turn_reason = reason; if partial.is_empty() { self.last_partial_turn_steps = 0; - self.last_partial_turn_limit = false; - self.last_partial_turn_budget = false; return; } // The session prepends the user message; the shell commits it here @@ -846,7 +863,6 @@ impl AiHandler { partial }; self.last_partial_turn_steps = tool_messages.len(); - self.last_partial_turn_limit = stopped_by_limit; if tool_messages.is_empty() { return; @@ -861,22 +877,28 @@ impl AiHandler { // synthetic assistant note so the next turn's model understands why // the transcript stops mid-task — and that the evidence above was // already executed, provider failure or deliberate stop alike. - let note = if stopped_by_limit { - "[turn stopped by the user at the tool-call iteration limit; the tool results above are completed evidence — do not re-run completed side effects]" - } else { - "[turn interrupted by a provider error before completion; the tool results above are partial evidence — do not re-run completed side effects]" + let note = match reason { + PartialTurnStopReason::IterationLimit => { + "[turn stopped by the user at the tool-call iteration limit; the tool results above are completed evidence — do not re-run completed side effects]" + } + PartialTurnStopReason::BudgetExhausted => { + "[turn stopped because the task budget was exhausted; the tool results above are completed evidence — do not re-run completed side effects]" + } + PartialTurnStopReason::ProviderError => { + "[turn interrupted by a provider error before completion; the tool results above are partial evidence — do not re-run completed side effects]" + } }; self.context_manager .add_message("assistant", note, MemoryType::Llm); self.context_manager.trim(); - let reason = if stopped_by_limit { - "iteration-limit stop" - } else { - "provider failure" + let reason_str = match reason { + PartialTurnStopReason::IterationLimit => "iteration-limit stop", + PartialTurnStopReason::BudgetExhausted => "budget-exhausted stop", + PartialTurnStopReason::ProviderError => "provider failure", }; tracing::info!( messages = tool_messages.len(), - reason, + reason = reason_str, "Persisted partial-turn tool evidence" ); } @@ -888,13 +910,6 @@ impl AiHandler { self.last_partial_turn_steps } - /// Whether the most recent partial turn was stopped at the iteration - /// limit (issue #572) rather than by a provider failure; used by the - /// shell to pick the matching user hint. - pub fn last_partial_turn_stopped_at_limit(&self) -> bool { - self.last_partial_turn_limit - } - /// Convert a turn's intermediate ChatMessages into persistable /// ContextMessages, applying secret redaction and a size cap to tool /// results before they enter long-lived context. @@ -3200,10 +3215,17 @@ mod tests { ChatMessage::tool_result("call_l1", "build finished"), ]; - handler.commit_partial_messages("do the task", &partial, true); + handler.commit_partial_messages( + "do the task", + &partial, + PartialTurnStopReason::IterationLimit, + ); - // The flag the shell hint reads is set. - assert!(handler.last_partial_turn_stopped_at_limit()); + // The reason the shell hint reads is recorded. + assert_eq!( + handler.last_partial_turn_reason(), + PartialTurnStopReason::IterationLimit + ); assert_eq!(handler.last_partial_turn_step_count(), 2); let context = handler.build_context_messages(); @@ -3238,9 +3260,16 @@ mod tests { ChatMessage::tool_result("call_f1", "match found"), ]; - handler.commit_partial_messages("do the task", &partial, false); + handler.commit_partial_messages( + "do the task", + &partial, + PartialTurnStopReason::ProviderError, + ); - assert!(!handler.last_partial_turn_stopped_at_limit()); + assert_eq!( + handler.last_partial_turn_reason(), + PartialTurnStopReason::ProviderError + ); let context = handler.build_context_messages(); let note = context .last() @@ -3249,6 +3278,61 @@ mod tests { assert!(note.contains("provider error"), "{note}"); } + /// Issue #569 review C3: a budget-exhausted stop must write the + /// budget-specific note into the persisted context — never the + /// iteration-limit wording. + #[test] + fn commit_partial_messages_uses_budget_note_on_budget_stop() { + let mut handler = test_handler(); + + let partial = [ + ChatMessage::user("do the task"), + { + let mut a = ChatMessage::assistant(""); + a.content = None; + a.tool_calls = Some(vec![aish_llm::ToolCall { + id: "call_b1".to_string(), + name: "bash".to_string(), + arguments: "{}".to_string(), + }]); + a + }, + ChatMessage::tool_result("call_b1", "done"), + ]; + + handler.commit_partial_messages( + "do the task", + &partial, + PartialTurnStopReason::BudgetExhausted, + ); + + assert_eq!( + handler.last_partial_turn_reason(), + PartialTurnStopReason::BudgetExhausted + ); + let context = handler.build_context_messages(); + let note = context + .last() + .and_then(|m| m.text_content()) + .unwrap_or_default(); + assert!(note.contains("task budget was exhausted"), "{note}"); + assert!(!note.contains("iteration limit"), "{note}"); + assert!(!note.contains("provider error"), "{note}"); + } + + /// Review C3: an empty partial must still record the stop reason + /// (steps reset to 0), so the shell hint never goes stale. + #[test] + fn commit_partial_messages_records_reason_even_when_partial_is_empty() { + let mut handler = test_handler(); + handler.commit_partial_messages("do the task", &[], PartialTurnStopReason::BudgetExhausted); + assert_eq!( + handler.last_partial_turn_reason(), + PartialTurnStopReason::BudgetExhausted + ); + assert_eq!(handler.last_partial_turn_step_count(), 0); + } + /// Bind a loopback server that replies to POST /v1/chat/completions with /// a canned OpenAI JSON completion (401 -> immediate, non-retryable /// failure; used to drive the manual-compact failure path fast). diff --git a/crates/aish-shell/src/app.rs b/crates/aish-shell/src/app.rs index 2b9c9499..de5aff07 100644 --- a/crates/aish-shell/src/app.rs +++ b/crates/aish-shell/src/app.rs @@ -3180,11 +3180,13 @@ impl AishShell { println!("{}", theme::warning(&t("shell.interrupted"))); } Err(e) => { - if !matches!( + if let aish_core::AishError::BudgetExhausted(dim) = &e { + // Issue #569 review C4: the main path must + // also offer the audited adjustment panel. + self.handle_budget_exhaustion(dim); + } else if !matches!( e, - aish_core::AishError::Llm(_) - | aish_core::AishError::IterationLimit - | aish_core::AishError::BudgetExhausted(_) + aish_core::AishError::Llm(_) | aish_core::AishError::IterationLimit ) { let msg = t("shell.error.llm_error_message") .replace("{error}", &e.to_string()); @@ -3379,11 +3381,13 @@ impl AishShell { println!("{}", theme::warning(&t("shell.interrupted"))); } Err(e) => { - if !matches!( + if let aish_core::AishError::BudgetExhausted(dim) = &e { + // Issue #569 review C4. + self.handle_budget_exhaustion(dim); + } else if !matches!( e, aish_core::AishError::Llm(_) | aish_core::AishError::IterationLimit - | aish_core::AishError::BudgetExhausted(_) ) { let msg = t("shell.error.llm_error_message") .replace("{error}", &e.to_string()); @@ -3760,11 +3764,12 @@ impl AishShell { println!("{}", theme::warning(&t("shell.interrupted"))); } Err(e) => { - if !matches!( + if let aish_core::AishError::BudgetExhausted(dim) = &e { + // Issue #569 review C4. + self.handle_budget_exhaustion(dim); + } else if !matches!( e, - aish_core::AishError::Llm(_) - | aish_core::AishError::IterationLimit - | aish_core::AishError::BudgetExhausted(_) + aish_core::AishError::Llm(_) | aish_core::AishError::IterationLimit ) { let msg = t("shell.error.llm_error_message") .replace("{error}", &e.to_string()); @@ -3854,15 +3859,18 @@ impl AishShell { if saved_steps == 0 { return false; } - // Issue #572: a deliberate iteration-limit stop is not a failure; - // issue #569: neither is a task-budget stop. Use the matching hint - // so the saved evidence is not misdescribed. - let key = if self.ai_handler.last_partial_turn_stopped_at_budget() { - "shell.error.partial_turn_saved_budget" - } else if self.ai_handler.last_partial_turn_stopped_at_limit() { - "shell.error.partial_turn_saved_limit" - } else { - "shell.error.partial_turn_saved" + // Issue #572/#569: deliberate stops are not failures; use the + // matching hint so the saved evidence is not misdescribed. + let key = match self.ai_handler.last_partial_turn_reason() { + crate::ai_handler::PartialTurnStopReason::BudgetExhausted => { + "shell.error.partial_turn_saved_budget" + } + crate::ai_handler::PartialTurnStopReason::IterationLimit => { + "shell.error.partial_turn_saved_limit" + } + crate::ai_handler::PartialTurnStopReason::ProviderError => { + "shell.error.partial_turn_saved" + } }; let msg = t(key).replace("{steps}", &saved_steps.to_string()); println!("{}", theme::warning(&msg)); @@ -3930,8 +3938,11 @@ impl AishShell { let mut new_limit_raw = String::new(); if std::io::stdin().read_line(&mut new_limit_raw).is_ok() { let parsed = new_limit_raw.trim().parse::().ok(); - let applied = parsed.filter(|v| *v > used).map(|v| (v, used)); - if let Some((new_limit, old_used)) = applied { + let applied = parsed.filter(|v| *v > used); + if let Some(new_limit) = applied { + // Review C5: the audit record must capture the LIMIT's + // before/after values, not the usage. + let old_limit = bound.map_or_else(|| "unbounded".to_string(), |b| b.to_string()); self.raise_budget_limit(dimension, new_limit); // Audit the explicit adjustment with before/after values. if let Some(ref audit) = self.audit_store { @@ -3940,7 +3951,10 @@ impl AishShell { Some(self.session_uuid.clone()), self.audit_user.clone(), self.audit_host.clone(), - format!("budget {} {} -> {}", dimension, old_used, new_limit), + format!( + "budget {} limit {} -> {} (used {})", + dimension, old_limit, new_limit, used + ), "budget".to_string(), 0, ); From d3c526e220d52caec5adf52dbc55dc8f4ac1c845 Mon Sep 17 00:00:00 2001 From: xuezhizone Date: Wed, 30 Sep 2026 13:45:58 +0800 Subject: [PATCH 3/4] fix(llm): pause parent clock and propagate remaining budget to spawned loops (issue #569 review) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit C6 (minor): the parent's wall-clock anchor kept running while spawn awaited the child loop, and the merge then added the child's own active_secs — double-counting the child interval. spawn now settles and pauses the parent clock before starting the child and resumes it after the merge (only when it was active). C7 (major): create_subsession reset both budget counters and limits, so a child loop ran without any budget admission check and could burn past the parent's cumulative cap. spawn now derives the child's limits from the parent's remaining budget (max minus used per dimension); child counters stay zero-based and fold into the parent on completion. Tests: child sees remaining limits via the configure hook; parent clock pauses before the child loop and resumes after the merge. aish-llm 351 tests pass. --- crates/aish-llm/src/agents/spawn.rs | 94 ++++++++++++++++++++++++++++- 1 file changed, 93 insertions(+), 1 deletion(-) diff --git a/crates/aish-llm/src/agents/spawn.rs b/crates/aish-llm/src/agents/spawn.rs index 3d863aa8..99a82c6b 100644 --- a/crates/aish-llm/src/agents/spawn.rs +++ b/crates/aish-llm/src/agents/spawn.rs @@ -93,7 +93,38 @@ pub async fn spawn( where F: FnOnce(&mut LlmSession), { + // Issue #569 review C6: pause the parent's wall-clock while the child + // runs. The child accrues its own active_secs (merged below); leaving + // the parent anchor running would double-count the child interval. + let parent_clock_active = parent.task_budget_state().last_accounted.is_some(); + if parent_clock_active { + parent.flush_task_budget_wall_clock(); // settle parent time so far + parent.task_budget_state_mut().last_accounted = None; + } + + // Issue #569 review C7: the child inherits the parent's REMAINING + // budget. Its counters stay zero-based — the merge below folds the + // delta into the parent — but the limits travel with it, so a spawned + // loop cannot burn past what the parent's task still allows. + let parent_limit = parent.task_budget_limit(); + let parent_state = parent.task_budget_state(); + let remaining_limits = crate::budget::TaskBudgetLimit { + max_rounds: parent_limit + .max_rounds + .map(|max| max.saturating_sub(parent_state.task_rounds)), + max_tool_calls: parent_limit + .max_tool_calls + .map(|max| max.saturating_sub(parent_state.tool_calls)), + max_tokens: parent_limit + .max_tokens + .map(|max| max.saturating_sub(parent_state.token_usage.total_tokens())), + max_duration_secs: parent_limit + .max_duration_secs + .map(|max| max.saturating_sub(parent_state.active_secs)), + }; + let mut sub = parent.create_subsession(); + sub.set_task_budget_limit(remaining_limits); configure(&mut sub); let parent_cancel = parent.cancellation_token_arc(); @@ -125,7 +156,11 @@ where // because the merge runs in the single completion path here. sub.flush_task_budget_wall_clock(); parent.merge_task_budget(&sub.task_budget_state()); - + // Review C6: resume the parent clock so its own post-spawn rounds + // keep accruing. + if parent_clock_active { + parent.task_budget_state_mut().start_turn(); + } SpawnResult { text: outcome.text, status: outcome.status, @@ -391,6 +426,63 @@ mod tests { .any(|s| s.function.name == "read_file")); } + #[tokio::test] + async fn test_spawn_child_inherits_remaining_budget() { + // Review C7: the child loop must see the parent's REMAINING limits + // (max minus used), so a near-exhausted task cannot burn past its + // cap through a spawned agent. The configure hook runs after the + // inheritance, so it observes the propagated limits. + let mut parent = LlmSession::new("http://localhost", "key", "model", None, None); + parent.register_tool(Box::new(MockTool::new("grep"))); + parent.set_task_budget_limit(crate::budget::TaskBudgetLimit { + max_rounds: Some(10), + max_tool_calls: Some(20), + max_tokens: None, + max_duration_secs: None, + }); + parent.merge_task_budget(&crate::budget::TaskBudgetState { + task_rounds: 3, + tool_calls: 8, + ..Default::default() + }); + + let registry = AgentRegistry::builtin(); + let _ = spawn_builtin(&parent, ®istry, "explore", "task", |sub, _specs| { + configure_spawn_test(sub, vec![Ok(mock_text_response("ok"))]); + let limit = sub.task_budget_limit(); + assert_eq!(limit.max_rounds, Some(7), "remaining rounds (10 - 3)"); + assert_eq!(limit.max_tool_calls, Some(12), "remaining calls (20 - 8)"); + }) + .await + .expect("spawn_builtin should succeed"); + } + + #[tokio::test] + async fn test_spawn_pauses_and_resumes_parent_clock() { + // Review C6: the parent clock pauses while the child runs (no + // double counting) and resumes after the merge. + let mut parent = LlmSession::new("http://localhost", "key", "model", None, None); + parent.register_tool(Box::new(MockTool::new("grep"))); + parent.task_budget_state_mut().start_turn(); + + let registry = AgentRegistry::builtin(); + let _ = spawn_builtin(&parent, ®istry, "explore", "task", |sub, _specs| { + configure_spawn_test(sub, vec![Ok(mock_text_response("ok"))]); + assert!( + sub.task_budget_state().last_accounted.is_none(), + "child clock must not start before its loop begins" + ); + }) + .await + .expect("spawn_builtin should succeed"); + + let state = parent.task_budget_state(); + assert!( + state.last_accounted.is_some(), + "parent clock must resume after the merge" + ); + } + #[tokio::test] async fn test_spawn_folds_subsession_budget_into_parent_task() { // Issue #569 acceptance 3: sub-agent rounds/tool-calls/tokens fold From 44090802ba2e7d3b1d7b8f2c0a8e60d023741c63 Mon Sep 17 00:00:00 2001 From: xuezhizone Date: Wed, 30 Sep 2026 14:09:17 +0800 Subject: [PATCH 4/4] fix(llm): enforce inherited budget in spawned loops before each request (issue #569 review) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The child loop accrued counters but never checked the inherited remaining limits — an exhausted budget still allowed every remaining child turn, and the excess usage merged only after the child finished. run_tool_loop_until_done now runs the same admission check as process_input at the loop top: settle the wall-clock, evaluate all dimensions against the inherited limits, and return a fatal LoopOutcome carrying AishError::BudgetExhausted before incrementing the round or issuing a model request. The parent merges the usage and stops cleanly on its own next check. Test: a child inheriting zero remaining rounds stops before its first request and surfaces error_category budget_exhausted. --- crates/aish-llm/src/agents/spawn.rs | 36 +++++++++++++++++++++++++ crates/aish-llm/src/agents/tool_loop.rs | 20 ++++++++++++++ 2 files changed, 56 insertions(+) diff --git a/crates/aish-llm/src/agents/spawn.rs b/crates/aish-llm/src/agents/spawn.rs index 99a82c6b..25a3b17e 100644 --- a/crates/aish-llm/src/agents/spawn.rs +++ b/crates/aish-llm/src/agents/spawn.rs @@ -426,6 +426,42 @@ mod tests { .any(|s| s.function.name == "read_file")); } + #[tokio::test] + async fn test_spawn_child_budget_exhaustion_returns_before_request() { + // Review C8: the child loop must enforce the inherited remaining + // budget BEFORE any model request; exhaustion surfaces as a fatal + // outcome carrying BudgetExhausted, and the parent merges the + // usage without issuing further requests. + let mut parent = LlmSession::new("http://localhost", "key", "model", None, None); + parent.register_tool(Box::new(MockTool::new("grep"))); + // Parent already consumed its whole remaining budget: the child + // inherits 0 remaining rounds and must stop immediately. + parent.set_task_budget_limit(crate::budget::TaskBudgetLimit { + max_rounds: Some(2), + max_tool_calls: None, + max_tokens: None, + max_duration_secs: None, + }); + parent.merge_task_budget(&crate::budget::TaskBudgetState { + task_rounds: 2, + ..Default::default() + }); + + let registry = AgentRegistry::builtin(); + let result = spawn_builtin(&parent, ®istry, "explore", "task", |sub, _specs| { + configure_spawn_test(sub, vec![Ok(mock_text_response("never reached"))]); + }) + .await + .expect("spawn_builtin returns a result, not an error"); + + assert_eq!(result.status, LoopStatus::Fatal); + assert_eq!( + result.error_category.as_deref(), + Some("budget_exhausted"), + "child exhaustion must surface as BudgetExhausted" + ); + } + #[tokio::test] async fn test_spawn_child_inherits_remaining_budget() { // Review C7: the child loop must see the parent's REMAINING limits diff --git a/crates/aish-llm/src/agents/tool_loop.rs b/crates/aish-llm/src/agents/tool_loop.rs index b3ac30fc..2c9bc92e 100644 --- a/crates/aish-llm/src/agents/tool_loop.rs +++ b/crates/aish-llm/src/agents/tool_loop.rs @@ -132,6 +132,26 @@ pub async fn run_tool_loop_until_done( ); return LoopOutcome::from_spawn_outcome(outcome, loop_messages, None); } + + // Task budget admission check (issue #569 review): the child loop + // enforces the inherited remaining limits BEFORE any model request + // — same contract as process_input. Exhaustion returns a fatal + // outcome carrying BudgetExhausted so the parent's merge folds the + // usage and the parent can stop cleanly. + { + let limit = session.task_budget_limit(); + let state = { + let mut state = session.task_budget_state_mut(); + // Settle the previous round's wall-clock before the check. + state.flush_wall_clock(); + state.clone() + }; + if let Some(dim) = limit.check(&state).exhausted { + let err = AishError::BudgetExhausted(dim.key().to_string()); + return LoopOutcome::fatal_with_messages(err, loop_messages); + } + } + iterations += 1; // Task budget accounting (issue #569): sub-agent loops accrue the // same counters as the main loop. Wall-clock stays anchored from