diff --git a/CONFIGURATION.md b/CONFIGURATION.md index 2bbc840c..0457652d 100644 --- a/CONFIGURATION.md +++ b/CONFIGURATION.md @@ -70,6 +70,7 @@ aish run --config ~/work/ai-shell-config.yaml | `context_token_budget` | integer/null | `null` | 可选的上下文 token 预算限制 | 如:4000,为 null 则仅使用消息数量限制 | | `enable_token_estimation` | boolean | `true` | 启用基于 tiktoken 的 token 估算 | true/false | | `context_auto_compact` | object | 见下方 | 自动上下文压缩配置,面向 shell/tool 输出控量 | 可选 | +| `task_budget` | object | 见下方 | 任务级累计预算(issue #569),任一维度达到上限即停止循环并提示(继续不会重置;调整预算需显式确认并写入审计) | 各维度 integer/null | ### 工具输出配置 @@ -203,6 +204,14 @@ max_llm_messages: 50 max_shell_messages: 20 context_token_budget: null enable_token_estimation: true + +# 任务级累计预算(可选;缺省不限,仅计数) +task_budget: + max_rounds: null # 累计工具循环轮次 + max_tool_calls: null # 累计工具调用次数 + max_tokens: null # 累计 token(输入+输出,不含缓存命中) + max_duration_secs: null # 累计活跃时长(秒) + context_auto_compact: enabled: true full_compact_enabled: true diff --git a/crates/aish-config/src/lib.rs b/crates/aish-config/src/lib.rs index a18818b1..9d936595 100644 --- a/crates/aish-config/src/lib.rs +++ b/crates/aish-config/src/lib.rs @@ -19,5 +19,5 @@ pub mod model; pub use loader::ConfigLoader; pub use model::{ compile_remote_danger_patterns, ApiAccountConfig, ConfigModel, InlineCompletionConfig, - MemoryConfig, OutputOffloadConfig, RegistrySource, SkillsConfig, ToolArgPreviewConfig, + MemoryConfig, OutputOffloadConfig, RegistrySource, SkillsConfig, TaskBudgetConfig, }; diff --git a/crates/aish-config/src/model.rs b/crates/aish-config/src/model.rs index 788120d1..ebcf4233 100644 --- a/crates/aish-config/src/model.rs +++ b/crates/aish-config/src/model.rs @@ -194,6 +194,37 @@ impl Default for ContextAutoCompactConfig { } } +// --------------------------------------------------------------------------- +// Task budget sub-config (issue #569) +// --------------------------------------------------------------------------- + +/// Task-level cumulative budget bounds. All dimensions are optional and +/// default to unlimited; the runtime still accrues counters for display. +/// A tripped budget stops the loop with `BudgetExhausted`; raising a limit +/// is an explicit, audit-logged action — never implied by "continue". +#[derive(Debug, Clone, Serialize, Deserialize, Default, PartialEq)] +#[serde(default)] +pub struct TaskBudgetConfig { + /// Cumulative tool-loop rounds for the whole task. + pub max_rounds: Option, + /// Cumulative tool executions (parallel calls each count). + pub max_tool_calls: Option, + /// Cumulative fresh tokens (input + output; cached reads excluded). + pub max_tokens: Option, + /// Cumulative active wall-clock seconds (turn time only, no idle). + pub max_duration_secs: Option, +} + +impl TaskBudgetConfig { + /// True when any dimension is bounded. + pub fn is_bounded(&self) -> bool { + self.max_rounds.is_some() + || self.max_tool_calls.is_some() + || self.max_tokens.is_some() + || self.max_duration_secs.is_some() + } +} + // --------------------------------------------------------------------------- // Inline AI completion sub-config // --------------------------------------------------------------------------- @@ -483,6 +514,11 @@ pub struct ConfigModel { #[serde(default)] pub context_token_budget: Option, + /// Task-level cumulative budget (issue #569). Every dimension is + /// optional; absent = unlimited (counters still accrue for /token). + #[serde(default)] + pub task_budget: TaskBudgetConfig, + /// Enable tiktoken-based token estimation for context trimming #[serde(default = "default_true")] pub enable_token_estimation: bool, @@ -582,6 +618,7 @@ impl Default for ConfigModel { context_token_budget: None, enable_token_estimation: default_true(), context_auto_compact: ContextAutoCompactConfig::default(), + task_budget: TaskBudgetConfig::default(), enable_scripts: default_true(), history_size: default_history_size(), terminal_resize_mode: default_terminal_resize_mode(), @@ -1018,4 +1055,27 @@ inline_completion: assert_eq!(m.inline_completion.debounce_ms, 400); assert_eq!(m.inline_completion.max_tokens, 512); } + + #[test] + fn task_budget_config_parses_from_yaml() { + let yaml = r#" +max_llm_messages: 30 +task_budget: + max_rounds: 200 + max_tokens: 2000000 +"#; + let config: ConfigModel = serde_yaml::from_str(yaml).unwrap(); + assert!(config.task_budget.is_bounded()); + assert_eq!(config.task_budget.max_rounds, Some(200)); + assert_eq!(config.task_budget.max_tokens, Some(2_000_000)); + assert_eq!(config.task_budget.max_tool_calls, None); + assert_eq!(config.task_budget.max_duration_secs, None); + } + + #[test] + fn task_budget_config_defaults_to_unbounded() { + let config = ConfigModel::default(); + assert!(!config.task_budget.is_bounded()); + assert_eq!(config.task_budget.max_rounds, None); + } } diff --git a/crates/aish-core/src/error.rs b/crates/aish-core/src/error.rs index eaceed96..367cfbf5 100644 --- a/crates/aish-core/src/error.rs +++ b/crates/aish-core/src/error.rs @@ -45,6 +45,11 @@ pub enum AishError { #[error("tool loop stopped at the iteration limit")] IterationLimit, + /// Task-level cumulative budget exhausted (issue #569). The field names + /// the dimension that tripped (rounds / tool_calls / tokens / duration). + #[error("task budget exhausted: {0}")] + BudgetExhausted(String), + #[error("operation timed out")] Timeout, } @@ -74,6 +79,7 @@ impl AishError { AishError::Shell(_) => "shell", AishError::Cancelled => "cancelled", AishError::IterationLimit => "iteration_limit", + AishError::BudgetExhausted(_) => "budget_exhausted", AishError::Parse(_) => "parse", AishError::Timeout => "timeout", } diff --git a/crates/aish-i18n/locales/de-DE.yaml b/crates/aish-i18n/locales/de-DE.yaml index 2bda0b36..3baa1da0 100644 --- a/crates/aish-i18n/locales/de-DE.yaml +++ b/crates/aish-i18n/locales/de-DE.yaml @@ -332,6 +332,15 @@ shell: no_evidence: "(nicht angegeben)" nl_detection: confirm_ask_ai: "Das sieht nach einer natürlichen Sprachfrage aus. KI fragen? (Y/n)" + budget: + exhausted_title: "⛔ Aufgabenbudget erschoepft" + exhausted_detail: "Dimension: {dimension} · Genutzt: {used} · Limit: {limit}" + exhausted_options: "Budget anpassen und fortfahren? Anpassungen werden auditiert (y=anpassen / n=stoppen)" + stopped_by_user: "Durch Budget gestoppt. Abgeschlossene Werkzeugschritte sind gespeichert; fortsetzen vom gespeicherten Punkt." + new_limit_prompt: "Neues Limit fuer diese Dimension eingeben:" + raised: "Budget angepasst: {dimension} -> {value}. Aufgabe erneut ausgeben, um fortzufahren." + invalid_limit: "Ungueltiger Wert (muss ueber der Nutzung liegen); Budget unveraendert." + error: load_script_failed: "Skript '{script}' konnte nicht geladen werden: {error}" pty_error: "PTY-Fehler: {error}" @@ -343,6 +352,7 @@ shell: llm_error_message: "Fehler: {error}" partial_turn_saved: "Die abgeschlossenen Werkzeugschritte dieses Durchgangs wurden gespeichert ({steps} Schritte). Die Arbeit vor dem Fehler geht nicht verloren; geben Sie eine neue Aufgabe ein oder ;retry, um ab dem gespeicherten Punkt fortzufahren." partial_turn_saved_limit: "Die abgeschlossenen Werkzeugschritte dieses Durchgangs wurden vor dem Stopp am Iterationslimit gespeichert ({steps} Schritte). Die Arbeit bleibt erhalten; geben Sie eine neue Aufgabe ein oder ;retry, um ab dem gespeicherten Punkt fortzufahren." + partial_turn_saved_budget: "Die abgeschlossenen Werkzeugschritte dieses Durchgangs wurden vor dem Stopp am Aufgabenbudgetlimit gespeichert ({steps} Schritte). Die Arbeit bleibt erhalten; geben Sie eine neue Aufgabe ein oder ;retry, um vom gespeicherten Punkt fortzufahren." llm_error_details_title: "Debug-Details" tool_args_json_invalid_hint: "❌ Anfrage fehlgeschlagen, bitte erneut versuchen." prompt: @@ -446,6 +456,13 @@ shell: output_tokens: "Ausgabe-Tokens" total: "Gesamt" api_calls: "API-Aufrufe" + budget_title: "Aufgabenbudget" + budget_rounds: "Runden (genutzt/Limit):" + budget_tool_calls: "Werkzeugaufrufe (genutzt/Limit):" + budget_tokens: "Token (genutzt/Limit):" + budget_duration: "Aktive Sekunden (genutzt/Limit):" + budget_cost: "Kosten:" + budget_cost_unavailable: "nicht verfuegbar (Anbieter melden keine Preise)" setup: no_config_manager: "Neukonfiguration nicht möglich: Konfigurationsmanager nicht verfügbar" cancelled: "Einrichtung abgebrochen; aktuelle Einstellungen werden beibehalten" @@ -559,6 +576,9 @@ shell: inline_enforce_json: label: "Inline Enforce JSON" desc: "Inline-Completion zwingen, striktes JSON zurückzugeben" + task_budget_max_rounds: + label: "Aufgabenbudget-Rundenlimit" + desc: "Kumulierte Werkzeugschleifen-Runden pro Aufgabe; leer = unbegrenzt (Weitermachen setzt nicht zurueck)" max_llm_messages: label: "Max LLM-Messages" desc: "Maximale Konversations-Turns im LLM-Verlauf" diff --git a/crates/aish-i18n/locales/en-US.yaml b/crates/aish-i18n/locales/en-US.yaml index e8c9ec46..0432eada 100644 --- a/crates/aish-i18n/locales/en-US.yaml +++ b/crates/aish-i18n/locales/en-US.yaml @@ -768,6 +768,15 @@ shell: nl_detection: confirm_ask_ai: "This looks like a natural language question. Ask AI? (Y/n)" + budget: + exhausted_title: "⛔ Task budget exhausted" + exhausted_detail: "Dimension: {dimension} · Used: {used} · Limit: {limit}" + exhausted_options: "Adjust the budget and continue? Adjustments are audit-logged (y=adjust / n=stop)" + stopped_by_user: "Stopped by budget. Completed tool evidence is saved; resume from the saved point later." + new_limit_prompt: "Enter the new limit for this dimension:" + raised: "Budget adjusted: {dimension} -> {value}. Re-issue the task to continue from the saved point." + invalid_limit: "Invalid value (must exceed usage); budget unchanged." + error: execution_error: "❌ Error during execution: {error}" pty_decode_error: "Caught PTY output decode error: {error}" @@ -779,6 +788,7 @@ shell: llm_error_message: "Error: {error}" partial_turn_saved: "Saved the completed tool steps of this turn ({steps} steps). Work done before the failure is not lost; enter a new task or ;retry to continue from the saved point." partial_turn_saved_limit: "Saved the completed tool steps of this turn ({steps} steps) before stopping at the iteration limit. The work is kept; enter a new task or ;retry to continue from the saved point." + partial_turn_saved_budget: "Saved the completed tool steps of this turn ({steps} steps) before stopping at the task budget limit. The work is kept; enter a new task or ;retry to continue from the saved point." llm_error_details_title: "Debug Details" tool_args_json_invalid_hint: "❌ Request failed, please retry." @@ -888,6 +898,13 @@ shell: output_tokens: "Output tokens" total: "Total" api_calls: "API calls" + budget_title: "Task budget" + budget_rounds: "Rounds (used/limit):" + budget_tool_calls: "Tool calls (used/limit):" + budget_tokens: "Tokens (used/limit):" + budget_duration: "Active seconds (used/limit):" + budget_cost: "Cost:" + budget_cost_unavailable: "unavailable (providers do not report pricing)" setup: no_config_manager: "Cannot reconfigure: configuration manager unavailable" @@ -983,6 +1000,9 @@ shell: inline_completion: label: "Inline AI Completion" desc: "Gray ghost-text AI suggestions while typing an AI prompt" + task_budget_max_rounds: + label: "Task Budget Round Limit" + desc: "Cumulative tool-loop rounds per task; blank = unlimited (continues never reset it)" max_llm_messages: label: "Max AI Messages" desc: "How many AI conversation turns to keep in context" diff --git a/crates/aish-i18n/locales/es-ES.yaml b/crates/aish-i18n/locales/es-ES.yaml index 0c1df33a..1d9cabec 100644 --- a/crates/aish-i18n/locales/es-ES.yaml +++ b/crates/aish-i18n/locales/es-ES.yaml @@ -332,6 +332,15 @@ shell: no_evidence: "(no proporcionada)" nl_detection: confirm_ask_ai: "Esto parece una pregunta en lenguaje natural. ¿Preguntar a IA? (Y/n)" + budget: + exhausted_title: "⛔ Presupuesto de tarea agotado" + exhausted_detail: "Dimensión: {dimension} · Usado: {used} · Límite: {limit}" + exhausted_options: "¿Ajustar el presupuesto y continuar? Los ajustes se registran en la auditoría (y=ajustar / n=detener)" + stopped_by_user: "Detenido por presupuesto. La evidencia de herramientas completadas está guardada; reanude desde el punto guardado." + new_limit_prompt: "Introduzca el nuevo límite para esta dimensión:" + raised: "Presupuesto ajustado: {dimension} -> {value}. Emita la tarea de nuevo para continuar." + invalid_limit: "Valor inválido (debe superar el uso); presupuesto sin cambios." + error: load_script_failed: "No se pudo cargar el script '{script}': {error}" pty_error: "Error de PTY: {error}" @@ -343,6 +352,7 @@ shell: llm_error_message: "Error: {error}" partial_turn_saved: "Se guardaron los pasos de herramienta completados de este turno ({steps} pasos). El trabajo previo al error no se pierde; escriba una nueva tarea o ;retry para continuar desde el punto guardado." partial_turn_saved_limit: "Se guardaron los pasos de herramienta completados de este turno ({steps} pasos) antes de detenerse en el límite de iteraciones. El trabajo se conserva; escriba una nueva tarea o ;retry para continuar desde el punto guardado." + partial_turn_saved_budget: "Se guardaron los pasos de herramientas completados de este turno ({steps} pasos) antes de detenerse en el límite de presupuesto de la tarea. El trabajo se conserva; escriba una nueva tarea o ;retry para continuar desde el punto guardado." llm_error_details_title: "Detalles de depuración" tool_args_json_invalid_hint: "❌ La solicitud falló; inténtalo de nuevo." prompt: @@ -446,6 +456,13 @@ shell: output_tokens: "Tokens de salida" total: "Total" api_calls: "Llamadas API" + budget_title: "Presupuesto de tarea" + budget_rounds: "Rondas (usado/límite):" + budget_tool_calls: "Llamadas de herramientas (usado/límite):" + budget_tokens: "Tokens (usado/límite):" + budget_duration: "Segundos activos (usado/límite):" + budget_cost: "Costo:" + budget_cost_unavailable: "no disponible (los proveedores no informan precios)" setup: no_config_manager: "No se puede reconfigurar: el gestor de configuración no está disponible" cancelled: "Configuración cancelada; se conservan los ajustes actuales" @@ -559,6 +576,9 @@ shell: inline_enforce_json: label: "Forzar JSON en línea" desc: "Forzar que el completado en línea devuelva JSON estricto" + task_budget_max_rounds: + label: "Límite de rondas del presupuesto" + desc: "Rondas acumuladas del bucle de herramientas por tarea; vacío = ilimitado (continuar nunca lo restablece)" max_llm_messages: label: "Mensajes LLM máx." desc: "Máximo de turnos de conversación conservados en el historial del LLM" diff --git a/crates/aish-i18n/locales/fr-FR.yaml b/crates/aish-i18n/locales/fr-FR.yaml index 1344c93a..13c0c440 100644 --- a/crates/aish-i18n/locales/fr-FR.yaml +++ b/crates/aish-i18n/locales/fr-FR.yaml @@ -332,6 +332,15 @@ shell: no_evidence: "(non fournie)" nl_detection: confirm_ask_ai: "Cela ressemble à une question en langage naturel. Demander à l'IA ? (Y/n)" + budget: + exhausted_title: "⛔ Budget de tache epuise" + exhausted_detail: "Dimension : {dimension} · Utilise : {used} · Limite : {limit}" + exhausted_options: "Ajuster le budget et continuer ? Les ajustements sont journalises (y=ajuster / n=arreter)" + stopped_by_user: "Arrete par budget. Les preuves d'outils completes sont sauvegardees ; reprenez depuis le point sauvegarde." + new_limit_prompt: "Saisissez la nouvelle limite pour cette dimension :" + raised: "Budget ajuste : {dimension} -> {value}. Relancez la tache pour continuer." + invalid_limit: "Valeur invalide (doit depasser l'utilisation) ; budget inchange." + error: load_script_failed: "Échec du chargement du script '{script}' : {error}" pty_error: "Erreur PTY : {error}" @@ -343,6 +352,7 @@ shell: llm_error_message: "Erreur : {error}" partial_turn_saved: "Les etapes d'outils completees de ce tour ont ete sauvegardees ({steps} etapes). Le travail precedent l'erreur n'est pas perdu ; saisissez une nouvelle tache ou ;retry pour reprendre depuis le point sauvegarde." partial_turn_saved_limit: "Les etapes d'outils completees de ce tour ont ete sauvegardees ({steps} etapes) avant l'arret a la limite d'iterations. Le travail est conserve ; saisissez une nouvelle tache ou ;retry pour reprendre depuis le point sauvegarde." + partial_turn_saved_budget: "Les etapes d'outils completees de ce tour ont ete sauvegardees ({steps} etapes) avant l'arret a la limite de budget de la tache. Le travail est conserve ; saisissez une nouvelle tache ou ;retry pour continuer depuis le point sauvegarde." llm_error_details_title: "Details de debogage" tool_args_json_invalid_hint: "❌ La requete a echoue, veuillez reessayer." prompt: @@ -446,6 +456,13 @@ shell: output_tokens: "Tokens de sortie" total: "Total" api_calls: "Appels API" + budget_title: "Budget de tache" + budget_rounds: "Tours (utilise/limite):" + budget_tool_calls: "Appels d'outils (utilise/limite):" + budget_tokens: "Tokens (utilise/limite):" + budget_duration: "Secondes actives (utilise/limite):" + budget_cost: "Cout:" + budget_cost_unavailable: "indisponible (les fournisseurs ne rapportent pas de prix)" setup: no_config_manager: "Reconfiguration impossible : gestionnaire de configuration indisponible" cancelled: "Configuration annulee, les parametres actuels sont conserves" @@ -559,6 +576,9 @@ shell: inline_enforce_json: label: "Forcer JSON en ligne" desc: "Forcer la complétion en ligne à renvoyer du JSON strict" + task_budget_max_rounds: + label: "Limite de tours du budget" + desc: "Tours cumules de la boucle d'outils par tache ; vide = illimite (continuer ne reinitialise jamais)" max_llm_messages: label: "Messages LLM max" desc: "Tours de conversation maximum conservés dans l'historique LLM" diff --git a/crates/aish-i18n/locales/ja-JP.yaml b/crates/aish-i18n/locales/ja-JP.yaml index a302ac10..f7caf130 100644 --- a/crates/aish-i18n/locales/ja-JP.yaml +++ b/crates/aish-i18n/locales/ja-JP.yaml @@ -332,6 +332,15 @@ shell: no_evidence: "(未提供)" nl_detection: confirm_ask_ai: "自然言語の質問のようです。AIに聞きますか?(Y/n)" + budget: + exhausted_title: "⛔ タスク予算が尽きました" + exhausted_detail: "次元: {dimension} · 使用: {used} · 上限: {limit}" + exhausted_options: "予算を調整して続行しますか?調整は監査ログに記録されます(y=調整 / n=停止)" + stopped_by_user: "予算により停止。完了したツール実行記録は保存されています。保存ポイントから再開できます。" + new_limit_prompt: "この次元の新しい上限を入力してください:" + raised: "予算を調整しました: {dimension} -> {value}。タスクを再度実行して続行できます。" + invalid_limit: "無効な値(使用量を超える必要)。予算は変更されません。" + error: load_script_failed: "スクリプト '{script}' の読み込みに失敗: {error}" pty_error: "PTY エラー: {error}" @@ -343,6 +352,7 @@ shell: llm_error_message: "エラー: {error}" partial_turn_saved: "このターンで完了したツール実行ステップを保存しました({steps} ステップ)。失敗前の作業は失われません。新しいタスクを入力するか ;retry で保存ポイントから続行できます。" partial_turn_saved_limit: "イテレーション上限で停止する前に、このターンで完了したツール実行ステップを保存しました({steps} ステップ)。完了した作業は失われません。新しいタスクを入力するか ;retry で保存ポイントから続行できます。" + partial_turn_saved_budget: "タスク予算上限で停止する前に、このターンで完了したツール実行ステップを保存しました({steps} ステップ)。完了した作業は失われません。新しいタスクを入力するか ;retry で保存ポイントから続行できます。" llm_error_details_title: "デバッグ詳細" tool_args_json_invalid_hint: "❌ リクエストに失敗しました。再試行してください。" prompt: @@ -446,6 +456,13 @@ shell: output_tokens: "出力 tokens" total: "合計" api_calls: "API 呼び出し回数" + budget_title: "タスク予算" + budget_rounds: "ラウンド(使用/上限):" + budget_tool_calls: "ツール呼び出し(使用/上限):" + budget_tokens: "Token(使用/上限):" + budget_duration: "アクティブ秒数(使用/上限):" + budget_cost: "費用:" + budget_cost_unavailable: "利用不可(プロバイダは価格を報告しません)" setup: no_config_manager: "再設定できません: 設定マネージャーを利用できません" cancelled: "セットアップをキャンセルしました。現在の設定を維持します" @@ -559,6 +576,9 @@ shell: inline_enforce_json: label: "インライン JSON 強制" desc: "インライン補完で厳密な JSON を返すよう強制" + task_budget_max_rounds: + label: "タスク予算ラウンド上限" + desc: "タスクごとの累積ツールループ上限。空欄 = 無制限(続行ではリセットされません)" max_llm_messages: label: "最大 LLM メッセージ数" desc: "LLM 履歴に保持する最大会話ターン数" diff --git a/crates/aish-i18n/locales/zh-CN.yaml b/crates/aish-i18n/locales/zh-CN.yaml index c2f2341d..bd60a5b9 100644 --- a/crates/aish-i18n/locales/zh-CN.yaml +++ b/crates/aish-i18n/locales/zh-CN.yaml @@ -768,6 +768,15 @@ shell: nl_detection: confirm_ask_ai: "这看起来像自然语言问题,是否让 AI 解答?(Y/n)" + budget: + exhausted_title: "⛔ 任务预算已耗尽" + exhausted_detail: "维度: {dimension} · 已用: {used} · 上限: {limit}" + exhausted_options: "是否调整预算并继续? 调整将被记录到审计日志 (y=调整 / n=停止)" + stopped_by_user: "已按预算停止。已完成的工具执行记录已保存,可稍后从保存点继续。" + new_limit_prompt: "请输入该维度的新上限数值:" + raised: "预算已调整: {dimension} -> {value}。重新发起任务以从保存点继续。" + invalid_limit: "无效数值(必须大于已用量),预算保持不变。" + error: execution_error: "❌ 执行过程中出错: {error}" pty_decode_error: "捕获 pty 输出解码错误: {error}" @@ -776,6 +785,7 @@ shell: llm_error_message: "错误: {error}" partial_turn_saved: "已保存本轮已完成的工具执行记录({steps} 步),失败前的工作不会丢失;输入新任务或 ;重试 可从保存点继续。" partial_turn_saved_limit: "已在达到轮次上限停止前保存本轮已完成的工具执行记录({steps} 步),已完成的工作不会丢失;输入新任务或 ;重试 可从保存点继续。" + partial_turn_saved_budget: "已在达到任务预算上限停止前保存本轮已完成的工具执行记录({steps} 步),已完成的工作不会丢失;输入新任务或 ;重试 可从保存点继续。" llm_error_details_title: "调试详情" tool_args_json_invalid_hint: "❌ 请求失败,请重试。" pty_error: "PTY 错误:{error}" @@ -888,6 +898,13 @@ shell: output_tokens: "输出 tokens" total: "总计" api_calls: "API 调用次数" + budget_title: "任务预算" + budget_rounds: "轮次(已用/上限):" + budget_tool_calls: "工具调用(已用/上限):" + budget_tokens: "Token(已用/上限):" + budget_duration: "活跃时长秒(已用/上限):" + budget_cost: "费用:" + budget_cost_unavailable: "不可用(提供商未回报价格数据)" setup: no_config_manager: "无法重新配置:配置管理器不可用" @@ -983,6 +1000,9 @@ shell: inline_completion: label: "行内 AI 补全" desc: "输入 AI 提示时显示灰色幽灵文本建议" + task_budget_max_rounds: + label: "任务预算轮次上限" + desc: "单个任务累计工具循环轮次上限,留空为不限(继续不重置)" max_llm_messages: label: "AI 最大消息数" desc: "上下文中保留的 AI 对话轮数" diff --git a/crates/aish-llm/src/agents/spawn.rs b/crates/aish-llm/src/agents/spawn.rs index 7e69b2f6..25a3b17e 100644 --- a/crates/aish-llm/src/agents/spawn.rs +++ b/crates/aish-llm/src/agents/spawn.rs @@ -93,7 +93,38 @@ pub async fn spawn( where F: FnOnce(&mut LlmSession), { + // Issue #569 review C6: pause the parent's wall-clock while the child + // runs. The child accrues its own active_secs (merged below); leaving + // the parent anchor running would double-count the child interval. + let parent_clock_active = parent.task_budget_state().last_accounted.is_some(); + if parent_clock_active { + parent.flush_task_budget_wall_clock(); // settle parent time so far + parent.task_budget_state_mut().last_accounted = None; + } + + // Issue #569 review C7: the child inherits the parent's REMAINING + // budget. Its counters stay zero-based — the merge below folds the + // delta into the parent — but the limits travel with it, so a spawned + // loop cannot burn past what the parent's task still allows. + let parent_limit = parent.task_budget_limit(); + let parent_state = parent.task_budget_state(); + let remaining_limits = crate::budget::TaskBudgetLimit { + max_rounds: parent_limit + .max_rounds + .map(|max| max.saturating_sub(parent_state.task_rounds)), + max_tool_calls: parent_limit + .max_tool_calls + .map(|max| max.saturating_sub(parent_state.tool_calls)), + max_tokens: parent_limit + .max_tokens + .map(|max| max.saturating_sub(parent_state.token_usage.total_tokens())), + max_duration_secs: parent_limit + .max_duration_secs + .map(|max| max.saturating_sub(parent_state.active_secs)), + }; + let mut sub = parent.create_subsession(); + sub.set_task_budget_limit(remaining_limits); configure(&mut sub); let parent_cancel = parent.cancellation_token_arc(); @@ -119,7 +150,17 @@ where // billed exactly like main-session turns). Totals only — see // TokenStats::merge_totals for why last_prompt_tokens is skipped. parent.merge_token_stats(&sub.token_stats()); - + // Issue #569 acceptance 3: fold the sub-agent's budget counters + // (rounds, tool calls, wall-clock seconds, tokens) into the parent + // task exactly once on completion — no double counting on retries, + // because the merge runs in the single completion path here. + sub.flush_task_budget_wall_clock(); + parent.merge_task_budget(&sub.task_budget_state()); + // Review C6: resume the parent clock so its own post-spawn rounds + // keep accruing. + if parent_clock_active { + parent.task_budget_state_mut().start_turn(); + } SpawnResult { text: outcome.text, status: outcome.status, @@ -385,6 +426,134 @@ mod tests { .any(|s| s.function.name == "read_file")); } + #[tokio::test] + async fn test_spawn_child_budget_exhaustion_returns_before_request() { + // Review C8: the child loop must enforce the inherited remaining + // budget BEFORE any model request; exhaustion surfaces as a fatal + // outcome carrying BudgetExhausted, and the parent merges the + // usage without issuing further requests. + let mut parent = LlmSession::new("http://localhost", "key", "model", None, None); + parent.register_tool(Box::new(MockTool::new("grep"))); + // Parent already consumed its whole remaining budget: the child + // inherits 0 remaining rounds and must stop immediately. + parent.set_task_budget_limit(crate::budget::TaskBudgetLimit { + max_rounds: Some(2), + max_tool_calls: None, + max_tokens: None, + max_duration_secs: None, + }); + parent.merge_task_budget(&crate::budget::TaskBudgetState { + task_rounds: 2, + ..Default::default() + }); + + let registry = AgentRegistry::builtin(); + let result = spawn_builtin(&parent, ®istry, "explore", "task", |sub, _specs| { + configure_spawn_test(sub, vec![Ok(mock_text_response("never reached"))]); + }) + .await + .expect("spawn_builtin returns a result, not an error"); + + assert_eq!(result.status, LoopStatus::Fatal); + assert_eq!( + result.error_category.as_deref(), + Some("budget_exhausted"), + "child exhaustion must surface as BudgetExhausted" + ); + } + + #[tokio::test] + async fn test_spawn_child_inherits_remaining_budget() { + // Review C7: the child loop must see the parent's REMAINING limits + // (max minus used), so a near-exhausted task cannot burn past its + // cap through a spawned agent. The configure hook runs after the + // inheritance, so it observes the propagated limits. + let mut parent = LlmSession::new("http://localhost", "key", "model", None, None); + parent.register_tool(Box::new(MockTool::new("grep"))); + parent.set_task_budget_limit(crate::budget::TaskBudgetLimit { + max_rounds: Some(10), + max_tool_calls: Some(20), + max_tokens: None, + max_duration_secs: None, + }); + parent.merge_task_budget(&crate::budget::TaskBudgetState { + task_rounds: 3, + tool_calls: 8, + ..Default::default() + }); + + let registry = AgentRegistry::builtin(); + let _ = spawn_builtin(&parent, ®istry, "explore", "task", |sub, _specs| { + configure_spawn_test(sub, vec![Ok(mock_text_response("ok"))]); + let limit = sub.task_budget_limit(); + assert_eq!(limit.max_rounds, Some(7), "remaining rounds (10 - 3)"); + assert_eq!(limit.max_tool_calls, Some(12), "remaining calls (20 - 8)"); + }) + .await + .expect("spawn_builtin should succeed"); + } + + #[tokio::test] + async fn test_spawn_pauses_and_resumes_parent_clock() { + // Review C6: the parent clock pauses while the child runs (no + // double counting) and resumes after the merge. + let mut parent = LlmSession::new("http://localhost", "key", "model", None, None); + parent.register_tool(Box::new(MockTool::new("grep"))); + parent.task_budget_state_mut().start_turn(); + + let registry = AgentRegistry::builtin(); + let _ = spawn_builtin(&parent, ®istry, "explore", "task", |sub, _specs| { + configure_spawn_test(sub, vec![Ok(mock_text_response("ok"))]); + assert!( + sub.task_budget_state().last_accounted.is_none(), + "child clock must not start before its loop begins" + ); + }) + .await + .expect("spawn_builtin should succeed"); + + let state = parent.task_budget_state(); + assert!( + state.last_accounted.is_some(), + "parent clock must resume after the merge" + ); + } + + #[tokio::test] + async fn test_spawn_folds_subsession_budget_into_parent_task() { + // Issue #569 acceptance 3: sub-agent rounds/tool-calls/tokens fold + // into the parent's task-level counters exactly once. + let mut parent = LlmSession::new("http://localhost", "key", "model", None, None); + parent.register_tool(Box::new(MockTool::new("grep"))); + let rounds_before = parent.task_budget_state().task_rounds; + let calls_before = parent.task_budget_state().tool_calls; + + let registry = AgentRegistry::builtin(); + let _ = spawn_builtin(&parent, ®istry, "explore", "task", |sub, _specs| { + configure_spawn_test( + sub, + vec![ + Ok(mock_tool_call_response_with_usage( + &[("c1", "grep", "{}")], + 500, + 40, + )), + Ok(mock_text_response_with_usage("done", 700, 60)), + ], + ); + }) + .await + .expect("spawn_builtin should succeed"); + + let budget = parent.task_budget_state(); + // The sub-agent ran 2 loop rounds and 1 tool call; the parent's own + // loop did not run, so the totals are exactly the sub's. + assert_eq!(budget.task_rounds, rounds_before + 2); + assert_eq!(budget.tool_calls, calls_before + 1); + assert_eq!(budget.token_usage.total_input, 1200); + assert_eq!(budget.token_usage.total_output, 100); + } + #[tokio::test] async fn test_spawn_merges_subsession_usage_into_parent() { // The sub-agent's billed usage must land in the parent's cumulative diff --git a/crates/aish-llm/src/agents/tool_loop.rs b/crates/aish-llm/src/agents/tool_loop.rs index 9f1487cf..2c9bc92e 100644 --- a/crates/aish-llm/src/agents/tool_loop.rs +++ b/crates/aish-llm/src/agents/tool_loop.rs @@ -97,6 +97,12 @@ pub async fn run_tool_loop_until_done( config: &ToolLoopConfig, ) -> LoopOutcome { let base_system = config.system_message.as_deref().unwrap_or(""); + // Wall-clock accounting for the whole spawn (issue #569): start once, + // flush every round before the parent merges the counters. The spawn + // path calls flush_task_budget_wall_clock() before merging, so no drop + // guard is needed here — the loop's own exits all leave the anchor in + // place for that flush. + session.task_budget_state_mut().start_turn(); let bundle = PromptAssembly::build(session, config.prompt_context.clone(), base_system); let mut messages: Vec = Vec::new(); if config.system_message.is_some() { @@ -126,7 +132,31 @@ pub async fn run_tool_loop_until_done( ); return LoopOutcome::from_spawn_outcome(outcome, loop_messages, None); } + + // Task budget admission check (issue #569 review): the child loop + // enforces the inherited remaining limits BEFORE any model request + // — same contract as process_input. Exhaustion returns a fatal + // outcome carrying BudgetExhausted so the parent's merge folds the + // usage and the parent can stop cleanly. + { + let limit = session.task_budget_limit(); + let state = { + let mut state = session.task_budget_state_mut(); + // Settle the previous round's wall-clock before the check. + state.flush_wall_clock(); + state.clone() + }; + if let Some(dim) = limit.check(&state).exhausted { + let err = AishError::BudgetExhausted(dim.key().to_string()); + return LoopOutcome::fatal_with_messages(err, loop_messages); + } + } + iterations += 1; + // Task budget accounting (issue #569): sub-agent loops accrue the + // same counters as the main loop. Wall-clock stays anchored from + // loop start; the spawn path flushes it before merging. + session.task_budget_state_mut().task_rounds += 1; messages = session.prepare_messages_for_send(messages).await; diff --git a/crates/aish-llm/src/budget.rs b/crates/aish-llm/src/budget.rs new file mode 100644 index 00000000..07852b4a --- /dev/null +++ b/crates/aish-llm/src/budget.rs @@ -0,0 +1,297 @@ +use std::time::Instant; + +use crate::usage::TokenStats; + +/// Task-level cumulative budget counters (issue #569). +/// +/// Unlike the per-segment iteration counter (reset when the user chooses to +/// continue), these counters accumulate for the lifetime of the task across +/// continues, compactions, and sub-agent merges. They are never reset by +/// "continue" — only persisted state or an explicit budget adjustment can +/// change their trajectory. +#[derive(Debug, Default, Clone)] +pub struct TaskBudgetState { + /// Cumulative tool-loop rounds across the whole task (continues included). + pub task_rounds: u64, + /// Cumulative tool executions across the whole task (parallel calls + /// within one round each count). + pub tool_calls: u64, + /// Active wall-clock seconds accumulated while a turn was in flight. + /// Advanced incrementally (`last_accounted += elapsed`) so continues, + /// pauses, and resumes never double-count. Idle time is not counted. + pub active_secs: u64, + /// Monotonic anchor for wall-clock accounting: the instant from which + /// active time is being accumulated. `None` while no turn is running. + pub last_accounted: Option, + /// Cumulative token usage. Cached-prefix reads are excluded from budget + /// math via `TokenStats::total_tokens()`. + pub token_usage: TokenStats, +} + +impl TaskBudgetState { + /// Begin (or resume) wall-clock accounting. + pub fn start_turn(&mut self) { + self.last_accounted = Some(Instant::now()); + } + + /// Flush pending wall-clock time into `active_secs` using the advancing + /// anchor pattern: the anchor moves forward by exactly the flushed amount, + /// so a later flush never double-counts and a pause/continue never resets + /// the baseline. Whole seconds only — sub-second remainder stays pending. + pub fn flush_wall_clock(&mut self) { + if let Some(anchor) = self.last_accounted { + let elapsed = anchor.elapsed().as_secs(); + if elapsed > 0 { + self.active_secs += elapsed; + self.last_accounted = Some(anchor + std::time::Duration::from_secs(elapsed)); + } + } + } + + /// Stop wall-clock accounting (turn ended). Flushes any pending whole + /// seconds first. + pub fn end_turn(&mut self) { + self.flush_wall_clock(); + self.last_accounted = None; + } + + /// Merge a sub-agent's cumulative counters into this parent task. + /// Each sub-agent reports exactly once on completion, so totals do not + /// double-count. Wall-clock anchors are not merged — they are process + /// local. + pub fn merge_task(&mut self, other: &TaskBudgetState) { + self.task_rounds += other.task_rounds; + self.tool_calls += other.tool_calls; + self.active_secs += other.active_secs; + self.token_usage.merge_totals(&other.token_usage); + } +} + +/// Near-threshold advisory state (issue #569 acceptance 3 / omp +/// `budgetReportedFor` pattern): the "you are approaching the budget" hint +/// fires once per threshold cycle and is re-armed by an explicit budget +/// adjustment. Tracked on the session, not the persisted counters. +#[derive(Debug, Default, Clone)] +pub struct NearThresholdState { + /// True once the hint has been shown since the last re-arm. + pub reported: bool, +} + +impl NearThresholdState { + /// True when usage crossed `NEAR_THRESHOLD_PERCENT` and the hint has + /// not been shown yet in this cycle. Idempotent: subsequent calls with + /// usage still above the threshold return false. + pub fn should_report(&mut self, check: &BudgetCheck) -> bool { + let crossed = check + .nearest_percent + .is_some_and(|pct| pct >= NEAR_THRESHOLD_PERCENT); + if crossed && !self.reported { + self.reported = true; + return true; + } + if !crossed { + // Usage dropped below the threshold (e.g. a limit raise): + // re-arm so a future crossing reports again. + self.reported = false; + } + false + } + + /// Explicit re-arm (budget adjustment path). + pub fn rearm(&mut self) { + self.reported = false; + } +} + +/// Configurable upper bounds for one task. `None` = unlimited for that +/// dimension. Limits come from `config.yaml` (`llm.task_budget`) and can only +/// be changed through the explicit budget-adjustment flow (never by "continue"). +#[derive(Debug, Clone, Default)] +pub struct TaskBudgetLimit { + pub max_rounds: Option, + pub max_tool_calls: Option, + pub max_tokens: Option, + pub max_duration_secs: Option, +} + +impl TaskBudgetLimit { + /// True when any dimension is bounded. + pub fn is_bounded(&self) -> bool { + self.max_rounds.is_some() + || self.max_tool_calls.is_some() + || self.max_tokens.is_some() + || self.max_duration_secs.is_some() + } +} + +/// Which dimension exhausted, in check order. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BudgetDimension { + Rounds, + ToolCalls, + Tokens, + Duration, +} + +impl BudgetDimension { + pub fn key(self) -> &'static str { + match self { + BudgetDimension::Rounds => "rounds", + BudgetDimension::ToolCalls => "tool_calls", + BudgetDimension::Tokens => "tokens", + BudgetDimension::Duration => "duration", + } + } +} + +/// Result of a budget check: which dimension is over (if any) and how close +/// the task is to the nearest threshold (for near-threshold UI, 0-100). +pub struct BudgetCheck { + pub exhausted: Option, + /// Highest usage/threshold ratio across bounded dimensions, scaled to + /// percent. `None` when nothing is bounded. + pub nearest_percent: Option, +} + +pub const NEAR_THRESHOLD_PERCENT: u64 = 80; + +impl TaskBudgetLimit { + /// Evaluate the current state against all bounds. Pure computation — + /// callers decide what to do with exhaustion (stop the loop) or + /// near-threshold pressure (checkpoint + UI hint). + pub fn check(&self, state: &TaskBudgetState) -> BudgetCheck { + // First exhausted dimension in a stable check order. + let exhausted_dim = [ + (state.task_rounds, self.max_rounds, BudgetDimension::Rounds), + ( + state.tool_calls, + self.max_tool_calls, + BudgetDimension::ToolCalls, + ), + ( + state.token_usage.total_tokens(), + self.max_tokens, + BudgetDimension::Tokens, + ), + ( + state.active_secs, + self.max_duration_secs, + BudgetDimension::Duration, + ), + ] + .into_iter() + .find(|(used, limit, _)| limit.is_some_and(|max| *used >= max)) + .map(|(_, _, dim)| dim); + + // Highest usage/threshold ratio across bounded dimensions. + let mut nearest: Option = None; + for (used, limit) in [ + (state.task_rounds, self.max_rounds), + (state.tool_calls, self.max_tool_calls), + (state.token_usage.total_tokens(), self.max_tokens), + (state.active_secs, self.max_duration_secs), + ] { + if let Some(max) = limit { + if let Some(pct) = used.saturating_mul(100).checked_div(max) { + nearest = Some(nearest.map_or(pct, |p: u64| p.max(pct))); + } + } + } + + BudgetCheck { + exhausted: exhausted_dim, + nearest_percent: nearest, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn advancing_wall_clock_never_double_counts() { + let mut state = TaskBudgetState::default(); + state.start_turn(); + state.flush_wall_clock(); + let first = state.active_secs; + state.flush_wall_clock(); + let second = state.active_secs; + assert!(second >= first, "flush must advance the counter"); + state.end_turn(); + assert!(state.last_accounted.is_none()); + } + + #[test] + fn exhaustion_is_detected_per_dimension_in_order() { + let mut state = TaskBudgetState::default(); + state.task_rounds = 10; + state.tool_calls = 50; + let limit = TaskBudgetLimit { + max_rounds: Some(10), + max_tool_calls: Some(50), + max_tokens: None, + max_duration_secs: None, + }; + assert_eq!(limit.check(&state).exhausted, Some(BudgetDimension::Rounds)); + + state.task_rounds = 9; + assert_eq!( + limit.check(&state).exhausted, + Some(BudgetDimension::ToolCalls) + ); + } + + #[test] + fn near_threshold_percent_is_the_max_ratio() { + let mut state = TaskBudgetState::default(); + state.task_rounds = 10; + state.tool_calls = 45; + let limit = TaskBudgetLimit { + max_rounds: Some(100), + max_tool_calls: Some(50), + max_tokens: None, + max_duration_secs: None, + }; + let check = limit.check(&state); + assert_eq!(check.nearest_percent, Some(90)); + assert!(check.exhausted.is_none()); + } + + #[test] + fn near_threshold_reports_once_then_rearms() { + let mut nt = NearThresholdState::default(); + let state = TaskBudgetState { + task_rounds: 85, + ..Default::default() + }; + let limit = TaskBudgetLimit { + max_rounds: Some(100), + ..Default::default() + }; + let check = limit.check(&state); + assert!(nt.should_report(&check), "first crossing reports"); + assert!( + !nt.should_report(&check), + "second call in the same cycle is deduplicated" + ); + nt.rearm(); + assert!(nt.should_report(&check), "rearm allows reporting again"); + } + + #[test] + fn near_threshold_ignores_unbounded_dimensions() { + let mut nt = NearThresholdState::default(); + let check = TaskBudgetLimit::default().check(&TaskBudgetState::default()); + assert!(!nt.should_report(&check)); + } + + #[test] + fn unbounded_limit_never_exhausts() { + let mut state = TaskBudgetState::default(); + state.task_rounds = u64::MAX; + let check = TaskBudgetLimit::default().check(&state); + assert!(check.exhausted.is_none()); + assert!(check.nearest_percent.is_none()); + } +} diff --git a/crates/aish-llm/src/lib.rs b/crates/aish-llm/src/lib.rs index 354fe797..39f7ad3d 100644 --- a/crates/aish-llm/src/lib.rs +++ b/crates/aish-llm/src/lib.rs @@ -16,6 +16,7 @@ pub mod agents; pub mod api; pub mod approval_memory; +pub mod budget; pub mod client; pub mod langfuse; pub mod llm_stream; diff --git a/crates/aish-llm/src/session.rs b/crates/aish-llm/src/session.rs index 0164c197..b5194628 100644 --- a/crates/aish-llm/src/session.rs +++ b/crates/aish-llm/src/session.rs @@ -110,7 +110,19 @@ pub struct LlmSession { /// `take_last_partial_turn` so the shell layer can persist the /// evidence. Empty unless the previous turn failed mid-loop. last_partial_turn: parking_lot::Mutex>, - /// Scripted chat completion responses for unit/integration tests (pop in order). + /// Task-level cumulative budget counters (issue #569). Accumulates + /// across continues, compactions, and sub-agent merges; never reset by + /// "continue". Guarded by a mutex because `execute_tool` (called from + /// async contexts) updates `tool_calls` mid-loop. + task_budget_state: parking_lot::Mutex, + /// Upper bounds for the task budget. `None`-dimension = unlimited. + task_budget_limit: parking_lot::Mutex, + /// Set when the loop stopped because a budget dimension was exhausted; + /// holds the dimension key for the shell's stop-note wording. + budget_exhausted_dimension: parking_lot::Mutex>, + /// Near-threshold advisory state: fires the "approaching budget" hint + /// once per cycle; re-armed by an explicit budget adjustment. + near_threshold_state: parking_lot::Mutex, #[cfg(test)] test_chat_responses: Option>>>>, } @@ -171,11 +183,67 @@ impl LlmSession { is_sub_agent: false, rotation: None, last_partial_turn: parking_lot::Mutex::new(Vec::new()), + task_budget_state: parking_lot::Mutex::new(crate::budget::TaskBudgetState::default()), + task_budget_limit: parking_lot::Mutex::new(crate::budget::TaskBudgetLimit::default()), + budget_exhausted_dimension: parking_lot::Mutex::new(None), + near_threshold_state: parking_lot::Mutex::new( + crate::budget::NearThresholdState::default(), + ), #[cfg(test)] test_chat_responses: None, } } + /// Set the task budget bounds (from `config.yaml llm.task_budget`). + pub fn set_task_budget_limit(&self, limit: crate::budget::TaskBudgetLimit) { + *self.task_budget_limit.lock() = limit; + } + + /// Fold a completed sub-agent's cumulative budget counters into this + /// session's task (issue #569 acceptance 3). Called once per sub-agent + /// completion; wall-clock anchors stay process-local and unmerged. + pub fn merge_task_budget(&self, other: &crate::budget::TaskBudgetState) { + self.task_budget_state.lock().merge_task(other); + } + + /// Snapshot the configured task budget bounds. + pub fn task_budget_limit(&self) -> crate::budget::TaskBudgetLimit { + self.task_budget_limit.lock().clone() + } + + /// Snapshot the cumulative task budget counters. + pub fn task_budget_state(&self) -> crate::budget::TaskBudgetState { + self.task_budget_state.lock().clone() + } + + /// Flush pending wall-clock seconds into the budget counters. Public + /// so the spawn path can settle a sub-session before folding it into + /// the parent task. + pub fn flush_task_budget_wall_clock(&self) { + self.task_budget_state.lock().end_turn(); + } + + /// Internal mutable access for loop accounting (tool_loop / spawn). + pub(crate) fn task_budget_state_mut( + &self, + ) -> parking_lot::MutexGuard<'_, crate::budget::TaskBudgetState> { + self.task_budget_state.lock() + } + + /// Seed cumulative counters on resume (issue #569 acceptance 5: budget + /// state must survive session restore so it cannot be bypassed by + /// restarting the process). Wall-clock anchor is never seeded — it is + /// process-local and starts with the next turn. + pub fn seed_task_budget_state(&self, mut state: crate::budget::TaskBudgetState) { + state.end_turn(); // drop any cross-process Instant anchor + *self.task_budget_state.lock() = state; + } + + /// Take the budget dimension that stopped the last turn, if any. + pub fn take_budget_exhausted_dimension(&self) -> Option { + self.budget_exhausted_dimension.lock().take() + } + /// Drain the tool-loop messages completed before the last mid-turn /// failure. Returns an empty vec when the previous turn succeeded (or /// failed before any tool ran). The caller is expected to persist these @@ -441,6 +509,12 @@ impl LlmSession { } fn record_usage(&self, usage: crate::usage::TokenUsage) { + // Mirror into the task budget counters first (issue #569): budget + // math must see the same usage the session footer reports. + self.task_budget_state + .lock() + .token_usage + .record(usage.clone()); self.token_stats.lock().unwrap().record(usage); } @@ -647,6 +721,21 @@ impl LlmSession { ) -> Result { self.cancellation_token.reset(); self.last_turn_compaction.lock().unwrap().take(); + + // Wall-clock settle guard (issue #569): every exit path of this + // function — success, error, cancel, iteration limit, budget stop — + // must flush pending wall-clock seconds and stop accounting, or the + // next round's start_turn() would reset the anchor and lose the + // elapsed time of the aborted round. + struct _WallClockGuard<'a>(&'a parking_lot::Mutex); + impl Drop for _WallClockGuard<'_> { + fn drop(&mut self) { + self.0.lock().end_turn(); + } + } + self.task_budget_state.lock().start_turn(); + let _wall_clock_guard = _WallClockGuard(&self.task_budget_state); + let turn_seq = self .turn_seq .fetch_add(1, std::sync::atomic::Ordering::Relaxed); @@ -772,6 +861,56 @@ impl LlmSession { return Err(AishError::IterationLimit); } } + + // Task-level cumulative budget check (issue #569): evaluated at + // the loop top, BEFORE any model request, so an exhausted budget + // never fires another call. The previous round's tools have all + // completed by now (rounds are serial), so the partial-turn + // evidence below is complete and consistent. + { + let (limit, state_snapshot) = { + let mut state = self.task_budget_state.lock(); + // Settle the previous round's wall-clock time BEFORE the + // check so duration exhaustion sees complete data. + state.flush_wall_clock(); + (self.task_budget_limit.lock().clone(), state.clone()) + }; + let check = limit.check(&state_snapshot); + if let Some(dim) = check.exhausted { + // Preserve the same evidence contract as the iteration + // limit / API error paths (#572/#452): user message + // prefix + the post-initial loop messages, call/result + // pairing intact. + if messages.len() > initial_len { + let mut partial: Vec = vec![user_msg.clone()]; + partial.extend(messages[initial_len..].iter().cloned()); + *self.last_partial_turn.lock() = partial; + } + let dim_key = dim.key().to_string(); + *self.budget_exhausted_dimension.lock() = Some(dim_key.clone()); + self.emit_event(LlmEvent { + event_type: LlmEventType::Error, + data: serde_json::json!({ + "error": "task budget exhausted", + "dimension": dim_key, + }), + timestamp: now_timestamp(), + metadata: None, + }); + self.emit_event(LlmEvent { + event_type: LlmEventType::OpEnd, + data: serde_json::json!({"reason": "budget_exhausted"}), + timestamp: now_timestamp(), + metadata: None, + }); + return Err(AishError::BudgetExhausted(dim_key)); + } + } + // Count this round. Wall-clock accounting is already running + // (started by the entry guard and flushed below before the + // budget check), so do NOT restart the anchor here — restarting + // it would discard the previous round's elapsed time. + self.task_budget_state.lock().task_rounds += 1; iterations += 1; messages = self.prepare_messages_for_send(messages).await; @@ -790,6 +929,7 @@ impl LlmSession { event_type: LlmEventType::GenerationStart, data: serde_json::json!({ "iteration": iterations, + "task_rounds": self.task_budget_state().task_rounds, "has_tools": has_tools, }), timestamp: now_timestamp(), @@ -947,6 +1087,8 @@ impl LlmSession { // Accumulate token usage from SSE chunks let mut stream_prompt_tokens: u64 = 0; let mut stream_completion_tokens: u64 = 0; + let mut stream_cache_read_tokens: u64 = 0; + let mut stream_cache_write_tokens: u64 = 0; while !stream_done { if self.cancellation_token.is_cancelled() { @@ -968,6 +1110,8 @@ impl LlmSession { if let Some(u) = chunk_usage { stream_prompt_tokens = u.prompt_tokens; stream_completion_tokens = u.completion_tokens; + stream_cache_read_tokens = u.cache_read_tokens; + stream_cache_write_tokens = u.cache_write_tokens; } // Content is emitted before ToolCallDelta in // the same parse batch, so look ahead: a @@ -1106,6 +1250,8 @@ impl LlmSession { self.record_usage(crate::usage::TokenUsage { prompt_tokens: stream_prompt_tokens, completion_tokens: stream_completion_tokens, + cache_read_tokens: stream_cache_read_tokens, + cache_write_tokens: stream_cache_write_tokens, }); } @@ -1260,10 +1406,17 @@ impl LlmSession { ) .await { + // Round finished — flush pending whole seconds of + // active wall-clock into the task budget (#569). + let mut state = self.task_budget_state.lock(); + state.end_turn(); + let check = self.task_budget_limit.lock().check(&state); + if self.near_threshold_state.lock().should_report(&check) { + self.emit_budget_near_threshold(&check); + } return Ok(pr); } - // Smart-trim old tool outputs to prevent unbounded growth smart_trim_tool_loop(&mut messages, initial_len); } } @@ -1313,6 +1466,9 @@ impl LlmSession { /// - Retries once on execution failure (matching Python's robustness). /// - Emits structured TOOL_EXECUTION_START / TOOL_EXECUTION_END events. async fn execute_tool(&self, tool_call: &ToolCall) -> ToolResult { + // Count every tool execution against the task budget (issue #569): + // parallel calls within one round each count. + self.task_budget_state.lock().tool_calls += 1; let args: serde_json::Value = serde_json::from_str(&tool_call.arguments).unwrap_or(serde_json::Value::Null); @@ -1684,6 +1840,12 @@ impl LlmSession { context_budget_policy: self.context_budget_policy.clone(), rotation: None, last_partial_turn: parking_lot::Mutex::new(Vec::new()), + task_budget_state: parking_lot::Mutex::new(crate::budget::TaskBudgetState::default()), + task_budget_limit: parking_lot::Mutex::new(crate::budget::TaskBudgetLimit::default()), + budget_exhausted_dimension: parking_lot::Mutex::new(None), + near_threshold_state: parking_lot::Mutex::new( + crate::budget::NearThresholdState::default(), + ), turn_seq: std::sync::atomic::AtomicU32::new(0), plan_state: Arc::new(Mutex::new(PlanModeState::default())), token_stats: std::sync::Mutex::new(crate::usage::TokenStats::default()), @@ -1858,6 +2020,27 @@ impl LlmSession { }); } + /// Issue #569: emit a one-shot advisory when cumulative usage crosses + /// the near-threshold percentage of any bounded dimension. Re-armed by + /// an explicit budget adjustment (`rearm_near_threshold`). + fn emit_budget_near_threshold(&self, check: &crate::budget::BudgetCheck) { + self.emit_event(LlmEvent { + event_type: LlmEventType::Error, + data: serde_json::json!({ + "error": "task budget near threshold", + "nearest_percent": check.nearest_percent, + "warning": true, + }), + timestamp: now_timestamp(), + metadata: None, + }); + } + + /// Re-arm the near-threshold advisory after an explicit budget raise. + pub fn rearm_near_threshold(&self) { + self.near_threshold_state.lock().rearm(); + } + pub fn emit_context_compaction_start(&self, scope: &str, mode: &str) { self.emit_event(LlmEvent { event_type: LlmEventType::ContextCompactionStart, @@ -4333,4 +4516,176 @@ mod tests { assert_eq!(called_ids, ids, "all completed call ids in order"); assert_eq!(called_ids, answered_ids, "no dangling tool_call_id"); } + + /// Issue #569: a token-only task budget stops the loop BEFORE the model + /// request that would exceed it, preserves the #572 evidence contract, + /// and reports the exhausted dimension. + #[tokio::test] + async fn task_budget_token_exhaustion_stops_loop_and_saves_evidence() { + use crate::agents::mock_tool_call_response_with_usage; + + let log = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); + // Three scripted rounds; each round burns 60 prompt + 40 output + // tokens. The 150-token budget must stop before round 3's request. + let responses: Vec> = (0..3) + .map(|i| { + Ok(mock_tool_call_response_with_usage( + &[(format!("c{}", i).as_str(), "grep", "{}")], + 60, + 40, + )) + }) + .collect(); + let session = partial_turn_session(responses, log.clone()); + session.set_task_budget_limit(crate::budget::TaskBudgetLimit { + max_tokens: Some(150), + ..Default::default() + }); + + let result = session + .process_input(&ChatMessage::user("do the task"), &[], Some("core"), false) + .await; + let err = result.expect_err("budget exhaustion must fail the turn"); + assert!( + matches!(&err, AishError::BudgetExhausted(d) if d == "tokens"), + "must surface BudgetExhausted(tokens), got {err:?}" + ); + assert_eq!( + session.take_budget_exhausted_dimension().as_deref(), + Some("tokens") + ); + // Rounds 1 and 2 ran (2×100=200 ≥ 150 after round 2); round 3 never + // sent a request, so the log shows exactly two tool executions. + assert_eq!(log.lock().unwrap().len(), 2, "round 3 must not run"); + let state = session.task_budget_state(); + assert_eq!(state.task_rounds, 2); + assert_eq!(state.tool_calls, 2); + assert_eq!(state.token_usage.total_tokens(), 200); + + // Evidence contract: user prefix + both completed call/result pairs. + let partial = session.take_last_partial_turn(); + let roles: Vec<&str> = partial.iter().map(|m| m.role.as_str()).collect(); + assert_eq!( + roles, + vec!["user", "assistant", "tool", "assistant", "tool"] + ); + let called_ids: Vec = partial + .iter() + .filter(|m| m.role == "assistant") + .flat_map(|m| m.tool_calls.iter().flatten().map(|tc| tc.id.clone())) + .collect(); + assert_eq!(called_ids, vec!["c0".to_string(), "c1".to_string()]); + } + + /// Issue #569: rounds-dimension exhaustion stops before the round that + /// would exceed it and reports the rounds dimension. + #[tokio::test] + async fn task_budget_rounds_exhaustion_stops_with_dimension() { + use crate::agents::mock_tool_call_response; + + let log = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); + let responses: Vec> = (0..5) + .map(|i| { + Ok(mock_tool_call_response(&[( + format!("c{}", i).as_str(), + "grep", + "{}", + )])) + }) + .collect(); + let session = partial_turn_session(responses, log.clone()); + session.set_task_budget_limit(crate::budget::TaskBudgetLimit { + max_rounds: Some(3), + ..Default::default() + }); + + let result = session + .process_input(&ChatMessage::user("do the task"), &[], Some("core"), false) + .await; + let err = result.expect_err("rounds exhaustion must fail the turn"); + assert!( + matches!(&err, AishError::BudgetExhausted(d) if d == "rounds"), + "must surface BudgetExhausted(rounds), got {err:?}" + ); + assert_eq!(log.lock().unwrap().len(), 3, "round 4 must not run"); + assert_eq!(session.task_budget_state().task_rounds, 3); + // Evidence: user prefix + 3 completed call/result pairs. + let partial = session.take_last_partial_turn(); + let assistants = partial.iter().filter(|m| m.role == "assistant").count(); + let tools = partial.iter().filter(|m| m.role == "tool").count(); + assert_eq!(assistants, 3); + assert_eq!(tools, 3); + } + + /// Issue #569: a bounded-but-unreached budget never stops the loop and + /// a completed turn keeps the partial buffer empty. + #[tokio::test] + async fn task_budget_within_bounds_completes_normally() { + use crate::agents::mock_tool_call_response; + + let log = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); + let session = partial_turn_session( + vec![ + Ok(mock_tool_call_response(&[("c1", "grep", "{}")])), + Ok(crate::agents::mock_text_response("done")), + ], + log.clone(), + ); + session.set_task_budget_limit(crate::budget::TaskBudgetLimit { + max_rounds: Some(100), + max_tool_calls: Some(100), + max_tokens: Some(1_000_000), + max_duration_secs: Some(3_600), + }); + + let result = session + .process_input(&ChatMessage::user("do the task"), &[], Some("core"), false) + .await; + assert!(result.is_ok(), "within-bounds turn must succeed"); + assert!(session.take_last_partial_turn().is_empty()); + assert_eq!(session.task_budget_state().task_rounds, 2); + assert_eq!(session.task_budget_state().tool_calls, 1); + } + + /// Issue #569 acceptance 2: "continue" resets only the segment counter, + /// never the cumulative task rounds. + #[tokio::test] + async fn task_rounds_accumulate_across_continue() { + use crate::agents::mock_tool_call_response; + + let log = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); + // Segment 1: 20 rounds then continue (segment counter resets), segment + // 2: 20 more rounds then stop. 40 rounds execute; the 41st scripted + // response is never consumed because the stop branch fires at the top + // of round 41 before any request. + let mut responses: Vec> = Vec::new(); + for i in 0..41 { + responses.push(Ok(mock_tool_call_response(&[( + format!("c{}", i).as_str(), + "grep", + "{}", + )]))); + } + let mut session = partial_turn_session(responses, log.clone()); + let first_vote = Arc::new(std::sync::atomic::AtomicBool::new(true)); + let vote = first_vote.clone(); + session.set_iteration_limit_callback(Arc::new(move |_| { + vote.swap(false, std::sync::atomic::Ordering::SeqCst) + })); + + let result = session + .process_input(&ChatMessage::user("do the task"), &[], Some("core"), false) + .await; + assert!(matches!(result, Err(AishError::IterationLimit))); + + // The task-level counter accumulated across the continue: 40 rounds. + assert_eq!( + session.task_budget_state().task_rounds, + 40, + "task rounds must accumulate across continues" + ); + assert_eq!(log.lock().unwrap().len(), 40); + // But the budget dimension was NOT the stopper. + assert!(session.take_budget_exhausted_dimension().is_none()); + } } diff --git a/crates/aish-llm/src/usage.rs b/crates/aish-llm/src/usage.rs index 31916911..831b3022 100644 --- a/crates/aish-llm/src/usage.rs +++ b/crates/aish-llm/src/usage.rs @@ -3,13 +3,18 @@ pub struct TokenUsage { pub prompt_tokens: u64, pub completion_tokens: u64, + /// Prompt tokens served from the provider's prompt cache (reused + /// prefix, not new work). OpenAI: `prompt_tokens_details.cached_tokens`; + /// Anthropic: `cache_read_input_tokens`. + pub cache_read_tokens: u64, + /// Prompt tokens written to the provider's prompt cache (billed as new + /// work). Anthropic: `cache_creation_input_tokens`; OpenAI-compatible + /// gateways usually fold this into `prompt_tokens` and report 0 here. + pub cache_write_tokens: u64, } impl TokenUsage { /// Extract token usage from an OpenAI-compatible API response JSON. - /// - /// Looks for `usage.prompt_tokens` and `usage.completion_tokens`. - /// Returns default (zeroed) if the fields are missing. pub fn from_response_json(json: &serde_json::Value) -> Self { let usage = json.get("usage"); Self { @@ -21,6 +26,12 @@ impl TokenUsage { .and_then(|u| u.get("completion_tokens")) .and_then(|v| v.as_u64()) .unwrap_or(0), + cache_read_tokens: usage + .and_then(|u| u.get("prompt_tokens_details")) + .and_then(|d| d.get("cached_tokens")) + .and_then(|v| v.as_u64()) + .unwrap_or(0), + cache_write_tokens: 0, } } @@ -36,6 +47,14 @@ impl TokenUsage { .and_then(|u| u.get("output_tokens")) .and_then(|v| v.as_u64()) .unwrap_or(0), + cache_read_tokens: usage + .and_then(|u| u.get("cache_read_input_tokens")) + .and_then(|v| v.as_u64()) + .unwrap_or(0), + cache_write_tokens: usage + .and_then(|u| u.get("cache_creation_input_tokens")) + .and_then(|v| v.as_u64()) + .unwrap_or(0), } } } @@ -46,6 +65,14 @@ pub struct TokenStats { pub total_input: u64, pub total_output: u64, pub request_count: u64, + /// Cumulative prompt tokens served from the provider's prompt cache. + /// Reused prefix, not new work — excluded from `total_tokens()` so + /// budgets measure fresh consumption only (same accounting as omp + /// goals: input + cacheWrite + output, cacheRead excluded). + pub total_cache_read: u64, + /// Cumulative prompt tokens written to the provider's prompt cache. + /// Billed as new work — included in `total_input`. + pub total_cache_write: u64, /// Prompt tokens from the most recent API call — the actual context /// window consumption at the current conversation depth. pub last_prompt_tokens: u64, @@ -56,6 +83,8 @@ impl TokenStats { pub fn record(&mut self, usage: TokenUsage) { self.total_input += usage.prompt_tokens; self.total_output += usage.completion_tokens; + self.total_cache_read += usage.cache_read_tokens; + self.total_cache_write += usage.cache_write_tokens; self.last_prompt_tokens = usage.prompt_tokens; self.request_count += 1; } @@ -67,10 +96,13 @@ impl TokenStats { pub fn merge_totals(&mut self, other: &TokenStats) { self.total_input += other.total_input; self.total_output += other.total_output; + self.total_cache_read += other.total_cache_read; + self.total_cache_write += other.total_cache_write; self.request_count += other.request_count; } - /// Total tokens consumed (input + output). + /// Total fresh tokens consumed (input + output; cached prefix reads are + /// excluded because they are reused context, not new work). pub fn total_tokens(&self) -> u64 { self.total_input + self.total_output } @@ -120,14 +152,52 @@ mod tests { stats.record(TokenUsage { prompt_tokens: 100, completion_tokens: 50, + cache_read_tokens: 0, + cache_write_tokens: 0, }); stats.record(TokenUsage { prompt_tokens: 200, completion_tokens: 80, + cache_read_tokens: 40, + cache_write_tokens: 10, }); assert_eq!(stats.total_input, 300); assert_eq!(stats.total_output, 130); + assert_eq!(stats.total_cache_read, 40); + assert_eq!(stats.total_cache_write, 10); assert_eq!(stats.request_count, 2); assert_eq!(stats.total_tokens(), 430); } + + #[test] + fn test_token_usage_openai_cached_tokens() { + let json = serde_json::json!({ + "usage": { + "prompt_tokens": 150, + "completion_tokens": 50, + "prompt_tokens_details": { "cached_tokens": 120 } + } + }); + let usage = TokenUsage::from_response_json(&json); + assert_eq!(usage.prompt_tokens, 150); + assert_eq!(usage.cache_read_tokens, 120); + assert_eq!(usage.cache_write_tokens, 0); + } + + #[test] + fn test_token_usage_anthropic_cache_fields() { + let json = serde_json::json!({ + "usage": { + "input_tokens": 30, + "output_tokens": 40, + "cache_read_input_tokens": 120, + "cache_creation_input_tokens": 25 + } + }); + let usage = TokenUsage::from_anthropic_json(&json); + assert_eq!(usage.prompt_tokens, 30); + assert_eq!(usage.completion_tokens, 40); + assert_eq!(usage.cache_read_tokens, 120); + assert_eq!(usage.cache_write_tokens, 25); + } } diff --git a/crates/aish-session/src/lib.rs b/crates/aish-session/src/lib.rs index 70c0d349..77017e86 100644 --- a/crates/aish-session/src/lib.rs +++ b/crates/aish-session/src/lib.rs @@ -18,6 +18,6 @@ pub mod store; pub use models::{ AuditEventRecord, AuditQuery, HistoryEntry, SessionContextMessage, SessionRecord, - SessionStateSnapshot, + SessionStateSnapshot, TaskBudgetSnapshot, }; pub use store::{AuditStore, SessionStore}; diff --git a/crates/aish-session/src/models.rs b/crates/aish-session/src/models.rs index 6cb7e526..37638970 100644 --- a/crates/aish-session/src/models.rs +++ b/crates/aish-session/src/models.rs @@ -42,6 +42,31 @@ pub struct SessionStateSnapshot { #[serde(default)] pub context_messages_snapshot: Vec, pub updated_at: Option>, + /// Task-level cumulative budget counters (issue #569), persisted so a + /// resumed session keeps accruing against the same task budget instead + /// of restarting it. Pure data: the runtime `Instant` anchor is + /// process-local and is never serialized. + #[serde(default)] + pub task_budget: Option, +} + +/// Serializable snapshot of the task budget state (issue #569). +#[derive(Debug, Clone, Serialize, Deserialize, Default)] +pub struct TaskBudgetSnapshot { + pub task_rounds: u64, + pub tool_calls: u64, + pub active_secs: u64, + pub total_input: u64, + pub total_output: u64, + pub total_cache_read: u64, + pub total_cache_write: u64, + pub request_count: u64, + /// Upper bounds persisted alongside the counters so resuming cannot + /// bypass the budget with a config that no longer matches. + pub max_rounds: Option, + pub max_tool_calls: Option, + pub max_tokens: Option, + pub max_duration_secs: Option, } impl SessionRecord { @@ -125,3 +150,47 @@ impl AuditQuery { Self::default() } } + +#[cfg(test)] +mod task_budget_snapshot_tests { + use super::*; + + #[test] + fn task_budget_snapshot_round_trips_through_json() { + let snapshot = SessionStateSnapshot { + cwd: Some("/tmp".into()), + summary_preview: None, + context_messages_snapshot: Vec::new(), + updated_at: None, + task_budget: Some(TaskBudgetSnapshot { + task_rounds: 25, + tool_calls: 31, + active_secs: 95, + total_input: 12_345, + total_output: 6_789, + total_cache_read: 999, + total_cache_write: 111, + request_count: 25, + max_rounds: Some(200), + max_tool_calls: None, + max_tokens: Some(2_000_000), + max_duration_secs: Some(3_600), + }), + }; + let json = serde_json::to_value(&snapshot).expect("serialize"); + // Legacy compatibility: an old snapshot without the field must parse. + let legacy = serde_json::json!({"cwd": "/tmp"}); + let parsed_legacy: SessionStateSnapshot = + serde_json::from_value(legacy).expect("legacy snapshot must parse"); + assert!(parsed_legacy.task_budget.is_none()); + + let parsed: SessionStateSnapshot = serde_json::from_value(json).expect("deserialize"); + let budget = parsed.task_budget.expect("budget present"); + assert_eq!(budget.task_rounds, 25); + assert_eq!(budget.tool_calls, 31); + assert_eq!(budget.active_secs, 95); + assert_eq!(budget.total_cache_read, 999); + assert_eq!(budget.max_rounds, Some(200)); + assert_eq!(budget.max_duration_secs, Some(3_600)); + } +} diff --git a/crates/aish-session/src/store.rs b/crates/aish-session/src/store.rs index 09c930ac..82a574c7 100644 --- a/crates/aish-session/src/store.rs +++ b/crates/aish-session/src/store.rs @@ -930,6 +930,7 @@ mod tests { reasoning_content: None, }], updated_at: Some(Utc::now()), + task_budget: None, }; store @@ -979,6 +980,7 @@ mod tests { summary_preview: Some("Rollback: systemctl revert nginx".to_string()), context_messages_snapshot: vec![summary_msg, tail_user], updated_at: Some(Utc::now()), + task_budget: None, }; store .update_session_state(&record.session_uuid, &snapshot) @@ -1338,6 +1340,7 @@ mod tests { reasoning_content: None, }], updated_at: Some(Utc::now()), + task_budget: None, }; store .update_session_state(&parent.session_uuid, &snapshot) diff --git a/crates/aish-shell/src/ai_handler.rs b/crates/aish-shell/src/ai_handler.rs index 14c6178d..c7c2714a 100644 --- a/crates/aish-shell/src/ai_handler.rs +++ b/crates/aish-shell/src/ai_handler.rs @@ -183,9 +183,23 @@ pub struct AiHandler { /// Tool-message count committed by the most recent `commit_partial_turn` /// (0 when the last turn succeeded or failed before any tool ran). last_partial_turn_steps: usize, - /// Whether the most recent partial turn was stopped at the iteration - /// limit (as opposed to a provider failure); selects the user hint. - last_partial_turn_limit: bool, + /// Why the most recent partial turn stopped; selects the user hint and + /// the synthetic note written into the persisted context. + last_partial_turn_reason: PartialTurnStopReason, +} + +/// Why a partial turn stopped (issue #569 review: replaces the +/// `stopped_by_limit` / `stopped_by_budget` boolean pair, whose stale-flag +/// interactions made the persisted note and user hint diverge). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum PartialTurnStopReason { + /// Provider died mid-turn (#452) — partial evidence. + #[default] + ProviderError, + /// User chose to stop at the 20-round iteration limit (#572). + IterationLimit, + /// Task-level cumulative budget exhausted (#569). + BudgetExhausted, } impl AiHandler { @@ -220,7 +234,7 @@ impl AiHandler { auto_search, secret_redactor: None, last_partial_turn_steps: 0, - last_partial_turn_limit: false, + last_partial_turn_reason: PartialTurnStopReason::ProviderError, } } @@ -631,7 +645,7 @@ impl AiHandler { // Clear the stale partial-turn count so a later error path // never reports evidence from an already-completed turn. self.last_partial_turn_steps = 0; - self.last_partial_turn_limit = false; + self.last_partial_turn_reason = PartialTurnStopReason::ProviderError; result } Err(err) => { @@ -643,6 +657,11 @@ impl AiHandler { // already done and must not blindly redo. if matches!(err, aish_core::AishError::IterationLimit) { self.commit_partial_turn_stopped_at_limit(&question_processed); + } else if matches!(err, aish_core::AishError::BudgetExhausted(_)) { + // Issue #569: task budget exhausted — same evidence + // contract, but the stop note must say budget, not + // provider failure or iteration limit. + self.commit_partial_turn_stopped_at_budget(&question_processed); } else { self.commit_partial_turn(&question_processed); } @@ -702,32 +721,138 @@ impl AiHandler { /// sees the evidence instead of dangling calls. fn commit_partial_turn(&mut self, question_processed: &str) { let partial = self.llm_session.take_last_partial_turn(); - let stopped_by_limit = false; - self.commit_partial_messages(question_processed, &partial, stopped_by_limit); + self.commit_partial_messages( + question_processed, + &partial, + PartialTurnStopReason::ProviderError, + ); } /// Issue #572 variant: commit a partial turn that ended because the - /// user chose to stop at the tool-call iteration limit. The synthetic - /// note and the user hint differ from the provider-failure case so the - /// persisted record is not misdescribed as a provider error. + /// user chose to stop at the tool-call iteration limit. fn commit_partial_turn_stopped_at_limit(&mut self, question_processed: &str) { let partial = self.llm_session.take_last_partial_turn(); - self.commit_partial_messages(question_processed, &partial, true); + self.commit_partial_messages( + question_processed, + &partial, + PartialTurnStopReason::IterationLimit, + ); + } + + /// Issue #569 variant: commit a partial turn that ended because the + /// task-level cumulative budget was exhausted. + fn commit_partial_turn_stopped_at_budget(&mut self, question_processed: &str) { + let partial = self.llm_session.take_last_partial_turn(); + self.commit_partial_messages( + question_processed, + &partial, + PartialTurnStopReason::BudgetExhausted, + ); + } + + /// Why the most recent partial turn stopped; used by the shell to pick + /// the user hint. + pub fn last_partial_turn_reason(&self) -> PartialTurnStopReason { + self.last_partial_turn_reason + } + /// Read-only access to the task budget state for /token and the footer. + pub fn llm_budget_state(&self) -> aish_llm::budget::TaskBudgetState { + self.llm_session.task_budget_state() + } + + /// Update the task budget bounds at runtime (explicit adjustment only). + pub fn set_llm_budget_limit(&self, limit: aish_llm::budget::TaskBudgetLimit) { + self.llm_session.set_task_budget_limit(limit); + } + + /// Install limits from `config.yaml llm.task_budget` (issue #569). + /// Absent dimensions stay unlimited; counters still accrue for display. + pub fn apply_task_budget_config(&self, config: &aish_config::TaskBudgetConfig) { + self.llm_session + .set_task_budget_limit(aish_llm::budget::TaskBudgetLimit { + max_rounds: config.max_rounds, + max_tool_calls: config.max_tool_calls, + max_tokens: config.max_tokens, + max_duration_secs: config.max_duration_secs, + }); + } + + /// Re-arm the near-threshold advisory after an explicit budget raise. + pub fn rearm_budget_advisory(&self) { + self.llm_session.rearm_near_threshold(); + } + + /// Read-only access to the task budget limits for /token and the footer. + pub fn llm_budget_limit(&self) -> aish_llm::budget::TaskBudgetLimit { + self.llm_session.task_budget_limit() + } + + /// Issue #569: serializable budget counters + limits for the session + /// snapshot. `None` when the process never bounded or accrued anything + /// (fresh session, no config limits) — keeps legacy snapshots identical. + pub fn task_budget_snapshot(&self) -> Option { + let state = self.llm_session.task_budget_state(); + let limit = self.llm_session.task_budget_limit(); + if !limit.is_bounded() && state.task_rounds == 0 && state.tool_calls == 0 { + return None; + } + Some(aish_session::TaskBudgetSnapshot { + task_rounds: state.task_rounds, + tool_calls: state.tool_calls, + active_secs: state.active_secs, + total_input: state.token_usage.total_input, + total_output: state.token_usage.total_output, + total_cache_read: state.token_usage.total_cache_read, + total_cache_write: state.token_usage.total_cache_write, + request_count: state.token_usage.request_count, + max_rounds: limit.max_rounds, + max_tool_calls: limit.max_tool_calls, + max_tokens: limit.max_tokens, + max_duration_secs: limit.max_duration_secs, + }) + } + + /// Issue #569: restore budget counters + limits from a session snapshot + /// on resume so the task keeps accruing against the same budget. + pub fn seed_task_budget(&self, snapshot: aish_session::TaskBudgetSnapshot) { + use aish_llm::usage::TokenStats; + let mut state = aish_llm::budget::TaskBudgetState::default(); + state.task_rounds = snapshot.task_rounds; + state.tool_calls = snapshot.tool_calls; + state.active_secs = snapshot.active_secs; + state.token_usage = TokenStats { + total_input: snapshot.total_input, + total_output: snapshot.total_output, + total_cache_read: snapshot.total_cache_read, + total_cache_write: snapshot.total_cache_write, + request_count: snapshot.request_count, + last_prompt_tokens: 0, + }; + self.llm_session.seed_task_budget_state(state); + self.llm_session + .set_task_budget_limit(aish_llm::budget::TaskBudgetLimit { + max_rounds: snapshot.max_rounds, + max_tool_calls: snapshot.max_tool_calls, + max_tokens: snapshot.max_tokens, + max_duration_secs: snapshot.max_duration_secs, + }); } /// Shared commit path: append the user message, the completed /// tool-call/tool-result pairs (redacted), and a synthetic assistant - /// note describing why the turn stopped. `stopped_by_limit` selects the - /// note text and the user-hint state exposed to the shell. + /// note describing why the turn stopped. `reason` selects the note text + /// and the user-hint state exposed to the shell. The reason is recorded + /// unconditionally — including the empty-partial case — so the flag can + /// never go stale. fn commit_partial_messages( &mut self, question_processed: &str, partial: &[ChatMessage], - stopped_by_limit: bool, + reason: PartialTurnStopReason, ) { + self.last_partial_turn_reason = reason; if partial.is_empty() { self.last_partial_turn_steps = 0; - self.last_partial_turn_limit = false; return; } // The session prepends the user message; the shell commits it here @@ -738,7 +863,6 @@ impl AiHandler { partial }; self.last_partial_turn_steps = tool_messages.len(); - self.last_partial_turn_limit = stopped_by_limit; if tool_messages.is_empty() { return; @@ -753,22 +877,28 @@ impl AiHandler { // synthetic assistant note so the next turn's model understands why // the transcript stops mid-task — and that the evidence above was // already executed, provider failure or deliberate stop alike. - let note = if stopped_by_limit { - "[turn stopped by the user at the tool-call iteration limit; the tool results above are completed evidence — do not re-run completed side effects]" - } else { - "[turn interrupted by a provider error before completion; the tool results above are partial evidence — do not re-run completed side effects]" + let note = match reason { + PartialTurnStopReason::IterationLimit => { + "[turn stopped by the user at the tool-call iteration limit; the tool results above are completed evidence — do not re-run completed side effects]" + } + PartialTurnStopReason::BudgetExhausted => { + "[turn stopped because the task budget was exhausted; the tool results above are completed evidence — do not re-run completed side effects]" + } + PartialTurnStopReason::ProviderError => { + "[turn interrupted by a provider error before completion; the tool results above are partial evidence — do not re-run completed side effects]" + } }; self.context_manager .add_message("assistant", note, MemoryType::Llm); self.context_manager.trim(); - let reason = if stopped_by_limit { - "iteration-limit stop" - } else { - "provider failure" + let reason_str = match reason { + PartialTurnStopReason::IterationLimit => "iteration-limit stop", + PartialTurnStopReason::BudgetExhausted => "budget-exhausted stop", + PartialTurnStopReason::ProviderError => "provider failure", }; tracing::info!( messages = tool_messages.len(), - reason, + reason = reason_str, "Persisted partial-turn tool evidence" ); } @@ -780,13 +910,6 @@ impl AiHandler { self.last_partial_turn_steps } - /// Whether the most recent partial turn was stopped at the iteration - /// limit (issue #572) rather than by a provider failure; used by the - /// shell to pick the matching user hint. - pub fn last_partial_turn_stopped_at_limit(&self) -> bool { - self.last_partial_turn_limit - } - /// Convert a turn's intermediate ChatMessages into persistable /// ContextMessages, applying secret redaction and a size cap to tool /// results before they enter long-lived context. @@ -3092,10 +3215,17 @@ mod tests { ChatMessage::tool_result("call_l1", "build finished"), ]; - handler.commit_partial_messages("do the task", &partial, true); + handler.commit_partial_messages( + "do the task", + &partial, + PartialTurnStopReason::IterationLimit, + ); - // The flag the shell hint reads is set. - assert!(handler.last_partial_turn_stopped_at_limit()); + // The reason the shell hint reads is recorded. + assert_eq!( + handler.last_partial_turn_reason(), + PartialTurnStopReason::IterationLimit + ); assert_eq!(handler.last_partial_turn_step_count(), 2); let context = handler.build_context_messages(); @@ -3130,9 +3260,16 @@ mod tests { ChatMessage::tool_result("call_f1", "match found"), ]; - handler.commit_partial_messages("do the task", &partial, false); + handler.commit_partial_messages( + "do the task", + &partial, + PartialTurnStopReason::ProviderError, + ); - assert!(!handler.last_partial_turn_stopped_at_limit()); + assert_eq!( + handler.last_partial_turn_reason(), + PartialTurnStopReason::ProviderError + ); let context = handler.build_context_messages(); let note = context .last() @@ -3141,6 +3278,61 @@ mod tests { assert!(note.contains("provider error"), "{note}"); } + /// Issue #569 review C3: a budget-exhausted stop must write the + /// budget-specific note into the persisted context — never the + /// iteration-limit wording. + #[test] + fn commit_partial_messages_uses_budget_note_on_budget_stop() { + let mut handler = test_handler(); + + let partial = [ + ChatMessage::user("do the task"), + { + let mut a = ChatMessage::assistant(""); + a.content = None; + a.tool_calls = Some(vec![aish_llm::ToolCall { + id: "call_b1".to_string(), + name: "bash".to_string(), + arguments: "{}".to_string(), + }]); + a + }, + ChatMessage::tool_result("call_b1", "done"), + ]; + + handler.commit_partial_messages( + "do the task", + &partial, + PartialTurnStopReason::BudgetExhausted, + ); + + assert_eq!( + handler.last_partial_turn_reason(), + PartialTurnStopReason::BudgetExhausted + ); + let context = handler.build_context_messages(); + let note = context + .last() + .and_then(|m| m.text_content()) + .unwrap_or_default(); + assert!(note.contains("task budget was exhausted"), "{note}"); + assert!(!note.contains("iteration limit"), "{note}"); + assert!(!note.contains("provider error"), "{note}"); + } + + /// Review C3: an empty partial must still record the stop reason + /// (steps reset to 0), so the shell hint never goes stale. + #[test] + fn commit_partial_messages_records_reason_even_when_partial_is_empty() { + let mut handler = test_handler(); + handler.commit_partial_messages("do the task", &[], PartialTurnStopReason::BudgetExhausted); + assert_eq!( + handler.last_partial_turn_reason(), + PartialTurnStopReason::BudgetExhausted + ); + assert_eq!(handler.last_partial_turn_step_count(), 0); + } + /// Bind a loopback server that replies to POST /v1/chat/completions with /// a canned OpenAI JSON completion (401 -> immediate, non-retryable /// failure; used to drive the manual-compact failure path fast). @@ -3723,4 +3915,58 @@ mod tests { aish_llm::ToolResult::success("probe ran") } } + + /// Issue #569 acceptance 5: seeding the budget from a snapshot restores + /// both the cumulative counters and the limits, and drops the process- + /// local wall-clock anchor so a resumed session never inherits a stale + /// Instant. + #[test] + fn seed_task_budget_restores_counters_and_limits() { + use aish_llm::budget::TaskBudgetLimit; + use aish_session::TaskBudgetSnapshot; + + let handler = test_handler(); + handler.seed_task_budget(TaskBudgetSnapshot { + task_rounds: 25, + tool_calls: 31, + active_secs: 95, + total_input: 12_345, + total_output: 6_789, + total_cache_read: 999, + total_cache_write: 111, + request_count: 25, + max_rounds: Some(200), + max_tool_calls: None, + max_tokens: Some(2_000_000), + max_duration_secs: Some(3_600), + }); + + let state = handler.llm_session.task_budget_state(); + assert_eq!(state.task_rounds, 25); + assert_eq!(state.tool_calls, 31); + assert_eq!(state.active_secs, 95); + assert_eq!(state.token_usage.total_input, 12_345); + assert_eq!(state.token_usage.total_cache_read, 999); + assert!(state.last_accounted.is_none(), "anchor must not be seeded"); + + let limit = handler.llm_session.task_budget_limit(); + assert_eq!(limit.max_rounds, Some(200)); + assert_eq!(limit.max_tokens, Some(2_000_000)); + assert_eq!(limit.max_duration_secs, Some(3_600)); + assert!(!TaskBudgetLimit::default().is_bounded()); + assert!(limit.is_bounded()); + + // The snapshot serializer round-trips the restored state. + let snapshot = handler.task_budget_snapshot().expect("bounded task"); + assert_eq!(snapshot.task_rounds, 25); + assert_eq!(snapshot.max_rounds, Some(200)); + } + + /// A fresh, unbounded handler yields no snapshot payload — legacy + /// sessions stay byte-identical. + #[test] + fn task_budget_snapshot_is_none_for_fresh_unbounded_session() { + let handler = test_handler(); + assert!(handler.task_budget_snapshot().is_none()); + } } diff --git a/crates/aish-shell/src/app.rs b/crates/aish-shell/src/app.rs index e3c80e6f..de5aff07 100644 --- a/crates/aish-shell/src/app.rs +++ b/crates/aish-shell/src/app.rs @@ -2201,6 +2201,9 @@ impl AishShell { context_budget_policy, config.skills.auto_search, ); + // Issue #569: install the configured task budget limits so the tool + // loop enforces them from the very first round. + ai_handler.apply_task_budget_config(&config.task_budget); // Redact secrets from tool outputs before they enter persistent // LLM context (same scanner as the audit path). @@ -2927,7 +2930,13 @@ impl AishShell { // Errors are already displayed via the LlmEventType::Error // event callback — avoid printing twice. Only handle // non-LLM errors that bypass the event system. - if !matches!( + if let aish_core::AishError::BudgetExhausted(dim) = &e { + // Issue #569 acceptance 5: the stop + // panel offers an explicit budget + // adjustment (recorded to audit) — + // plain "continue" never raises it. + self.handle_budget_exhaustion(dim); + } else if !matches!( e, aish_core::AishError::Llm(_) | aish_core::AishError::IterationLimit @@ -2942,11 +2951,6 @@ impl AishShell { } } - // InputGuard pre-check for AI prompts - if !self.screen_ai_prompt(&question) { - continue; - } - // Security gate: detect secrets in AI input if !self.check_security_gate(&mut question) { continue; @@ -3176,7 +3180,11 @@ impl AishShell { println!("{}", theme::warning(&t("shell.interrupted"))); } Err(e) => { - if !matches!( + if let aish_core::AishError::BudgetExhausted(dim) = &e { + // Issue #569 review C4: the main path must + // also offer the audited adjustment panel. + self.handle_budget_exhaustion(dim); + } else if !matches!( e, aish_core::AishError::Llm(_) | aish_core::AishError::IterationLimit ) { @@ -3373,7 +3381,10 @@ impl AishShell { println!("{}", theme::warning(&t("shell.interrupted"))); } Err(e) => { - if !matches!( + if let aish_core::AishError::BudgetExhausted(dim) = &e { + // Issue #569 review C4. + self.handle_budget_exhaustion(dim); + } else if !matches!( e, aish_core::AishError::Llm(_) | aish_core::AishError::IterationLimit @@ -3753,7 +3764,10 @@ impl AishShell { println!("{}", theme::warning(&t("shell.interrupted"))); } Err(e) => { - if !matches!( + if let aish_core::AishError::BudgetExhausted(dim) = &e { + // Issue #569 review C4. + self.handle_budget_exhaustion(dim); + } else if !matches!( e, aish_core::AishError::Llm(_) | aish_core::AishError::IterationLimit ) { @@ -3845,12 +3859,18 @@ impl AishShell { if saved_steps == 0 { return false; } - // Issue #572: a deliberate iteration-limit stop is not a failure; - // use the matching hint so the saved evidence is not misdescribed. - let key = if self.ai_handler.last_partial_turn_stopped_at_limit() { - "shell.error.partial_turn_saved_limit" - } else { - "shell.error.partial_turn_saved" + // Issue #572/#569: deliberate stops are not failures; use the + // matching hint so the saved evidence is not misdescribed. + let key = match self.ai_handler.last_partial_turn_reason() { + crate::ai_handler::PartialTurnStopReason::BudgetExhausted => { + "shell.error.partial_turn_saved_budget" + } + crate::ai_handler::PartialTurnStopReason::IterationLimit => { + "shell.error.partial_turn_saved_limit" + } + crate::ai_handler::PartialTurnStopReason::ProviderError => { + "shell.error.partial_turn_saved" + } }; let msg = t(key).replace("{steps}", &saved_steps.to_string()); println!("{}", theme::warning(&msg)); @@ -3858,6 +3878,119 @@ impl AishShell { true } + /// Issue #569: interactive panel shown when the task budget stops a + /// turn. Offers: stop (default) / adjust the budget explicitly / show + /// the saved-evidence hint. An adjustment is recorded to the audit + /// store with its before/after values — "continue" alone never raises + /// the limit. + fn handle_budget_exhaustion(&mut self, dimension: &str) { + let limit = self.ai_handler.llm_budget_limit(); + let state = self.ai_handler.llm_budget_state(); + let (used, bound) = match dimension { + "rounds" => (state.task_rounds, limit.max_rounds), + "tool_calls" => (state.tool_calls, limit.max_tool_calls), + "tokens" => (state.token_usage.total_tokens(), limit.max_tokens), + "duration" => (state.active_secs, limit.max_duration_secs), + _ => (0, None), + }; + println!(); + println!( + "{}", + theme::warning(&aish_i18n::t("shell.budget.exhausted_title")) + ); + println!( + " {}", + aish_i18n::t_with_args("shell.budget.exhausted_detail", &{ + let mut m = std::collections::HashMap::new(); + m.insert("dimension".to_string(), dimension.to_string()); + m.insert("used".to_string(), format_number(used)); + m.insert( + "limit".to_string(), + bound.map_or_else(|| "∞".to_string(), format_number), + ); + m + }) + ); + println!(" {}", aish_i18n::t("shell.budget.exhausted_options")); + print!(" "); + let _ = std::io::Write::flush(&mut std::io::stdout()); + + let choice = read_raw_confirmation(); + if !choice { + let msg = aish_i18n::t("shell.budget.stopped_by_user"); + println!("{}", theme::warning(&msg)); + return; + } + + // Explicit increase: prompt for the new value for the tripped + // dimension only. + println!( + " {}", + aish_i18n::t_with_args("shell.budget.new_limit_prompt", &{ + let mut m = std::collections::HashMap::new(); + m.insert("dimension".to_string(), dimension.to_string()); + m + }) + ); + print!(" "); + let _ = std::io::Write::flush(&mut std::io::stdout()); + + let mut new_limit_raw = String::new(); + if std::io::stdin().read_line(&mut new_limit_raw).is_ok() { + let parsed = new_limit_raw.trim().parse::().ok(); + let applied = parsed.filter(|v| *v > used); + if let Some(new_limit) = applied { + // Review C5: the audit record must capture the LIMIT's + // before/after values, not the usage. + let old_limit = bound.map_or_else(|| "unbounded".to_string(), |b| b.to_string()); + self.raise_budget_limit(dimension, new_limit); + // Audit the explicit adjustment with before/after values. + if let Some(ref audit) = self.audit_store { + let event = aish_core::AuditEvent::command( + chrono::Utc::now(), + Some(self.session_uuid.clone()), + self.audit_user.clone(), + self.audit_host.clone(), + format!( + "budget {} limit {} -> {} (used {})", + dimension, old_limit, new_limit, used + ), + "budget".to_string(), + 0, + ); + audit.record(event); + } + let msg = aish_i18n::t_with_args("shell.budget.raised", &{ + let mut m = std::collections::HashMap::new(); + m.insert("dimension".to_string(), dimension.to_string()); + m.insert("value".to_string(), format_number(new_limit)); + m + }); + println!("{}", theme::accent(&msg)); + } else { + let msg = aish_i18n::t("shell.budget.invalid_limit"); + println!("{}", theme::error(&msg)); + } + } + } + + /// Raise one budget dimension's limit in-place (also updates the LLM + /// session so the loop sees it immediately). + fn raise_budget_limit(&mut self, dimension: &str, new_limit: u64) { + let mut limit = self.ai_handler.llm_budget_limit(); + match dimension { + "rounds" => limit.max_rounds = Some(new_limit), + "tool_calls" => limit.max_tool_calls = Some(new_limit), + "tokens" => limit.max_tokens = Some(new_limit), + "duration" => limit.max_duration_secs = Some(new_limit), + _ => {} + } + self.ai_handler.set_llm_budget_limit(limit); + self.ai_handler.rearm_budget_advisory(); + } + + /// Issue #569: budget counters ride the session snapshot; limits are + /// persisted too so resuming cannot bypass the budget. fn session_state_snapshot( &self, updated_at: chrono::DateTime, @@ -3868,6 +4001,7 @@ impl AishShell { summary_preview: summary_preview_from_context(&context_messages), context_messages_snapshot: context_messages, updated_at: Some(updated_at), + task_budget: self.ai_handler.task_budget_snapshot(), } } @@ -7144,6 +7278,13 @@ impl AishShell { self.ai_handler .restore_context_messages(restored_context.clone()); + // Issue #569 acceptance 5: seed the persisted task-budget counters + // and limits so a resumed session keeps accruing against the same + // budget instead of restarting it. + if let Some(budget) = &snapshot.task_budget { + self.ai_handler.seed_task_budget(budget.clone()); + } + // Issue #530: a legacy session (snapshot but no transcript rows) // resumed here must not let its first new turn become the only // exported message — seed the snapshot into the transcript once. @@ -8611,7 +8752,16 @@ impl AishShell { self.refresh_config_dependent_tools(); } LiveEffect::ToolsRefresh => self.refresh_config_dependent_tools(), - LiveEffect::None => {} + LiveEffect::None => { + // Issue #569: the task-budget limit is read from the + // session on every loop round — push it immediately so + // /setting changes take effect without a restart. + if matches!(key, crate::settings_panel::SettingKey::TaskBudgetMaxRounds) { + self.ai_handler + .apply_task_budget_config(&self.config.task_budget); + self.ai_handler.rearm_budget_advisory(); + } + } } } @@ -8923,6 +9073,47 @@ impl AishShell { aish_i18n::t("shell.token.api_calls"), format_number(stats.request_count) ); + + // Issue #569: task budget usage alongside token totals. Cost is not + // reported by providers, so it is explicitly shown as unavailable. + let limit = self.ai_handler.llm_budget_limit(); + let state = self.ai_handler.llm_budget_state(); + if limit.is_bounded() { + println!(); + println!( + "{}", + theme::accent(&aish_i18n::t("shell.token.budget_title")) + ); + let row = |label: String, used: u64, limit: Option| { + let rest = match limit { + Some(max) => format!("{} / {}", format_number(used), format_number(max)), + None => format_number(used), + }; + println!(" {} {}", label, rest); + }; + row( + aish_i18n::t("shell.token.budget_rounds"), + state.task_rounds, + limit.max_rounds, + ); + row( + aish_i18n::t("shell.token.budget_tool_calls"), + state.tool_calls, + limit.max_tool_calls, + ); + row( + aish_i18n::t("shell.token.budget_tokens"), + state.token_usage.total_tokens(), + limit.max_tokens, + ); + row( + aish_i18n::t("shell.token.budget_duration"), + state.active_secs, + limit.max_duration_secs, + ); + let cost = aish_i18n::t("shell.token.budget_cost_unavailable"); + println!(" {} {}", aish_i18n::t("shell.token.budget_cost"), cost); + } println!(); } diff --git a/crates/aish-shell/src/settings_panel.rs b/crates/aish-shell/src/settings_panel.rs index 918c77ee..b65dd86c 100644 --- a/crates/aish-shell/src/settings_panel.rs +++ b/crates/aish-shell/src/settings_panel.rs @@ -138,6 +138,7 @@ pub enum SettingKey { InlineDisableThinking, InlineEnforceJson, MaxLlmMessages, + TaskBudgetMaxRounds, EnableTokenEstimation, // Security InputGuardEnabled, @@ -200,6 +201,7 @@ impl SettingKey { SettingKey::InlineDisableThinking => "inline_disable_thinking", SettingKey::InlineEnforceJson => "inline_enforce_json", SettingKey::MaxLlmMessages => "max_llm_messages", + SettingKey::TaskBudgetMaxRounds => "task_budget_max_rounds", SettingKey::EnableTokenEstimation => "enable_token_estimation", SettingKey::InputGuardEnabled => "input_guard_enabled", SettingKey::EnableSandbox => "enable_sandbox", @@ -345,6 +347,11 @@ pub const SETTINGS: &[SettingDef] = &[ category: SettingCategory::Ai, kind: SettingKind::Int, }, + SettingDef { + key: SettingKey::TaskBudgetMaxRounds, + category: SettingCategory::Ai, + kind: SettingKind::Int, + }, SettingDef { key: SettingKey::EnableTokenEstimation, category: SettingCategory::Ai, @@ -698,6 +705,11 @@ pub fn current_raw(cfg: &ConfigModel, key: SettingKey) -> String { SettingKey::InlineDisableThinking => bool_str(cfg.inline_completion.disable_thinking), SettingKey::InlineEnforceJson => bool_str(cfg.inline_completion.enforce_json), SettingKey::MaxLlmMessages => cfg.max_llm_messages.to_string(), + SettingKey::TaskBudgetMaxRounds => cfg + .task_budget + .max_rounds + .map(|n| n.to_string()) + .unwrap_or_default(), SettingKey::EnableTokenEstimation => bool_str(cfg.enable_token_estimation), SettingKey::InputGuardEnabled | SettingKey::EnableSandbox @@ -855,6 +867,13 @@ pub fn apply(cfg: &mut ConfigModel, key: SettingKey, value: &str) -> Result<(), cfg.inline_completion.enforce_json = parse_bool(value)?; } SettingKey::MaxLlmMessages => cfg.max_llm_messages = parse_usize(value)?, + SettingKey::TaskBudgetMaxRounds => { + cfg.task_budget.max_rounds = if value.is_empty() { + None + } else { + Some(parse_usize(value)? as u64) + }; + } SettingKey::EnableTokenEstimation => cfg.enable_token_estimation = parse_bool(value)?, SettingKey::InputGuardEnabled | SettingKey::EnableSandbox @@ -1120,6 +1139,7 @@ mod tests { SettingKey::InlineDisableThinking, SettingKey::InlineEnforceJson, SettingKey::MaxLlmMessages, + SettingKey::TaskBudgetMaxRounds, SettingKey::EnableTokenEstimation, SettingKey::InputGuardEnabled, SettingKey::EnableSandbox, @@ -1163,6 +1183,24 @@ mod tests { } } + #[test] + fn task_budget_max_rounds_round_trips() { + // Unset renders as "" and blank input clears; a number applies. + let mut cfg = default_cfg(); + assert_eq!(current_raw(&cfg, SettingKey::TaskBudgetMaxRounds), ""); + + apply(&mut cfg, SettingKey::TaskBudgetMaxRounds, "200").unwrap(); + assert_eq!(cfg.task_budget.max_rounds, Some(200)); + assert_eq!(current_raw(&cfg, SettingKey::TaskBudgetMaxRounds), "200"); + + apply(&mut cfg, SettingKey::TaskBudgetMaxRounds, "").unwrap(); + assert_eq!(cfg.task_budget.max_rounds, None); + + // Rejects non-numeric input without mutating the live value. + assert!(apply(&mut cfg, SettingKey::TaskBudgetMaxRounds, "abc").is_err()); + assert_eq!(cfg.task_budget.max_rounds, None); + } + #[test] fn choice_current_value_is_case_normalized() { // Security enums always render as canonical lowercase options.