From 2a470fc2d217eccc3c51560cf19bc7c38a9ae276 Mon Sep 17 00:00:00 2001 From: Eric Lee Date: Thu, 24 Sep 2026 18:45:24 -0700 Subject: [PATCH] fix(agents): subagents run at their definition's or the parent's effort; Explore caps at low Delegating to a subagent took minutes at /effort max. A "what is this repo" question on the ChatGPT subscription (gpt-6-astra) spent 122 s inside one Explore subagent that read a handful of files. Two causes, both measured live: 1. run_agent built the subagent's QueryParams with no thinking_effort, so the wire boundary fell back to the saved settings.effort for EVERY subagent. The built-in Explore agent, whose job is fast read-only search, reasoned at max on every turn. An effort: in agent frontmatter was parsed and dropped. A session-only level (headless --effort, an unsaved /effort) never reached subagents. 2. The OpenAI provider had no subagent tier table, so Explore's `haiku` inherited the session model, the slowest GPT-6 tier. Fix: - query() captures its level on ToolContext.thinking_effort, beside rendered_system_prompt. - run_agent.resolve_subagent_effort: an authored definition's level is used as written, otherwise the parent's level (TS runAgent.ts:514-518). - A BUILT-IN definition's level is a ceiling: min(own, level in force). It adds nothing when nothing is configured, because the field alone is a 400 on Groq's Llama models and older Grok. - EXPLORE_AGENT.effort = "low". TS disables thinking for every non-fork subagent; that is deliberately not copied, because custom agents doing long proofs rely on the session level. - OpenAI subagent_tier_models (haiku->gpt-6-luna, sonnet->gpt-6-sol, opus/fable->gpt-6-astra) apply ONLY on the ChatGPT subscription, whose availability gate reads the account's own catalog. API-key and custom-base-URL sessions inherit, as TS agent.ts:97-112 does. Explore on the same task, 3 live runs each: 118-124 s before; 48-64 s after with model "inherit"; 31-35 s after with the model left to the definition (gpt-6-luna). Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 26 ++ src/agent/agent_definitions.py | 19 +- src/agent/agent_model.py | 42 ++- src/agent/run_agent.py | 43 +++ src/providers/__init__.py | 18 ++ src/query/query.py | 10 + src/tool_system/context.py | 10 + tests/test_ch08_subagents_round4.py | 73 ++++- tests/test_subagent_effort.py | 346 ++++++++++++++++++++++ tests/test_team_runtime_e2e.py | 18 ++ tests/workflow/test_runner_integration.py | 28 ++ 11 files changed, 620 insertions(+), 13 deletions(-) create mode 100644 tests/test_subagent_effort.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 7a3962a26..3b0f5e821 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,32 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Fixed + +- **Delegating to a subagent no longer takes minutes at `/effort max`.** A + subagent's requests carried no effort of their own, so every subagent fell + back to the saved `settings.effort`: the built-in Explore agent — the fast, + read-only search agent — reasoned at `max` on every turn; an `effort:` in an + agent definition's frontmatter was parsed and then ignored; and a + session-only level (headless `--effort`, or `/effort` on a client that does + not save preferences) never reached its subagents. A subagent now runs at + its definition's `effort`, else the level its parent runs at (the + reference's precedence). Explore caps its level at `low` — a ceiling, so a + session with no effort configured still sends none. On the ChatGPT + subscription (`gpt-6-astra`, `/effort max`), a "what is this repo" Explore + delegation fell from ~121 s to ~57 s (three runs each). This applies where + the wire takes an effort level; on first-party Anthropic, Explore runs on + `claude-haiku-4-5`, which takes none. Custom agents without an `effort:` + keep the session's level. +- **Subagent tier aliases resolve on the ChatGPT subscription.** `haiku` — the + tier the Explore agent asks for — now runs on `gpt-6-luna` (`sonnet` → + `gpt-6-sol`, `opus`/`fable` → `gpt-6-astra`) when your plan's model catalog + lists it. Before, every alias fell back to the session model (`gpt-6-astra` + in the reported session); with the model left to the agent definition the + same delegation takes ~33 s. API-key and custom-endpoint OpenAI sessions + keep inheriting the session model, as do an explicit `model: "inherit"` and + agents that name no model. + ## [1.7.0] - 2026-09-23 ### Added diff --git a/src/agent/agent_definitions.py b/src/agent/agent_definitions.py index 4444e78f0..47c08f4b1 100644 --- a/src/agent/agent_definitions.py +++ b/src/agent/agent_definitions.py @@ -165,10 +165,23 @@ def _explore_system_prompt(**_kwargs: Any) -> str: # ch08 round-4 (critic M1) — Explore is the fast/cheap read-only agent; # TS exploreAgent.ts:77 runs it on Haiku. get_agent_model resolves this # per provider via the ``subagent_tier_models`` tables (anthropic → - # claude-haiku-4-5, deepseek → deepseek-v4-flash) and inherits on - # providers without a haiku-class mapping, so it is cross-provider - # safe. + # claude-haiku-4-5, deepseek → deepseek-flash, openai on the ChatGPT + # subscription → gpt-6-luna) and inherits the session model everywhere + # else (API-key openai, custom endpoints, providers without a table). model="haiku", + # A CEILING, not a level of its own (run_agent.resolve_subagent_effort): + # Explore never reasons harder than low, and never adds an effort field + # to a session that configured none. Without it the fast agent reasoned + # at the session's /effort max on every search turn — a quick repo + # survey took ~121 s on gpt-6-astra vs ~57 s at low (same task and + # tools, three runs each, 2026-09-24). Only wires that take an effort + # level are affected: on first-party Anthropic Explore runs on + # claude-haiku-4-5, which takes none (its cost is its thinking budget). + # TS goes further — runAgent.ts:720-725 runs every non-fork subagent + # with thinking DISABLED — which this port deliberately does not copy: + # custom agents doing long proofs rely on the session level. A + # user/project ``Explore`` definition replaces this one. + effort="low", get_system_prompt=_explore_system_prompt, ) diff --git a/src/agent/agent_model.py b/src/agent/agent_model.py index f10e125ab..71d447aa8 100644 --- a/src/agent/agent_model.py +++ b/src/agent/agent_model.py @@ -10,12 +10,13 @@ * ``subagent_tier_models`` — the resolution targets for bare tier aliases (anthropic haiku → ``claude-haiku-4-5``; deepseek haiku → - ``deepseek-v4-flash``). This is the port of the TS reference's + ``deepseek-flash``; openai haiku → ``gpt-6-luna``, on the ChatGPT + subscription only). This is the port of the TS reference's per-provider ``getDefault{Opus,Sonnet,Haiku}Model()`` functions. * ``subagent_model`` — the model subagents run on when neither the Agent tool call nor the agent definition names one (anthropic: ``claude-haiku-4-5``, the cheapest current-gen tier; deepseek: - ``deepseek-v4-flash``), overridable per provider via + ``deepseek-flash``), overridable per provider via ``providers..subagent_model`` in config.json. NOTE this is a DELIBERATE DIVERGENCE from both references, by explicit user directive (cheap fan-outs): TS ``getDefaultSubagentModel()`` returns ``'inherit'`` @@ -38,7 +39,9 @@ the session model rather than 400-ing the request, and an unspecified model inherits. A custom Anthropic-compatible endpoint (proxy / self-hosted) also inherits rather than trusting the first-party table, mirroring TS -``checkIsClaudeNativeProvider``. +``checkIsClaudeNativeProvider``. The openai table applies only on the +ChatGPT subscription, the one route whose availability gate reads the +account's own catalog; API-key and custom-endpoint openai sessions inherit. Deliberate asymmetry: a KNOWN alias whose canonical target is retired degrades to inherit through the availability gate (``h35``, @@ -118,19 +121,44 @@ def _is_custom_anthropic(session_provider: Any) -> bool: return False +def _is_openai_without_live_catalog(session_provider: Any) -> bool: + """An openai provider that is NOT on the ChatGPT subscription route. + + The openai tier targets are trusted only where the availability gate is + a real per-account check: the subscription's Codex catalog + (``openai_subscription_models``). Everywhere else + ``get_available_models()`` is the static registry list, so the gate + passes every listed id — an API key whose project may not use + ``gpt-6-luna``, and a custom base URL (LiteLLM, vLLM, Azure, + ``$OPENAI_BASE_URL``) that serves none of them, would both be sent a + model they cannot serve. Those keep the reference behavior and inherit + (TS agent.ts:97-112 inherits haiku/sonnet on non-Claude-native + providers). The subscription only ever activates against the first-party + endpoint (``OpenAIProvider.__init__``). Read with an ``isinstance`` + check, like the agent-server's effort-options probe: a mock answers any + attribute with a truthy object. + """ + if _provider_id(session_provider) != "openai": + return False + active = getattr(_unwrap(session_provider), "_subscription_active", None) + return not (isinstance(active, bool) and active) + + def _provider_info_row(session_provider: Any) -> dict[str, Any]: """The session provider's PROVIDER_INFO row, or ``{}``. - Empty when the provider carries no ``provider_id`` (unregistered) or is - an anthropic provider on a custom endpoint — both mean "no subagent - table", so every lookup falls through to the reference (inherit) - behavior. + Empty when the provider carries no ``provider_id`` (unregistered), is an + anthropic provider on a custom endpoint, or is an openai provider off + the subscription route — each means "no subagent table", so every lookup + falls through to the reference (inherit) behavior. """ provider_id = _provider_id(session_provider) if not provider_id: return {} if _is_custom_anthropic(session_provider): return {} + if _is_openai_without_live_catalog(session_provider): + return {} try: from src.providers import PROVIDER_INFO diff --git a/src/agent/run_agent.py b/src/agent/run_agent.py index 86cae5370..9b3bcf5e2 100644 --- a/src/agent/run_agent.py +++ b/src/agent/run_agent.py @@ -179,6 +179,45 @@ def _build_permission_context( ) +def resolve_subagent_effort( + agent_definition: AgentDefinition, + parent_context: ToolContext, +) -> str | None: + """The reasoning-effort level a subagent's query runs at. + + TS runAgent.ts:514-518: ``agentDefinition.effort ?? state.effortValue`` + — the definition's own level, else the SESSION's. Here the session level + is whatever the spawning query runs at (``ToolContext.thinking_effort``, + captured by query() at turn entry). ``None`` leaves the wire boundary's + fallback to the persisted ``settings.effort`` — the same place the + parent's own requests land when it has no explicit level. + + A BUILT-IN definition's level is a ceiling on the level already in + force, not a level of its own: with nothing configured it must keep + sending nothing, because the field's mere presence is a hard 400 on some + wires (Groq's Llama models, older Grok) where the parent sends none. A + user/project/plugin definition's level is the author's explicit choice + and is used as written (TS parity). + + A definition level off the ladder (frontmatter also accepts integers, + TS's numeric effort, which no wire here takes) is treated as unset so + it falls through to the parent's level rather than to the settings + fallback. + """ + from ..query.query import VALID_THINKING_EFFORT_LEVELS, resolve_thinking_effort + + parent = getattr(parent_context, "thinking_effort", None) + own = (agent_definition.effort or "").strip().lower() + if own not in VALID_THINKING_EFFORT_LEVELS: + return parent + if not is_built_in_agent(agent_definition): + return own + in_force = resolve_thinking_effort(parent, None, clamp_xhigh=False) + if in_force is None: + return parent + return min(own, in_force, key=VALID_THINKING_EFFORT_LEVELS.index) + + def filter_incomplete_tool_calls(messages: list[Message]) -> list[Message]: """Remove assistant messages that contain incomplete tool_use blocks. @@ -436,6 +475,10 @@ async def run_agent(params: RunAgentParams) -> AsyncGenerator[Message, None]: abort_controller=abort_controller, query_source=effective_query_source, max_turns=max_turns, + # Unset, every subagent fell back to the persisted settings.effort: + # an ``effort:`` in the definition was dead, and a session-only level + # (headless --effort, an unsaved /effort) never reached the subagent. + thinking_effort=resolve_subagent_effort(agent_def, params.parent_context), ) terminal = TerminalHolder() diff --git a/src/providers/__init__.py b/src/providers/__init__.py index f83639b39..8687107a3 100644 --- a/src/providers/__init__.py +++ b/src/providers/__init__.py @@ -88,6 +88,24 @@ class ProviderInfo(_ProviderInfoOptional): "label": "OpenAI GPT", "default_base_url": "https://api.openai.com/v1", "default_model": "gpt-5.4", + # Tier aliases → GPT-6 tiers, applied ONLY on the ChatGPT + # subscription route (see agent_model._is_openai_without_live_catalog): + # that is the one place the availability gate checks the account's + # own catalog. An API key's gate is this static list and a custom + # base URL serves none of these, so both keep inheriting — which is + # also what TS does for Explore on OpenAI-shaped providers + # (agent.ts:97-112); this row is a deliberate divergence for the + # subscription. Without it Explore's ``haiku`` inherited the session + # model: ~57 s for a repo survey on gpt-6-astra vs ~33 s on + # gpt-6-luna, both at effort low (measured 2026-09-24). Deliberately + # NO ``subagent_model``: an agent with no model keeps the session + # model on this provider. + "subagent_tier_models": { + "fable": "gpt-6-astra", + "opus": "gpt-6-astra", + "sonnet": "gpt-6-sol", + "haiku": "gpt-6-luna", + }, "available_models": [ # https://developers.openai.com/api/docs/models (2026-09-24) # GPT-6 — Astra is the frontier tier, Sol the flagship, Luna the diff --git a/src/query/query.py b/src/query/query.py index 14365505b..2d4f86d2a 100644 --- a/src/query/query.py +++ b/src/query/query.py @@ -1738,6 +1738,16 @@ async def query( params.tool_use_context.rendered_system_prompt = params.system_prompt except Exception: # noqa: BLE001 — read-only stub context pass + # Same capture for the effort level: run_agent gives a subagent its + # definition's ``effort`` else THIS value (TS runAgent.ts:514-518). + # Without it a subagent's query carried no effort at all and fell back + # to the persisted ``settings.effort`` — a session-only level (headless + # ``--effort``, an unsaved ``/effort``) never reached it. Unconditional: + # ``None`` is meaningful (the parent itself runs on that fallback). + try: + params.tool_use_context.thinking_effort = params.thinking_effort + except Exception: # noqa: BLE001 — read-only stub context + pass # How much of the session-lifetime outbox predates THIS query. Entries # below the mark belong to earlier prompts and must not count as "the diff --git a/src/tool_system/context.py b/src/tool_system/context.py index f22d7d48b..55529bce2 100644 --- a/src/tool_system/context.py +++ b/src/tool_system/context.py @@ -283,6 +283,16 @@ class ToolContext: # _call_model_sync assembly → byte-identical wire prefix. rendered_system_prompt: "str | list[dict[str, Any]] | None" = None + # The explicit reasoning-effort level of the query running on this + # context (``QueryParams.thinking_effort`` — the session's ``/effort``, + # headless ``--effort``); ``None`` = unset, so the wire boundary falls + # back to the persisted ``settings.effort``. POPULATED by query() at turn + # entry beside ``rendered_system_prompt``, so a subagent spawned during + # the turn inherits the level its parent actually runs at — TS + # runAgent.ts:514-518 reads the session ``effortValue`` from app state + # for the same purpose (see ``run_agent.resolve_subagent_effort``). + thinking_effort: str | None = None + def __post_init__(self) -> None: self.workspace_root = Path(self.workspace_root).resolve() if self.cwd is None: diff --git a/tests/test_ch08_subagents_round4.py b/tests/test_ch08_subagents_round4.py index b14c0adde..0cb7798ea 100644 --- a/tests/test_ch08_subagents_round4.py +++ b/tests/test_ch08_subagents_round4.py @@ -129,6 +129,7 @@ def setUp(self): "ANTHROPIC_DEFAULT_SONNET_MODEL", "ANTHROPIC_DEFAULT_HAIKU_MODEL", "ANTHROPIC_BASE_URL", + "OPENAI_BASE_URL", ): os.environ.pop(var, None) # Hermetic: the resolver consults the user's real config.json for @@ -153,6 +154,72 @@ def _deepseek(model="deepseek-flash"): return DeepSeekProvider(api_key="test-key", model=model) + @staticmethod + def _openai(model="gpt-6-astra", **kwargs): + from src.providers.openai_provider import OpenAIProvider + + return OpenAIProvider(api_key="test-key", model=model, **kwargs) + + def _openai_subscription( + self, model="gpt-6-astra", catalog=("gpt-6-astra", "gpt-6-sol", "gpt-6-luna"), + ): + # A ChatGPT-subscription session without reading the developer's real + # login: the route flag plus the account catalog the gate consults. + p = self._openai(model) + p._subscription_active = True + p.get_available_models = lambda: list(catalog) + return p + + def test_openai_subscription_haiku_tier_uses_luna(self): + # Explore pins 'haiku'. With no openai row the alias went through the + # global Claude-id table, which OpenAI does not serve, and inherited + # the session model — every Explore ran on gpt-6-astra (~57 s vs + # ~33 s on luna for the same repo survey, 2026-09-24). + self.assertEqual( + get_agent_model(None, "haiku", self._openai_subscription()), "gpt-6-luna", + ) + + def test_openai_subscription_catalog_without_the_tier_inherits(self): + # The live gate: a login whose Codex catalog lacks luna must not be + # sent a model its account cannot use. + p = self._openai_subscription(catalog=("gpt-6-astra", "gpt-5.5")) + self.assertEqual(get_agent_model(None, "haiku", p), "gpt-6-astra") + + def test_openai_unspecified_model_still_inherits(self): + # Deliberately no openai ``subagent_model``: general-purpose and + # custom agents with no model keep the session model. + self.assertEqual( + get_agent_model(None, None, self._openai_subscription()), "gpt-6-astra", + ) + + def test_openai_explicit_inherit_beats_the_explore_tier(self): + # The slow session passed model: "inherit" on its Explore call; the + # tool param still wins over the definition's haiku. + self.assertEqual( + get_agent_model("inherit", "haiku", self._openai_subscription()), + "gpt-6-astra", + ) + + def test_openai_api_key_inherits(self): + # On an API key the gate is the static registry list, which passes + # gpt-6-luna whether or not the key's project may use it — so the + # table stays off and Explore keeps the session model it ran before. + self.assertEqual(get_agent_model(None, "haiku", self._openai()), "gpt-6-astra") + + def test_openai_custom_base_url_inherits(self): + # LiteLLM / vLLM / Azure proxies serve none of the GPT-6 ids. + p = self._openai("my-litellm-model", base_url="http://localhost:4000/v1") + self.assertEqual(get_agent_model(None, "haiku", p), "my-litellm-model") + + def test_openai_env_base_url_inherits(self): + # The other base-URL channel, with no key (the case the provider's + # own OAuth guard exists for): still no table. + from src.providers.openai_provider import OpenAIProvider + + os.environ["OPENAI_BASE_URL"] = "https://api.deepseek.com/v1" + p = OpenAIProvider(api_key="", model="deepseek-v4-pro") + self.assertEqual(get_agent_model(None, "haiku", p), "deepseek-v4-pro") + def test_anthropic_unspecified_uses_default_subagent_model(self): # Goal ask #1: on the anthropic provider the default subagent model # is claude-haiku-4-5 (the cheapest current-gen tier), NOT an @@ -321,9 +388,9 @@ def test_registry_subagent_targets_are_available(self): f"{name} does not list — the runtime gate would " "silently ignore it", ) - # Today: anthropic + deepseek. If this drops to zero the tables were - # deleted and this test should go with them. - self.assertGreaterEqual(checked, 2) + # Today: anthropic + deepseek + openai. If this drops to zero the + # tables were deleted and this test should go with them. + self.assertGreaterEqual(checked, 3) def test_tier_env_pin_beats_table(self): os.environ["ANTHROPIC_DEFAULT_HAIKU_MODEL"] = "my-bedrock-haiku" diff --git a/tests/test_subagent_effort.py b/tests/test_subagent_effort.py new file mode 100644 index 000000000..be917e050 --- /dev/null +++ b/tests/test_subagent_effort.py @@ -0,0 +1,346 @@ +"""Subagent reasoning effort: which level a spawned agent's requests carry. + +The bug these lock down (2026-09-24). ``run_agent`` built the subagent's +``QueryParams`` with no ``thinking_effort``, so the wire boundary fell back to +the persisted ``settings.effort`` for EVERY subagent: + +* the built-in Explore agent — "a fast agent that returns output as quickly + as possible" — reasoned at the session's max on every search turn. A quick + "what is this repo" delegation took ~121 s on gpt-6-astra at max vs ~57 s + at low (three live runs each, same task and tools); +* an agent definition's ``effort:`` frontmatter was parsed and then dropped; +* a session-only level — headless ``--effort``, or ``/effort`` on a client + that does not save preferences — never reached its subagents, which used + the persisted setting instead. + +TS runAgent.ts:514-518 resolves ``agentDefinition.effort ?? state.effortValue`` +(the session's level). The port now does the same: query() captures its level +on the ToolContext, and run_agent takes the definition's level, else that one — +except that a BUILT-IN definition's level (Explore's ``low``) is only a ceiling +on the level already in force, so a session that configured no effort keeps +sending none (the field alone is a 400 on some wires, e.g. Groq's Llama models). +""" + +from __future__ import annotations + +import asyncio +import tempfile +import unittest +from pathlib import Path +from types import SimpleNamespace +from unittest import mock +from unittest.mock import MagicMock + +from src.agent.agent_definitions import ( + EXPLORE_AGENT, + GENERAL_PURPOSE_AGENT, + PLAN_AGENT, + AgentDefinition, +) +from src.agent.run_agent import RunAgentParams, resolve_subagent_effort, run_agent +from src.providers.base import ChatResponse +from src.providers.openrouter_provider import OpenRouterProvider +from src.query.query import QueryParams, query +from src.tool_system.context import ToolContext +from src.tool_system.defaults import build_default_registry +from src.types.messages import UserMessage +from src.utils.abort_controller import AbortController + + +def _settings_effort(level: str): + """Pin the persisted ``settings.effort`` — the fallback a subagent used to + land on — so each test controls what "the old behavior" would send.""" + return mock.patch( + "src.settings.settings.get_settings", + return_value=SimpleNamespace(effort=level), + ) + + +def _text(content: str = "done") -> ChatResponse: + return ChatResponse( + content=content, + model="openai/gpt-5.6-luna", + usage={"input_tokens": 1, "output_tokens": 1}, + finish_reason="stop", + tool_uses=None, + ) + + +def _openai_compat_mock(*responses: ChatResponse) -> MagicMock: + """Takes the OpenAI-compatible wire branch (``reasoning_effort`` in + ``extra_body``). Streaming is forced into the ``chat()`` fallback so every + request's kwargs are on ``chat.call_args_list``. No ``provider_id`` table + applies, so a subagent keeps this same instance (no model clone).""" + provider = MagicMock(spec=OpenRouterProvider) + provider.model = "openai/gpt-5.6-luna" + provider.base_url = "https://openrouter.ai/api/v1" + provider.chat_stream_response.side_effect = NotImplementedError() + provider.chat.side_effect = list(responses) or [_text()] + return provider + + +def _wire_effort(call) -> str | None: + return (call.kwargs.get("extra_body") or {}).get("reasoning_effort") + + +class TestResolveSubagentEffort(unittest.TestCase): + """The precedence itself: an authored definition level, else the parent's + level, with a built-in's level capping — never adding — a level.""" + + def setUp(self): + # The ceiling reads the level in force, which falls back to + # settings.effort; pin it so a developer's own config can't leak in. + self._settings = _settings_effort("") + self._settings.start() + + def tearDown(self): + self._settings.stop() + + def _ctx(self, effort): + ctx = ToolContext(workspace_root=Path(tempfile.gettempdir())) + ctx.thinking_effort = effort + return ctx + + def test_authored_definition_effort_wins(self): + # A user/project agent's ``effort:`` is used as written (TS parity), + # above or below the session's level, and even with none configured. + agent = AgentDefinition(agent_type="t", when_to_use="t", source="project", effort="medium") + self.assertEqual(resolve_subagent_effort(agent, self._ctx("max")), "medium") + self.assertEqual(resolve_subagent_effort(agent, self._ctx("low")), "medium") + self.assertEqual(resolve_subagent_effort(agent, self._ctx(None)), "medium") + + def test_builtin_level_is_a_ceiling(self): + agent = AgentDefinition(agent_type="t", when_to_use="t", source="built-in", effort="medium") + self.assertEqual(resolve_subagent_effort(agent, self._ctx("max")), "medium") + self.assertEqual(resolve_subagent_effort(agent, self._ctx("low")), "low") + + def test_builtin_level_adds_nothing_to_an_unconfigured_session(self): + # Nothing set anywhere → the parent sends no effort field, so Explore + # must not either: the field alone is a 400 on some wires. + self.assertIsNone(resolve_subagent_effort(EXPLORE_AGENT, self._ctx(None))) + + def test_builtin_ceiling_sees_the_settings_fallback(self): + # The parent has no explicit level but runs on settings.effort=max + # (what an unsaved-level session looks like) — Explore still caps. + with _settings_effort("max"): + self.assertEqual(resolve_subagent_effort(EXPLORE_AGENT, self._ctx(None)), "low") + + def test_no_definition_effort_inherits_the_parent_level(self): + self.assertEqual( + resolve_subagent_effort(GENERAL_PURPOSE_AGENT, self._ctx("xhigh")), + "xhigh", + ) + + def test_parent_without_explicit_level_stays_unset(self): + # None is not "no effort": it means the settings fallback, which is + # exactly where the parent's own requests land. + self.assertIsNone(resolve_subagent_effort(GENERAL_PURPOSE_AGENT, self._ctx(None))) + + def test_off_ladder_definition_level_falls_through_to_the_parent(self): + # Frontmatter accepts integers (TS's numeric effort). No wire here + # takes one, and resolve_thinking_effort would treat it as unset and + # jump to settings — skipping the parent's explicit level. + agent = AgentDefinition(agent_type="t", when_to_use="t", effort="7") + self.assertEqual(resolve_subagent_effort(agent, self._ctx("high")), "high") + + def test_explore_declares_low_and_plan_inherits(self): + self.assertEqual(EXPLORE_AGENT.effort, "low") + self.assertEqual(resolve_subagent_effort(EXPLORE_AGENT, self._ctx("max")), "low") + # Plan does real design work; it keeps the session's level. + self.assertIsNone(PLAN_AGENT.effort) + self.assertEqual(resolve_subagent_effort(PLAN_AGENT, self._ctx("max")), "max") + + +class TestQueryCapturesItsEffort(unittest.TestCase): + """query() records its level on the context — the only way a subagent + spawned mid-turn can learn what its parent runs at.""" + + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.registry = build_default_registry() + self.context = ToolContext(workspace_root=Path(self.tmp.name)) + + def tearDown(self): + self.tmp.cleanup() + + def _drive(self, **extra): + params = QueryParams( + messages=[UserMessage(content="hi")], + system_prompt="hello", + tools=self.registry.list_tools(), + tool_registry=self.registry, + tool_use_context=self.context, + provider=_openai_compat_mock(), + abort_controller=AbortController(), + max_turns=1, + **extra, + ) + + async def run(): + async for _ in query(params): + pass + + asyncio.run(run()) + + def test_explicit_level_is_captured(self): + with _settings_effort("max"): + self._drive(thinking_effort="low") + self.assertEqual(self.context.thinking_effort, "low") + + def test_unset_level_overwrites_a_stale_capture(self): + # A context reused across turns must not keep the previous turn's + # level after the session drops back to the settings fallback. + self.context.thinking_effort = "xhigh" + with _settings_effort("max"): + self._drive() + self.assertIsNone(self.context.thinking_effort) + + +class TestSubagentEffortOnTheWire(unittest.TestCase): + """What a subagent's request actually carries.""" + + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.registry = build_default_registry() + + def tearDown(self): + self.tmp.cleanup() + + def _run(self, agent: AgentDefinition, parent_effort: str | None) -> MagicMock: + provider = _openai_compat_mock() + parent = ToolContext(workspace_root=Path(self.tmp.name)) + parent.thinking_effort = parent_effort + params = RunAgentParams( + parent_context=parent, + agent_definition=agent, + prompt="look around", + available_tools=self.registry.list_tools(), + tool_registry=self.registry, + provider=provider, + ) + + async def drain(): + async for _ in run_agent(params): + pass + + asyncio.run(drain()) + self.assertEqual(provider.chat.call_count, 1) + return provider + + def test_session_level_beats_the_persisted_setting(self): + # The precedence inversion: /effort high (or --effort high) in the + # session, max saved in settings. The subagent used to send max. + with _settings_effort("max"): + provider = self._run(GENERAL_PURPOSE_AGENT, "high") + self.assertEqual(_wire_effort(provider.chat.call_args), "high") + + def test_explore_sends_low_under_a_max_session(self): + # The reported slowness: Explore reasoned at the session's max. + with _settings_effort("max"): + provider = self._run(EXPLORE_AGENT, "max") + self.assertEqual(_wire_effort(provider.chat.call_args), "low") + + def test_explore_sends_low_even_when_only_settings_say_max(self): + # The TUI's common case: the session level equals settings.effort. + with _settings_effort("max"): + provider = self._run(EXPLORE_AGENT, None) + self.assertEqual(_wire_effort(provider.chat.call_args), "low") + + def test_explore_adds_no_field_to_an_unconfigured_session(self): + # The Groq/Llama shape: no effort anywhere, and a wire that 400s on + # the field's mere presence. Before the ceiling, Explore sent "low". + with _settings_effort(""): + provider = self._run(EXPLORE_AGENT, None) + self.assertIsNone(_wire_effort(provider.chat.call_args)) + + def test_definition_frontmatter_effort_reaches_the_wire(self): + agent = AgentDefinition( + agent_type="prover", + when_to_use="proofs", + tools=["*"], + source="project", + effort="xhigh", + get_system_prompt=lambda: "prove it", + ) + with _settings_effort("low"): + provider = self._run(agent, "medium") + self.assertEqual(_wire_effort(provider.chat.call_args), "xhigh") + + def test_no_level_anywhere_still_omits_the_field(self): + # Unchanged default path: nothing set → nothing sent. + with _settings_effort(""): + provider = self._run(GENERAL_PURPOSE_AGENT, None) + self.assertIsNone(_wire_effort(provider.chat.call_args)) + + +class TestEffortThroughTheAgentTool(unittest.TestCase): + """The whole chain: a main turn at one level delegates through the real + Agent tool, and the subagent's request carries the level it resolves to. + Requests are sequential (the sync Agent call blocks the main turn), so + chat.call_args_list reads [main, subagent, main].""" + + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + + def tearDown(self): + self.tmp.cleanup() + + def _delegate(self, subagent_type: str, main_effort: str) -> list: + provider = _openai_compat_mock( + ChatResponse( + content="", + model="openai/gpt-5.6-luna", + usage={"input_tokens": 1, "output_tokens": 1}, + finish_reason="tool_calls", + tool_uses=[{ + "id": "call_1", + "name": "Agent", + "input": { + "description": "Survey the repo", + "prompt": "What is this repo?", + "subagent_type": subagent_type, + }, + }], + ), + _text("the subagent's report"), + _text("the answer"), + ) + registry = build_default_registry(provider=provider) + context = ToolContext(workspace_root=Path(self.tmp.name)) + # Hermetic: the Agent tool would otherwise read agent definitions from + # the real config dirs, where a user's own Explore.md or + # general-purpose.md replaces the built-in under test. + context.options.agent_definitions = { + "active_agents": [GENERAL_PURPOSE_AGENT, EXPLORE_AGENT, PLAN_AGENT], + } + params = QueryParams( + messages=[UserMessage(content="what is this repo?")], + system_prompt="hello", + tools=registry.list_tools(), + tool_registry=registry, + tool_use_context=context, + provider=provider, + abort_controller=AbortController(), + max_turns=3, + thinking_effort=main_effort, + ) + + async def run(): + async for _ in query(params): + pass + + with _settings_effort("max"): + asyncio.run(run()) + calls = provider.chat.call_args_list + self.assertEqual(len(calls), 3, "expected main → subagent → main") + return [_wire_effort(call) for call in calls] + + def test_general_purpose_follows_the_session_level(self): + self.assertEqual(self._delegate("general-purpose", "medium"), ["medium", "medium", "medium"]) + + def test_explore_runs_low_while_the_session_stays_max(self): + self.assertEqual(self._delegate("Explore", "max"), ["max", "low", "max"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_team_runtime_e2e.py b/tests/test_team_runtime_e2e.py index a918eb7a2..e75896f6c 100644 --- a/tests/test_team_runtime_e2e.py +++ b/tests/test_team_runtime_e2e.py @@ -592,3 +592,21 @@ def test_automatic_task_claim_is_atomic_across_competing_workers(team): claimed = [task["id"] for task in claims if task] assert len(claimed) == len(set(claimed)) == len(ids) assert set(claimed) == ids + + +def test_teammate_turns_run_at_the_leaders_level(team, monkeypatch): + # A teammate turn is a run_agent over the leader's context, so it takes the + # level the leader's query captured there — not the persisted + # settings.effort that every subagent used to fall back to. + provider, context, registry = team + effort_seen = [] + original = provider.chat + + def recording_chat(messages, tools=None, **kwargs): + effort_seen.append((kwargs.get("extra_body") or {}).get("reasoning_effort")) + return original(messages, tools=tools, **kwargs) + + monkeypatch.setattr(provider, "chat", recording_chat) + context.thinking_effort = "high" + spawn(team, "alice") + assert effort_seen and set(effort_seen) == {"high"} diff --git a/tests/workflow/test_runner_integration.py b/tests/workflow/test_runner_integration.py index 852bf8480..80dfb875c 100644 --- a/tests/workflow/test_runner_integration.py +++ b/tests/workflow/test_runner_integration.py @@ -198,3 +198,31 @@ def _mk(resolver): ) assert captured == ["inherit", "haiku", "opus"] + + +async def test_runner_agents_run_at_the_parent_level(tmp_path): + """A workflow agent's requests carry the level of the query that started + the workflow (``ToolContext.thinking_effort``, captured by query()), not + the persisted settings.effort that every subagent fell back to before.""" + effort_seen: list = [] + + class _Recording(_ScriptedProvider): + def chat(self, messages, tools=None, **kwargs): + effort_seen.append((kwargs.get("extra_body") or {}).get("reasoning_effort")) + return super().chat(messages, tools=tools, **kwargs) + + provider = _Recording([_resp("done")]) + registry = build_default_registry(provider=provider) + ctx = ToolContext(workspace_root=tmp_path) + ctx.thinking_effort = "high" + runner = LiveAgentRunner( + provider=provider, + tool_registry=registry, + parent_context=ctx, + base_tools=list(registry.list_tools()), + resolve_agent=lambda _t: GENERAL_PURPOSE_AGENT, + run_id="wf_etest", + max_turns=2, + ) + await runner.run(AgentSpec(prompt="p"), abort=create_abort_controller(), index="0") + assert effort_seen == ["high"]