Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
102 changes: 102 additions & 0 deletions core/agent_runtime/evidence_ledger.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,102 @@
"""Advisory no-progress reminders — the evidence half of loop-breaking.

``repeat_guard`` watches *call* repetition: identical consecutive calls earn an
escalating reminder. That misses the other classic loop, where the model
interleaves different calls (A, B, A, B, ...) and every attempt hands back the
same evidence it already had — no call is ever repeated consecutively, yet the
run is learning nothing.

``EvidenceLedger`` keys on the call *and its result*: for each canonical call
signature it counts the result fingerprints already seen. When one pair comes
back ``threshold`` times the call provably is not making progress, and a
reminder is emitted.

The fingerprint is the normalized tool result — the exact content the model
read, truncation included. Two results the model cannot tell apart are the same
evidence, which is the property that matters here.

Same discipline as ``repeat_guard``: advisory only, it never delays, rewrites,
or blocks a call, the decision stays entirely with the model, and it is the
FIRST line of defense in front of any hard stop (``max_iterations``,
``should_stop_callback``). The reminder never quotes tool output: results are
unbounded and may be attacker-controlled, so they stay out of the prompt. The
runner injects these through its existing reminder channel, skipping any call
``repeat_guard`` already flagged, so one iteration injects at most one reminder
per call.
"""

from __future__ import annotations

import hashlib
import json
from typing import Any

DEFAULT_NO_PROGRESS_THRESHOLD = 3
# A long run can call with unbounded argument variety, and only recent evidence
# matters for spotting a stall — so the ledger keeps a bounded window of calls.
_MAX_TRACKED_CALLS = 256


def _canonicalize(value: Any) -> str:
"""Order must not defeat detection (same key as ``repeat_guard``)."""
try:
return json.dumps(value, sort_keys=True, ensure_ascii=False, default=str)
except (TypeError, ValueError):
return repr(value)


def _fingerprint(result: Any) -> str:
"""Fingerprint what the model read, whatever shape the tool handed back.

Results cross a dynamic boundary: most tools return text, but a tool is free
to return structured content (the Goal tools return dicts) and ``content``
carries it through untouched. A non-text result therefore gets the same
canonical form as a call signature — the run must never fail just because a
reminder could not be computed.
"""
text = result if isinstance(result, str) else _canonicalize(result)
return hashlib.sha256(text.encode("utf-8", "replace")).hexdigest()[:16]


def _validated(threshold: int) -> int:
if isinstance(threshold, bool) or not isinstance(threshold, int) or threshold < 2:
raise ValueError("no-progress threshold must be an int >= 2")
return threshold


def _no_progress_reminder(tool_name: str, count: int) -> str:
return (
f"Reminder: `{tool_name}` has now returned the same result {count} times "
"in this run, counting attempts separated by other calls. Repeating it "
"is unlikely to produce new evidence — change the approach instead: vary "
"the arguments, use a different tool, or say what you are blocked on."
)


class EvidenceLedger:
"""Counts how often each (canonical call, result) pair has been observed."""

def __init__(self, threshold: int = DEFAULT_NO_PROGRESS_THRESHOLD) -> None:
self.threshold = _validated(threshold)
# call signature -> {result fingerprint: count}, kept in recency order
# so the least recently used call is the one evicted at the bound.
self._evidence: dict[tuple[str, str], dict[str, int]] = {}

def observe(self, tool_name: str, arguments: Any, result: Any) -> str | None:
"""Record one finished call; return a reminder when it repeats evidence."""
signature = (tool_name, _canonicalize(arguments))
seen = self._evidence.pop(signature, None)
if seen is None:
seen = {}
self._evidence[signature] = seen
while len(self._evidence) > _MAX_TRACKED_CALLS:
self._evidence.pop(next(iter(self._evidence)))
fingerprint = _fingerprint(result)
count = seen.get(fingerprint, 0) + 1
seen[fingerprint] = count
if count == self.threshold:
return _no_progress_reminder(tool_name, count)
return None


__all__ = ["DEFAULT_NO_PROGRESS_THRESHOLD", "EvidenceLedger"]
54 changes: 47 additions & 7 deletions core/agent_runtime/runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -43,6 +43,10 @@
DEFAULT_TOKEN_METER_FACTORY,
TokenMeter,
)
from core.agent_runtime.evidence_ledger import (
DEFAULT_NO_PROGRESS_THRESHOLD,
EvidenceLedger,
)
from core.agent_runtime.hook import AgentHook, AgentHookContext
from core.agent_runtime.pruner import ToolResultPruner
from core.agent_runtime.repeat_guard import (
Expand Down Expand Up @@ -156,6 +160,14 @@ class AgentRunSpec:
# run lengths of identical consecutive tool calls that earn an escalating
# reminder instead of a hard stop. ``None`` disables the guard.
repeat_call_thresholds: tuple[int, ...] | None = DEFAULT_REPEAT_THRESHOLDS
# Advisory no-progress reminders (see core.agent_runtime.evidence_ledger):
# how often one call may hand back the SAME result — consecutive or
# interleaved — before a reminder is injected. Complements
# ``repeat_call_thresholds``, which only sees consecutive identical calls
# and is therefore structurally blind to an A, B, A, B, ... stall. A call
# the tracker already flagged is left to the tracker, so one iteration
# injects at most one reminder per call. ``None`` disables the ledger.
evidence_ledger_threshold: int | None = DEFAULT_NO_PROGRESS_THRESHOLD
# Model-visible means logged (the dsh session-log rule): mid-turn messages
# the runner itself adds to the PERSISTED model history — injected
# sub-agent results, Goal updates, repeat-call reminders, length-recovery
Expand Down Expand Up @@ -470,6 +482,11 @@ async def run(self, spec: AgentRunSpec) -> AgentRunResult:
if spec.repeat_call_thresholds is not None
else None
)
evidence_ledger = (
EvidenceLedger(spec.evidence_ledger_threshold)
if spec.evidence_ledger_threshold is not None
else None
)
empty_content_retries = 0
length_recovery_count = 0
overflow_recoveries = 0
Expand Down Expand Up @@ -651,18 +668,41 @@ async def record_compaction_response(response: LLMResponse) -> None:
}
messages.append(tool_message)
completed_tool_results.append(tool_message)
if repeat_tracker is not None:
if repeat_tracker is not None or evidence_ledger is not None:
# Observed at the result boundary so denied and failed
# calls count too — a model hammering a rejected call is
# exactly the loop worth interrupting. The reminder rides
# a user message AFTER the results, so the model reads
# what happened and then why it should change course.
reminders = [
reminder
for tc in response.tool_calls
if (reminder := repeat_tracker.observe(tc.name, tc.arguments))
is not None
]
reminders: list[str] = []
flagged: set[str] = set()
if repeat_tracker is not None:
for tc in response.tool_calls:
reminder = repeat_tracker.observe(tc.name, tc.arguments)
if reminder is not None:
reminders.append(reminder)
flagged.add(tc.id)
if evidence_ledger is not None:
# The evidence half: a call that keeps returning the
# same result without ever repeating consecutively is
# invisible to the tracker above. Calls it already
# flagged stay with it, so each call contributes at
# most one reminder to this batch. Both halves share
# this one message (and so one note source): stacking
# consecutive user turns would not survive providers
# that require alternating roles.
for tc, result_message in zip(
response.tool_calls, completed_tool_results
):
if tc.id in flagged:
continue
reminder = evidence_ledger.observe(
tc.name,
tc.arguments,
result_message.get("content", ""),
)
if reminder is not None:
reminders.append(reminder)
if reminders:
reminder_text = "\n\n".join(reminders)
messages.append(
Expand Down
Loading
Loading