Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 2 additions & 3 deletions docs/envs/domino/continuous-perception.md
Original file line number Diff line number Diff line change
Expand Up @@ -89,9 +89,8 @@ contiguous recorded motion. Re-time an episode after the merge.
`after_step` returns `obs` **unchanged**. Shipping is a pure write-only side
effect, so *when* it happens is unobservable to the rollout — deferring every
chunk to the end produces a bit-identical twin trajectory. Everything reading
state mid-episode (`subgoal_annotations` monitor,
`agent_bilevel_max_execution_replans`, `terminate_on_goal_reached`) reads the
twin's own deterministic simulation either way.
state mid-episode (`terminate_on_goal_reached`) reads the twin's own
deterministic simulation either way.

**No protocol change is needed.** `execute_chunks` already packs a list of chunks
into one `StepRequest` (`real_robot_bridge.py:176-185`), and `_split_actions` is
Expand Down
31 changes: 10 additions & 21 deletions predicators/agent_sdk/belief_probe.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,15 +38,15 @@
Sequence, Set, Tuple, Union

from predicators import utils
from predicators.agent_sdk.config import RefinementConfig, ToolSurfaceConfig, \
ValidationConfig
from predicators.agent_sdk.config import RefinementConfig, ToolSurfaceConfig
from predicators.agent_sdk.parallel_rollouts import prefetch_parallel
from predicators.agent_sdk.tools.context import absolute_rollout_seed, \
decorrelated_rollout_seed
from predicators.agent_sdk.tools.scene import apply_state_modifications, \
draw_pybullet_annotation, render_pybullet_image, render_scene_image
from predicators.agent_sdk.tools.verdicts import _EvalStateCollector, \
evaluate_states_with, load_ground_sampler_fns, make_solved_check
_policy_source_path, evaluate_states_with, load_ground_sampler_fns, \
make_solved_check
from predicators.structs import State, Task, excluded_object_type_names

if TYPE_CHECKING:
Expand All @@ -62,29 +62,17 @@ class ProbeBudgetExceeded(Exception):
"""A probe call ran past a wall-clock budget.

Raised cooperatively at probe checkpoints (every sim call) when the
run_python per-call limit or the solve attempt's wall clock has
expired. ``run_python`` catches it specially: the code's printed
run_python per-call limit has expired. ``run_python`` catches it
specially: the code's printed
output so far is returned with the budget message appended, so a
stopped sweep still hands the agent its partial results.
"""


def _check_time_budget(ctx: "ToolContext") -> None:
"""Raise :class:`ProbeBudgetExceeded` when a wall-clock budget is up.

Never fires during the final-submission nudge
(``ctx.capture_best_effort_plan``): with the budget spent, the one
thing left is submitting, and blocking that would forfeit the task.
"""
if ctx.capture_best_effort_plan:
return
"""Raise :class:`ProbeBudgetExceeded` when the run_python call's time limit
is up."""
now = time.monotonic()
attempt_dl = ctx.attempt_deadline
if attempt_dl is not None and now > attempt_dl:
raise ProbeBudgetExceeded(
"the attempt's wall-clock exploration budget is exhausted. Stop "
"exploring NOW and submit your single best plan via "
"submit_plan on the current task (omit task_idx).")
call_dl = ctx.python_call_deadline
if call_dl is not None and now > call_dl:
call_timeout = ToolSurfaceConfig.from_cfg().python_call_timeout
Expand Down Expand Up @@ -930,7 +918,9 @@ def _option_model(self) -> Any:

def _fresh_scope(self) -> Optional[Callable[..., Any]]:
"""Select isolation for the model this probe actually executes."""
if not ValidationConfig.from_cfg().fresh_env:
# pylint: disable-next=import-outside-toplevel
from predicators.settings import CFG
if not CFG.agent_plan_validation_fresh_env:
return None
ctx = self._ctx
if ctx.probe_option_model_provider is not None:
Expand Down Expand Up @@ -2561,7 +2551,6 @@ def run_policy(

from predicators.agent_sdk.policy_execution import \
build_policy_option_fn, execute_policy_forward
from predicators.agent_sdk.tools.testing import _policy_source_path
from predicators.settings import CFG

# pylint: enable=import-outside-toplevel
Expand Down
8 changes: 0 additions & 8 deletions predicators/agent_sdk/bilevel_sketch.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,9 +8,6 @@
- ``sketch_types``: shared dataclasses (``GroundSampler``,
``SketchStep``) that parsing constructs and refinement/execution
consume.
- ``sketch_prompts``: ``build_solve_system_prompt`` and
``build_solve_prompt``, the solve/explore system-prompt and query
builders (rendered from ``prompts/*.md``).
- ``sketch_parsing``: the sketch-line grammar - step/plan formatters
and the parsers for subgoal / ``~`` ground-sampler annotations and
continuous params.
Expand All @@ -27,8 +24,6 @@
parse_region_annotations, parse_sketch_from_text, \
parse_subgoal_annotations, strip_code_fences, strip_region_annotations, \
strip_subgoal_annotations
from predicators.agent_sdk.sketch_prompts import build_early_stop_note, \
build_solve_prompt, build_solve_system_prompt
from predicators.agent_sdk.sketch_refinement import DeepestFailure, \
InfoScorer, RefineOutcome, StepProbeSuggestion, ground_step, \
refine_and_validate_report, refine_sketch, resolve_refine_timeout, \
Expand All @@ -44,9 +39,6 @@
"SketchStep",
"StepOutcome",
"StepProbeSuggestion",
"build_early_stop_note",
"build_solve_prompt",
"build_solve_system_prompt",
"execute_plan_forward",
"format_plan_lines",
"format_sketch_lines",
Expand Down
49 changes: 1 addition & 48 deletions predicators/agent_sdk/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -49,64 +49,17 @@ def from_cfg(cls) -> "SessionConfig":

@dataclass(frozen=True)
class RefinementConfig:
"""Plan-sketch refinement: search budgets, gates, and ground samplers.

Consumed at handler entry by ``submit_plan`` (tools/testing.py) and
by the probe's ``refine`` (belief_probe.py).
"""
"""Plan-sketch refinement settings the probe's ``refine`` reads."""
ground_samplers: bool
refinement_timeout_per_step: float
refinement_timeout_min: float
max_samples_per_step: int
check_subgoals: bool
log_state: bool
use_llm_initial_params: bool

@classmethod
def from_cfg(cls) -> "RefinementConfig":
"""Read the refinement flags from the live ``CFG``."""
# Flags keep their names for experiment-yaml compatibility.
return cls(
ground_samplers=CFG.agent_bilevel_ground_samplers,
refinement_timeout_per_step=(
CFG.agent_bilevel_refinement_timeout_per_step),
refinement_timeout_min=CFG.agent_bilevel_refinement_timeout_min,
max_samples_per_step=CFG.agent_bilevel_max_samples_per_step,
check_subgoals=CFG.agent_bilevel_check_subgoals,
log_state=CFG.agent_bilevel_log_state,
use_llm_initial_params=CFG.agent_bilevel_use_llm_initial_params,
)


@dataclass(frozen=True)
class ValidationConfig:
"""Capture-validation rollouts and the cross-attempt journal.

Consumed at handler entry by ``submit_plan`` (tools.py) and the
probe's ``run(trials=N)`` (belief_probe.py); ``use_journal`` gates
the journal / attempt-log channel.
"""
rollouts: int
rollouts_after_flaky: int
fresh_env: bool
physics_margin: bool
rule_param_margin: bool
necessity: bool
use_journal: bool

@classmethod
def from_cfg(cls) -> "ValidationConfig":
"""Read the validation flags from the live ``CFG``."""
# Flags keep their names for experiment-yaml compatibility.
return cls(
rollouts=CFG.agent_plan_validation_rollouts,
rollouts_after_flaky=(
CFG.agent_plan_validation_rollouts_after_flaky),
fresh_env=CFG.agent_plan_validation_fresh_env,
physics_margin=CFG.agent_plan_validation_physics_margin,
rule_param_margin=CFG.agent_plan_validation_rule_param_margin,
necessity=CFG.agent_plan_validation_necessity,
use_journal=CFG.agent_solve_use_journal,
)


Expand Down
167 changes: 28 additions & 139 deletions predicators/agent_sdk/journal.py
Original file line number Diff line number Diff line change
@@ -1,35 +1,24 @@
"""Persistent per-run solve journal and attempt log.

Two markdown files in the sandbox that carry knowledge across solve
attempts, test tasks, and learning cycles:

- ``journal.md`` is the AGENT's notebook. Solve and learn sessions
append to it with the ordinary file tools (no dedicated tool): short
factual entries - what was tried with exact parameters, what was
measured, what to try differently. The prompts ask for facts and
measurements rather than verdicts: a recorded "X is impossible"
from a failed attempt would re-import exactly the anchoring a fresh
context is meant to shed, while "tried yaws 0-15 deg at x in
[0.50, 0.54], all stopped >=5 cm short" steers the next attempt
without foreclosing it.
- ``attempts.md`` is the HARNESS's log, never edited by the agent:
each task's goal + initial state (once per task) and each
attempt's outcome and captured or best refused plan, so the
essentials of every attempt are on record even when the agent
writes nothing.

Fresh-context solve sessions read both from their prompt (tail-capped
so recent attempts stay intact), so knowledge travels through these
curated channels instead of raw transcript history.

Phase lifecycle: learning-phase content persists for the whole run
and accumulates across online-learning cycles, so every evaluation
starts from all learning knowledge so far. Test-phase additions live
only for their own evaluation: at ``end_test_phase`` the approach
archives both files to the run's log dir (outside the sandbox, so the
agent cannot read them) and rolls them back to their pre-test content
via :func:`read_raw` / :func:`restore` - entries written while
solving one evaluation's test tasks must not leak into the next.
"""Persistent per-run journal and round log.

Two markdown files in the sandbox that carry knowledge across the rounds
and levels of a continual run:

- ``journal.md`` is the AGENT's notebook. The agent appends to it with
the ordinary file tools (no dedicated tool): short factual entries -
what was tried with exact parameters, what was measured, what to try
differently. The prompts ask for facts and measurements rather than
verdicts: a recorded "X is impossible" from a failed attempt would
re-import exactly the anchoring a fresh context is meant to shed,
while "tried yaws 0-15 deg at x in [0.50, 0.54], all stopped >=5 cm
short" steers the next attempt without foreclosing it.
- ``attempts.md`` is the HARNESS's log, never edited by the agent: one
entry per round (what the agent did in the environment) and per model
change, so the essentials of every round are on record even when the
agent writes nothing.

Each round's query injects both (tail-capped so recent entries stay
intact), so knowledge travels through these curated channels instead of
raw transcript history.
"""

from __future__ import annotations
Expand All @@ -38,133 +27,33 @@
from typing import Optional

JOURNAL_FILENAME = "journal.md"
# The harness-owned attempt log (task contexts, attempt outcomes).
# The harness-owned round log.
ATTEMPTS_FILENAME = "attempts.md"

# Per-entry cap for harness attempt-log entries: the first entry per
# task embeds the init-state feature dict (the prompt's own
# representation) and a captured plan. The writer orders the layout
# block last, so tail truncation at this cap can only ever cut layout,
# never the outcome or the captured plan.
# Per-entry cap for harness log entries.
MAX_ENTRY_CHARS = 4000
MAX_AUTO_ENTRY_CHARS = MAX_ENTRY_CHARS
# Cap on how much of each file is injected into a solve prompt.
# Tail-biased: recent attempts (usually the same task) matter most.
# Cap on how much of each file is injected into a query. Tail-biased:
# recent rounds matter most.
MAX_PROMPT_CHARS = 6000

# The learn-phase-maintained domain strategy document. Unlike the
# append-only journal (facts and measurements), strategy.md is a LIVING
# document the learn agent rewrites freely each cycle: its best current
# natural-language account of how to solve tasks in this domain. Solve
# prompts inject it as explicitly-advisory reference.
STRATEGY_FILENAME = "strategy.md"

# Cap on how much strategy is injected into a solve prompt. Head-biased
# (unlike the journal): the document is curated, so its lead carries the
# headline strategy and a tail truncation only cuts detail.
MAX_STRATEGY_PROMPT_CHARS = 4000


def journal_path(sandbox_dir: str) -> str:
"""Host path of the run's journal file."""
return os.path.join(sandbox_dir, JOURNAL_FILENAME)


def attempts_path(sandbox_dir: str) -> str:
"""Host path of the run's harness-owned attempt log."""
return os.path.join(sandbox_dir, ATTEMPTS_FILENAME)


def strategy_path(sandbox_dir: str) -> str:
"""Host path of the run's domain strategy document."""
return os.path.join(sandbox_dir, STRATEGY_FILENAME)


def read_strategy(sandbox_dir: Optional[str],
max_chars: int = MAX_STRATEGY_PROMPT_CHARS) -> str:
"""Strategy document content for prompt injection ("" when absent).

Head-biased truncation: the document is curated by the learn agent,
so the front holds the headline strategy; a truncation notice marks
the cut so readers know detail was dropped.
"""
if not sandbox_dir:
return ""
path = strategy_path(sandbox_dir)
if not os.path.isfile(path):
return ""
with open(path, "r", encoding="utf-8") as f:
content = f.read().strip()
if len(content) > max_chars:
# Cut at a line boundary, never mid-word.
head = content[:max_chars]
cut = head.rfind("\n")
if cut > 0:
head = head[:cut]
content = (head.rstrip() +
"\n[strategy truncated at the prompt cap - read "
f"./{STRATEGY_FILENAME} for the rest]")
return content


def append_entry(sandbox_dir: str,
header: str,
body: str,
max_chars: int = MAX_ENTRY_CHARS,
filename: str = ATTEMPTS_FILENAME) -> Optional[str]:
"""Append one harness entry; returns a truncation notice or None.
filename: str = ATTEMPTS_FILENAME) -> None:
"""Append one harness entry.

``header`` becomes a ``### <header>`` line; ``body`` is written
verbatim below it, truncated at ``max_chars`` (default
:data:`MAX_ENTRY_CHARS`; harness auto-entries pass
:data:`MAX_AUTO_ENTRY_CHARS`).
verbatim below it, truncated at ``max_chars``.
"""
os.makedirs(sandbox_dir, exist_ok=True)
note: Optional[str] = None
body = body.strip()
if len(body) > max_chars:
body = body[:max_chars].rstrip()
body += "\n[entry truncated at the per-entry size cap]"
note = (f"entry truncated to {max_chars} chars - keep journal "
"entries short and factual")
with open(os.path.join(sandbox_dir, filename), "a", encoding="utf-8") as f:
f.write(f"### {header.strip()}\n{body}\n\n")
return note


def read_raw(sandbox_dir: Optional[str],
filename: str = JOURNAL_FILENAME) -> Optional[str]:
"""Exact file content, or None if the file does not exist.

Unlike :func:`read_journal` there is no prompt trimming and the
absent-file case is distinguishable from an empty file, so the
result is a faithful snapshot for :func:`restore`.
"""
if not sandbox_dir:
return None
path = os.path.join(sandbox_dir, filename)
if not os.path.isfile(path):
return None
with open(path, "r", encoding="utf-8") as f:
return f.read()


def restore(sandbox_dir: str,
snapshot: Optional[str],
filename: str = JOURNAL_FILENAME) -> None:
"""Reset the file to a :func:`read_raw` snapshot.

A ``None`` snapshot means the file did not exist, so it is removed
if present.
"""
path = os.path.join(sandbox_dir, filename)
if snapshot is None:
if os.path.isfile(path):
os.remove(path)
return
os.makedirs(sandbox_dir, exist_ok=True)
with open(path, "w", encoding="utf-8") as f:
f.write(snapshot)


def read_journal(sandbox_dir: Optional[str],
Expand Down
Loading
Loading